crypton 2.1.7 → 2.1.8
raw patch · 238 files changed
+67189/−2160 lines, 238 filesPVP: major bump suggested
API removals or changes: PVP suggests a major version bump
API changes (from Hackage documentation)
- Crypto.ECC: instance Control.DeepSeq.NFData Crypto.ECC.SharedSecret
- Crypto.ECC: instance Data.ByteArray.Types.ByteArrayAccess Crypto.ECC.SharedSecret
- Crypto.ECC: instance GHC.Base.Monoid Crypto.ECC.SharedSecret
- Crypto.ECC: instance GHC.Base.Semigroup Crypto.ECC.SharedSecret
- Crypto.ECC: instance GHC.Classes.Eq Crypto.ECC.SharedSecret
+ Crypto.Error: CryptoError_PublicKeyStructureInvalid :: CryptoError
+ Crypto.KEM: -- Both are secret, and both determine the shared secret completely.
+ Crypto.KEM: -- bytes; in DHKEM it is the ephemeral secret key, a scalar of the group.
+ Crypto.KEM: -- caller supply it. In ML-KEM it is <tt>m</tt> of FIPS 203, a string of
+ Crypto.KEM: -- | The randomness <a>encapsulate</a> draws, for the instances that let a
+ Crypto.KEM: SharedSecret :: ScrubbedBytes -> SharedSecret
+ Crypto.KEM: class KEM kem where {
+ Crypto.KEM: decapsulate :: KEM kem => proxy kem -> DecapsulationKey kem -> Ciphertext kem -> CryptoFailable SharedSecret
+ Crypto.KEM: encapsulate :: (KEM kem, MonadRandom m) => proxy kem -> EncapsulationKey kem -> m (CryptoFailable (Ciphertext kem, SharedSecret))
+ Crypto.KEM: encapsulateWith :: KEM kem => proxy kem -> EncapsulationKey kem -> Coins kem -> CryptoFailable (Ciphertext kem, SharedSecret)
+ Crypto.KEM: generateKeyPair :: (KEM kem, MonadRandom m) => proxy kem -> m (EncapsulationKey kem, DecapsulationKey kem)
+ Crypto.KEM: instance Control.DeepSeq.NFData Crypto.KEM.SharedSecret
+ Crypto.KEM: instance Data.ByteArray.Types.ByteArrayAccess Crypto.KEM.SharedSecret
+ Crypto.KEM: instance GHC.Base.Monoid Crypto.KEM.SharedSecret
+ Crypto.KEM: instance GHC.Base.Semigroup Crypto.KEM.SharedSecret
+ Crypto.KEM: instance GHC.Classes.Eq Crypto.KEM.SharedSecret
+ Crypto.KEM: instance GHC.Show.Show Crypto.KEM.SharedSecret
+ Crypto.KEM: newtype SharedSecret
+ Crypto.KEM: type Ciphertext kem;
+ Crypto.KEM: type Coins kem;
+ Crypto.KEM: type DecapsulationKey kem;
+ Crypto.KEM: type EncapsulationKey kem;
+ Crypto.KEM: }
+ Crypto.PubKey.MLDSA: MLDSA44 :: MLDSA44
+ Crypto.PubKey.MLDSA: MLDSA65 :: MLDSA65
+ Crypto.PubKey.MLDSA: MLDSA87 :: MLDSA87
+ Crypto.PubKey.MLDSA: class MLDSA p
+ Crypto.PubKey.MLDSA: context :: ByteArrayAccess ba => ba -> CryptoFailable Context
+ Crypto.PubKey.MLDSA: data Context
+ Crypto.PubKey.MLDSA: data MLDSA44
+ Crypto.PubKey.MLDSA: data MLDSA65
+ Crypto.PubKey.MLDSA: data MLDSA87
+ Crypto.PubKey.MLDSA: data Mu
+ Crypto.PubKey.MLDSA: data MuContext
+ Crypto.PubKey.MLDSA: data Signature p
+ Crypto.PubKey.MLDSA: data SigningKey p
+ Crypto.PubKey.MLDSA: data VerificationKey p
+ Crypto.PubKey.MLDSA: emptyContext :: Context
+ Crypto.PubKey.MLDSA: generateKeyPair :: forall p proxy m. (MLDSA p, MonadRandom m) => proxy p -> m (VerificationKey p, SigningKey p)
+ Crypto.PubKey.MLDSA: generateKeyPairAndSeed :: forall p proxy m. (MLDSA p, MonadRandom m) => proxy p -> m (VerificationKey p, SigningKey p, ScrubbedBytes)
+ Crypto.PubKey.MLDSA: instance Control.DeepSeq.NFData (Crypto.PubKey.MLDSA.Signature p)
+ Crypto.PubKey.MLDSA: instance Control.DeepSeq.NFData (Crypto.PubKey.MLDSA.SigningKey p)
+ Crypto.PubKey.MLDSA: instance Control.DeepSeq.NFData (Crypto.PubKey.MLDSA.VerificationKey p)
+ Crypto.PubKey.MLDSA: instance Control.DeepSeq.NFData Crypto.PubKey.MLDSA.Context
+ Crypto.PubKey.MLDSA: instance Control.DeepSeq.NFData Crypto.PubKey.MLDSA.Mu
+ Crypto.PubKey.MLDSA: instance Crypto.Debug.DebugShow (Crypto.PubKey.MLDSA.SigningKey p)
+ Crypto.PubKey.MLDSA: instance Crypto.PubKey.MLDSA.MLDSA Crypto.PubKey.MLDSA.MLDSA44
+ Crypto.PubKey.MLDSA: instance Crypto.PubKey.MLDSA.MLDSA Crypto.PubKey.MLDSA.MLDSA65
+ Crypto.PubKey.MLDSA: instance Crypto.PubKey.MLDSA.MLDSA Crypto.PubKey.MLDSA.MLDSA87
+ Crypto.PubKey.MLDSA: instance Data.ByteArray.Types.ByteArrayAccess (Crypto.PubKey.MLDSA.Signature p)
+ Crypto.PubKey.MLDSA: instance Data.ByteArray.Types.ByteArrayAccess (Crypto.PubKey.MLDSA.SigningKey p)
+ Crypto.PubKey.MLDSA: instance Data.ByteArray.Types.ByteArrayAccess (Crypto.PubKey.MLDSA.VerificationKey p)
+ Crypto.PubKey.MLDSA: instance Data.ByteArray.Types.ByteArrayAccess Crypto.PubKey.MLDSA.Context
+ Crypto.PubKey.MLDSA: instance Data.ByteArray.Types.ByteArrayAccess Crypto.PubKey.MLDSA.Mu
+ Crypto.PubKey.MLDSA: instance GHC.Classes.Eq (Crypto.PubKey.MLDSA.Signature p)
+ Crypto.PubKey.MLDSA: instance GHC.Classes.Eq (Crypto.PubKey.MLDSA.SigningKey p)
+ Crypto.PubKey.MLDSA: instance GHC.Classes.Eq (Crypto.PubKey.MLDSA.VerificationKey p)
+ Crypto.PubKey.MLDSA: instance GHC.Classes.Eq Crypto.PubKey.MLDSA.Context
+ Crypto.PubKey.MLDSA: instance GHC.Classes.Eq Crypto.PubKey.MLDSA.MLDSA44
+ Crypto.PubKey.MLDSA: instance GHC.Classes.Eq Crypto.PubKey.MLDSA.MLDSA65
+ Crypto.PubKey.MLDSA: instance GHC.Classes.Eq Crypto.PubKey.MLDSA.MLDSA87
+ Crypto.PubKey.MLDSA: instance GHC.Classes.Eq Crypto.PubKey.MLDSA.Mu
+ Crypto.PubKey.MLDSA: instance GHC.Show.Show (Crypto.PubKey.MLDSA.Signature p)
+ Crypto.PubKey.MLDSA: instance GHC.Show.Show (Crypto.PubKey.MLDSA.SigningKey p)
+ Crypto.PubKey.MLDSA: instance GHC.Show.Show (Crypto.PubKey.MLDSA.VerificationKey p)
+ Crypto.PubKey.MLDSA: instance GHC.Show.Show Crypto.PubKey.MLDSA.Context
+ Crypto.PubKey.MLDSA: instance GHC.Show.Show Crypto.PubKey.MLDSA.MLDSA44
+ Crypto.PubKey.MLDSA: instance GHC.Show.Show Crypto.PubKey.MLDSA.MLDSA65
+ Crypto.PubKey.MLDSA: instance GHC.Show.Show Crypto.PubKey.MLDSA.MLDSA87
+ Crypto.PubKey.MLDSA: instance GHC.Show.Show Crypto.PubKey.MLDSA.Mu
+ Crypto.PubKey.MLDSA: keyPairFromSeed :: forall p proxy ba. (MLDSA p, ByteArrayAccess ba) => proxy p -> ba -> CryptoFailable (VerificationKey p, SigningKey p)
+ Crypto.PubKey.MLDSA: maxContextLength :: Int
+ Crypto.PubKey.MLDSA: messageRepresentative :: (MLDSA p, ByteArrayAccess msg) => VerificationKey p -> Context -> msg -> Mu
+ Crypto.PubKey.MLDSA: mu :: ByteArrayAccess ba => ba -> CryptoFailable Mu
+ Crypto.PubKey.MLDSA: muFinalize :: MuContext -> Mu
+ Crypto.PubKey.MLDSA: muInit :: MLDSA p => VerificationKey p -> Context -> MuContext
+ Crypto.PubKey.MLDSA: muSize :: Int
+ Crypto.PubKey.MLDSA: muUpdate :: ByteArrayAccess msg => MuContext -> msg -> MuContext
+ Crypto.PubKey.MLDSA: muUpdates :: ByteArrayAccess msg => MuContext -> [msg] -> MuContext
+ Crypto.PubKey.MLDSA: seedSize :: Int
+ Crypto.PubKey.MLDSA: sign :: (MLDSA p, MonadRandom m, ByteArrayAccess msg) => SigningKey p -> Context -> msg -> m (Signature p)
+ Crypto.PubKey.MLDSA: signDeterministic :: (MLDSA p, ByteArrayAccess msg) => SigningKey p -> Context -> msg -> Signature p
+ Crypto.PubKey.MLDSA: signExternalMu :: (MLDSA p, MonadRandom m) => SigningKey p -> Mu -> m (Signature p)
+ Crypto.PubKey.MLDSA: signExternalMuDeterministic :: MLDSA p => SigningKey p -> Mu -> Signature p
+ Crypto.PubKey.MLDSA: signExternalMuWith :: (MLDSA p, ByteArrayAccess rnd) => SigningKey p -> Mu -> rnd -> CryptoFailable (Signature p)
+ Crypto.PubKey.MLDSA: signWith :: (MLDSA p, ByteArrayAccess msg, ByteArrayAccess rnd) => SigningKey p -> Context -> msg -> rnd -> CryptoFailable (Signature p)
+ Crypto.PubKey.MLDSA: signature :: (MLDSA p, ByteArrayAccess ba) => ba -> CryptoFailable (Signature p)
+ Crypto.PubKey.MLDSA: signatureSize :: MLDSA p => proxy p -> Int
+ Crypto.PubKey.MLDSA: signingKey :: (MLDSA p, ByteArrayAccess ba) => ba -> CryptoFailable (SigningKey p)
+ Crypto.PubKey.MLDSA: signingKeySize :: MLDSA p => proxy p -> Int
+ Crypto.PubKey.MLDSA: signingRandomnessSize :: Int
+ Crypto.PubKey.MLDSA: toPublic :: MLDSA p => SigningKey p -> VerificationKey p
+ Crypto.PubKey.MLDSA: verificationKey :: (MLDSA p, ByteArrayAccess ba) => ba -> CryptoFailable (VerificationKey p)
+ Crypto.PubKey.MLDSA: verificationKeySize :: MLDSA p => proxy p -> Int
+ Crypto.PubKey.MLDSA: verify :: (MLDSA p, ByteArrayAccess msg) => VerificationKey p -> Context -> msg -> Signature p -> Bool
+ Crypto.PubKey.MLDSA: verifyExternalMu :: MLDSA p => VerificationKey p -> Mu -> Signature p -> Bool
+ Crypto.PubKey.MLKEM: -- Both are secret, and both determine the shared secret completely.
+ Crypto.PubKey.MLKEM: -- bytes; in DHKEM it is the ephemeral secret key, a scalar of the group.
+ Crypto.PubKey.MLKEM: -- caller supply it. In ML-KEM it is <tt>m</tt> of FIPS 203, a string of
+ Crypto.PubKey.MLKEM: -- | The randomness <a>encapsulate</a> draws, for the instances that let a
+ Crypto.PubKey.MLKEM: MLKEM1024 :: MLKEM1024
+ Crypto.PubKey.MLKEM: MLKEM512 :: MLKEM512
+ Crypto.PubKey.MLKEM: MLKEM768 :: MLKEM768
+ Crypto.PubKey.MLKEM: SharedSecret :: ScrubbedBytes -> SharedSecret
+ Crypto.PubKey.MLKEM: ciphertext :: (MLKEM p, ByteArrayAccess ba) => ba -> CryptoFailable (Ciphertext p)
+ Crypto.PubKey.MLKEM: ciphertextSize :: MLKEM p => proxy p -> Int
+ Crypto.PubKey.MLKEM: class KEM kem where {
+ Crypto.PubKey.MLKEM: class (KEM p, EncapsulationKey p ~ MLKEMEncapsulationKey p, DecapsulationKey p ~ MLKEMDecapsulationKey p, Ciphertext p ~ MLKEMCiphertext p, Coins p ~ ScrubbedBytes) => MLKEM p
+ Crypto.PubKey.MLKEM: data MLKEM1024
+ Crypto.PubKey.MLKEM: data MLKEM512
+ Crypto.PubKey.MLKEM: data MLKEM768
+ Crypto.PubKey.MLKEM: decapsulate :: KEM kem => proxy kem -> DecapsulationKey kem -> Ciphertext kem -> CryptoFailable SharedSecret
+ Crypto.PubKey.MLKEM: decapsulationKey :: (MLKEM p, ByteArrayAccess ba) => ba -> CryptoFailable (DecapsulationKey p)
+ Crypto.PubKey.MLKEM: decapsulationKeySize :: MLKEM p => proxy p -> Int
+ Crypto.PubKey.MLKEM: encapsulate :: (KEM kem, MonadRandom m) => proxy kem -> EncapsulationKey kem -> m (CryptoFailable (Ciphertext kem, SharedSecret))
+ Crypto.PubKey.MLKEM: encapsulateWith :: KEM kem => proxy kem -> EncapsulationKey kem -> Coins kem -> CryptoFailable (Ciphertext kem, SharedSecret)
+ Crypto.PubKey.MLKEM: encapsulationCoinsSize :: Int
+ Crypto.PubKey.MLKEM: encapsulationKey :: (MLKEM p, ByteArrayAccess ba) => ba -> CryptoFailable (EncapsulationKey p)
+ Crypto.PubKey.MLKEM: encapsulationKeySize :: MLKEM p => proxy p -> Int
+ Crypto.PubKey.MLKEM: generateKeyPair :: (KEM kem, MonadRandom m) => proxy kem -> m (EncapsulationKey kem, DecapsulationKey kem)
+ Crypto.PubKey.MLKEM: generateKeyPairAndSeed :: forall p proxy m. (MLKEM p, MonadRandom m) => proxy p -> m (MLKEMEncapsulationKey p, DecapsulationKey p, ScrubbedBytes)
+ Crypto.PubKey.MLKEM: instance Control.DeepSeq.NFData (Crypto.PubKey.MLKEM.MLKEMCiphertext p)
+ Crypto.PubKey.MLKEM: instance Control.DeepSeq.NFData (Crypto.PubKey.MLKEM.MLKEMDecapsulationKey p)
+ Crypto.PubKey.MLKEM: instance Control.DeepSeq.NFData (Crypto.PubKey.MLKEM.MLKEMEncapsulationKey p)
+ Crypto.PubKey.MLKEM: instance Crypto.Debug.DebugShow (Crypto.PubKey.MLKEM.MLKEMDecapsulationKey p)
+ Crypto.PubKey.MLKEM: instance Crypto.KEM.KEM Crypto.PubKey.MLKEM.MLKEM1024
+ Crypto.PubKey.MLKEM: instance Crypto.KEM.KEM Crypto.PubKey.MLKEM.MLKEM512
+ Crypto.PubKey.MLKEM: instance Crypto.KEM.KEM Crypto.PubKey.MLKEM.MLKEM768
+ Crypto.PubKey.MLKEM: instance Crypto.PubKey.MLKEM.MLKEM Crypto.PubKey.MLKEM.MLKEM1024
+ Crypto.PubKey.MLKEM: instance Crypto.PubKey.MLKEM.MLKEM Crypto.PubKey.MLKEM.MLKEM512
+ Crypto.PubKey.MLKEM: instance Crypto.PubKey.MLKEM.MLKEM Crypto.PubKey.MLKEM.MLKEM768
+ Crypto.PubKey.MLKEM: instance Data.ByteArray.Types.ByteArrayAccess (Crypto.PubKey.MLKEM.MLKEMCiphertext p)
+ Crypto.PubKey.MLKEM: instance Data.ByteArray.Types.ByteArrayAccess (Crypto.PubKey.MLKEM.MLKEMDecapsulationKey p)
+ Crypto.PubKey.MLKEM: instance Data.ByteArray.Types.ByteArrayAccess (Crypto.PubKey.MLKEM.MLKEMEncapsulationKey p)
+ Crypto.PubKey.MLKEM: instance GHC.Classes.Eq (Crypto.PubKey.MLKEM.MLKEMCiphertext p)
+ Crypto.PubKey.MLKEM: instance GHC.Classes.Eq (Crypto.PubKey.MLKEM.MLKEMDecapsulationKey p)
+ Crypto.PubKey.MLKEM: instance GHC.Classes.Eq (Crypto.PubKey.MLKEM.MLKEMEncapsulationKey p)
+ Crypto.PubKey.MLKEM: instance GHC.Classes.Eq Crypto.PubKey.MLKEM.MLKEM1024
+ Crypto.PubKey.MLKEM: instance GHC.Classes.Eq Crypto.PubKey.MLKEM.MLKEM512
+ Crypto.PubKey.MLKEM: instance GHC.Classes.Eq Crypto.PubKey.MLKEM.MLKEM768
+ Crypto.PubKey.MLKEM: instance GHC.Show.Show (Crypto.PubKey.MLKEM.MLKEMCiphertext p)
+ Crypto.PubKey.MLKEM: instance GHC.Show.Show (Crypto.PubKey.MLKEM.MLKEMDecapsulationKey p)
+ Crypto.PubKey.MLKEM: instance GHC.Show.Show (Crypto.PubKey.MLKEM.MLKEMEncapsulationKey p)
+ Crypto.PubKey.MLKEM: instance GHC.Show.Show Crypto.PubKey.MLKEM.MLKEM1024
+ Crypto.PubKey.MLKEM: instance GHC.Show.Show Crypto.PubKey.MLKEM.MLKEM512
+ Crypto.PubKey.MLKEM: instance GHC.Show.Show Crypto.PubKey.MLKEM.MLKEM768
+ Crypto.PubKey.MLKEM: keyPairFromSeed :: forall p proxy ba. (MLKEM p, ByteArrayAccess ba) => proxy p -> ba -> CryptoFailable (MLKEMEncapsulationKey p, DecapsulationKey p)
+ Crypto.PubKey.MLKEM: newtype SharedSecret
+ Crypto.PubKey.MLKEM: seedSize :: Int
+ Crypto.PubKey.MLKEM: sharedSecretSize :: Int
+ Crypto.PubKey.MLKEM: type Ciphertext kem;
+ Crypto.PubKey.MLKEM: type Coins kem;
+ Crypto.PubKey.MLKEM: type DecapsulationKey kem;
+ Crypto.PubKey.MLKEM: type EncapsulationKey kem;
+ Crypto.PubKey.MLKEM: }
Files
- CHANGELOG.md +851/−2113
- Crypto/Cipher/Twofish/Primitive.hs +5/−6
- Crypto/ConstructHash/MiyaguchiPreneel.hs +2/−3
- Crypto/ECC.hs +1/−12
- Crypto/Error/Types.hs +4/−0
- Crypto/KEM.hs +121/−0
- Crypto/Number/F2m.hs +3/−4
- Crypto/OTP.hs +4/−5
- Crypto/PubKey/Internal.hs +2/−3
- Crypto/PubKey/MLDSA.hs +656/−0
- Crypto/PubKey/MLKEM.hs +448/−0
- Crypto/PubKey/RSA/OAEP.hs +2/−3
- Crypto/PubKey/RSA/PKCS15.hs +2/−3
- Crypto/PubKey/Rabin/OAEP.hs +2/−3
- Crypto/Random/Types.hs +12/−0
- cbits/crypton_cpu.c +19/−2
- cbits/mldsa/COMMIT +2/−0
- cbits/mldsa/LICENSE +305/−0
- cbits/mldsa/README.md +60/−0
- cbits/mldsa/crypton_mldsa.c +31/−0
- cbits/mldsa/crypton_mldsa.h +41/−0
- cbits/mldsa/crypton_mldsa_asm.S +14/−0
- cbits/mldsa/import.sh +40/−0
- cbits/mldsa/mldsa_native.c +803/−0
- cbits/mldsa/mldsa_native.h +956/−0
- cbits/mldsa/mldsa_native_asm.S +830/−0
- cbits/mldsa/mldsa_native_config.h +855/−0
- cbits/mldsa/src/cbmc.h +233/−0
- cbits/mldsa/src/common.h +301/−0
- cbits/mldsa/src/context.h +152/−0
- cbits/mldsa/src/ct.c +21/−0
- cbits/mldsa/src/ct.h +373/−0
- cbits/mldsa/src/debug.c +75/−0
- cbits/mldsa/src/debug.h +125/−0
- cbits/mldsa/src/fips202/fips202.c +270/−0
- cbits/mldsa/src/fips202/fips202.h +224/−0
- cbits/mldsa/src/fips202/fips202x4.c +187/−0
- cbits/mldsa/src/fips202/fips202x4.h +125/−0
- cbits/mldsa/src/fips202/keccakf1600.c +510/−0
- cbits/mldsa/src/fips202/keccakf1600.h +110/−0
- cbits/mldsa/src/fips202/native/aarch64/auto.h +85/−0
- cbits/mldsa/src/fips202/native/aarch64/src/fips202_native_aarch64.h +69/−0
- cbits/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +378/−0
- cbits/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +207/−0
- cbits/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +262/−0
- cbits/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +1080/−0
- cbits/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +990/−0
- cbits/mldsa/src/fips202/native/aarch64/src/keccakf1600_round_constants.c +47/−0
- cbits/mldsa/src/fips202/native/aarch64/x1_scalar.h +27/−0
- cbits/mldsa/src/fips202/native/aarch64/x1_v84a.h +36/−0
- cbits/mldsa/src/fips202/native/aarch64/x2_v84a.h +40/−0
- cbits/mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h +32/−0
- cbits/mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +37/−0
- cbits/mldsa/src/fips202/native/api.h +129/−0
- cbits/mldsa/src/fips202/native/auto.h +35/−0
- cbits/mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +34/−0
- cbits/mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h +45/−0
- cbits/mldsa/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +488/−0
- cbits/mldsa/src/fips202/native/x86_64/src/keccakf1600_constants.c +52/−0
- cbits/mldsa/src/native/aarch64/meta.h +314/−0
- cbits/mldsa/src/native/aarch64/src/aarch64_zetas.c +248/−0
- cbits/mldsa/src/native/aarch64/src/arith_native_aarch64.h +367/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_intt_aarch64_asm.S +786/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_ntt_aarch64_asm.S +686/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_pointwise_montgomery_aarch64_asm.S +106/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_poly_caddq_aarch64_asm.S +69/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_poly_chknorm_aarch64_asm.S +76/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_poly_decompose_32_aarch64_asm.S +108/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_poly_decompose_88_aarch64_asm.S +108/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_poly_use_hint_32_aarch64_asm.S +125/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_poly_use_hint_88_aarch64_asm.S +133/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S +157/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S +173/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S +205/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_polyz_unpack_17_aarch64_asm.S +103/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_polyz_unpack_19_aarch64_asm.S +100/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_rej_uniform_aarch64_asm.S +222/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_rej_uniform_eta2_aarch64_asm.S +170/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_rej_uniform_eta4_aarch64_asm.S +163/−0
- cbits/mldsa/src/native/aarch64/src/polyz_unpack_table.c +52/−0
- cbits/mldsa/src/native/aarch64/src/rej_uniform_eta_table.c +547/−0
- cbits/mldsa/src/native/aarch64/src/rej_uniform_table.c +63/−0
- cbits/mldsa/src/native/api.h +617/−0
- cbits/mldsa/src/native/meta.h +24/−0
- cbits/mldsa/src/native/x86_64/meta.h +323/−0
- cbits/mldsa/src/native/x86_64/src/arith_native_x86_64.h +330/−0
- cbits/mldsa/src/native/x86_64/src/consts.c +157/−0
- cbits/mldsa/src/native/x86_64/src/consts.h +27/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_intt_avx2_asm.S +2333/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_ntt_avx2_asm.S +2405/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_nttunpack_avx2_asm.S +254/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_pointwise_acc_l4_avx2_asm.S +173/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_pointwise_acc_l5_avx2_asm.S +189/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_pointwise_acc_l7_avx2_asm.S +221/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_pointwise_avx2_asm.S +158/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_poly_caddq_avx2_asm.S +199/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S +176/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S +490/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S +489/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S +123/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S +125/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S +355/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S +355/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_rej_uniform_avx2_asm.S +132/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S +205/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S +176/−0
- cbits/mldsa/src/native/x86_64/src/rej_uniform_table.c +161/−0
- cbits/mldsa/src/packing.c +213/−0
- cbits/mldsa/src/packing.h +277/−0
- cbits/mldsa/src/params.h +153/−0
- cbits/mldsa/src/poly.c +1066/−0
- cbits/mldsa/src/poly.h +464/−0
- cbits/mldsa/src/poly_kl.c +910/−0
- cbits/mldsa/src/poly_kl.h +367/−0
- cbits/mldsa/src/polyvec.c +509/−0
- cbits/mldsa/src/polyvec.h +435/−0
- cbits/mldsa/src/polyvec_lazy.c +311/−0
- cbits/mldsa/src/polyvec_lazy.h +652/−0
- cbits/mldsa/src/randombytes.h +26/−0
- cbits/mldsa/src/reduce.h +144/−0
- cbits/mldsa/src/rounding.h +265/−0
- cbits/mldsa/src/sign.c +1720/−0
- cbits/mldsa/src/sign.h +850/−0
- cbits/mldsa/src/symmetric.h +68/−0
- cbits/mldsa/src/sys.h +327/−0
- cbits/mldsa/src/zetas.inc +55/−0
- cbits/mlkem/COMMIT +2/−0
- cbits/mlkem/LICENSE +312/−0
- cbits/mlkem/README.md +59/−0
- cbits/mlkem/crypton_mlkem.c +31/−0
- cbits/mlkem/crypton_mlkem.h +41/−0
- cbits/mlkem/crypton_mlkem_asm.S +14/−0
- cbits/mlkem/import.sh +42/−0
- cbits/mlkem/mlkem_native.c +692/−0
- cbits/mlkem/mlkem_native.h +464/−0
- cbits/mlkem/mlkem_native_asm.S +716/−0
- cbits/mlkem/mlkem_native_config.h +683/−0
- cbits/mlkem/src/cbmc.h +222/−0
- cbits/mlkem/src/common.h +296/−0
- cbits/mlkem/src/compress.c +763/−0
- cbits/mlkem/src/compress.h +613/−0
- cbits/mlkem/src/context.h +51/−0
- cbits/mlkem/src/debug.c +64/−0
- cbits/mlkem/src/debug.h +121/−0
- cbits/mlkem/src/fips202/fips202.c +249/−0
- cbits/mlkem/src/fips202/fips202.h +144/−0
- cbits/mlkem/src/fips202/fips202x4.c +207/−0
- cbits/mlkem/src/fips202/fips202x4.h +81/−0
- cbits/mlkem/src/fips202/keccakf1600.c +499/−0
- cbits/mlkem/src/fips202/keccakf1600.h +98/−0
- cbits/mlkem/src/fips202/native/aarch64/auto.h +78/−0
- cbits/mlkem/src/fips202/native/aarch64/src/fips202_native_aarch64.h +80/−0
- cbits/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +377/−0
- cbits/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +206/−0
- cbits/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +261/−0
- cbits/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +1079/−0
- cbits/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +989/−0
- cbits/mlkem/src/fips202/native/aarch64/src/keccakf1600_round_constants.c +47/−0
- cbits/mlkem/src/fips202/native/aarch64/x1_scalar.h +26/−0
- cbits/mlkem/src/fips202/native/aarch64/x1_v84a.h +35/−0
- cbits/mlkem/src/fips202/native/aarch64/x2_v84a.h +38/−0
- cbits/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h +31/−0
- cbits/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +36/−0
- cbits/mlkem/src/fips202/native/api.h +117/−0
- cbits/mlkem/src/fips202/native/auto.h +29/−0
- cbits/mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +33/−0
- cbits/mlkem/src/fips202/native/x86_64/src/fips202_native_x86_64.h +44/−0
- cbits/mlkem/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +487/−0
- cbits/mlkem/src/fips202/native/x86_64/src/keccakf1600_constants.c +52/−0
- cbits/mlkem/src/indcpa.c +675/−0
- cbits/mlkem/src/indcpa.h +172/−0
- cbits/mlkem/src/kem.c +474/−0
- cbits/mlkem/src/kem.h +353/−0
- cbits/mlkem/src/native/aarch64/README.md +16/−0
- cbits/mlkem/src/native/aarch64/meta.h +166/−0
- cbits/mlkem/src/native/aarch64/src/aarch64_zetas.c +184/−0
- cbits/mlkem/src/native/aarch64/src/arith_native_aarch64.h +184/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_intt_aarch64_asm.S +635/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_ntt_aarch64_asm.S +565/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_poly_mulcache_compute_aarch64_asm.S +130/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_poly_reduce_aarch64_asm.S +153/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_poly_tobytes_aarch64_asm.S +124/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_poly_tomont_aarch64_asm.S +102/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S +264/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S +317/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S +371/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_rej_uniform_aarch64_asm.S +226/−0
- cbits/mlkem/src/native/aarch64/src/rej_uniform_table.c +543/−0
- cbits/mlkem/src/native/api.h +651/−0
- cbits/mlkem/src/native/meta.h +30/−0
- cbits/mlkem/src/native/x86_64/README.md +4/−0
- cbits/mlkem/src/native/x86_64/meta.h +324/−0
- cbits/mlkem/src/native/x86_64/src/arith_native_x86_64.h +327/−0
- cbits/mlkem/src/native/x86_64/src/compress_consts.c +115/−0
- cbits/mlkem/src/native/x86_64/src/compress_consts.h +53/−0
- cbits/mlkem/src/native/x86_64/src/consts.c +102/−0
- cbits/mlkem/src/native/x86_64/src/consts.h +25/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_intt_avx2_asm.S +743/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_ntt_avx2_asm.S +661/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_nttfrombytes_avx2_asm.S +217/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_ntttobytes_avx2_asm.S +205/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_nttunpack_avx2_asm.S +190/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_compress_d10_avx2_asm.S +413/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_compress_d11_avx2_asm.S +479/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_compress_d4_avx2_asm.S +194/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_compress_d5_avx2_asm.S +251/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_decompress_d10_avx2_asm.S +257/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_decompress_d11_avx2_asm.S +307/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_decompress_d4_avx2_asm.S +209/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_decompress_d5_avx2_asm.S +222/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_mulcache_compute_avx2_asm.S +118/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S +536/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S +784/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S +1032/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_reduce_avx2_asm.S +234/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_rej_uniform_avx2_asm.S +136/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_tomont_avx2_asm.S +172/−0
- cbits/mlkem/src/native/x86_64/src/rej_uniform_table.c +545/−0
- cbits/mlkem/src/params.h +76/−0
- cbits/mlkem/src/poly.c +581/−0
- cbits/mlkem/src/poly.h +296/−0
- cbits/mlkem/src/poly_k.c +522/−0
- cbits/mlkem/src/poly_k.h +622/−0
- cbits/mlkem/src/randombytes.h +52/−0
- cbits/mlkem/src/sampling.c +371/−0
- cbits/mlkem/src/sampling.h +111/−0
- cbits/mlkem/src/symmetric.h +70/−0
- cbits/mlkem/src/sys.h +320/−0
- cbits/mlkem/src/verify.c +20/−0
- cbits/mlkem/src/verify.h +447/−0
- cbits/mlkem/src/zetas.inc +30/−0
- cbits/tests/ct/ct_mlkem.c +45/−0
- cbits/tests/ct/run.sh +40/−2
- crypton.cabal +110/−1
- tests/PubKey/MLDSASpec.hs +251/−0
- tests/PubKey/MLDSAVectors.hs +207/−0
- tests/PubKey/MLKEMSpec.hs +184/−0
- tests/PubKey/MLKEMVectors.hs +195/−0
CHANGELOG.md view
@@ -1,2115 +1,853 @@ # CHANGELOG for crypton -## 2.1.7--One fix: RSA-PSS verification was accepting a signature RFC 8017 says to-refuse. It is a conformance fault rather than a forgery -- producing such a-signature takes the private key, since it is the signer who chooses the-encoding, and a third party holding a valid signature cannot turn it into-one of these.--* fix(pss): RSA-PSS verification refuses an encoding with a bit set outside- `emBits`, as RFC 8017 9.1.2 step 6 requires. Step 9 clears those bits in- DB and crypton did that; clearing is not checking, so an encoding the- standard calls inconsistent verified as though it were sound -- the bit- that made it wrong was thrown away before anything looked at it. Only the- signer can produce such a signature, since it takes the private key to- sign a chosen encoding, so this is conformance rather than forgery. Found- with tlsfuzzer, in the `xor 0x80 at 0` case of- `test-tls13-certificate-verify.py`, while testing hs-tls--## 2.1.6--Two things a caller could walk into, the licence field saying what the tree-actually holds, and the C building without a warning.--* fix(pbkdf2): an output length of zero no longer takes the process down.- `tryFastPBKDF2_*` passed it through to C, where `assert(out && nout)`- aborted -- crypton's C is built without `NDEBUG`, so its assertions are- live in a release. `Crypto.KDF.PBKDF2.tryGenerate` has always answered- with an empty result for the same request, and the fast paths now agree-* fix(p256): `Crypto.PubKey.ECC.P256.scalarInv` returns on a zero scalar- rather than looping for ever. `scalarFromBinary` accepts any 256 bits, so- a zero scalar is easy to come by, and the binary extended Euclid behind- `scalarInv` has no exit for it: zero stays even and is halved for ever.- A hang inside a foreign call is not interruptible, so `System.Timeout` was- no help either. It now answers zero, which is what the function already- answered for the other input with no inverse, and what `scalarInvSafe`- answers for both. crypton's own ECDSA was never exposed: it rejects a- zero scalar before inverting, and uses `scalarInvSafe`-* doc(cabal): the `license:` field says what the tree holds --- `BSD-3-Clause AND MIT AND ISC` -- and `license-files:` lists the five- licence texts, where a tool looking for licences will find them. The- parts of `cbits/aes/gcm_fused_x86.c` that follow picotls's `fusion` now- carry its MIT notice beside the file. **Nothing is required of a user- that was not required before**: crypton's own code is BSD-3-Clause as it- always was, and the MIT and ISC code was already in the tree -- the field- was silent about it. Raised by Joey Hess in #232, and settled with the- help of Kazuho Oku, who divided `fusion` between what derives from- OpenSSL and what does not, and rewrote the former upstream-* fix(c): the sanitizer build is quiet again. `crypton_sha256_finalize` and- `crypton_sha512_finalize` say that their pointers are never null, which- they never were. gcc's `-Wstringop-overflow` had been reporting them as- writing "into a region of size 0" at "address zero" under- `-fsanitize=undefined`: UndefinedBehaviorSanitizer inserts a null check- before `memcpy`, because glibc declares `memcpy` nonnull, and the check- puts a null path in front of the warning pass. Stating the contract- removes the path rather than the warning, and costs nothing -- compiled as- the package compiles it, the assembly is identical either way-* fix(pbkdf2): an instantiation whose digest is larger than its block is- refused where it is written rather than after it has overflowed. The- macro that builds the three PBKDF2 variants shortens a long key by hashing- it into a buffer the size of the block, and asserted afterwards that the- result fitted -- afterwards being too late, since the write has already- happened. The check is a compile-time one now. The three that exist are- unaffected: SHA-1, SHA-256 and SHA-512 have digests of 20, 32 and 64 bytes- against blocks of 64, 64 and 128--## 2.1.5--crypton 2.1.3 and 2.1.4 cannot be built with GCC 14 or newer; it was-reported from a Fedora 43 system, which ships GCC 15. This release is that-fix, and two things that came with it.--* fix(x86): the C builds with GCC 14 and newer again.- `crypton_sha1_x86_do_chunk` was declared taking `const uint32_t buf[16]`- while every caller passes a `const uint8_t *`, and GCC 14 made- `-Wincompatible-pointer-types` an error by default where GCC 13 only warns.- The declaration now says `const uint8_t buf[64]`, which is the block size- SHA-1 actually takes and what the two sibling functions already said.- Reported as #282 and fixed in #284, both by @tbidne, who bisected it to- the commit that introduced the declaration. CI now builds the C with- gcc-14 as well, so the next one of these is caught before release-* perf(armv8): AES-GCM is about a quarter faster on AArch64, which puts it- ahead of OpenSSL 4.0.3 rather than behind it -- 1.15 at AES-128 and 1.06- at AES-256 on an Apple M4, from 0.86. The GHASH no longer keeps H the way- GCM writes it; it is twisted once at key setup so that GCM's bit- reflection is already undone, which turns a reduction of some twenty-five- shifts and XORs into two PMULL and six EOR, and makes Karatsuba worth- taking -- three multiplications a block rather than four. Against the- previous code over 16 KiB messages: 1.34 at AES-128 and 1.23 at AES-256 on- an M4, and 1.25 across the three key sizes on a Neoverse N2. The scheme- is ARM's, from the BSD-3-Clause part of- https://github.com/ARM-software/AArch64cryptolib-* test(armv8): the constant-time harness runs on AArch64, where it never had.- It was pinned to one x86-64 job, so the AArch64 AES and GHASH had never- been put to it; they are now, and they let no secret decide a branch or an- address. Running it somewhere new also found a fault in the harness- itself: it counted the frame `--track-origins` prints to say where a value- came from as a place that branched on a secret, which invented a finding- rather than hiding one--## 2.1.4--2.1.3 could not be built from Hackage at all in the default configuration,-and is deprecated there. This release is that fix and three more.--* fix(cabal): the source distribution carries `cbits/p256/p256_verify.h`.- No field named it, so it was absent from the 2.1.3 tarball, and- `cbits/p256/p256_ec.c` includes it whenever `support_s2n_bignum` is on --- which is every x86-64 and aarch64 machine that leaves the flag alone. The- package built perfectly from a git checkout and not at all from Hackage.- Reported as #270 by Laurent P. Rene de Cotret on the day 2.1.3 went out,- fixed by gev in #271-* fix(armv8): the C builds with gcc before 13 again. A function that uses an- AArch64 extension says so with `__attribute__((target(...)))`, and the- spelling used -- `target("+sha3")` -- is one clang has always taken and gcc- learned in 13. Before that the extension never reaches the function and an- `always_inline` intrinsic that needs it cannot be inlined, which stops the- build rather than slowing it. Naming the architecture beside the extension- is understood by both compilers at every version, so gcc is given that- spelling. Reported as #273 by gev, against 2.1.1, 2.1.2 and 2.1.3-* fix(sha256): SHA-256 on AArch64 is no longer five times slower than it was- in 2.1.2 -- 644 MB/s against 3396 over 16 KiB on an Apple M4. The- CRYPTOGAMS assembly picks its path from `crypton_armcap_P` rather than from- a flag in the C, and the bit was set only while that flag was still- unresolved; 2.1.3 added a constructor that resolves it before anything- runs, so the bit was never set and the assembly took its generic path on- every processor. SHA-1 was unaffected, its constructor setting the- corresponding bit itself, and the SHA-512 assembly is x86 only. No test- could have caught this: the answers were right all along, only slow-* test(ci): three jobs for the three ways the above went unnoticed. One- builds the source distribution and then builds the library from it- somewhere other than the checkout, since nothing had ever built a tarball- and listing one is not building it. One installs gcc-12 and builds the C- with it, the runner's own gcc being 13, which is why a report covering- three releases never reproduced here. Both are verified against the bug- they exist for: each was red before its fix and green after--# CHANGELOG for crypton--## 2.1.3--* fix(number): the arithmetic crypton falls back to when it is built without- GMP no longer walks off the end of a buffer. `fillPtr` writes a number out- starting at its last byte and stops at offset zero, so a number of no bytes- -- which is what zero is -- started it at minus one, and it never stopped.- The same build also had `exponentiation` fail to terminate on a negative- exponent, stepping between -1 and -2 until the stack ran out, where- `expFast` and `expSafe` both reach it; `numBits` fail to terminate on a- negative number; `numBits 0` answer one where GMP answers zero, so that- `numBytes 0` claimed a byte that is not there; and a modulus of one answer- one rather than zero in two places. Every corner is now held to what the- GMP build answers on the same input, and `-integer-gmp` has a CI job so- that it stays that way-* fix(cabal): `-fsupport_sse` no longer selects the SSE BLAKE2 and Argon2- sources on architectures that have no SSE, where they cannot compile, and- `-fold_toolchain_inliner` links again -- one of the sixteen inline- definitions in `cbits/decaf/include/word.h` had lost its `static`, which- `-fgnu89-inline` turns into a duplicate symbol. Both flags now have a CI- job-* fix(p256): `crypton_p256_shl` and `crypton_p256_shr` no longer shift a- digit by its own width, which the standard leaves undefined and which x86- and ARM answer differently. Nothing in the library calls either, which is- why the sanitizers had not reported them: they can only report what runs-* fix(headers): `crypton_skein256.h` and `crypton_skein512.h` declared six- functions under a misspelling of the package's own name, so the six the C- defines had no prototype at all. `crypton_chacha.h` and `crypton_salsa.h`- each typedef'd a union to the bare name `block`, so no translation unit- could include both; they are `crypton_chacha_block` and- `crypton_salsa_block` now-* security(c): seven places that erase key material and then let the memory- die -- five that `memset` a buffer and `free` it on the next line, two that- clear a recoded scalar in a local going out of scope -- now write through a- volatile pointer, which a compiler may not remove. clang at -O2 was- keeping all seven, but that is its choice rather than a guarantee. RSA-2048- signing is unchanged at 1479.3 microseconds against 1480.0 on an Apple M4-* perf(ed448): the scalar arithmetic on Apple Silicon runs on 64-bit limbs- rather than 32-bit ones. decaf sized its field limbs from the architecture- and its scalar limbs from a macro it worked out from the compiler, and the- two disagreed wherever `uint_fast32_t` is four bytes, so the same aarch64- CPU took 64-bit field limbs and 32-bit scalar limbs on macOS and 64-bit for- both on Linux. Ed448 signing is 5.5% quicker for it, 20.69 to 19.56- microseconds on an M4-* perf(p256): the one addition in each scalar multiplication that can be a- point added to itself goes through a complete formula rather than being- detected and worked around. Kyle Butt pointed out that the comb's table is- affine, so the same three numbers are the point in projective coordinates- and in Jacobian ones, which is what lets a comb built on Jacobian- arithmetic step into the formula for one addition. It costs about a third- of a percent, and buys an argument a reader had to follow becoming a- comparison a machine can run-* doc(hash): the Haddock for `Context` says that it is not erased when it- is finished with, what survives in it, and why scrubbing every one is not- done -- it costs about 70% of a 32-byte hash, where the allocation is most- of the work-* test(c): the C is now checked in CI for things the test suite cannot ask- about: whether it is the same on a 32-bit machine and on a big-endian one,- whether a private key ever decides a branch or an address, what a secret- leaves behind in memory, and whether anything breaks on generated input.- Each of the last four carries a probe that is deliberately wrong and has to- be reported, because a check that has quietly stopped working otherwise- reads as a clean bill of health -- which, four times over the course of- this work, is exactly what it did--* doc(cabal): every dependency has an upper bound, which `cabal check` had- been asking for, and `bytestring` is no longer listed twice in the same- `build-depends`--* fix(c): the table that says which AES implementation to call is filled in- once, before there is a second thread, rather than on every- `crypton_aes_initkey`. Two threads taking a key at the same time were- writing the whole table at the same time, which ThreadSanitizer reports- forty-four times over for eight threads doing nothing else. The same for- the flags that say whether to use the ARMv8 SHA-1, SHA-256 and SHA-512- instructions. Nothing has ever come of it -- the values written are the- same ones every time and the table starts out holding valid generic- implementations -- but it is a race the standard gives no meaning to.- Taking an AES key is 6.7% quicker for not doing the work again: 72.1 to- 67.3 nanoseconds on an Apple M4-* fix(c): the C no longer reads a word off a caller's pointer by casting it,- which the standard leaves undefined at an address the word type is not- aligned for and which UndefinedBehaviorSanitizer reported on seventy-eight- lines. `crypton_align.h`'s accessors go through `memcpy`, the ten hash- implementations read their block a word at a time rather than pointing at- it as though it were an array of words, and `block128` is packed so that a- caller's pointer may be one. The non-aligned trampolines those hashes kept- for the case are gone with it. Measured on an Apple M4, every hash and- AES-GCM is where it was, within a tenth of a per cent. The sanitizers now- run in CI with nothing turned off-* fix(c): two left shifts the C standard leaves undefined, found by building- the C with UndefinedBehaviorSanitizer and running the test suite. One- shifted a carry into the sign bit of a signed 64-bit digit in- `cbits/p256/p256.c`, the other shifted a negative value in- `cbits/decaf/ed448goldilocks/decaf.c`. Neither miscomputes on any compiler- crypton is built with, and the values are unchanged; what they were was a- licence the standard gives the compiler and no reason to give it-* security(cipher): a message of 2^32 bytes or more no longer has its length- truncated on the way to the C, which takes its lengths as `uint32_t`. It- used to be: `Crypto.Cipher.ChaCha.combine` given 2^32 + 64 bytes enciphered- 64 of them and returned the rest as it found the buffer -- 4 GiB of zeros- where the ciphertext should have been, with nothing returned to say so.- `Crypto.Hash` has cut its work into 2 GiB pieces for this reason since- before crypton; nothing else did.-- Where the C carries its state in a context and can simply be called again,- the work is now cut up the same way and a long message is enciphered:- `Crypto.Cipher.ChaCha`, `Crypto.Cipher.Salsa`, `Crypto.Cipher.XSalsa`,- `Crypto.Cipher.RC4`, `Crypto.MAC.Poly1305`, and AES-GCM's incremental- interface -- which is what `Crypto.Cipher.ChaChaPoly1305` and- `Crypto.MAC.KMAC` reach it through.-- Where it cannot -- the one-call AEADs, which do the whole message in one- call, and the AES modes, which are handed the IV and do not hand it back --- the message is refused: `CryptoError_ParameterInvalid` from- `Crypto.Cipher.AES.GCM` and `Crypto.Cipher.ChaCha.Poly1305`, and an error- from AES ECB, CBC, CTR, XTS, OCB and CCM. ECB, CBC and XTS count blocks- rather than bytes, so their limit is sixteen times further out.- `Crypto.Cipher.AESGCMSIV` already refused, and still does-* perf(p256): the comb that multiplies the base point takes five bits of the- scalar at a time from each of two blocks rather than four, over the signed- all-bits-set representation, so the table stays the same size while the- loop goes from 32 steps to 26: 25 doublings and 52 mixed additions against- 31 and 64, which is 19% fewer field operations. An ECDSA P-256 signature- goes 25.52 to 22.47 microseconds on an Apple M4. This is the C path, which- x86-64 and AArch64 do not take -- they have the vendored assembly -- so it- is for i386, armv7, riscv64, ppc64le, s390x and the rest. The arrangement- was suggested by Kyle Butt.-- One scalar below the order makes the last addition of the comb add a point- to itself, which the formulas there cannot do; it was found by searching- the sign patterns rather than by sampling, and the recoder now reports it- and the answer is the doubling. Every other addition is ruled out, the 48- from step two upwards by parity and size and the two of step one by- exhausting the 2^20 sign patterns that could reach them-* security(aes): `Crypto.Cipher.AES.GCM` refuses a nonce of no bytes, which- its four functions used to accept. SP 800-38D 5.2.1.1 asks for at least- one byte, and with none GCM's pre-counter block is zero, so the tag of a- message is `GHASH_H(A, C) XOR E(K, 0^128)` -- and `E(K, 0^128)` is the- GHASH key `H` itself. One full tag therefore gives `H` away; `H` belongs- to the key rather than to the nonce, so an attacker who has it, and one- genuine message under any nonce, can forge a tag that verifies for data of- their own under that nonce, twelve-byte ones included. `encrypt` and- `decryptWithTag` now throw `CryptoError_IvSizeInvalid`, `decrypt` gives- `Nothing` and `encryptWithMask` gives `False` and writes nothing. Only the- empty nonce is refused; every other length stays allowed. The general- interface has refused it since 2.1.0 and this module did not. Affects- 2.1.0, 2.1.1 and 2.1.2. Reported by arybczak in- [#249](https://github.com/kazu-yamamoto/crypton/issues/249)-* perf(rsa): the multiply-accumulate at the bottom of the modular- exponentiation takes four limbs to an iteration on AArch64, with the flags- carrying through two long chains -- one for the low halves of the four- products, one for the high halves a place up -- where the compiler writes- a chain per limb and spends two instructions moving each carry between the- flags and a register. Measured on an Apple M4, an RSA-2048 signature goes- 593.0 to 515.9 microseconds through the Haskell API and the exponentiation- itself 590.1 to 514.9. The ragged end of a row -- three limbs, two or- one -- is written out the same way rather than handed back to C, which- matters because mont_sqr asks for every length from n-1 down to 1: that is- a further two per cent of the exponentiation, 514.2 to 504.9. The- arrangement follows addMulVVWx in Go's crypto/internal/fips140/bigmod,- which is BSD-3-Clause as this library is; cbits/LICENSE.go carries its- notice. Three other arrangements were tried and measured worse, and the- comment above the function says which and why, so that they are not tried- again-* perf(rsa): a Montgomery multiplication no longer clears its 2n-limb- scratch first. The first row of the product lands on empty space, so it- is written rather than added to, and every row after it reads only limbs- an earlier row has already put there. Worth between a half and one per- cent of an RSA-2048 signature on an Apple M4, which is less than it- sounds like it should be: the clearing was cheap, and what it cost was- mostly the row that had to add to zeros-* perf(rsa): the R^2 that Montgomery arithmetic needs before it can start is- built by squaring rather than by doubling a bit at a time. Write a value- as 2^(lgR + d) mod m; a Montgomery squaring divides by R, so it takes that- to 2^(lgR + 2d) and doubles d. The climb from R to R^2 is then the binary- expansion of lgR -- ten squarings for a 1024-bit modulus, where there were- a thousand and twenty-five doublings. On an Apple M4 the setup goes 24.74- to 2.07 microseconds, and an RSA-2048 signature does it twice, once for- each CRT half: the signature goes 512.3 to 466.9 through the Haskell API- and the two exponentiations 502.6 to 457.0. Curves that go through- crypton_ecc.c pay the same setup once per context and gain the same.- Public-key operations still go through GMP and do not change-* perf(ecdsa): P-256 verification multiplies both scalars at once, in- variable time, where it used to do two constant-time multiplications and- add the results. Everything a verification touches is public -- the- message, the signature and the public key -- so the constant-time work- there was paid for nothing, and the two multiplications can share their- doublings besides. Both scalars go into non-adjacent form, width 7 for- the base point, whose odd multiples are a table in the library, and width- 5 for the public key, whose eight are built per call; one pass down the- digits does a doubling at every step and an addition where a digit is not- zero. Measured through the Haskell API on an Apple M4, a verification- goes 30.37 to 25.48 microseconds, and the multiplication itself 29.94 to- 25.34; on an x86-64, 85.04 to 71.53. Signing is untouched. s2n-bignum's- point addition is correct except when its two arguments are the same- point, which is the side condition its proof carries, so a sum that comes- out as the point at infinity from arguments that were not is given up on- and the constant-time pair answers instead -- 1203 cases covering that,- including 41 that take the fallback, agree with the old answers on both- architectures-* perf(gcm): AES-GCM uses the 512-bit form of the AES and carry-less- multiply instructions where the processor has them and they are worth- having, which is Ice Lake and Zen 5 onwards. Four blocks to an- instruction where the 256-bit form takes two; a group is thirty-two- blocks in eight registers and its GHASH is two passes of sixteen, since- sixteen is how many powers of H the table holds. Measured on GitHub's- runners over 16 KiB, AES-128-GCM and AES-256-GCM in MB/s: an EPYC 9V45- goes 9616 to 14268 and 8422 to 12674, a Xeon 6973P-C 8095 to 9848 and- 7187 to 8323, a Xeon 8573C 6983 to 8447 and 6166 to 7113. Zen 4 is left- on the 256-bit path -- there the 512-bit instructions are two passes- through a 256-bit datapath, and AES-GCM measured slightly slower -- and- the run-time check asks for three more bits of XCR0 as well as the- instruction bits, since a machine can report these and still fault on- them-* perf(ed25519): the base point multiplication goes through s2n-bignum,- which signing does twice -- once for the nonce's point and once for the- public key, which `sign` derives from the secret key every time rather- than trusting the one it is handed. Measured through the Haskell API on- an Apple M4, signing goes 11.92 to 7.27 microseconds and `toPublic` 5.97- to 3.52; at the C level on an x86-64 without ADX, where the `_alt` form- runs, a public key goes 13.25 to 9.93 and a signature with the key in- hand 14.74 to 11.34. Verification is untouched: it multiplies two- scalars at once and crypton's variable-time code is ahead of the- constant-time assembly there. ed25519-donna's table stays for every- other architecture, and 5000 key and signature pairs agree between the- two paths on both architectures-* perf(rsa): the masked scan of the exponentiation's table goes four limbs- at a time on x86-64 with AVX2. Every entry of the table is read and a- mask keeps the one the window asks for, which is what keeps the address- stream off the exponent, and at RSA-2048's CRT size that is two kilobytes- read per window with 256 windows to an exponentiation -- 11% of the whole- where s2n-bignum's multiplication runs, measured by taking the scan out- altogether. With the vector form almost all of it comes back: one- RSA-2048 CRT private operation goes 812.4 to 720.8 microseconds against a- 707.1 floor with no scan at all, and on the same machine without ADX- 1646.1 to 1599.3. AArch64 keeps the scalar form, where the same- measurement puts the scan at 1% and there is nothing to win. The- processor is asked once per call, and without the AVX2 bit, or built- without `use_target_attributes`, the scalar scan is what runs-* perf(rsa): the modular exponentiation squares into the other of its two- buffers and swaps them, rather than squaring into one and copying it back.- That is a copy of the modulus' width saved 1280 times per exponentiation,- and which of the three buffers a pointer names is nobody's secret, so- nothing about the timing changes. Measured by thread CPU time, best of- many, on one RSA-2048 CRT private operation: an Apple M4 goes 604.3 to- 595.9 microseconds and an x86-64 without ADX 1678.0 to 1665.6. Where- s2n-bignum's multiplication runs the difference is below the noise, that- multiplication being most of the time there. Also writes down what a- five-bit window is worth, which was measured and is not taken: 3.5% on the- M4, 1% the wrong way on an older x86-64, and 7% the wrong way wherever the- assembly runs, because a table twice as long is a masked scan twice as long--## 2.1.2--* perf(gcm): AES-GCM uses the 256-bit form of the AES and carry-less- multiply instructions where the processor has them, which is Zen 3 and Ice- Lake onwards. Measured on an EPYC 7763 over 16 KiB, in the library as- cabal builds it, AES-128-GCM goes 4245 to 6042 MB/s and AES-256-GCM 3932- to 5463 -- 1.42 and 1.39 times, and ahead of OpenSSL 3.0.13 on that- machine, which reaches 4266 and 3946. `crypton_cpu.c` asks the processor- first and everything else is unchanged-* perf(x25519): X25519 goes through s2n-bignum, and key generation reads a- table rather than multiplying the base point 9 the general way, which is- what crypton did for want of anything else. Measured through the Haskell- API on an Apple M4, the shared secret goes 19.87 to 13.78 microseconds and- a key generation 19.92 to 3.83 -- on x86-64 the C level is 41.19 to 26.96- and 41.18 to 8.54. The RFC 7748 vectors say the answers are the same-* perf(ecdsa): inverting modulo the group order, which ECDSA does once per- signature and once per verification, takes a fixed number of division- steps instead of a whole exponentiation. Measured on an Apple M4:- P-256 6.02 to 0.80 microseconds, P-384 31.3 to 1.20, P-521 63.2 to 2.05.- Through the Haskell API that is ECDSA P-256 signing 11.77 to 7.10, which- is faster than OpenSSL on that machine, verification 36.75 to 32.10, and- P-384 signing 171.6 to 151.3. `inverseSafe` carries it, so DSA, ElGamal,- RSA's `qinv` and `Crypto.PubKey.ECC.Prim` get it too; it checks its answer- by multiplying out, as it always has, and falls back where that fails-* perf(rsa): the modular exponentiation's Montgomery multiplication and- square go through s2n-bignum on x86-64 with ADX, which is twice as fast as- the C there because the C cannot form the two carry chains `ADCX` and- `ADOX` give: measured on an EPYC 7763 at 1024 bits, 0.4832 to 0.2479- microseconds for a multiplication and 0.3984 to 0.1994 for a square. The- window, the table and its masked scan are untouched, and so is every other- size and architecture -- on AArch64 the C measures faster than the- assembly, so nothing is even vendored for it-* perf(p256): ECDSA signing is 2.4 times faster and verification 2.2, which- finishes what the two entries below began. Signing and key generation go- through s2n-bignum's fixed-base routine, which reads a table of multiples- of the base point that `cbits/p256/gen_base_table.py` builds and that the- test suite checks against crypton's own answer; verification goes through- that one and the variable-point one, with the addition of the two left- where it was. Measured through the Haskell API on an Apple M4, signing- goes 28.55 to 12.03 microseconds and verification 82.68 to 37.73, which- leaves both within a tenth of OpenSSL on that machine where they were at- four tenths of it. The table costs 52 KiB of constant data, against the- 2.4 KiB of the one it replaces-* perf(ecc): P-384 and P-521 go through s2n-bignum as well, where the curve- is exactly one of those two. Measured through the Haskell API on an Apple- M4, ECDH P-384 goes 549.5 to 75.4 microseconds and ECDSA P-384- verification 772.4 to 295.8; at the C level P-384 is 7x and P-521 between- 9 and 10x, on both architectures. Signing does not move: there is no- fixed-base routine upstream for these two, so it keeps crypton's comb.- Every other curve `crypton_ecc_mul` is asked about, including these two- named with a different a or b, goes on to the C as before-* perf(p256): ECDH is 2.7 times faster on x86-64 and AArch64. The- variable-point scalar multiplication now goes through AWS's- [s2n-bignum](https://github.com/awslabs/s2n-bignum), vendored in- `cbits/s2n`: hand-written assembly, constant-time, and carrying a- machine-checked proof in HOL-Light that it computes what it says. Measured- through the Haskell API on an Apple M4, ECDH P-256 goes 57.50 to 21.32- microseconds; at the C level it is 2.6x there, 3.2x on an x86-64 with ADX- and 2.6x on the `_alt` path taken where `CPUID` does not report it. Its licence is `Apache-2.0 OR ISC OR MIT-0`,- which is what makes this possible at all -- OpenSSL's and BoringSSL's- `ecp_nistz256` is Apache-2.0 only. Every other architecture keeps the C,- as does Windows for now, and `-f-support_s2n_bignum` turns it off-* perf(p256): the field inversion takes the least squarings an exponent of- 256 bits can be done in. It built the low 94 ones of p-2 in a second- accumulator and multiplied the two at the end, which cost 287 squarings- where 255 will do; the exponent is now built left to right from the shape- of p-2 in one pass. Measured on an Apple M4 the inversion goes 4.224 ->- 3.765 microseconds, about 11%, which is 0.9% of an ECDH since that is where- the inversion sits. The same 13 multiplications either way--## 2.1.1--* docs(rsa): the haddock says what the optional blinder covers -- that the- private exponent is not what is at risk, `expSafe` keeping its value out of- the work, and that what a blinder covers is the input, which without one is- the ciphertext as it arrived and so a number an attacker may have chosen.- The eight places taking a `Maybe Blinder` point at t'Blinder' rather than- repeating half of it; the four in `Crypto.PubKey.RSA.PSS` said nothing at- all before--* feat(chachapoly): `Crypto.Cipher.ChaCha.Poly1305`, which does a whole- ChaCha20-Poly1305 message in one call, the shape `Crypto.Cipher.AES.GCM`- has. `Crypto.Cipher.ChaChaPoly1305` takes a message in steps, which is- right when it arrives in pieces and is eight foreign calls and the- allocations between them when it was already whole. Measured on an Apple- M4 through the Haskell interface, a 100-byte message goes 0.97 -> 0.415- microseconds and a 1400-byte one 2.89 -> 2.36. The nonce is the twelve- bytes RFC 8439 defines; eight is the other ChaCha construction and is- refused rather than quietly encrypted under a scheme nobody asked for--* feat(gcm): `Crypto.Cipher.AES.GCM.decryptWithTag`, which decrypts and hands- back the tag it computed rather than comparing it. For a protocol that- carries the tag apart from the ciphertext, where `decrypt` -- which wants- the two together -- does not fit. It returns an `AuthTag`, whose `Eq` is a- constant-time comparison, so the safe way to use it is also the obvious one--* feat(ecdsa): `Crypto.PubKey.ECDSA` gains the deterministic nonce of RFC- 6979, which `Crypto.PubKey.ECC.ECDSA` already had. The fast module was the- one without it, so moving to it for the speed meant giving up the one- protection against the mistake that hands over an ECDSA private key. Three- new names: `deterministicNonce`, and `signDeterministic` and- `signDigestDeterministic` over it. Held to the implementation in- `Crypto.PubKey.ECC.ECDSA`, which is itself held to the vectors in the RFC,- on P-256, P-384 and P-521 with SHA-1 through SHA-512--## 2.1.0--* perf(p256): a signed five-bit window for the variable-point scalar- multiplication, which is what ECDH and ECDSA signing spend their time in.- The scalar is recoded into 52 digits, every one of them odd, so the table- holds only the odd multiples P, 3P, ..., 31P and a negative digit costs a- negation of y, which is free. The main loop goes from 252 doublings and 64- additions to 255 and 51, and -- because no digit is zero and no partial sum- is the infinity -- it drops the masks that stood in for infinity on every- iteration. The table is built so that each pair of neighbouring odd- multiples comes out of one doubling and one addition that shares everything- but a squaring and a multiplication between X+P and X-P. Counted exactly,- the field multiplications and squarings go 3477 -> 3326. Measured on an- idle Intel Haswell, thirty runs each, ECDH is 158.5 -> 153.3 microseconds,- about 4%; on an Apple M4 under desktop load the difference did not come out- of the noise. One scalar, 30, would have reached the last addition with- the accumulator equal to the point being added, which these formulas cannot- do; the recoder detects that from the scalar and the last iteration doubles- instead. Suggested by Kyle Butt--* perf(gcm): GHASH takes the ciphertext from the output buffer. A group's- multiplies are issued between the rounds of the group after it, and the- blocks were copied into six registers' worth of scratch to wait there --- six stores a group for bytes that had just been written to the output- anyway. The multiplies read the output instead, which is what picotls's- fusion does. On an Intel Haswell this is worth two to three points against- fusion between 400 and 1440 bytes, and it removes the queue from the- AES-128 path--* perf(gcm): decryption takes the fused path too, on both x86-64 and- AArch64. It had been left on the generic framing, so a received packet- cost what a sent one did before any of this: measured at 100 bytes, three- times what encrypting the same packet cost on either. It is the simpler of- the two -- what GHASH absorbs is the ciphertext, and the ciphertext is the- input, so the multiplies need not wait on the AES and nothing is carried- between groups. On an Intel Haswell, 100 bytes goes 0.165 -> 0.050- microseconds and 1440 bytes 0.362 -> 0.290; on an Apple M4, 0.230 -> 0.077- and 0.396 -> 0.321. Decrypting is now about what encrypting is rather than- three times it. The tag is still compared a byte at a time over its whole- length whichever way the answer goes--* perf(gcm): let the one-call interface specialise. `gcmFullEncrypt`,- `gcmFullDecrypt` and `gcmFullEncryptMask` take three `ByteArrayAccess`- arguments and were marked `NOINLINE`, which is this module's habit and is- right for a wrapper that is called once; these are called once a packet.- With no specialisation every `withByteArray` and every `length` went through- a dictionary, and on an Apple M4 that measured **0.15 of the 0.265- microseconds** a 100-byte packet cost through the Haskell interface -- more- than the encryption it wrapped. Marked `INLINABLE` so the caller can- specialise them, 100 bytes falls to 0.128, where the same work measured in C- is 0.114: the Haskell layer costs 0.014 rather than 0.15--* perf(gcm): build the length block and the initial counter in registers.- The length block -- the two bit counts GHASH ends on -- was assembled by- sixteen byte stores to the stack and read back, which measured about 9 of- the 58 nanoseconds a 100-byte packet cost. Reversed the way every block is- on its way to GHASH, that block is just the two counts as little endian- words with the message's in the low half, so one `_mm_set_epi64x` makes it- and no shuffle is needed. The initial counter likewise: the nonce is read- where it lies and masked, rather than in three pieces of four bytes. On an- Intel Haswell a 100-byte packet goes from 0.058 to 0.049 microseconds. With- this every length measured is at 90 per cent of fusion's speed or better --- 100 bytes 90 and 95 with the header protection mask, 200 bytes 96, 1440- bytes 94, 16 KiB 95 -- where the series began at 31 per cent for 100 bytes--* perf(gcm): read a short block without going through the stack. Zeroing- sixteen bytes, copying the block in and loading them back is three trips to- memory with a store the load must wait for, and at packet lengths that was a- tenth of the call. The sixteen bytes are read where they lie and what is- above the length masked off, which is safe everywhere except at the end of a- page -- and a block near the end of a page whose own bytes stop short of it- is read aligned, which cannot leave the page, and shuffled down. This is- how picotls's fusion does it. On an Intel Haswell a 100-byte packet goes- from 0.0625 to 0.058 microseconds, 79 per cent of fusion's speed against 73,- and with the header protection mask 83-* perf(gcm): the tail of a message gets what the groups already had. The- six-wide pass that finishes a message was still running its rounds from a- loop over a count held in the key, so the compiler could not place the- waiting multiplies between them, and the blocks it produced went through an- array indexed by a loop variable, which it cannot see through -- each block- then reloaded its own keystream from memory. Written out for the ten rounds- of AES-128, with the multiplies at slots named at compile time, and the- blocks taken from the registers the pass left them in. The pass itself- falls from about 29 to 6 nanoseconds; on an Intel Haswell 112 bytes is 13- per cent faster and 400 bytes goes from 81 to 86 per cent of fusion's speed--* perf(gcm): give E(K,Y0) a lane that would otherwise sit idle, and stop- copying the last short block through the stack. Six blocks are in flight- whatever the message length, so a message leaving a tail of four blocks or- fewer has lanes to spare; the block that masks the tag rides in one of them- instead of taking ten rounds nothing overlaps, which at 100 bytes measured- 9.3 of 78.9 nanoseconds. And the last short block was stored to the stack- and copied back, when the tag that follows it is about to overwrite the- bytes above it anyway -- where there are sixteen to spare, one store does.- On an Intel Haswell, 100 bytes goes from 60 to 69 per cent of fusion's speed- and 200 bytes from 68 to 82--* perf(gcm): build the counter block without leaving the vector registers,- and keep each power of H beside its Karatsuba term. Both came from reading- what picotls's `fusion` does differently. The counter was being stepped in- a general register, byte swapped there and inserted into a vector one, which- is a move across register files for every lane and six to a group; held- byte reversed in a vector register instead, `_mm_add_epi32` steps the low- thirty-two bits and wraps them where GCM wants, and a shuffle puts the bytes- back. The powers were in two arrays 256 bytes apart, so a multiply touched- two cache lines for operands it always wants together; they are now- adjacent. On an Intel Haswell a 1200-byte message goes from 0.311 to 0.281- microseconds and 1440 bytes from 79 to 90 per cent of fusion's speed. The- multiplies are also now genuinely issued between the AES rounds, which the- comment claimed and the generated code did not do: a test before each one- ended the basic block the scheduler works inside, and the first group is- peeled so that there is nothing to test--* perf(gcm): the same for AArch64, where what costs is the framing rather- than a missing fast path. `armv8_impl.c` already encrypts eight blocks at a- time and folds their GHASH into one reduction; what sat outside it was the- additional data, the tag and the counter, each reached through the branch- table so that the running state went back to memory between them and a- one-block header paid a reduction of its own. Measured on an Apple M4, that- framing was 0.07 of the 0.112 microseconds a 100-byte packet cost -- more- than the encryption of the packet itself. Taking the whole message in one- call, with the tag and the counter in registers from end to end and the- additional data and the length block riding in the same batches as the- ciphertext: a 100-byte packet 3.0x, 200 bytes 2.6x, 400 bytes 1.6x, 1200- bytes 1.30x, 1440 bytes 1.23x, and level from about 6 KB up. With the QUIC- header protection mask, 100 bytes is 3.3x. Unlike x86-64 there is no length- above which something else is faster, because there is no vendored assembly- on this side to hand a long message to, so every length goes this way. Held- against the interface it replaces on two key sizes, seven lengths of- additional data, fifteen message lengths up to 16 KB, three tag lengths and- every sample offset that fits--* perf(gcm): a fused AES-GCM for x86-64, for messages short enough that the- stitched assembly will not take them. That assembly refuses anything under- 288 bytes, so until now a QUIC packet paid for the AES key schedule and the- GHASH one block at a time, through a branch table that put the 128-bit state- back in memory at every step: a 100-byte packet cost 0.153 us of which the- encryption was a small part. This is the design Kazuho Oku sets out for- picotls's `fusion` -- keep AES-NI issuing every clock, six blocks in flight,- and fit the additional data, the tag and the QUIC header protection mask- into the gaps between the rounds -- written in C with intrinsics, for the- reason he gives: what is complicated here is the scheduling, and it has to- stay readable to stay correct. On an Intel Haswell a 100-byte packet goes- from 0.153 to 0.082 us and a 1200-byte one from 0.373 to 0.317, and the- header protection mask becomes **free** wherever it can be taken: 400 bytes- is 0.131 with it and 0.131 without, against 0.181 and 0.177, because it- rides in a lane of the AES pipeline that the message length leaves idle- rather than taking a block of its own. It can be taken there only when the- sample lies in output already written and the two key schedules are the same- length, which is what TLS and QUIC do; otherwise it is computed after the- tag, where everything it may cover exists. Above- 1536 bytes the assembly is faster -- by 12 per cent at 3 KB and 20 at 16 KB- -- so longer messages still go there and nothing about TLS-sized records- changes. The powers of H are built once per key, sixteen of them, which- adds 512 bytes to what a key holds and no parameter to any interface: a- power per block of the message would fold the whole GHASH into one reduction- but would make that state grow with the longest message a caller might send.- Held against the incremental interface on every combination of three key- sizes, seven lengths of additional data, twelve message lengths and three- tag lengths--* fix(cpu): stop reading Intel's SDBG bit as AMD's XOP, which crashed SHA-512- and ChaCha20 on Broadwell and later. The vendored assembly dispatches on a- capability word this library fills, and reads bit 11 of its second dword as- XOP -- which is CPUID leaf 0x80000001 ECX bit 11, an AMD extended leaf.- crypton put the raw leaf 1 ECX there, whose bit 11 is SDBG, the silicon- debug interface, which Intel has reported since Broadwell. On such a- processor `sha512-x86_64.S` took `.Lxop_shortcut` and- `chacha-x86_64.S` took `.Lcrypton_chacha20_asm_4xop`, and the first- `vprotq` is an invalid opcode: SIGILL, for SHA-512 and SHA-384, and for- ChaCha20 and so ChaCha20-Poly1305. OpenSSL clears leaf 1's bit 11 before- merging the real flag in; crypton now clears it and leaves it clear, so the- XOP paths are never taken. Nothing is lost -- XOP ran from AMD's Bulldozer- to Excavator and Zen dropped it -- and it is the reason the AVX-512 bits- beside it are cleared as well. Reported by @lucasdicioccio, who- disassembled the trap- [#202](https://github.com/kazu-yamamoto/crypton/issues/202)--* feat(aes): `encryptWithMask`, for the QUIC header protection mask. QUIC- takes the sample for its header protection from the ciphertext, so the mask- cannot be had before the encryption -- but it can be had before coming back.- `newHeaderKey` builds the second key schedule once, where `quic` builds it- per packet, and `encryptWithMask` seals the message and writes the sixteen- bytes of mask, both into buffers the caller already has, so that nothing is- allocated for either. On an Apple M4 the mask then costs about 0.02 us- against 0.09 to 0.11 asked for separately: a 1440-byte packet goes from- 0.473 to 0.382 us and a 100-byte one from 0.343 to 0.260. Most of that is- the allocations rather than the crossing -- a version returning the two as- bytearrays was measured at 0.419 and 0.299, so it recovered less than a- third of it -- and the AES block itself is under two nanoseconds--* feat(aes): `Crypto.Cipher.AES.GCM`, for many short messages under one key.- The interface in `Crypto.Cipher.Types` builds a state from the key *and* the- nonce and then walks it through appending the additional data, encrypting- and finalizing, copying the 320-byte state at each step. For a stream that- is nothing next to the encryption; for a datagram it is most of the work.- Measured on an Apple M4, a 1440-byte packet with a 20-byte header took- 1.30 us, of which 0.17 us was the encryption: the AES key schedule and the- table of multiples of `H` were rebuilt for every nonce although both depend- on the key alone, and three state copies and four foreign calls carried the- rest. A `Context` now holds what the key determines and is built once, and- `encrypt` takes a nonce and a whole message and answers in one call, giving- the ciphertext with the tag after it -- the shape a packet wants. `decrypt`- takes that shape back and compares the tag itself, in C, looking at every- byte either way. 1440 bytes: 1.30 to 0.37 us, 3.5x; 100 bytes 0.83 to 0.23;- a 16 KiB TLS record 2.86 to 2.20, where the saving is the setup rather than- the call. A message of 4 KiB or less goes through an unsafe foreign call,- which is worth 0.075 us and is only right because such a call is over- quickly; anything longer keeps the safe one. This computes what the general- interface computes, which the tests hold it to on the same vectors-* Breaking change: fix(chachapoly1305): take a checked key, so that- initializing cannot fail. The nonce was already a checked type, built by- `nonce8`, `nonce12` or `nonce24`, so the key length was the only way- `initialize` and `initializeX` could fail -- and callers answered that with- `throwCryptoError`, `tls` among them, where- `noFail (ChaChaPoly1305.nonce12 nonce >>= ChaChaPoly1305.initialize key)`- re-checked a length once per record for a key fixed for the connection.- There is now a `Key` with `key` to build one, and- `initialize :: Key -> Nonce -> State` and- `initializeX :: Key -> XNonce -> State` are total.- `aeadChacha20poly1305Init` is unchanged and still reports a bad key length.- This also closes the last of #28: `initFromRootState` wrapped a- `throwCryptoError` around a `B.take 32`, and the Poly1305 key type now has a- home in a hidden module so the modules here that know the length can say so- [#193](https://github.com/kazu-yamamoto/crypton/issues/193)--* feat(hash): Skein with the digest size as a type parameter. Skein is- defined for any digest size and the C here has always taken one -- the- length goes to `crypton_skein512_init` and `crypton_skein512_finalize`, and- the output is produced in counter mode for as many blocks as are asked for- -- but Haskell could only reach the four sizes that had a type of their own.- `Skein256 (bitlen :: Nat)` and `Skein512 (bitlen :: Nat)` take any, in the- manner `SHAKE128` and `SHAKE256` already did; `Skein512 512` is- `Skein512_512`, which the tests hold it to, and the named types are- untouched. This also brought back the `Skein256-160` and `Skein512-160`- known-answer vectors, which had been commented out of the test suite for- want of a type to run them against. One large digest is a good deal cheaper- than the same bytes from repeated small ones: 512 KiB at 947 MB/s in one- digest against 172 MB/s as 8192 separate `Skein512_512` ones, on an M4- [#56](https://github.com/kazu-yamamoto/crypton/issues/56)--## 2.0.0--* fix(docs): export the names the documentation already referred to.- `Crypto.Number.ModArithmetic` throws `CoprimesAssertionError` and- `ModulusAssertionError` from `inverseCoprimes` and `squareRoot`, and said so- in the haddock, without exporting either, so a caller could not name the- exception it was told to expect; `Crypto.PubKey.Rabin.Types.generatePrimes`- takes a `PrimeCondition` in its exported signature and that synonym was not- exported either. All three are now exported- [#195](https://github.com/kazu-yamamoto/crypton/pull/195)--* Breaking change: fix(bcrypt): refuse a cost bcrypt does not have rather than- substituting one. A cost below 4 came back as a cost-10 hash and a cost- above 31 as a cost-31 one, with nothing said either way, so a caller asking- for something bcrypt does not do was answered with something else and had no- way to tell -- `hashPassword 3` and `hashPassword 10` returned the same- thing. Both ends are now reported as `CryptoError_ParameterInvalid`, which- is what every other KDF here already did for a refused parameter.- `hashPassword` can therefore fail where it could not before, so- `tryHashPassword` is added beside it; the salt it generates is always the- right length, so the cost is the only thing it can report- [#59](https://github.com/kazu-yamamoto/crypton/issues/59)--* Breaking change: fix(poly1305): take a checked key, so that initializing- cannot fail. A Poly1305 key is thirty-two bytes and nothing else about it- can be wrong, so `initialize` returning a `CryptoFailable` put an error case- in front of every caller for a length most of them know is right -- and they- answered it with `throwCryptoError`, this library included: the one in- `Crypto.Cipher.ChaChaPoly1305` guarded a `B.take 32`, and `auth` did not- even do that, it called `error`. There is now a `Key` with `key` to build- one, the length is checked there, and `initialize :: Key -> State` and- `auth :: Key -> ba -> Auth` are total. A caller checks once and then- initializes as often as it likes with nothing to handle. `initialize k`- becomes `initialize <$> key k` where the length is unknown, and where it is- known the check moves to where the key is made. `Key` has no `Show`, as- key material should not- [#28](https://github.com/kazu-yamamoto/crypton/issues/28)--* fix(pubkey): stop printing private keys. `Show` is what `print`, a message- built with `error`, an exception and a test framework's failure output all- reach for, so it is the instance a key travels on when nobody meant to send- it anywhere; the library already kept that promise for the secret keys held- in a `ScrubbedBytes`, and the documentation for `ScrubbedBytes` advertises- it, while ten other types printed theirs in full. Those ten --- `RSA.PrivateKey` and `RSA.KeyPair`, `DSA.PrivateKey` and `DSA.KeyPair`,- `ECDSA.PrivateKey` and `ECDSA.KeyPair`, `DH.PrivateNumber`, and the- `PrivateKey` of `Rabin.Basic`, `Rabin.Modified` and `Rabin.RW` -- now render- the public part and `<secret>` for the rest. Nothing else about them- changes: `Read`, `Eq`, `Data`, `Generic` and `NFData` are all still derived.- The new `Crypto.Debug` exports a class `DebugShow` whose `debugShow` returns- exactly what the derived `Show` used to return, for those ten and for the- five `ScrubbedBytes` secret keys as well, which never had a `Show` that- spoke. **Code that serialized a key through `show` has to say `debugShow`- instead**; `read (debugShow k) == k` still holds, but `read` given the- output of `show` will now fail at run time, which the compiler cannot point- at- [#72](https://github.com/kazu-yamamoto/crypton/issues/72)--* deprecate(ecc): the curves over a binary field. They are obsolete, they are- the curves a cofactor makes delicate, and pyca/cryptography deprecated them- for removal in the release that fixed CVE-2026-26007. A `DEPRECATED` pragma- now covers the eighteen `SEC_t*` constructors of `CurveName` and the- eighteen types of the same names in `Crypto.ECC.Simple.Types`; nothing is- removed, so the only effect is a warning where one of them is named, and- they will go in a later major version. The prime curves with a cofactor,- `SEC_p112r2` and `SEC_p128r2`, are not deprecated: the check above covers- them- [#66](https://github.com/kazu-yamamoto/crypton/issues/66)--* fix(ecc): refuse a public point outside the prime-order subgroup. A point- that satisfies the curve equation is not necessarily in the subgroup the- base point generates; the two coincide only where the cofactor is 1. Of the- curves in `CurveName` twenty have a cofactor -- the eighteen binary ones,- and `SEC_p112r2` and `SEC_p128r2`, whose cofactor is 4 -- and on those the- other party could offer a point of small order, at which point the value- that came back depended on our private number only through its residue- modulo that order, and offering it and watching the answer handed them those- bits. This is the flaw pyca/cryptography fixed as CVE-2026-26007; what is- fixed here is the same one, found by following that report.- `Crypto.PubKey.ECC.DH.getShared` and `tryGetShared`, and- `Crypto.ECC.Simple.Prim.pointFromIntegers`, now require the point to be in- the subgroup and report `CryptoError_PointSubgroupInvalid` when it is not.- The check is `isPointInSubgroup`, newly exported from both prim modules:- where the cofactor is 1 it answers without work, and otherwise it multiplies- by the group order and requires the point at infinity, which is what- OpenSSL's `EC_KEY_check_key` does and costs one further scalar- multiplication -- an exchange on an affected curve is about twice the price,- and one on every other curve is unchanged. The typed `Crypto.ECC` interface- offers only cofactor-1 curves and dedicated implementations, so nothing- reaching elliptic curves through it, `tls` among them, was affected--* perf(p256): inline the field arithmetic on AArch64. `felem_mul` and- `felem_square` end in `felem_reduce_degree`, a carry chain the whole width- of the number, and that chain is what their latency is: one product feeding- the next costs 18.1 ns on an Apple M4, while four independent ones cost 11.4- ns each. The curve arithmetic has independent products to offer -- the two- squarings that open a point doubling, the multiplication and the squaring- that close it -- but only if the compiler inlines the reduction instead of- calling it, since a call is a fence. Plain `inline` does not change its- mind; `always_inline` does, and it is worth asking where there are registers- to hold two carry chains at once and not where there are not: 1.23x on an- M4, 1.12x and 1.05x on a Neoverse under clang and gcc, and 0.95x and 0.82x- on x86-64, whose fifteen general-purpose registers are not enough. So it is- gated on the architecture and x86-64 is left byte-identical. ECDH P-256 on- an M4: 69.16 to 56.24 us, against openssl's 24.68, so 0.36 becomes 0.44;- ECDSA P-256 signing and verification move with it. The cost is code, 30 to- 116 kilobytes of it- [#188](https://github.com/kazu-yamamoto/crypton/pull/188)-* perf(number): count bytes from the bit count, not from base 256. `numBytes`- asked GMP how many base-256 digits a number has, and GHC's bignum answers- that by dividing the number down to nothing, one digit at a time, where the- same question in base two is a look at the highest limb. On a 2048-bit- `Integer` that is 1.65 us against 0.01. Every serialization here asks for- the size before it allocates and `i2ospOf` asks twice, so the cost landed on- every RSA, DSA and DH operation leaving the `Integer` world: `i2ospOf_` at- 256 bytes goes from 2.87 to 0.09 us and RSA-2048 verification from 18.4 to- 15.5 us on an M4. Signing moves by a percent; it is two exponentiations and- hardly touches this- [#187](https://github.com/kazu-yamamoto/crypton/pull/187)-* perf(ecc): stop sharing the doublings in the double multiplication.- `pointAddTwoMuls` was Shamir's trick, one pass over the bits of both scalars- at once in `Integer` arithmetic, which is the right trade when the two- multiplications would cost the same. They have not for a while: `pointMul`- goes to C, and over a prime field it multiplies the base point through a- table of its multiples at about a third of the price -- and the base point- is one of the two, since ECDSA verification is the only caller. Doing them- separately and adding: ECDSA P-384 verification 1397 to 698 us on an M4 in- the typed API, 9839 to 738 in the older one, and a curve over a binary field- 641 ms to 2.6. P-256 keeps the double multiplication it has in C- [#186](https://github.com/kazu-yamamoto/crypton/pull/186)-* perf(sha3): take the CRYPTOGAMS Keccak for x86-64 as well. The same module- as [#181](https://github.com/kazu-yamamoto/crypton/pull/181) on the other- architecture, and the reason it was not taken at the time was a measurement- taken on the wrong machine: crypton on Apple silicon against openssl on- x86-64, which said there was nothing to gain. Measured on one machine there- was: SHA3-256 on an EPYC 7763 goes from 109 to 421 MB/s, against openssl's- 426, so 0.26 becomes 0.99- [#184](https://github.com/kazu-yamamoto/crypton/pull/184)-* perf(sha1): take the CRYPTOGAMS SHA-1 for AArch64. The instructions are the- ones [#170](https://github.com/kazu-yamamoto/crypton/pull/170) put in, and- the arrangement is what the module has over them: the message schedule of- the next four rounds runs against the rounds of this one, which is not- something a C function is going to be made to do --- [#179](https://github.com/kazu-yamamoto/crypton/pull/179) tried the one- thing C can do here, handing over a run of blocks, and on this processor it- measured nothing at all. The entry point for processors that have the- instructions is not exported, so the module's own dispatch picks it and the- answer to the runtime question goes into the word that dispatch reads. On- Apple silicon: 3155 to 3379 MB/s at 16 KiB, against openssl's 3350 on the- same machine, so where this was at 0.94 it is now a shade ahead. The- intrinsics stay for the block a message ends with, and for any processor- that has the instructions but is built without the assembly- [#182](https://github.com/kazu-yamamoto/crypton/pull/182)-* perf(sha3): take the CRYPTOGAMS Keccak for AArch64. The instructions are- the ones [#171](https://github.com/kazu-yamamoto/crypton/pull/171) put in --- EOR3, RAX1, XAR and BCAX -- and what this module does with them is take a- run of blocks rather than one at a time, and schedule the round it is in- against the next one. The absorb loop hands over the whole run, which also- drops the alignment trampoline on that path: the assembly reads the message- as bytes. SHA3-256 on Apple silicon: 1002 to 1104 MB/s at 16 KiB, against- openssl's 1058 on the same machine. Only the absorb side is handed over;- the squeeze, which SHAKE uses to produce output, is entangled with this- side's buffer bookkeeping and is not where the time goes- [#181](https://github.com/kazu-yamamoto/crypton/pull/181)-* perf(xts): double the XTS tweak in the integer registers. The tweak- advances by doubling in GF(2^128) once per block, and it was doing that in a- vector register: six operations on the same units that are running the- rounds and the exclusive ors, in a chain where each waits for the one- before. On a processor whose AES is fast that is not a detail -- taking the- doubling out of a diagnostic build, which gives the wrong answer but says- where the time goes, left XTS running at the speed of ECB. It costs three- integer operations instead, and the integer units have nothing else to do- here; what crosses over is one move per block. On x86-64 that also gets- eight values out of a register file with sixteen entries, so the round keys- stay where they were. AES-128-XTS at 16 KiB: 9644 to 18646 MB/s on Apple- silicon and 4504 to 7660 on a Haswell-generation x86-64, against openssl's- 17382 and 6997 on the same machines, so both are now a little ahead where- they were at 0.55 and 0.64. No assembly: the AArch64 module in CRYPTOGAMS- has no XTS, and the x86-64 one's is inside a module this does not otherwise- want- [#180](https://github.com/kazu-yamamoto/crypton/pull/180)-* perf(sha1): hand the SHA-1 block loop a run of blocks rather than one at a- time. A block at a time means the state goes out to memory and comes back- either side of every block, with the two shuffles that put it in the order- the instructions want; against the hundred-odd cycles a block costs with the- SHA extensions that is most of what stood between this and openssl. On an- EPYC 7763: 1364 to 1677 MB/s, against openssl's 1670 on the same machine,- and on a Xeon 8370C 1506 to 1619. On Apple silicon it measures nothing at- all -- that processor hides the cost -- and is kept there only so the two- paths have one shape. The intended file for this was CRYPTOGAMS'- `sha1-x86_64.pl`, which turns out to be the 2006 scalar implementation: no- SSSE3, no AVX, no SHA extensions. Processors without the extensions are- therefore where they were, 664 MB/s against openssl's 791 on a Haswell- [#179](https://github.com/kazu-yamamoto/crypton/pull/179)-* perf(sha2): take the CRYPTOGAMS SHA-256 and SHA-512 for x86-64. One- generator gives both, as on AArch64, and each dispatches on what the- processor has: the SHA extensions, AVX2, AVX, SSSE3 or plain integer code.- That replaces everything written here for x86-64 -- `sha256_x86.c` and- `sha512_x86.c` go -- since it is ahead of all of it either way. At 16 KiB:- on an EPYC 7763, SHA-256 1430 to 1584 MB/s and SHA-512 423 to 769; on a- Haswell-generation part, SHA-256 318 to 379 and SHA-512 488 to 593, where- openssl reports 378 and 589. The block loops hand over the whole run of- blocks rather than one at a time, and the alignment trampoline goes with it.- This also fixes a bug in the capability word- [#176](https://github.com/kazu-yamamoto/crypton/pull/176) added: bit 29 of- leaf 7 EBX is the SHA extensions, not an AVX-512 bit, and was being cleared- along with them -- which cost the SHA-256 assembly two thirds of its speed- on a processor that has them, and which no machine here could have shown,- since none has them- [#178](https://github.com/kazu-yamamoto/crypton/pull/178)-* perf(chacha): take the CRYPTOGAMS ChaCha20 for x86-64 as well. The C here- vectorises from eight blocks up and takes anything shorter one block at a- time, so a message of a few hundred bytes -- a QUIC packet, a small TLS- record -- ran at a fifth of the bulk rate. The module has vector code for- those lengths and is a few per cent ahead in bulk besides: 494 to 1091 MB/s- at 256 bytes, 477 to 701 at 128, and 2201 to 2374 at 16 KiB, which is- openssl's 2389 on the same machine. It is handed everything from one block- up, where the AArch64 module is handed nothing below three, that one's- scalar path measuring level with the C. Keystream generation is still the- C on both, having no input to exclusive-or- [#177](https://github.com/kazu-yamamoto/crypton/pull/177)-* perf(poly1305): take the CRYPTOGAMS Poly1305 for x86-64 as well. The same- module for the other architecture, through the same three functions, so what- this adds is the capability word: where the AArch64 one reads- `crypton_armcap_P`, this one reads `crypton_ia32cap_P`, which is cpuid's own- words in the order OpenSSL keeps them, filled with the bits for anything the- operating system will not preserve cleared. What it brings over the AVX2- written here is a hand-scheduled scalar path, which is what a message of a- few hundred bytes actually uses, and an AVX path for machines with no AVX2:- on a Haswell-generation x86-64, 4298 to 5345 MB/s at 16 KiB and 556 to 1573- at 64 bytes, against the roughly 5270 openssl reaches there. `poly1305_avx2.c`- goes the way the NEON did. The module's AVX-512 paths are not taken: the- generator chooses what to emit from the version of the assembler it is told- about, and it is now told one that predates them, no machine here being able- to run them and an assembler still in use being unable to assemble them.- Pinning that version also makes the checked-in assembly independent of the- host that produced it- [#176](https://github.com/kazu-yamamoto/crypton/pull/176)-* perf(sha256): take the CRYPTOGAMS SHA-256 for AArch64. The instructions are- the ones the intrinsics here already use; what the module does with them is- schedule them across a whole run of blocks rather than one at a time, and- keep the message schedule of the next block moving while the rounds of this- one are still going, which a function that is handed one block and returns- cannot do whatever it is written in. So the block loop hands over the whole- run, which also drops the alignment trampoline on this path -- the assembly- reads the message as bytes and wants neither the alignment nor the copy. On- Apple silicon: 2637 to 3279 MB/s at 16 KiB, against openssl's 3323 on the- same machine, and 1576 to 1966 at 64 bytes. SHA-512, which the same- generator emits, is not taken: 1876 here against openssl's 1880, the- ARMv8.2 instructions for it having gone in with #110- [#175](https://github.com/kazu-yamamoto/crypton/pull/175)-* perf(poly1305): take the CRYPTOGAMS Poly1305 for AArch64. One- multiplication modulo 2^130 - 5 depends on the one before it, so what there- is to win is in how the multiplies and the carries are laid against each- other, and in keeping the accumulator in whichever base costs less: the- module works in base 2^64 while the message is short and switches to base- 2^26 for the four-way vector loop, deciding that for itself. Unlike the- other two it is the whole of the arithmetic rather than a bulk loop bolted- to the side, so the context now holds either the 26-bit limbs the C works in- or the 192 bytes the assembly keeps, as a union, and grows from 84 bytes to- 232. On Apple silicon: 4269 to 8060 MB/s at 16 KiB and 1542 to 4355 at 64- bytes, and ChaCha20-Poly1305 together, which is what this is for, 1416 to- 2284 against openssl's 2180 on the same machine. `poly1305_neon.c`, which- [#169](https://github.com/kazu-yamamoto/crypton/pull/169) added, goes: the- assembly is faster at every length on every target that gets it, and the- scalar C remains for the targets that do not. The tests came first and- found that the chunking property here had been testing nothing -- it used- the all-zero key, whose r is zero, so both sides were the nonce whatever- they did, which is how it came to feed the chunks to `update` in reverse- order and pass- [#174](https://github.com/kazu-yamamoto/crypton/pull/174)-* perf(chacha): take the CRYPTOGAMS ChaCha20 for AArch64. The vector- registers hold four ChaCha states and there is no room for a fifth, so once- four blocks are in flight the only place further parallelism can come from- is the integer side: that module runs a fifth block through the general- registers alongside four in the vector ones, and above 512 bytes two- alongside six. Which register holds which word is the whole of the trick- and C has no way to say it, which is why the intrinsics here sat at about- 0.63 of what openssl gets out of this very file. On Apple silicon, a- message per call: 1911 to 3069 MB/s at 512 bytes, 1913 to 3093 at 4 KiB and- 2056 to 3112 at 64 KiB, against openssl 3.6's 3164 on the same machine,- which is this code. It is handed only the states it fits -- twenty rounds,- a 256-bit key, and as many blocks as the 32-bit counter has room for, since- crypton's counter is 64 bits wide and carries where the assembly wraps --- and nothing below 192 bytes, where its own vector path starts. The tests- came first: the properties here generated one shape of state, so the- 256-bit constants were never exercised by them- [#173](https://github.com/kazu-yamamoto/crypton/pull/173)-* perf(gcm): take the CRYPTOGAMS stitched AES-GCM for x86-64, which is the- first assembly in the package. Counter-mode AES and GHASH do not compete- for the same execution ports, so a loop that interleaves them at- instruction granularity runs both in about the time the rounds alone take;- written in C that interleaving does not survive the compiler, which sinks- every multiply to the end of the group, and the disassembly of what- [#160](https://github.com/kazu-yamamoto/crypton/pull/160) produced says so.- On a Haswell-generation x86-64, a message per call, AES-128-GCM: 2672 to- 3172 MB/s at 1152 bytes, 3394 to 4271 at 4 KiB and 3657 to 5110 at 16 KiB,- where openssl speed on the same machine reports 4896; decryption within a- couple of points of that, and AES-256-GCM 3147 to 4297 at 16 KiB against- openssl's 4206. `cbits/asm` holds the module, the translator it needs and the- generated assembly, one file per object format, so that building needs no- perl; `cbits/asm/README.md` records where it came from and what was done to- it, which is to rename the entry points, a program linking both crypton- and openssl being entitled to object to two definitions of- `aesni_gcm_encrypt`. What the assembly reads is laid out OpenSSL's way and- is built per message in `cbits/aes/gcm_x86_asm.c`, the powers of H being- the ones crypton already has, shifted up a bit. Short messages are not- handed over at all. `cabal-version` is now 3.0, for `asm-sources`- [#172](https://github.com/kazu-yamamoto/crypton/pull/172)-* perf(sha3): use the ARMv8.2 SHA-3 instructions. Keccak was the plain C- everywhere, a round at a time over tables of rotation amounts and lane- positions, at half of what openssl manages on the same machine. EOR3,- RAX1, XAR and BCAX exist for exactly this permutation and take a round from- around a hundred and fifty operations to sixty-six; they come with the- SHA-512 extension the tree already asks for. Rho and pi move one lane of- every row into every other row, so the round cannot be done in place, and- four rounds go in an iteration, which is worth a fifth over one. SHA3-256- 551 to 991 MB/s (openssl 1064), SHAKE128 700 to 1166, Keccak-256 559 to- 944. The body is generated from the definitions in FIPS 202 rather than- copied in, and the script that worked out the rotations and the lane- permutation checked itself against the published digests of the empty- string and of "abc" before emitting any C, which is how a first attempt- with chi reading lanes another row had already overwritten was caught. x86- is untouched: nothing there has instructions for this- [#171](https://github.com/kazu-yamamoto/crypton/pull/171)-* perf(sha1): use the ARMv8 SHA-1 instructions. The AArch64 paths for- SHA-256 and SHA-512 went in with #104 and #110 and x86 got its SHA-1- instructions in [#165](https://github.com/kazu-yamamoto/crypton/pull/165),- but the AArch64 SHA-1 ones were never used -- and they are part of the same- optional feature as the SHA-256 ones, so every processor that has those has- these. SHA1C, SHA1P and SHA1M each do four rounds with one of the three- round functions, SHA1H carries E from one group to the next, and SHA1SU0- and SHA1SU1 do the message schedule between them. On Apple silicon: 1272- to 3180 MB/s, against openssl 3.6's 3350 on the same machine. Checked- against the hardware rather than through an emulation of the instructions,- the machine here having them: the digests agree with the generic- implementation over every message length from 0 to 2000, with each input- split in two updates- [#170](https://github.com/kazu-yamamoto/crypton/pull/170)-* perf(poly1305): four blocks at a time with NEON. AArch64 had only the- scalar loop, whose five 26-bit limbs and 32-bit multiplies are the shape a- 32-bit machine wants. This is the arithmetic of the AVX2 path in NEON,- written as a transliteration of that file rather than a fresh formulation,- since the maths there is already pinned by the known-answer tests; what- differs is the width, AVX2 holding four 64-bit products in a register where- NEON holds two, so each product becomes a pair and the limbs are packed- back into four 32-bit lanes before the next multiply. On Apple silicon:- Poly1305 2783 to 4840 MB/s, and ChaCha20-Poly1305 together 1164 to 1368.- Checked against the scalar implementation over forty keys and every message- length from 0 to 400, with each input split in two updates. Also measured- and left alone: BLAKE2b at 1612 MB/s against openssl's 1378, the reference- C being the faster of the two- [#169](https://github.com/kazu-yamamoto/crypton/pull/169)-* perf(modes): stop the generic cipher modes allocating per byte. Counter- mode with a cipher whose modes are not in C ran at a third of what the same- cipher managed in ECB, and at an eighth for Blowfish, for two reasons- outside the cipher. The counters were built one at a time by `ivAdd`,- which allocates a block and walks the whole width of the counter from the- original for each of them; they are now one buffer filled in place. And- the exclusive or was `Data.ByteArray`'s, which walks a byte at a time- through an IO applicative -- 420 MB of heap for 8 MiB of counter mode,- against 17 MB for the same data through ECB, which is fifty bytes allocated- per byte produced and cost more than the cipher did. There is a- `crypton_memxor` to call instead, a pass of words, which the modes and CMAC- use. The serial modes also took each block as a copy and take shared- slices now. On Apple silicon, counter mode: Camellia-128 79.7 to 285.9- MB/s, Blowfish 60.4 to 282.6, DES 46.2 to 114.9, CAST5 40.2 to 86.6,- Twofish-128 37.4 to 57.6, 3DES 25.2 to 36.3, and CBC and CMAC by a third to- a half as much again. AES is unchanged: its modes are in C and never came- this way- [#168](https://github.com/kazu-yamamoto/crypton/pull/168)-* build: compile the C at -O3, which is what came of looking at P-256 against- openssl. The comparison in the problem list was wrong -- a base point- multiplication here against openssl's ECDH, which is a variable point one --- and measured properly P-256 is 2.8 to 3.3 times slower rather than the 1.27- claimed. The time is in the field arithmetic, five 51-bit limbs in- Montgomery form at 44.4 ns a multiplication, against hand-written assembly- using `mulx`, `adcx` and `adox`; a four-limb saturated Montgomery- multiplication written in C to see what a compiler would give measured 41.3- ns, so that is not the way in. What did move is the optimisation level GHC- passes: a P-256 base point multiplication goes from 71.0 to 59.8 us on x86-64- and 26.0 to 24.3 on Apple silicon, AES-128-GCM from 3455 to 3708 MB/s and- AES-128-OCB from 2187 to 2484, with ChaCha20, Poly1305, SHA-1 and MD5 within- a couple of per cent either way. The masked selections in the curve and- field code compile to no conditional jumps at either level- [#167](https://github.com/kazu-yamamoto/crypton/pull/167)-* refactor(aes): drop the keystream generator nobody can call. `genCTR` and- `genCounter` are exported from a module in `other-modules`, so nothing- outside the library could reach them and nothing inside used them; the only- mention left was a test commented out since the cryptonite days. They were- also the slowest thing in the file, a block at a time through the- single-block entry point at 715 MB/s where counter mode does 5788, and the- three ways of fixing that are each worse than removing them: counter mode- over zeros costs Apple silicon a fifth, counters through ECB costs both, and- a keystream loop written out per key size is eighty lines for an API no- caller can see. Also declares `crypton_aes_encrypt_ctr` and- `crypton_aes_encrypt_c32` in the header, which had them defined and imported- but never declared- [#166](https://github.com/kazu-yamamoto/crypton/pull/166)-* perf(sha1): use the Intel SHA extensions on x86-64. The extension that- carries the SHA-256 instructions carries four for SHA-1 as well, and the same- cpuid bit answers for both, so this is one file and one branch. On an AMD- EPYC 7763: 725.1 to 1363.8 MB/s, against openssl's 1668.2 on the same- machine. 1.9x, where the SHA-256 instructions were worth 4.8x -- SHA-1's- rounds are cheaper to begin with, so there is less for an instruction to- replace. The sequence was checked by replacing the four instructions with C- that follows the SDM and comparing against the generic implementation over- every length from 0 to 1024, which found the same missing schedule step- [#155](https://github.com/kazu-yamamoto/crypton/pull/155) had- [#165](https://github.com/kazu-yamamoto/crypton/pull/165)-* docs(sidechannel): say what the modules that still work in `Integer` keep- from the clock, and fix the two places where something could be done about- it. ElGamal inverted the shared secret with the extended Euclidean- algorithm, whose steps follow the bits it is given -- the modulus is prime,- so Fermat reaches it. Its signing inverts the ephemeral value modulo an even- number, where Fermat does not reach, so `sign` blinds instead: the algorithm- is handed that value times a fresh random unit and the blinder divided out- afterwards. What is left is written down rather than fixed -- the Jacobi- symbols Rabin takes modulo its private primes, and the cost of `Integer`- arithmetic following the size of the numbers -- and `Crypto.Cipher.AES` now- says which implementation a machine gets and that the fallback, being- table-driven, is not constant time- [#164](https://github.com/kazu-yamamoto/crypton/pull/164)-* perf(poly1305): shorten the carry chain and stop the AVX2 loop spilling. The- carries go in pairs, since the two halves of that chain do not depend on each- other; the powers of r are read from memory, there being sixteen registers- and ten of them wanted for the accumulator and the products; and the message- is added limb by limb as the block comes apart rather than five limbs being- formed first. 4203 to 4452 MB/s, and ChaCha20-Poly1305 together from 1412 to- 1488. What is left is the instruction count: 107 per 64 bytes, of which 25- are the multiply- [#163](https://github.com/kazu-yamamoto/crypton/pull/163)-* perf(chacha): combine as the keystream comes out of the registers. All three- vector implementations wrote it to a buffer on the stack and read it back to- exclusive-or it with the input, which is a pass over every byte for something- the registers were already holding. 2128 to 2230 MB/s on x86-64, and nothing- on Apple silicon, where the round trip was free. Measured while doing it:- the AVX2 path already did eight blocks at a time, and what is left of the gap- to openssl in this AEAD is Poly1305 rather than the cipher- [#162](https://github.com/kazu-yamamoto/crypton/pull/162)-* perf(sha): compute the message schedule in vector registers on x86. SHA-512- has no instruction there and SHA-256 has none on a processor older than- Goldmont or Zen, which includes the Ice Lake and Cascade Lake server parts.- The rounds are a chain and stay where they are; the schedule is a quarter of- the work, comes out four words at a time and depends on nothing but the- message, so it goes into the vector registers and runs alongside rounds that- need the general ones. SHA-256 228 to 314 MB/s, SHA-512 354 to 480- [#161](https://github.com/kazu-yamamoto/crypton/pull/161)-* perf(gcm): take the GHASH of the group before, alongside this group's rounds.- Held a group apart the multiply and the rounds run through each other, where- in step neither could start until the other finished. With it, the multiply- called directly rather than through a branch pointer the compiler cannot see- through, and the round keys read from memory rather than spilled: AES-128-GCM- 2797 to 3458 MB/s and AES-256-GCM 2458 to 3053. openssl does 4895 and 4205- on the same machine; the rest of that is instruction-level interleaving,- which does not survive being written in intrinsics- [#160](https://github.com/kazu-yamamoto/crypton/pull/160)-* perf(ocb): drive OCB through the ECB paths a group at a time. It ran one- block at a time through the single-block entry point and so cost four times- what GCM costs, for a mode that does less work than GCM. The offsets have to- be worked out in order but the block cipher calls under them do not depend on- each other, so eight go through ECB together. OCB-128 1130 to 3500 MB/s on- Apple silicon and 694 to 2173 on x86-64, the authenticated data 1141 to 5900- and 692 to 2526. CCM is unchanged and stays that way: what is left there is- CBC-MAC, where each block waits for the one before it- [#159](https://github.com/kazu-yamamoto/crypton/pull/159)-* test(aes): run the XTS vectors, and add OCB and CCM at 192 and 256 bits. The- XTS known-answer tests never ran: the call was commented out and the test it- would have called did not compile, so vectors at both key sizes sat in the- tree unused. XTS is defined only for a 128-bit block, which the general KAT- runner cannot promise, so it gains a counterpart for a cipher that can. OCB- and CCM had vectors at 128 bits only. 2613 examples to 2679- [#158](https://github.com/kazu-yamamoto/crypton/pull/158)-* perf(aes): build the AArch64 key schedule with the instructions rather than- the S-box table. The AArch64 path expanded a key by calling the generic- implementation and then inverting the round keys, so every schedule went- through sixteen lookups at addresses derived from the key -- a small thing- next to the per-block indexing the extensions exist to remove, but a key- schedule is what an attacker most wants out of a cache, and x86 has never- needed the table. AArch64 has no counterpart to AESKEYGENASSIST, but AESE- against a zero key is SubBytes and ShiftRows, and a word given to it in all- four columns comes back as SubWord in each of them. The words stay in- vector registers throughout, which is what makes it free: moving each one to- a general register for the instruction and back cost more than the- instruction did, 87 to 144 ns for an AES-128 schedule, where keeping them in- registers gives 81.4- [#157](https://github.com/kazu-yamamoto/crypton/pull/157)-* perf(aes): AES-192 through the processor's AES instructions. Every 192-bit- slot in the branch table was left at the generic code, on x86 and on AArch64- alike, so a 192-bit key got the table-driven software AES while 128 and 256- got the instructions. It was 164 times slower for counter mode on the x86- machine measured and 62 on Apple silicon, and it was also the only key size- whose data path indexes a table with bytes derived from the key -- a caller- who picks AES-192 over AES-128 for a wider margin was quietly given a weaker- one. Counter mode then GCM, before and after: Apple silicon 152.7 to 9452.0- MB/s and 112.3 to 7049.3, x86-64 40.4 to 6635.1 and 39.9 to 2633.5. Both- implementations were already written once per key size, so this instantiates- them again at twelve rounds; x86 also needed the 192-bit schedule, which- does not fall into 128-bit pieces the way the other two do- [#156](https://github.com/kazu-yamamoto/crypton/pull/156)-* perf(sha256): use the Intel SHA extensions on x86-64, which is what issue- [#31](https://github.com/kazu-yamamoto/crypton/issues/31) reports -- SHA-256- four to eight times slower than sha256sum and openssl, both of which use the- processor's instructions. AArch64 got its instructions in #104 and is at- parity with them; x86 had nothing. SHA256RNDS2 does two rounds at a time and- SHA256MSG1 and SHA256MSG2 help with the message schedule, so a block costs- four groups of sixteen instructions instead of sixty-four rounds of scalar- work. On an AMD EPYC 9V74: 338.3 to 1612.7 MB/s, against openssl's 1783.8 on- the same machine. The extensions arrived with Goldmont and Ice Lake at Intel- and with Zen at AMD, far later than AES-NI, so a processor without them is- ordinary rather than ancient: the code sits behind a target attribute and a- cpuid question, and the plain C stays for everything else- [#155](https://github.com/kazu-yamamoto/crypton/pull/155)-* perf(bcrypt): Blowfish, and the key setup bcrypt wraps it in, in C. bcrypt- is a cost parameter and a promise that the cost is paid, and what pays it is- the Blowfish key schedule; in Haskell that cost about twice what the usual- implementations charge, so a hash of a given length of time had to be asked- for with a lower cost than elsewhere. Cost 8 goes from 25.97 to 9.98 ms,- cost 10 from 102.01 to 39.79, cost 12 from 418.78 to 159.41, `bcrypt_pbkdf`- from 109.76 to 40.38, and Blowfish over 4 KiB from 0.05 to 0.01 -- at cost- 10 that is 39.8 ms against the 52 `htpasswd` takes on the same machine. The- Haskell cipher goes with it, so there is one implementation rather than two,- and nothing exposed changes- [#154](https://github.com/kazu-yamamoto/crypton/pull/154)-* perf(prime): fewer Miller-Rabin rounds for a candidate nobody chose. A- number handed over may have been built to pass, and against that the only- thing to go on is that a round catches three quarters of the composites- there are, so `isProbablyPrime`, `findPrimeFrom` and `findPrimeFromWith`,- which all take their number from the caller, keep their thirty rounds. A- candidate drawn here is the case Damgard, Landrock and Pomerance worked out- and Table 4.4 of the Handbook of Applied Cryptography tabulates:- `generatePrime` and `generateSafePrime` now use twice what it asks for one- chance in 2^80, capped at the thirty they had, which leaves the chance far- under one in 2^100 at every size. With the candidates held fixed,- `generatePrime 1024` goes from 28.5 to 16.9 ms and an RSA-2048 key from 52.9- to 39.2- [#153](https://github.com/kazu-yamamoto/crypton/pull/153)-* fix(rsa): work the private exponent out without the extended Euclidean- algorithm. The modulus is the secret there, so multiplying the value by a- random number hides nothing; what does is that `e` is public. Whatever `d`- is, `e * d = 1 + k * phi` for some `k` under `e`, and reading that modulo- `e` gives `k` as an inverse modulo a number of a handful of bits, which for- a prime `e` is Fermat; `d` is then an exact division. What phi touches is a- remainder and a division, and nothing in either follows it- [#152](https://github.com/kazu-yamamoto/crypton/pull/152)-* perf(f2m): ask aarch64 for its carry-less multiply as well. #148 used PMULL- only where the compiler had been told the machine has the crypto- extensions, which is so on Apple and not on a Linux built for the bare- ARMv8 baseline, though every processor that runs such a build has it. It is- now compiled behind an attribute and the machine asked at run time, through- the auxiliary vector on Linux and Android, elf_aux_info on FreeBSD and a- sysctl on Apple: sect283k1 421.2 to 168.9 us there, sect571r1 2289.2 to- 579.9- [#151](https://github.com/kazu-yamamoto/crypton/pull/151)-* refactor(ecc): one multiplication for both of the curve APIs, and one place- for each buffer's size. Which path a point multiplication takes was written- out twice, and the copy in `Crypto.ECC.Simple.Prim` cannot be reached from- outside the library on a curve over a binary field, so the suite never ran- it; it moves to the internal module both already share, which makes the copy- nobody can call the same code everybody runs. The two buffers for a C call- that still had their size written out separately from the offsets into them- now take both from one list, as the one that was wrong in #141 does -- the- note there records that neither valgrind nor the debug RTS catches that- mistake, both having been tried- [#150](https://github.com/kazu-yamamoto/crypton/pull/150)-* perf(f2m): use the x86 carry-less multiply where the processor has it.- PCLMULQDQ is not part of the x86-64 baseline, so the cpuid the package- already runs for AES-NI reports one more bit and the multiplication that- uses the instruction sits behind an attribute. Measured through Rosetta,- which translates rather than runs it, so the ratio is what to read:- sect283k1 560.4 to 177.5 us, sect571r1 3049.6 to 623.9- [#149](https://github.com/kazu-yamamoto/crypton/pull/149)-* perf(f2m): do the binary field arithmetic in C. The ladder of #142 spent- nearly all its time on one thing -- a carry-less multiplication, which- ordinary arithmetic does not give and which in Haskell was `Integer` shifts- and exclusive ors, about 4 us for a 283-bit multiplication. The field and- the ladder over it are now C, with the processor's instruction where there- is one and four interleaved groups of bits where there is not, folding for- the reduction and Fermat for the inverse. sect163k1 3431 to 67.4 us with- the instruction and 117.3 without, sect283k1 10181 to 157.6 and 386.2,- sect571r1 40469 to 521.6 and 2101.0. It is also constant time, which the- Haskell ladder was not- [#148](https://github.com/kazu-yamamoto/crypton/pull/148)-* perf(bignum): start the doubling for `R^2 mod m` at the highest power of two- under the modulus rather than at one, which for a modulus that fills its- limbs is half the steps. Two to three percent of a curve operation, and- every curve operation and every `expSafe` pays for it once. Folding instead- of Montgomery for the primes shaped `2^k - c` was written and measured- alongside it and is not here: it is slower in this representation, 87.8 ns- against 76.8 for a 521-bit multiplication, because the shift down by `k`- costs more than the reduction pass it replaces when `k` does not land on a- limb boundary- [#147](https://github.com/kazu-yamamoto/crypton/pull/147)-* perf(ecc): keep a table of the multiples of each curve's base point, which- is the point signing and making a key multiply and the only one worth a- table. A multiplication with it is one addition per four bits and no- doublings: secp256k1 211.1 to 59.8 us, secp384r1 519.3 to 144.2, secp521r1- 1047.7 to 283.0, and ECDSA P-384 signing 556.4 to 179.9 on both elliptic- curve APIs, which share the table. A table is built when a curve is first- asked for one -- 2.8 ms for secp256k1, 5.5 for secp384r1, 10.6 for secp521r1- -- and is 221 KB and 456 KB for the last two, so it pays for itself after- about fifteen multiplications- [#146](https://github.com/kazu-yamamoto/crypton/pull/146)-* perf(bignum): take the limbs four and two at a time as well as eight in the- loop every modular multiplication is built out of. Four and six limbs, which- is what most of the curves want, fell entirely to the one-at-a-time tail- before: a field multiplication at six limbs goes from about 58 to 49 ns,- secp384r1 scalar multiplication from 596.7 to 519.3 us and ECDSA P-384- signing from 645.0 to 556.4. Specialising the sizes further, which is what a- generated implementation would do, measures about 4% more and is not here- [#145](https://github.com/kazu-yamamoto/crypton/pull/145)-* fix(rsa): keep the blinding factor out of the extended Euclidean algorithm.- The blinder is a random number and its inverse, and the inverse went through- an algorithm whose steps follow the number handed to it -- the number the- blinding rests on, and unlike the other inverses this one is worked out once- per operation rather than once per key. `n` being composite leaves no- Fermat to fall back on, so the algorithm is handed the factor multiplied by- sixteen fresh random bytes and its answer multiplied by them again, which- leaves the inverse wanted and shows the algorithm nothing to do with it. In- IO, where every draw of randomness goes to the system, `generateBlinder`- goes from 74 to about 120 us and a PKCS#1 v1.5 `signSafer` from 719 to about- 765; under a DRG the caller carries, 24.7 to 24.9- [#144](https://github.com/kazu-yamamoto/crypton/pull/144)-* fix(rsa): work `qinv` out without the extended Euclidean algorithm. Making- a key inverts one prime modulo the other and both of them are the key- itself, so that inverse is now Fermat's little theorem through `expSafe`,- which the other prime being prime allows: 308.6 us against 10.1, on a key- that takes tens of milliseconds to make. Making a key cannot be constant- time -- the search for the primes takes as long as it takes -- but what that- leaks is about the search rather than about the primes it settles on, and- the haddock now says which is which- [#143](https://github.com/kazu-yamamoto/crypton/pull/143)-* perf(ecc): a ladder for the curves over a binary field. These were the last- multiplication whose cost followed the scalar: an affine double-and-add, one- addition for every bit that was set and none for the others, which on- sect283k1 ran from 9665 us for a scalar with two bits set to 17958 for one- with 270. It is now Montgomery's ladder, which carries the multiples of two- consecutive numbers -- their difference being the point is what lets it- carry only their x coordinates -- and spends one addition and one doubling- on every bit whichever way it goes, working the y out at the end from the- two x it is left with, so one division does for the whole multiplication- where the affine code had one per step. The multiplication is now flat, and- quicker: sect163k1 4428 to 3431 us, sect233r1 9304 to 6806, sect283k1 13797- to 10181, sect409k1 30826 to 20584, sect571r1 61931 to 40469. Uniform is- not constant time -- these are `Integer` operations, whose cost follows the- values -- and the point with no x, which is its own negation, keeps the code- that was there- [#142](https://github.com/kazu-yamamoto/crypton/pull/142)-* perf(ecc): multiply points in C on curves over a prime field. P-256 has had- a C implementation all along; every other prime curve -- P-384, P-521,- secp256k1 and the rest -- multiplied points with `Integer` arithmetic, which- cannot be constant time, since what an `Integer` operation costs follows the- value it is given. The C walks four bits of scalar at a time, taking the- multiple to add from a table of sixteen that it reads by touching every- entry and keeping one with a mask, and its addition and doubling are the- complete formulas of Renes, Costello and Batina, which answer for every pair- of points with no case to choose between. A P-384 multiplication goes from- 1557 to 585 us and no longer follows the scalar, ECDSA P-384 signing from- 1700 to 636 us, P-521 from 1942 to 1123. Binary curves are unchanged, and a- point that is not on the curve keeps the answer the Haskell gives it- [#141](https://github.com/kazu-yamamoto/crypton/pull/141)-* fix(ecc): add at every bit in the prime-curve multiplication, which laziness- was skipping. The multiplication adds at every bit, set or not, so that its- cost follows the width of the curve's order rather than the scalar -- but- the addition was a binding only one branch of the following `if` used, so at- a bit that was not set it stayed a thunk and was never worked out. The cost- followed the number of bits set in the scalar, which is the nonce when- signing and the private key in ECDH: on P-384, 765.8 us for a scalar with- two bits set against 1671.5 for one with 383, in a straight line between.- Both copies of the multiplication had it, so both elliptic curve APIs were- affected on every prime curve but P-256- [#140](https://github.com/kazu-yamamoto/crypton/pull/140)-* fix(ecdsa): keep the P-256 signature out of `Integer` arithmetic. The- scalar handed to the C implementation was reduced with `mod`, a division,- whose steps follow the number being divided -- the nonce when signing, the- private key in ECDH. Twice the order is more than 256 bits hold, so a- scalar that fits is brought under the order by one masked subtraction- instead. The second half of a signature, `kInv * (z + r * d)`, was- `Integer` arithmetic as well, and now goes through `scalarAdd` and- `scalarMul`, which on P-256 are the C implementation's fixed-width- arithmetic. What is left on that curve is the conversion between `Integer`- and fixed-width scalars, which is also what it costs: signing goes from 34.1- to 39.3 us, and ECDH and the other curves are unchanged- [#139](https://github.com/kazu-yamamoto/crypton/pull/139)-* fix(dsa,ecdsa): invert the signing nonce without a side channel. Both- inverted it with the extended Euclidean algorithm, whose step count and- branches follow the bits of what it is given -- and a handful of signatures- whose nonces are partly known give the private key away, so the nonce is- worth as much as the key. `Crypto.Number.ModArithmetic.inverseSafe` works- the inverse out with Fermat's little theorem through `expSafe` instead,- falling back on `inverse` when the modulus turns out not to be prime, so- every answer is the one it was. On P-256 the C implementation does it.- Signing costs a little more: ECDSA P-256 30.6 to 34.1 us, ECDSA P-384 678.7- to 703.4, DSA-2048 422.5 to 437.1. Verification inverts a value that- arrives in the signature and is left alone- [#138](https://github.com/kazu-yamamoto/crypton/pull/138)-* perf(number): square, and multiply, faster in `expSafe`. The product and- the Montgomery reduction are now a full product followed by a reduction- rather than interleaved, built out of one loop that takes its limbs eight at- a time, and squaring works out only the products on one side of the diagonal- and doubles their sum. At 2048 bits the constant-time exponentiation goes- from 3.59 to 2.15 ms, which is 1.4x GMP's own rather than 2.4x; RSA-2048- signing goes from 0.80 to 0.63 ms and DH-2048 `getShared` from 2.43 to 1.67- [#137](https://github.com/kazu-yamamoto/crypton/pull/137)-* fix(number): make `expSafe` hide the exponent again. It asked integer-gmp- for `powModSecInteger` and fell back on the ordinary `powModInteger` when- that was missing; since integer-gmp 1.1 it is always missing, so on every- GHC this package supports `expSafe` was the same windowed exponentiation as- `expFast`, table indexed by the exponent's bits, for RSA, DSA, DH, ElGamal- and Rabin alike. It now goes to C: four bits of exponent at a time, the- table of sixteen read by touching every entry and keeping one with a mask,- and a Montgomery multiplication whose final subtraction is masked too. The- exponent's length is still visible, rounded up to a whole 64-bit word, which- is what GMP's own `mpz_powm_sec` lets slip. Hiding the exponent costs 1.6x- at 512 bits and 2.4x at 2048: RSA-2048 signing goes from 0.46 to 0.80 ms and- DH-2048 `getShared` from 1.04 to 2.43- [#136](https://github.com/kazu-yamamoto/crypton/pull/136)-* perf(prime): stop running a Fermat test that Miller-Rabin subsumes. Every- candidate was tested to base 2 before the Miller-Rabin rounds, which begin- with the same base and prove more; the primes it passed paid for it twice- and the composites it caught were nearly all caught by trial division first.- RSA-2048 key generation goes from 55.0 to 32.8 ms- [#135](https://github.com/kazu-yamamoto/crypton/pull/135)-* perf(f2m): reduce the binary field by folding the top back in rather than- taking a step per bit of excess, square a byte at a time through a table of- the patterns a byte spreads into, and take four bits of a multiplier at a- time rather than one. On the 283-bit field, squaring goes from 5440 to 2068- ns and multiplication from 7526 to 4086. A scalar multiplication there is- still affine, so it inverts once per addition, which is where its time now- goes- [#134](https://github.com/kazu-yamamoto/crypton/pull/134)-* perf(ecc): fold instead of dividing in the generic prime-curve arithmetic,- and add the point being multiplied as the affine point it is. These primes- are `2^k - c` with `c` far smaller, so the top half of a product folds back- in with a shift, a multiplication and an addition, where dividing costs four- times as much -- above 256 bits, below which the folding costs more than it- saves. P-521 scalar multiplication goes from 892 to 492 us and P-384 from- 684 to 572- [#133](https://github.com/kazu-yamamoto/crypton/pull/133)-* perf(ecc): route P-256 through the C implementation the library already had.- `Crypto.PubKey.ECDSA` reached `cbits/p256`; `Crypto.PubKey.ECC.*`, the older- and more widely used API, never did. ECDSA signing goes from 590 to 28.7 us,- verification from 726 to 91.4, and `getShared` from 1177 to 96. On P-256- that multiplication is now constant time, where the generic code branches on- the scalar at every bit- [#132](https://github.com/kazu-yamamoto/crypton/pull/132)-* Breaking change: perf(camellia): put Camellia in C, 40 to 321 MiB/s. The- round function ran a byte at a time in Haskell; generating the tables that- take a byte straight to its contribution gained 14%, and the rest was the- language. Input that is not a whole number of blocks now raises, where the- tail of the answer used to be uninitialised memory- [#131](https://github.com/kazu-yamamoto/crypton/pull/131)-* Breaking change: perf(twofish): walk the blocks once and carry them in words- rather than appending each result to what came before and going through lists- per block. 2 MiB goes from 0.14 to 56 MiB/s, and the rate no longer falls as- the message grows. Input that is not a whole number of blocks now raises,- where it used to come back longer than it went in- [#130](https://github.com/kazu-yamamoto/crypton/pull/130)-* perf(modes): cut the message without copying the rest of it in the generic- block cipher modes, which every cipher but AES uses, and hand whole slices to- the cipher in the modes whose blocks do not depend on one another. Camellia- in CBC goes from 1.8 to 22.9 MiB/s at 1 MiB, DES CBC decryption from 0.5 to- 83, and every figure is now flat in the message length where it used to fall- [#129](https://github.com/kazu-yamamoto/crypton/pull/129)-* Breaking change: perf(des): put DES in C. It was carried over lists of- `Bool`, one cons cell per bit, with the key schedule recomputed for every- block: 0.04 MiB/s, and 3DES 0.013, against 105 and 41 for OpenSSL. They are- now 112 and 37. Input that is not a whole number of blocks now raises, where- the tail of the answer used to be uninitialised memory- [#128](https://github.com/kazu-yamamoto/crypton/pull/128)-* perf(cmac): slice the message rather than copying what is left of it once per- block, and chain through CBC, which is what CMAC's chaining is. A MAC over- 4 MiB goes from 0.36 to 1628 MiB/s, which is the speed of AES-CBC itself- [#127](https://github.com/kazu-yamamoto/crypton/pull/127)-* fix(rabin): decode OAEP without early exits, as- `Crypto.PubKey.RSA.OAEP.unpad` has since #91. The difference is not- measurable against the cost of mask generation, and is structural: the scan- across the padding no longer depends on the data- [#126](https://github.com/kazu-yamamoto/crypton/pull/126)-* Breaking change: fix(rabin): refuse a ciphertext or a signature that is not- below the modulus, and a ciphertext carrying a leading zero octet. Squaring- and the square roots that undo it work modulo n, so Basic and Rabin-Williams- decrypted `c + n` to whatever `c` decrypted to, and all three schemes verified- `s + n`, and `-s`, wherever they verified `s`. `Basic.signWith` also refuses a- padding whose first octet is zero, which the signature cannot carry: about one- signature in 256 was one its own `verify` rejected- [#125](https://github.com/kazu-yamamoto/crypton/pull/125)-* fix(prime): derive the Miller-Rabin witnesses from the number being tested and- from a secret drawn once per process. They came from one generator made once- and shared by every call, so the witnesses for one number were the witnesses- for every number, and testing a number again told the caller nothing it had- not already been told. This is the path every GHC since 9.0 takes, integer-gmp- 1.1 having no Miller-Rabin of its own- [#124](https://github.com/kazu-yamamoto/crypton/pull/124)-* docs(elgamal): say what `signWith` requires of its ephemeral value: the range- is 1 to p-2, not the "between 0 and p-1" the haddock claimed, and the value is- a private key that a signature discloses if it is reused or revealed- [#123](https://github.com/kazu-yamamoto/crypton/pull/123)-* Breaking change: fix(afis): give `split` and `merge` one answer for a parameter- they cannot use. They had four between them, including a division by zero for- an expand count of zero and, for a count of one, handing the diffused data back- as though it were the secret- [#122](https://github.com/kazu-yamamoto/crypton/pull/122)-* Breaking change: fix(rsa): refuse a ciphertext or a signature whose integer- representative is not below the modulus, which RFC 8017 requires in sections- 5.1.2 and 5.2.2. `PKCS15.decrypt` and `OAEP.decrypt` decrypted `c + n` to the- same message as `c`, and `PSS.verifyDigest` accepted `s + n` wherever it- accepted `s`- [#121](https://github.com/kazu-yamamoto/crypton/pull/121)-* fix(otp): search the HOTP resynchronization window without early exits. The- time taken read out both where in the window the client's counter was found- and how many of the submitted values were right -- the second of which the- answer itself does not give, being `Nothing` either way. A call now costs one- HMAC per counter in the window plus one per extra value, every time- [#120](https://github.com/kazu-yamamoto/crypton/pull/120)-* Breaking change: fix(kdf): report a refused parameter as a `CryptoError` rather- than as an `ErrorCall` carrying a string, with a `'`-suffixed variant of each- entry point returning `CryptoFailable`. PBKDF2 had no validation at all: a- negative output length reached `memSet` and killed the process with SIGBUS, and- an iteration count of zero returned 32 bytes of zeroes- [#119](https://github.com/kazu-yamamoto/crypton/pull/119)-* perf(xts): take eight blocks at a time on AArch64 and x86-64, and dispatch XTS- decryption through the branch table, which it had never used. AArch64 goes- from 1200 to 7742 MiB/s encrypting and 1166 to 7763 decrypting, x86-64 from- 1220 to 3464 and from 594 to 3461- [#118](https://github.com/kazu-yamamoto/crypton/pull/118)-* perf(poly1305): take four blocks at a time with AVX2 on x86-64, folding the- lanes back together weighted by the powers of r. 1347 to 4137 MiB/s- [#117](https://github.com/kazu-yamamoto/crypton/pull/117)-* perf(ecc): work in Jacobian coordinates in both generic prime-field scalar- multiplications, and say in `Crypto.ECC` which curves branch on a secret- scalar. P-384 and P-521 ECDSA are 2.3x: signing goes from 3.36 to 1.46 ms and- from 5.96 to 2.61 ms. P-256, which has its own C implementation, is unaffected- [#116](https://github.com/kazu-yamamoto/crypton/pull/116)-* Breaking change: fix(padding): bound PKCS#7 padding by the block rather than by- the whole input, which had let a block of sixteen accept a claim of twenty, and- refuse a `ZERO` size of zero rather than dividing by it. What `ZERO` can and- cannot undo is now written down- [#115](https://github.com/kazu-yamamoto/crypton/pull/115)-* perf(gcm): give x86 its own GCM decryption loop. It fell to the generic one,- which calls the block function once per block, and ran at a quarter the speed- of encryption; both directions now take eight blocks at a time and fold their- GHASH into one reduction. AES-256-GCM decryption goes from 561 to 2733 MiB/s- and AES-128 from 667 to 3150- [#114](https://github.com/kazu-yamamoto/crypton/pull/114)-* perf(chacha): take eight blocks at a time with AVX2 where the machine has it,- with the cpuid and XGETBV checks that decide. ChaCha20 on x86-64 goes from- 900 to 2074 MiB/s- [#113](https://github.com/kazu-yamamoto/crypton/pull/113)-* perf(chacha): do four blocks at a time with SSE2 on x86-64, where the cipher- had no vector code at all. ChaCha20 goes from 493 to 900 MiB/s- [#112](https://github.com/kazu-yamamoto/crypton/pull/112)-* perf(chacha): do four blocks at a time with NEON on AArch64. ChaCha20 goes- from 1025 to 1955 MiB/s- [#111](https://github.com/kazu-yamamoto/crypton/pull/111)-* feat(sha512): use the ARMv8.2 SHA-512 instructions on AArch64, which SHA-384- and the truncated SHA-512/t variants share. Hashing 1 MiB goes from 1.53 ms- to 597 us. The extension is optional, so it is asked for at runtime on both- Apple and Linux rather than assumed- [#110](https://github.com/kazu-yamamoto/crypton/pull/110)-* perf(gcm): drive GCM from AArch64 rather than the generic loop, with a group- of eight blocks folding into a single GHASH reduction. AES-128-GCM goes from- 4030 to 8266 MiB/s and AES-256 from 4043 to 7172- [#109](https://github.com/kazu-yamamoto/crypton/pull/109)-* perf(aes): specialise the AArch64 code by key size and interleave eight- blocks, and give CTR its own loop. AES-256 ECB goes from 3886 to 15991- MiB/s, CTR from 2935 to 13567 and CBC decryption from 4366 to 15807- [#108](https://github.com/kazu-yamamoto/crypton/pull/108)--* perf(aes): build the AES-NI paths on Windows, which was missing from the list of- systems that compile them. Windows builds have been doing AES, and GHASH with it,- in the generic C- [#107](https://github.com/kazu-yamamoto/crypton/pull/107)-* fix(armv8): compile the AArch64 sources on a toolchain whose baseline lacks the- crypto extensions. They had not built with GCC on AArch64 Linux since #100; CI now- builds and tests there- [#106](https://github.com/kazu-yamamoto/crypton/pull/106)-* perf(gcm): fold four GHASH blocks into one reduction. AES-256-GCM is 1.6x at 1 KiB- and 2.6x at 64 KiB on Apple silicon, and the x86 paths gain the same structure- [#105](https://github.com/kazu-yamamoto/crypton/pull/105)-* perf(sha256): use the ARMv8 SHA-2 instructions on AArch64. SHA-256 and SHA-224 are- 5.5x- [#104](https://github.com/kazu-yamamoto/crypton/pull/104)-* ci: keep the macOS jobs from queueing behind each other, and supersede a branch's- earlier run- [#103](https://github.com/kazu-yamamoto/crypton/pull/103)-* perf(aes): use PMULL for GHASH on AArch64- [#102](https://github.com/kazu-yamamoto/crypton/pull/102)-* ci: ask cabal where its caches live rather than assuming, and keep the build- products in the cache- [#101](https://github.com/kazu-yamamoto/crypton/pull/101)-* perf(aes): use the ARMv8 cryptographic extensions on AArch64. With the GHASH work- in #102 and #105, AES-256-ECB goes from 121 to 2992 MiB/s and AES-256-GCM from 92 to- 2318 MiB/s on Apple silicon- [#100](https://github.com/kazu-yamamoto/crypton/pull/100)-* build(bench): move the benchmarks from gauge, which is no longer maintained, to- tasty-bench, and let them resolve on a current GHC- [#99](https://github.com/kazu-yamamoto/crypton/pull/99)-* Breaking change: fix(padding): reject a `PKCS7` block size outside 1..255. `pad`- raises and `unpad` returns `Nothing`, where both previously narrowed the size to a- `Word8` and silently agreed on the wrong value- [#98](https://github.com/kazu-yamamoto/crypton/pull/98)-* feat(elgamal): fix `Crypto.PubKey.ElGamal` and expose it- [#97](https://github.com/kazu-yamamoto/crypton/pull/97)-* docs(bcrypt): say that only the first 72 bytes of a password count- [#96](https://github.com/kazu-yamamoto/crypton/pull/96)-* test: move the test suite from tasty to hspec, with hspec-discover. `cabal-version`- is now 2.0- [#95](https://github.com/kazu-yamamoto/crypton/pull/95)--* feat(aead): add `tryAeadSimpleDecrypt`, which takes the tag length as its own argument instead of reading it off the supplied tag- [#94](https://github.com/kazu-yamamoto/crypton/pull/94)-* Breaking change: feat(dh): add `tryGetShared` to `Crypto.PubKey.DH` and `Crypto.PubKey.ECC.DH`, reporting a rejected peer value as `CryptoFailable`; `getShared` is now defined in terms of it and so raises a `CryptoError` rather than an `ErrorCall`- [#93](https://github.com/kazu-yamamoto/crypton/pull/93)-* fix(otp): compare TOTP candidates without an early exit- [#92](https://github.com/kazu-yamamoto/crypton/pull/92)-* fix(rsa): drop the early exits from PKCS#1 v1.5 and OAEP unpadding- [#91](https://github.com/kazu-yamamoto/crypton/pull/91)-* Breaking change: fix(argon2): report invalid options as `CryptoFailed` rather than raising, adding `CryptoError_ParameterInvalid` to `CryptoError`- [#90](https://github.com/kazu-yamamoto/crypton/pull/90)-* Breaking change: fix(dh): validate the peer public number, and size the shared secret from `p` rather than `params_bits`- [#89](https://github.com/kazu-yamamoto/crypton/pull/89)-* fix(dsa): do not crash on values that are not invertible modulo `q`- [#88](https://github.com/kazu-yamamoto/crypton/pull/88)-* Breaking change: fix(ecdh): validate the peer point before the exchange- [#87](https://github.com/kazu-yamamoto/crypton/pull/87)-* Breaking change: fix(pkcs15): reject PKCS#1 v1.5 signatures of the wrong length or out of range- [#86](https://github.com/kazu-yamamoto/crypton/pull/86)-* Breaking change: fix(otp): require a digest long enough for RFC 4226 dynamic truncation, which was reading past the end of the MAC- [#85](https://github.com/kazu-yamamoto/crypton/pull/85)-* fix(ecc): accept zero-x P-256 shared secret- [#84](https://github.com/kazu-yamamoto/crypton/pull/84)-* fix(p256): accept valid edge-case points- [#83](https://github.com/kazu-yamamoto/crypton/pull/83)-* Breaking change: fix(hkdf): enforce output length limit- [#82](https://github.com/kazu-yamamoto/crypton/pull/82)-* Breaking change: fix(ed25519): reject non-canonical signatures- [#81](https://github.com/kazu-yamamoto/crypton/pull/81)-* Support GHC 9.14; `tested-with` now covers 9.10.2, 9.12.4 and 9.14.1- [#74](https://github.com/kazu-yamamoto/crypton/pull/74)--### API changes--* New exports: `Crypto.OTP.minimumDigestSize`, `Crypto.PubKey.DH.tryGetShared`,- `Crypto.PubKey.ECC.DH.tryGetShared`, `Crypto.Cipher.Types.AEAD.tryAeadSimpleDecrypt`,- `Crypto.Number.ModArithmetic.inverseSafe`, `Crypto.PubKey.ECC.Prim.scalarInverse`,- `scalarAdd` and `scalarMul`, `Crypto.PubKey.ECC.P256.scalarReduce`,- and the whole of `Crypto.PubKey.ElGamal`, which was present but not exposed.- The variant of an entry point that reports a refusal rather than raising is- named `try` followed by the name it varies, `tryExpand` beside `expand`. A- trailing apostrophe was the obvious spelling and is what these were called- until shortly before release; it collides too easily, since a caller that- imports one of these modules unqualified and has its own `expand'` or- `split'` no longer compiles, and `tls` did. `Safe` was considered and set- aside: this library already uses that suffix for something else, in- `Crypto.Number.ModArithmetic.expSafe` and `inverseSafe` and in- `Crypto.PubKey.ECC.P256.scalarInvSafe`, where it means the value being- worked on stays out of the timing.- The KDFs gained a variant of each entry point that can refuse its parameters,- returning `CryptoFailable` instead of raising: `Crypto.KDF.Scrypt.tryGenerate`,- `Crypto.KDF.BCrypt.tryBcrypt`, `Crypto.KDF.BCryptPBKDF.tryGenerate` and- `tryHashInternal`, `Crypto.KDF.HKDF.tryExpand`, `Crypto.KDF.PBKDF2.tryGenerate` and- `tryFastPBKDF2_SHA1`, `tryFastPBKDF2_SHA256` and `tryFastPBKDF2_SHA512`, and- `Crypto.Data.AFIS.trySplit` and `tryMerge`. These are additions and break nothing.-* Breaking change: `CryptoError_ParameterInvalid` is added to `CryptoError`. It is- appended, so the `Enum` values of the existing constructors are unchanged, but an- exhaustive `case` without a wildcard will warn. Adding a constructor to an exported- datatype is what requires a major version bump under the PVP, which would have been- 1.2.0; this release goes to 2.0.0. Everything else below changes behaviour rather- than types.-* Breaking change: `getShared` in both DH modules raises a `CryptoError` where it- previously raised an `ErrorCall`, since it is now defined in terms of `tryGetShared`.- The same is now true of `Crypto.KDF.Scrypt.generate`, `Crypto.KDF.BCrypt.bcrypt`,- `Crypto.KDF.BCryptPBKDF.generate` and `hashInternal`, and `Crypto.Data.AFIS.split`- and `merge`, each of which is defined in terms of the variant above.-* Breaking change: input that used to be accepted is now rejected -- a digest shorter- than 20 bytes in `Crypto.OTP.hotp`, a signature of the wrong length or out of range- in `Crypto.PubKey.RSA.PKCS15.verify`, an off-curve peer point or a peer public number- outside `1 < y < p-1` in `getShared`, an output beyond 255 blocks in- `Crypto.KDF.HKDF.expand`, a non-canonical Ed25519 signature, and `Options` the- implementation refuses in `Crypto.KDF.Argon2.hash`.-* Breaking change: a value at or above the modulus is now rejected where it used to be- reduced and accepted -- a ciphertext in `Crypto.PubKey.RSA.PKCS15.decrypt` and- `Crypto.PubKey.RSA.OAEP.decrypt`, a signature in `Crypto.PubKey.RSA.PSS.verify`, and- both, along with a negated signature and a ciphertext with a leading zero octet, in- the three `Crypto.PubKey.Rabin.*` schemes.-* Breaking change: parameters that used to be accepted are now refused -- an iteration- count below one or a negative output length in `Crypto.KDF.PBKDF2`, an expand count- below two or a secret of no bytes in `Crypto.Data.AFIS`, a `PKCS7` claim longer than- the block and a `ZERO` size of zero in `Crypto.Data.Padding`, and a signature padding- whose first octet is zero in `Crypto.PubKey.Rabin.Basic.signWith`.-* Breaking change: DES, 3DES, Twofish and Camellia now raise on input that is not a- whole number of blocks, as AES already did. Before, DES and Camellia returned an- answer whose tail was never written -- uninitialised memory -- and Twofish returned- more than it was given, the missing bytes read as zero.-* Breaking change: `Crypto.Data.Padding.pad` raises on a `PKCS7` block size outside- 1..255, and `unpad` returns `Nothing` for one, where both used to narrow the size to- a `Word8` and hand back something other than what was padded.-* No exported function changed its signature.--## 1.1.5--* fix(aead): reject undersized tags- [#80](https://github.com/kazu-yamamoto/crypton/pull/80)-* fix(aes): refuse a zero-length AES-GCM IV- [#79](https://github.com/kazu-yamamoto/crypton/pull/79)-* fix(p256): prevent crashes when validating valid points- [#78](https://github.com/kazu-yamamoto/crypton/pull/78)-* feat(asn1): add SHA-3 HashAlgorithmASN1 instances for PKCS#1 v1.5- [#77](https://github.com/kazu-yamamoto/crypton/pull/77)-* OCB3 conformance- [#76](https://github.com/kazu-yamamoto/crypton/pull/76)--## 1.1.4--* Generic instance for RSA PublicKey and PrivateKey--## 1.1.3--* Ensure that `pointAdd` in `PubKey.ECC.P256` treats the point at infinity as the additive identity.- [#73](https://github.com/kazu-yamamoto/crypton/pull/73)--## 1.1.2--* Preparing `ram` v0.22.-* Generalizing RSA encrypt/decrypt to manipulate ScrubbedBytes directly.--## 1.1.1--* On iOS, ScrubbedBytes based hashing is used for seedNew. On other- plateforms, entropy is used directly as used to be.- [#71](https://github.com/kazu-yamamoto/crypton/pull/71)--## 1.1.0--* Removing "basement" and "memory".- [#67](https://github.com/kazu-yamamoto/crypton/pull/67)---## 1.0.7--* Stop depending on basement, use upstream dependencies instead-* Stop transitively depending on basement by depending on ram.--## 1.0.6--* Fix test failures on less common 64-bit arches.- [#65](https://github.com/kazu-yamamoto/crypton/pull/65)--## 1.0.5--* Setter/Getter for ChaCha counter.- [#63](https://github.com/kazu-yamamoto/crypton/pull/63)-* Add simple interface to generate full blocks- [#60](https://github.com/kazu-yamamoto/crypton/pull/60)-* Avoid `ghc-prim` dependency.- [#61](https://github.com/kazu-yamamoto/crypton/pull/61)--## 1.0.4--* Ed448.sign: avoid extra re-derive of public key.- [#48](https://github.com/kazu-yamamoto/crypton/pull/48)--## 1.0.3--* Make sign of Ed25519/Ed448 safer. The public key parameter is- ignored and its public key is generated from the secret key- parameter to prevent Double Public Key Signing Function Oracle- Attack.- [#47](https://github.com/kazu-yamamoto/crypton/pull/47)--## 1.0.2--* Deterministic Nonce Generation for ECDSA- [#46](https://github.com/kazu-yamamoto/crypton/pull/46)-* ECDSA Signature Normalization.- [#45](https://github.com/kazu-yamamoto/crypton/pull/45)-* Add Full Test Suite from RFC 6979.- [#44](https://github.com/kazu-yamamoto/crypton/pull/44)-* ECDSA with Public Key Recovery.- [#43](https://github.com/kazu-yamamoto/crypton/pull/43)-* Providing necessary features for HPKE.- [#42](https://github.com/kazu-yamamoto/crypton/pull/42)--## 1.0.1--* Update decaf library.- [#38](https://github.com/kazu-yamamoto/crypton/pull/38)-* Add TypeOperators language extension to EdDSA.hs.- [#36](https://github.com/kazu-yamamoto/crypton/pull/36)--## 1.0.0--* Versions follow the standard version policy.-* Removing pthread stuff.- [#32](https://github.com/kazu-yamamoto/crypton/pull/32)--## 0.34--* Hashing getRandomBytes before using as Seed for ChaChaDRG- [#24](https://github.com/kazu-yamamoto/crypton/pull/24)-* Add support for XChaCha and XChaChaPoly1305- [#18](https://github.com/kazu-yamamoto/crypton/pull/18)-* Strict byteArray of IV c- [#16](https://github.com/kazu-yamamoto/crypton/pull/16)--## 0.33--* Add "crypton_" prefix to the final C symbols.- [#9](https://github.com/kazu-yamamoto/crypton/pull/9)--## 0.32--* All C symbols now have the "crypton_" prefix.- [#7](https://github.com/kazu-yamamoto/crypton/pull/7)- [#8](https://github.com/kazu-yamamoto/crypton/pull/8)--## 0.31--* Crypton is forked from cryptonite with the original authors permission.-* Ignoring exceptons from hClose to read the next entropy- [#1](https://github.com/kazu-yamamoto/crypton/pull/1)-* Enabling the support_pclmuldq flag by default.--## 0.30--* Fix some C symbol blake2b prefix to be cryptonite_ prefix (fix mixing with other C library)-* add hmac-lazy-* Fix compilation with GHC 9.2-* Drop support for GHC8.0, GHC8.2, GHC8.4, GHC8.6--## 0.29--* advance compilation with gmp breakage due to change upstream-* Add native EdDSA support--## 0.28--* Add hash constant time capability-* Prevent possible overflow during hashing by hashing in 4GB chunks--## 0.27--* Optimise AES GCM and CCM-* Optimise P256R1 implementation-* Various AES-NI building improvements-* Add better ECDSA support-* Add XSalsa derive-* Implement square roots for ECC binary curve-* Various tests and benchmarks--## 0.26--* Add Rabin cryptosystem (and variants)-* Add bcrypt_pbkdf key derivation function-* Optimize Blowfish implementation-* Add KMAC (Keccak Message Authentication Code)-* Add ECDSA sign/verify digest APIs-* Hash algorithms with runtime output length-* Update blake2 to latest upstream version-* RSA-PSS with arbitrary key size-* SHAKE with output length not divisible by 8-* Add Read and Data instances for Digest type-* Improve P256 scalar primitives-* Fix hash truncation bug in DSA-* Fix cost parsing for bcrypt-* Fix ECC failures on arm64-* Correction to PKCS#1 v1.5 padding-* Use powModSecInteger when available-* Drop GHC 7.8 and GHC 7.10 support, refer to pkg-guidelines-* Optimise GCM mode-* Add little endian serialization of integer--## 0.25--* Improve digest binary conversion efficiency-* AES CCM support-* Add MonadFailure instance for CryptoFailable-* Various misc improvements on documentation-* Edwards25519 lowlevel arithmetic support-* P256 add point negation-* Improvement in ECC (benchmark, better normalization)-* Blake2 improvements to context size-* Use gauge instead of criterion-* Use haskell-ci for CI scripts-* Improve Digest memory representation to be 2 less Ints and one less boxing- moving from `UArray` to `Block`--## 0.24--* Ed25519: generateSecret & Documentation updates-* Repair tutorial-* RSA: Allow signing digest directly-* IV add: fix overflow behavior-* P256: validate point when decoding-* Compilation fix with deepseq disabled-* Improve Curve448 and use decaf for Ed448-* Compilation flag blake2 sse merged in sse support-* Process unaligned data better in hashes and AES, on architecture needing alignment-* Drop support for ghc 7.6-* Add ability to create random generator Seed from binary data and- loosen constraint on ChaChaDRG seed from ByteArray to ByteArrayAccess.-* Add 3 associated types with the HashAlgorithm class, to get- access to the constant for BlockSize, DigestSize and ContextSize at the type level.- the related function that this replaced will be deprecated in later release, and- eventually removed.--API CHANGES:--* Improve ECDH safety to return failure for bad inputs (e.g. public point in small order subgroup).- To go back to previous behavior you can replace `ecdh` by `ecdhRaw`. It's recommended to- use `ecdh` and handle the error appropriately.-* Users defining their own HashAlgorithm needs to define the- HashBlockSize, HashDigest, HashInternalContextSize associated types--## 0.23--* Digest memory usage improvement by using unpinned memory-* Fix generateBetween to generate within the right bounds-* Add pure Twofish implementation-* Fix memory allocation in P256 when using a temp point-* Consolidate hash benchmark code-* Add Nat-length Blake2 support (GHC > 8.0)-* Update tutorial--## 0.22--* Add Argon2 (Password Hashing Competition winner) hash function-* Update blake2 to latest upstream version-* Add extra blake2 hashing size-* Add faster PBKDF2 functions for SHA1/SHA256/SHA512-* Add SHAKE128 and SHAKE256-* Cleanup prime generation, and add tests-* Add Time-based One Time Password (TOTP) and HMAC-based One Time Password (HOTP)-* Rename Ed448 module name to Curve448, old module name still valid for now--## 0.21--* Drop automated tests with GHC 7.0, GHC 7.4, GHC 7.6. support dropped, but probably still working.-* Improve non-aligned support in C sources, ChaCha and SHA3 now probably work on arch without support for unaligned access. not complete or tested.-* Add another ECC framework that is more flexible, allowing different implementations to work instead of- the existing Pure haskell NIST implementation.-* Add ECIES basic primitives-* Add XSalsa20 stream cipher-* Process partial buffer correctly with Poly1305--## 0.20--* Fixed hash truncation used in ECDSA signature & verification (Olivier Chéron)-* Fix ECDH when scalar and coordinate bit sizes differ (Olivier Chéron)-* Speed up ECDSA verification using Shamir's trick (Olivier Chéron)-* Fix rdrand on windows--## 0.19--* Add tutorial (Yann Esposito)-* Derive Show instance for better interaction with Show pretty printer (Eric Mertens)--## 0.18--* Re-used standard rdrand instructions instead of bytedump of rdrand instruction-* Improvement to F2m, including lots of tests (Andrew Lelechenko)-* Add error check on salt length in bcrypt--## 0.17--* Add Miyaguchi-Preneel construction (Kei Hibino)-* Fix buffer length in scrypt (Luke Taylor)-* build fixes for i686 and arm related to rdrand--## 0.16--* Fix basepoint for Ed448--* Enable 64-bit Curve25519 implementation--## 0.15--* Fix serialization of DH and ECDH--## 0.14--* Reduce size of SHA3 context instead of allocating all-size fit memory. save- up to 72 bytes of memory per context for SHA3-512.-* Add a Seed capability to the main DRG, to be able to debug/reproduce randomized program- where you would want to disable the randomness.-* Add support for Cipher-based Message Authentication Code (CMAC) (Kei Hibino)-* *CHANGE* Change the `SharedKey` for `Crypto.PubKey.DH` and `Crypto.PubKey.ECC.DH`,- from an Integer newtype to a ScrubbedBytes newtype. Prevent mistake where the- bytes representation is generated without the right padding (when needed).-* *CHANGE* Keep The field size in bits, in the `Params` in `Crypto.PubKey.DH`,- moving from 2 elements to 3 elements in the structure.--## 0.13--* *SECURITY* Fix buffer overflow issue in SHA384, copying 16 extra bytes from- the SHA512 context to the destination memory pointer leading to memory- corruption, segfault. (Mikael Bung)--## 0.12--* Fix compilation issue with Ed448 on 32 bits machine.--## 0.11--* Truncate hashing correctly for DSA-* Add support for HKDF (RFC 5869)-* Add support for Ed448-* Extends support for Blake2s to 224 bits version.-* Compilation workaround for old distribution (RHEL 4.1)-* Compilation fix for AIX-* Compilation fix with AESNI and ghci compiling C source in a weird order.-* Fix example compilation, typo, and warning--## 0.10--* Add reference implementation of blake2 for non-SSE2 platform-* Add support\_blake2\_sse flag--## 0.9--* Quiet down unused module imports-* Move Curve25519 over to Crypto.Error instead of using Either String.-* Add documentation for ChaChaPoly1305-* Add missing documentation for various modules-* Add a way to create Poly1305 Auth tag.-* Added support for the BLAKE2 family of hash algorithms-* Fix endianness of incrementNonce function for ChaChaPoly1305--## 0.8--* Add support for ChaChaPoly1305 Nonce Increment (John Galt)-* Move repository to the haskell-crypto organisation--## 0.7--* Add PKCS5 / PKCS7 padding and unpadding methods-* Fix ChaChaPoly1305 Decryption-* Add support for BCrypt (Luke Taylor)--## 0.6--* Add ChaChaPoly1305 AE cipher-* Add instructions in README for building on old OSX-* Fix blocking /dev/random Andrey Sverdlichenko--## 0.5--* Fix all strays exports to all be under the cryptonite prefix.--## 0.4--* Add a System DRG that represent a referentially transparent of evaluated bytes- while using lazy evaluation for future entropy values.--## 0.3--* Allow drgNew to run in any MonadRandom, providing cascading initialization-* Remove Crypto.PubKey.HashDescr in favor of just having the algorithm- specified in PKCS15 RSA function.-* Fix documentation in cipher sub section (Luke Taylor)-* Cleanup AES dead functions (Luke Taylor)-* Fix Show instance of Digest to display without quotes similar to cryptohash-* Use scrubbed bytes instead of bytes for P256 scalar--## 0.2--* Fix P256 compilation and exactness, + add tests-* Add a raw memory number serialization capability (i2osp, os2ip)-* Improve tests for number serialization-* Improve tests for ECC arithmetics-* Add Ord instance for Digest (Nicolas Di Prima)-* Fix entropy compilation on windows 64 bits.--## 0.1--* Initial release+## 2.1.8++* chore: stop hiding foldl' from Prelude+ [#294](https://github.com/kazu-yamamoto/crypton/pull/294)+* chore: ask for hidden visibility only where the format has it+ [#295](https://github.com/kazu-yamamoto/crypton/pull/295)+* feat: ML-KEM and ML-DSA, through mlkem-native and mldsa-native+ [#297](https://github.com/kazu-yamamoto/crypton/pull/297)++## 2.1.7++RSA-PSS verification was accepting an encoding RFC 8017 says to refuse. It+is a conformance fault rather than a forgery: producing such a signature takes+the private key, so a third party holding a valid signature cannot turn it+into one of these.++* chore(p256): drop two declarations nothing defines+ [#292](https://github.com/kazu-yamamoto/crypton/pull/292)+* fix(pss): refuse an encoding with a bit set outside emBits+ [#293](https://github.com/kazu-yamamoto/crypton/pull/293)++## 2.1.6++The `license:` field now says what the tree holds -- `BSD-3-Clause AND MIT+AND ISC` -- and `license-files:` lists all five texts. **Nothing is required+of a user that was not required before**; the field was simply incomplete.++* Carry the MIT notice for the parts that follow fusion, and say what the tree holds+ [#267](https://github.com/kazu-yamamoto/crypton/pull/267)+* fix: two preconditions the Haskell layer did not enforce+ [#288](https://github.com/kazu-yamamoto/crypton/pull/288)+* fix(x86): stop casting a packed block to __m128i *+ [#289](https://github.com/kazu-yamamoto/crypton/pull/289)+* fix(c): say that the digest pointers are never null+ [#290](https://github.com/kazu-yamamoto/crypton/pull/290)+* fix(pbkdf2): refuse a digest larger than its block at compile time+ [#291](https://github.com/kazu-yamamoto/crypton/pull/291)++## 2.1.5++2.1.3 and 2.1.4 cannot be built with GCC 14 or newer. This release is+that fix.++* Watch for a primitive fallen off its fast path+ [#277](https://github.com/kazu-yamamoto/crypton/pull/277)+* The performance tables for 2.1.4+ [#278](https://github.com/kazu-yamamoto/crypton/pull/278)+* The x86-64 table, on the machine the README names+ [#279](https://github.com/kazu-yamamoto/crypton/pull/279)+* perf(armv8): GHASH against a twisted H, a quarter faster+ [#281](https://github.com/kazu-yamamoto/crypton/pull/281)+* ct: run the constant-time harness on AArch64, and put its AES to it+ [#283](https://github.com/kazu-yamamoto/crypton/pull/283)+* Fix crypton_sha1_x86_do_chunk signature+ [#284](https://github.com/kazu-yamamoto/crypton/pull/284)+* ci: build the C with a compiler stricter than this matrix has+ [#285](https://github.com/kazu-yamamoto/crypton/pull/285)++## 2.1.4++2.1.3 could not be built from Hackage in the default configuration, and is+deprecated there. This release is that fix.++* Add p256 header files to cabal extra-source-files+ [#271](https://github.com/kazu-yamamoto/crypton/pull/271)+* Build what is published, not only what is checked out+ [#272](https://github.com/kazu-yamamoto/crypton/pull/272)+* SHA-256 on AArch64 is 5.4x slower in 2.1.3 than in 2.1.2+ [#274](https://github.com/kazu-yamamoto/crypton/pull/274)+* The target attribute spelling gcc before 13 understands+ [#276](https://github.com/kazu-yamamoto/crypton/pull/276)++## 2.1.3++* fix(aead): reject missing and oversized authentication tags+ [#233](https://github.com/kazu-yamamoto/crypton/pull/233)+* fix(chacha): remove redundant Word8 import+ [#234](https://github.com/kazu-yamamoto/crypton/pull/234)+* test(number): cover the two modulus sizes the assembly runs at+ [#235](https://github.com/kazu-yamamoto/crypton/pull/235)+* perf(rsa): swap the buffers instead of copying them back+ [#236](https://github.com/kazu-yamamoto/crypton/pull/236)+* perf(rsa): scan the exponentiation's table four limbs at a time+ [#237](https://github.com/kazu-yamamoto/crypton/pull/237)+* fix(gcm): write the field doubling from its definition+ [#238](https://github.com/kazu-yamamoto/crypton/pull/238)+* doc(gcm): carry the MIT notice for the parts that follow fusion+ [#239](https://github.com/kazu-yamamoto/crypton/pull/239)+* perf(ed25519): the base point multiplication through s2n-bignum+ [#240](https://github.com/kazu-yamamoto/crypton/pull/240)+* perf(gcm): AES-GCM through the 512-bit VAES and VPCLMULQDQ+ [#241](https://github.com/kazu-yamamoto/crypton/pull/241)+* doc: the README said AVX-512 was not used, and it is+ [#242](https://github.com/kazu-yamamoto/crypton/pull/242)+* perf(ecdsa): P-256 verification multiplies both scalars at once+ [#243](https://github.com/kazu-yamamoto/crypton/pull/243)+* perf(rsa): four limbs and two carry chains on AArch64+ [#244](https://github.com/kazu-yamamoto/crypton/pull/244)+* perf(rsa): write out the ragged end of the AArch64 row+ [#245](https://github.com/kazu-yamamoto/crypton/pull/245)+* perf(rsa): build R^2 by squaring, not by doubling+ [#246](https://github.com/kazu-yamamoto/crypton/pull/246)+* perf(rsa): stop clearing the scratch a Montgomery multiply writes over+ [#247](https://github.com/kazu-yamamoto/crypton/pull/247)+* doc(gcm): write down what the AArch64 GHASH is short of+ [#248](https://github.com/kazu-yamamoto/crypton/pull/248)+* security(aes): refuse a nonce of no bytes in Crypto.Cipher.AES.GCM+ [#250](https://github.com/kazu-yamamoto/crypton/pull/250)+* ci: build and test the C the other architectures use+ [#251](https://github.com/kazu-yamamoto/crypton/pull/251)+* perf(p256): five teeth to a comb block, over the signed representation+ [#252](https://github.com/kazu-yamamoto/crypton/pull/252)+* security(cipher): stop truncating message lengths on the way to the C+ [#253](https://github.com/kazu-yamamoto/crypton/pull/253)+* fix(c): two left shifts the standard leaves undefined+ [#254](https://github.com/kazu-yamamoto/crypton/pull/254)+* ci: run the C under the sanitizers+ [#255](https://github.com/kazu-yamamoto/crypton/pull/255)+* fix(c): read words out of a block rather than pointing at it+ [#256](https://github.com/kazu-yamamoto/crypton/pull/256)+* fix(internal): drop an import nothing uses any more+ [#257](https://github.com/kazu-yamamoto/crypton/pull/257)+* fix(c): decide the dispatch table once, not on every key+ [#258](https://github.com/kazu-yamamoto/crypton/pull/258)+* Fix the three cabal flag settings that were broken+ [#259](https://github.com/kazu-yamamoto/crypton/pull/259)+* Run the C that only 32-bit architectures get, and fix what that found+ [#260](https://github.com/kazu-yamamoto/crypton/pull/260)+* Let the last addition of each scalar multiplication be a complete one+ [#261](https://github.com/kazu-yamamoto/crypton/pull/261)+* Ask whether the secrets decide anything+ [#262](https://github.com/kazu-yamamoto/crypton/pull/262)+* Ask a big-endian machine the same questions+ [#263](https://github.com/kazu-yamamoto/crypton/pull/263)+* Ask what the secrets leave behind+ [#264](https://github.com/kazu-yamamoto/crypton/pull/264)+* Stop taking eighteen runner slots to test three things+ [#265](https://github.com/kazu-yamamoto/crypton/pull/265)+* Make the scrubs ones the compiler cannot drop, and finish round ten+ [#266](https://github.com/kazu-yamamoto/crypton/pull/266)+* Feed the parsers bytes nobody chose+ [#268](https://github.com/kazu-yamamoto/crypton/pull/268)++## 2.1.2++* perf(p256): 255 squarings for the field inversion, not 287+ [#223](https://github.com/kazu-yamamoto/crypton/pull/223)+* perf(p256): ECDH through s2n-bignum, 2.7x+ [#224](https://github.com/kazu-yamamoto/crypton/pull/224)+* perf(ecc): P-384 and P-521 through s2n-bignum, 7x and 9x+ [#225](https://github.com/kazu-yamamoto/crypton/pull/225)+* perf(p256): ECDSA signing 2.4x and verification 2.2x+ [#226](https://github.com/kazu-yamamoto/crypton/pull/226)+* perf(rsa): the Montgomery multiplication through s2n-bignum on x86-64+ [#227](https://github.com/kazu-yamamoto/crypton/pull/227)+* perf(ecdsa): invert modulo the order in division steps, not an exponentiation+ [#228](https://github.com/kazu-yamamoto/crypton/pull/228)+* perf(x25519): X25519 through s2n-bignum, and a table for key generation+ [#229](https://github.com/kazu-yamamoto/crypton/pull/229)+* perf(gcm): AES-GCM through VAES and VPCLMULQDQ+ [#230](https://github.com/kazu-yamamoto/crypton/pull/230)+* perf(gcm): compile the wide loop once per key length+ [#231](https://github.com/kazu-yamamoto/crypton/pull/231)++## 2.1.1++* feat(ecdsa): RFC 6979 deterministic nonces for Crypto.PubKey.ECDSA+ [#219](https://github.com/kazu-yamamoto/crypton/pull/219)+* feat(gcm): a decrypt that hands back the tag instead of comparing it+ [#220](https://github.com/kazu-yamamoto/crypton/pull/220)+* feat(chachapoly): ChaCha20-Poly1305 a message at a time+ [#221](https://github.com/kazu-yamamoto/crypton/pull/221)+* docs(rsa): say in the haddock what the optional blinder covers+ [#222](https://github.com/kazu-yamamoto/crypton/pull/222)++## 2.1.0++* fix(cpu): stop reading Intel's SDBG bit as AMD's XOP+ [#204](https://github.com/kazu-yamamoto/crypton/pull/204)+* fix(bench): build the benchmark against the checked ChaCha20-Poly1305 key+ [#205](https://github.com/kazu-yamamoto/crypton/pull/205)+* ci: key the cache on the package version+ [#206](https://github.com/kazu-yamamoto/crypton/pull/206)+* ci: build the benchmarks+ [#207](https://github.com/kazu-yamamoto/crypton/pull/207)+* perf(gcm): a fused AES-GCM for x86-64+ [#208](https://github.com/kazu-yamamoto/crypton/pull/208)+* perf(gcm): a fused AES-GCM for AArch64+ [#209](https://github.com/kazu-yamamoto/crypton/pull/209)+* perf(gcm): build the counter in vector registers+ [#210](https://github.com/kazu-yamamoto/crypton/pull/210)+* perf(gcm): a spare lane for E(K,Y0), and a cheaper short block+ [#211](https://github.com/kazu-yamamoto/crypton/pull/211)+* perf(gcm): unroll the tail pass, and take its blocks from registers+ [#212](https://github.com/kazu-yamamoto/crypton/pull/212)+* perf(gcm): read a short block where it lies+ [#213](https://github.com/kazu-yamamoto/crypton/pull/213)+* perf(gcm): the length block and the counter, in registers+ [#214](https://github.com/kazu-yamamoto/crypton/pull/214)+* perf(gcm): let the one-call interface specialise+ [#215](https://github.com/kazu-yamamoto/crypton/pull/215)+* perf(gcm): decryption takes the fused path too+ [#216](https://github.com/kazu-yamamoto/crypton/pull/216)+* perf(gcm): GHASH takes the ciphertext from the output buffer+ [#217](https://github.com/kazu-yamamoto/crypton/pull/217)+* perf(p256): a signed five-bit window for the variable-point multiply+ [#218](https://github.com/kazu-yamamoto/crypton/pull/218)++## 2.0.1++* feat(hash): Skein with the digest size as a type parameter+ [#197](https://github.com/kazu-yamamoto/crypton/pull/197)+* fix(chachapoly1305): take a checked key, so that initializing cannot fail+ [#198](https://github.com/kazu-yamamoto/crypton/pull/198)+* feat(aes): Crypto.Cipher.AES.GCM, for many short messages under one key+ [#199](https://github.com/kazu-yamamoto/crypton/pull/199)+* build: say which platforms the fallback AES sources are for+ [#200](https://github.com/kazu-yamamoto/crypton/pull/200)+* feat(aes): encryptWithMask, for the QUIC header protection mask+ [#201](https://github.com/kazu-yamamoto/crypton/pull/201)+* fix(cpu): stop reading Intel's SDBG bit as AMD's XOP+ [#203](https://github.com/kazu-yamamoto/crypton/pull/203)++## 2.0.0++**Breaking changes.** Input that used to be accepted is now refused: a value+at or above an RSA or Rabin modulus, a signature of the wrong length or out of+range, a digest too short for HOTP's dynamic truncation, a non-canonical+Ed25519 signature, a PKCS#7 block size outside 1..255, and block cipher input+that is not a whole number of blocks. A refused KDF, Argon2 or bcrypt+parameter is reported as a `CryptoError` rather than raised as an `ErrorCall`,+and `CryptoError_ParameterInvalid` is appended to `CryptoError`;+`tryGetShared` is added beside `getShared`. No exported function changed its+signature.++**Deprecated.** The eighteen curves over a binary field in+`Crypto.ECC.Simple.Types`. They are obsolete, they are the curves whose+cofactor is not 1, and they will go in a later major version. Prefer a prime+curve, or X25519.++* Add GHC 9.14 to CI+ [#74](https://github.com/kazu-yamamoto/crypton/pull/74)+* fix(ed25519): reject non-canonical signatures+ [#81](https://github.com/kazu-yamamoto/crypton/pull/81)+* fix(hkdf): enforce RFC 5869 output limit+ [#82](https://github.com/kazu-yamamoto/crypton/pull/82)+* fix(p256): accept valid edge-case points+ [#83](https://github.com/kazu-yamamoto/crypton/pull/83)+* fix(ecc): accept zero-x P-256 shared secrets+ [#84](https://github.com/kazu-yamamoto/crypton/pull/84)+* fix(otp): require a digest long enough for dynamic truncation+ [#85](https://github.com/kazu-yamamoto/crypton/pull/85)+* fix(pkcs15): reject malformed PKCS#1 v1.5 signatures+ [#86](https://github.com/kazu-yamamoto/crypton/pull/86)+* fix(ecdh): validate the peer point before the exchange+ [#87](https://github.com/kazu-yamamoto/crypton/pull/87)+* fix(dsa): do not crash on non-invertible values+ [#88](https://github.com/kazu-yamamoto/crypton/pull/88)+* fix(dh): validate the peer public number+ [#89](https://github.com/kazu-yamamoto/crypton/pull/89)+* fix(argon2): report invalid options as CryptoFailed+ [#90](https://github.com/kazu-yamamoto/crypton/pull/90)+* fix(rsa): drop the early exits from PKCS#1 v1.5 and OAEP unpadding+ [#91](https://github.com/kazu-yamamoto/crypton/pull/91)+* fix(otp): compare TOTP candidates without an early exit+ [#92](https://github.com/kazu-yamamoto/crypton/pull/92)+* feat(dh): add getShared' reporting rejections as CryptoFailable+ [#93](https://github.com/kazu-yamamoto/crypton/pull/93)+* feat(aead): add aeadSimpleDecrypt' taking the tag length+ [#94](https://github.com/kazu-yamamoto/crypton/pull/94)+* test: move the suite to hspec, with hspec-discover+ [#95](https://github.com/kazu-yamamoto/crypton/pull/95)+* docs(bcrypt): say that only the first 72 bytes of a password count+ [#96](https://github.com/kazu-yamamoto/crypton/pull/96)+* feat(elgamal): fix and expose Crypto.PubKey.ElGamal+ [#97](https://github.com/kazu-yamamoto/crypton/pull/97)+* fix(padding): reject a PKCS7 block size outside 1..255+ [#98](https://github.com/kazu-yamamoto/crypton/pull/98)+* build(bench): move the benchmarks from gauge to tasty-bench+ [#99](https://github.com/kazu-yamamoto/crypton/pull/99)+* feat(aes): use the ARMv8 cryptographic extensions on AArch64+ [#100](https://github.com/kazu-yamamoto/crypton/pull/100)+* ci: stop throwing the cache away, and keep the build products in it+ [#101](https://github.com/kazu-yamamoto/crypton/pull/101)+* feat(aes): use PMULL for GHASH on AArch64+ [#102](https://github.com/kazu-yamamoto/crypton/pull/102)+* ci: cut the macOS queueing and supersede stale branch runs+ [#103](https://github.com/kazu-yamamoto/crypton/pull/103)+* feat(sha256): use the ARMv8 SHA-2 instructions on AArch64+ [#104](https://github.com/kazu-yamamoto/crypton/pull/104)+* perf(gcm): fold four GHASH blocks into one reduction+ [#105](https://github.com/kazu-yamamoto/crypton/pull/105)+* ci: build and test on aarch64 Linux+ [#106](https://github.com/kazu-yamamoto/crypton/pull/106)+* fix(cabal): build the AES-NI paths on Windows too+ [#107](https://github.com/kazu-yamamoto/crypton/pull/107)+* perf(aes): specialise by key size and interleave eight blocks on AArch64+ [#108](https://github.com/kazu-yamamoto/crypton/pull/108)+* perf(gcm): drive GCM from AArch64 rather than the generic loop+ [#109](https://github.com/kazu-yamamoto/crypton/pull/109)+* feat(sha512): use the ARMv8.2 SHA-512 instructions on AArch64+ [#110](https://github.com/kazu-yamamoto/crypton/pull/110)+* perf(chacha): do four blocks at a time with NEON on AArch64+ [#111](https://github.com/kazu-yamamoto/crypton/pull/111)+* perf(chacha): do four blocks at a time with SSE2 on x86-64+ [#112](https://github.com/kazu-yamamoto/crypton/pull/112)+* perf(chacha): take eight blocks with AVX2 where the machine has it+ [#113](https://github.com/kazu-yamamoto/crypton/pull/113)+* perf(gcm): give x86 its own decryption loop, and eight blocks either way+ [#114](https://github.com/kazu-yamamoto/crypton/pull/114)+* fix(padding): bound PKCS7 padding by the block, and check ZERO's size+ [#115](https://github.com/kazu-yamamoto/crypton/pull/115)+* ecc: say which curves branch on a secret scalar, and work in Jacobian coordinates+ [#116](https://github.com/kazu-yamamoto/crypton/pull/116)+* perf(poly1305): take four blocks at a time with AVX2 on x86-64+ [#117](https://github.com/kazu-yamamoto/crypton/pull/117)+* perf(xts): drive XTS eight blocks at a time, and dispatch its decryption+ [#118](https://github.com/kazu-yamamoto/crypton/pull/118)+* Report refused KDF parameters as CryptoError, and fix a PBKDF2 SIGBUS+ [#119](https://github.com/kazu-yamamoto/crypton/pull/119)+* Search the HOTP resynchronization window without early exits+ [#120](https://github.com/kazu-yamamoto/crypton/pull/120)+* Refuse an RSA representative that is not below the modulus+ [#121](https://github.com/kazu-yamamoto/crypton/pull/121)+* Give AFIS one answer for a parameter it cannot use+ [#122](https://github.com/kazu-yamamoto/crypton/pull/122)+* Say what ElGamal's signWith requires of k+ [#123](https://github.com/kazu-yamamoto/crypton/pull/123)+* Draw Miller-Rabin witnesses per number, not once per process+ [#124](https://github.com/kazu-yamamoto/crypton/pull/124)+* Refuse Rabin values that are not below the modulus, and keep the padding that was signed+ [#125](https://github.com/kazu-yamamoto/crypton/pull/125)+* Decode Rabin's OAEP without early exits+ [#126](https://github.com/kazu-yamamoto/crypton/pull/126)+* Make CMAC linear, and chain it through CBC+ [#127](https://github.com/kazu-yamamoto/crypton/pull/127)+* Put DES in C+ [#128](https://github.com/kazu-yamamoto/crypton/pull/128)+* Make the generic block cipher modes linear, and bulk where the blocks allow+ [#129](https://github.com/kazu-yamamoto/crypton/pull/129)+* Walk Twofish's blocks once, and carry them in words+ [#130](https://github.com/kazu-yamamoto/crypton/pull/130)+* Put Camellia in C+ [#131](https://github.com/kazu-yamamoto/crypton/pull/131)+* Route P-256 through the C implementation it already had+ [#132](https://github.com/kazu-yamamoto/crypton/pull/132)+* Fold instead of dividing in the generic curve arithmetic+ [#133](https://github.com/kazu-yamamoto/crypton/pull/133)+* Reduce the binary field by folding, and work a byte and a nibble at a time+ [#134](https://github.com/kazu-yamamoto/crypton/pull/134)+* Stop running a Fermat test Miller-Rabin subsumes+ [#135](https://github.com/kazu-yamamoto/crypton/pull/135)+* Make expSafe hide the exponent again+ [#136](https://github.com/kazu-yamamoto/crypton/pull/136)+* Square, and multiply, faster in expSafe+ [#137](https://github.com/kazu-yamamoto/crypton/pull/137)+* Invert the signing nonce without a side channel+ [#138](https://github.com/kazu-yamamoto/crypton/pull/138)+* Keep the P-256 signature out of Integer arithmetic+ [#139](https://github.com/kazu-yamamoto/crypton/pull/139)+* Add at every bit in the prime-curve multiplication, which laziness was skipping+ [#140](https://github.com/kazu-yamamoto/crypton/pull/140)+* Multiply points in C on curves over a prime field+ [#141](https://github.com/kazu-yamamoto/crypton/pull/141)+* A ladder for the curves over a binary field+ [#142](https://github.com/kazu-yamamoto/crypton/pull/142)+* Work RSA's qinv out without the extended Euclidean algorithm+ [#143](https://github.com/kazu-yamamoto/crypton/pull/143)+* Keep the RSA blinding factor out of the extended algorithm+ [#144](https://github.com/kazu-yamamoto/crypton/pull/144)+* Unroll the inner loop at four and two as well+ [#145](https://github.com/kazu-yamamoto/crypton/pull/145)+* Keep a table for each curve's base point+ [#146](https://github.com/kazu-yamamoto/crypton/pull/146)+* Start R squared at the top of the modulus, and why folding did not pay+ [#147](https://github.com/kazu-yamamoto/crypton/pull/147)+* Do the binary field arithmetic in C+ [#148](https://github.com/kazu-yamamoto/crypton/pull/148)+* Use the x86 carry-less multiply where the processor has it+ [#149](https://github.com/kazu-yamamoto/crypton/pull/149)+* Close the two testing gaps: one multiplication for both APIs, one place for each buffer's size+ [#150](https://github.com/kazu-yamamoto/crypton/pull/150)+* Ask aarch64 for its carry-less multiply as well+ [#151](https://github.com/kazu-yamamoto/crypton/pull/151)+* Work the RSA private exponent out without the extended algorithm+ [#152](https://github.com/kazu-yamamoto/crypton/pull/152)+* Fewer Miller-Rabin rounds for a candidate nobody chose+ [#153](https://github.com/kazu-yamamoto/crypton/pull/153)+* Blowfish, and the key setup bcrypt wraps it in, in C+ [#154](https://github.com/kazu-yamamoto/crypton/pull/154)+* perf(sha256): use the Intel SHA extensions on x86-64+ [#155](https://github.com/kazu-yamamoto/crypton/pull/155)+* perf(aes): AES-192 through the processor's AES instructions+ [#156](https://github.com/kazu-yamamoto/crypton/pull/156)+* perf(aes): build the AArch64 key schedule with AESE, not the S-box table+ [#157](https://github.com/kazu-yamamoto/crypton/pull/157)+* test(aes): run the XTS vectors, and OCB and CCM at 192 and 256 bits+ [#158](https://github.com/kazu-yamamoto/crypton/pull/158)+* perf(ocb): drive OCB through the ECB paths a group at a time+ [#159](https://github.com/kazu-yamamoto/crypton/pull/159)+* perf(gcm): take the GHASH of the group before, alongside this group's rounds+ [#160](https://github.com/kazu-yamamoto/crypton/pull/160)+* perf(sha): compute the message schedule in vector registers on x86+ [#161](https://github.com/kazu-yamamoto/crypton/pull/161)+* perf(chacha): combine as the keystream comes out of the registers+ [#162](https://github.com/kazu-yamamoto/crypton/pull/162)+* perf(poly1305): shorten the carry chain and stop spilling the loop+ [#163](https://github.com/kazu-yamamoto/crypton/pull/163)+* docs(sidechannel): say what the prime-field modules keep from the clock+ [#164](https://github.com/kazu-yamamoto/crypton/pull/164)+* perf(sha1): use the Intel SHA extensions on x86-64+ [#165](https://github.com/kazu-yamamoto/crypton/pull/165)+* refactor(aes): drop the keystream generator nobody can call+ [#166](https://github.com/kazu-yamamoto/crypton/pull/166)+* build: compile the C at -O3+ [#167](https://github.com/kazu-yamamoto/crypton/pull/167)+* perf(modes): stop the generic cipher modes allocating per byte+ [#168](https://github.com/kazu-yamamoto/crypton/pull/168)+* perf(poly1305): four blocks at a time with NEON+ [#169](https://github.com/kazu-yamamoto/crypton/pull/169)+* perf(sha1): use the ARMv8 SHA-1 instructions+ [#170](https://github.com/kazu-yamamoto/crypton/pull/170)+* perf(sha3): use the ARMv8.2 SHA-3 instructions+ [#171](https://github.com/kazu-yamamoto/crypton/pull/171)+* perf(gcm): the CRYPTOGAMS stitched AES-GCM on x86-64+ [#172](https://github.com/kazu-yamamoto/crypton/pull/172)+* perf(chacha): the CRYPTOGAMS ChaCha20 on AArch64+ [#173](https://github.com/kazu-yamamoto/crypton/pull/173)+* perf(poly1305): the CRYPTOGAMS Poly1305 on AArch64+ [#174](https://github.com/kazu-yamamoto/crypton/pull/174)+* perf(sha256): the CRYPTOGAMS SHA-256 on AArch64+ [#175](https://github.com/kazu-yamamoto/crypton/pull/175)+* perf(poly1305): the CRYPTOGAMS Poly1305 on x86-64 too+ [#176](https://github.com/kazu-yamamoto/crypton/pull/176)+* perf(chacha): the CRYPTOGAMS ChaCha20 on x86-64 too+ [#177](https://github.com/kazu-yamamoto/crypton/pull/177)+* perf(sha2): the CRYPTOGAMS SHA-256 and SHA-512 on x86-64+ [#178](https://github.com/kazu-yamamoto/crypton/pull/178)+* perf(sha1): hand the SHA-1 block loop a run of blocks, not one at a time+ [#179](https://github.com/kazu-yamamoto/crypton/pull/179)+* perf(xts): double the tweak in the integer registers+ [#180](https://github.com/kazu-yamamoto/crypton/pull/180)+* perf(sha3): take the CRYPTOGAMS Keccak for AArch64+ [#181](https://github.com/kazu-yamamoto/crypton/pull/181)+* perf(sha1): take the CRYPTOGAMS SHA-1 for AArch64+ [#182](https://github.com/kazu-yamamoto/crypton/pull/182)+* docs: put the performance tables in the README+ [#183](https://github.com/kazu-yamamoto/crypton/pull/183)+* perf(sha3): take the CRYPTOGAMS Keccak for x86-64 as well+ [#184](https://github.com/kazu-yamamoto/crypton/pull/184)+* docs: rebuild the performance tables+ [#185](https://github.com/kazu-yamamoto/crypton/pull/185)+* perf(ecc): stop sharing the doublings in the double multiplication+ [#186](https://github.com/kazu-yamamoto/crypton/pull/186)+* perf(number): count bytes from the bit count, not from base 256+ [#187](https://github.com/kazu-yamamoto/crypton/pull/187)+* perf(p256): inline the field arithmetic on AArch64+ [#188](https://github.com/kazu-yamamoto/crypton/pull/188)+* fix(api): name the reporting variants try..., not with an apostrophe+ [#189](https://github.com/kazu-yamamoto/crypton/pull/189)+* fix(ecc): require a public point to be in the prime-order subgroup+ [#190](https://github.com/kazu-yamamoto/crypton/pull/190)+* fix(pubkey): stop printing private keys, and add Crypto.Debug+ [#191](https://github.com/kazu-yamamoto/crypton/pull/191)+* fix(poly1305): take a checked key, so that initializing cannot fail+ [#192](https://github.com/kazu-yamamoto/crypton/pull/192)+* fix(bcrypt): refuse a cost bcrypt does not have rather than substituting one+ [#194](https://github.com/kazu-yamamoto/crypton/pull/194)+* chore: build without a warning+ [#195](https://github.com/kazu-yamamoto/crypton/pull/195)+* docs: build the documentation without a warning+ [#196](https://github.com/kazu-yamamoto/crypton/pull/196)++## 1.1.5++* fix(aead): reject undersized tags+ [#80](https://github.com/kazu-yamamoto/crypton/pull/80)+* fix(aes): refuse a zero-length AES-GCM IV+ [#79](https://github.com/kazu-yamamoto/crypton/pull/79)+* fix(p256): prevent crashes when validating valid points+ [#78](https://github.com/kazu-yamamoto/crypton/pull/78)+* feat(asn1): add SHA-3 HashAlgorithmASN1 instances for PKCS#1 v1.5+ [#77](https://github.com/kazu-yamamoto/crypton/pull/77)+* OCB3 conformance+ [#76](https://github.com/kazu-yamamoto/crypton/pull/76)++## 1.1.4++* Generic instance for RSA PublicKey and PrivateKey++## 1.1.3++* Ensure that `pointAdd` in `PubKey.ECC.P256` treats the point at infinity as the additive identity.+ [#73](https://github.com/kazu-yamamoto/crypton/pull/73)++## 1.1.2++* Preparing `ram` v0.22.+* Generalizing RSA encrypt/decrypt to manipulate ScrubbedBytes directly.++## 1.1.1++* On iOS, ScrubbedBytes based hashing is used for seedNew. On other+ plateforms, entropy is used directly as used to be.+ [#71](https://github.com/kazu-yamamoto/crypton/pull/71)++## 1.1.0++* Removing "basement" and "memory".+ [#67](https://github.com/kazu-yamamoto/crypton/pull/67)+++## 1.0.7++* Stop depending on basement, use upstream dependencies instead+* Stop transitively depending on basement by depending on ram.++## 1.0.6++* Fix test failures on less common 64-bit arches.+ [#65](https://github.com/kazu-yamamoto/crypton/pull/65)++## 1.0.5++* Setter/Getter for ChaCha counter.+ [#63](https://github.com/kazu-yamamoto/crypton/pull/63)+* Add simple interface to generate full blocks+ [#60](https://github.com/kazu-yamamoto/crypton/pull/60)+* Avoid `ghc-prim` dependency.+ [#61](https://github.com/kazu-yamamoto/crypton/pull/61)++## 1.0.4++* Ed448.sign: avoid extra re-derive of public key.+ [#48](https://github.com/kazu-yamamoto/crypton/pull/48)++## 1.0.3++* Make sign of Ed25519/Ed448 safer. The public key parameter is+ ignored and its public key is generated from the secret key+ parameter to prevent Double Public Key Signing Function Oracle+ Attack.+ [#47](https://github.com/kazu-yamamoto/crypton/pull/47)++## 1.0.2++* Deterministic Nonce Generation for ECDSA+ [#46](https://github.com/kazu-yamamoto/crypton/pull/46)+* ECDSA Signature Normalization.+ [#45](https://github.com/kazu-yamamoto/crypton/pull/45)+* Add Full Test Suite from RFC 6979.+ [#44](https://github.com/kazu-yamamoto/crypton/pull/44)+* ECDSA with Public Key Recovery.+ [#43](https://github.com/kazu-yamamoto/crypton/pull/43)+* Providing necessary features for HPKE.+ [#42](https://github.com/kazu-yamamoto/crypton/pull/42)++## 1.0.1++* Update decaf library.+ [#38](https://github.com/kazu-yamamoto/crypton/pull/38)+* Add TypeOperators language extension to EdDSA.hs.+ [#36](https://github.com/kazu-yamamoto/crypton/pull/36)++## 1.0.0++* Versions follow the standard version policy.+* Removing pthread stuff.+ [#32](https://github.com/kazu-yamamoto/crypton/pull/32)++## 0.34++* Hashing getRandomBytes before using as Seed for ChaChaDRG+ [#24](https://github.com/kazu-yamamoto/crypton/pull/24)+* Add support for XChaCha and XChaChaPoly1305+ [#18](https://github.com/kazu-yamamoto/crypton/pull/18)+* Strict byteArray of IV c+ [#16](https://github.com/kazu-yamamoto/crypton/pull/16)++## 0.33++* Add "crypton_" prefix to the final C symbols.+ [#9](https://github.com/kazu-yamamoto/crypton/pull/9)++## 0.32++* All C symbols now have the "crypton_" prefix.+ [#7](https://github.com/kazu-yamamoto/crypton/pull/7)+ [#8](https://github.com/kazu-yamamoto/crypton/pull/8)++## 0.31++* Crypton is forked from cryptonite with the original authors permission.+* Ignoring exceptons from hClose to read the next entropy+ [#1](https://github.com/kazu-yamamoto/crypton/pull/1)+* Enabling the support_pclmuldq flag by default.++## 0.30++* Fix some C symbol blake2b prefix to be cryptonite_ prefix (fix mixing with other C library)+* add hmac-lazy+* Fix compilation with GHC 9.2+* Drop support for GHC8.0, GHC8.2, GHC8.4, GHC8.6++## 0.29++* advance compilation with gmp breakage due to change upstream+* Add native EdDSA support++## 0.28++* Add hash constant time capability+* Prevent possible overflow during hashing by hashing in 4GB chunks++## 0.27++* Optimise AES GCM and CCM+* Optimise P256R1 implementation+* Various AES-NI building improvements+* Add better ECDSA support+* Add XSalsa derive+* Implement square roots for ECC binary curve+* Various tests and benchmarks++## 0.26++* Add Rabin cryptosystem (and variants)+* Add bcrypt_pbkdf key derivation function+* Optimize Blowfish implementation+* Add KMAC (Keccak Message Authentication Code)+* Add ECDSA sign/verify digest APIs+* Hash algorithms with runtime output length+* Update blake2 to latest upstream version+* RSA-PSS with arbitrary key size+* SHAKE with output length not divisible by 8+* Add Read and Data instances for Digest type+* Improve P256 scalar primitives+* Fix hash truncation bug in DSA+* Fix cost parsing for bcrypt+* Fix ECC failures on arm64+* Correction to PKCS#1 v1.5 padding+* Use powModSecInteger when available+* Drop GHC 7.8 and GHC 7.10 support, refer to pkg-guidelines+* Optimise GCM mode+* Add little endian serialization of integer++## 0.25++* Improve digest binary conversion efficiency+* AES CCM support+* Add MonadFailure instance for CryptoFailable+* Various misc improvements on documentation+* Edwards25519 lowlevel arithmetic support+* P256 add point negation+* Improvement in ECC (benchmark, better normalization)+* Blake2 improvements to context size+* Use gauge instead of criterion+* Use haskell-ci for CI scripts+* Improve Digest memory representation to be 2 less Ints and one less boxing+ moving from `UArray` to `Block`++## 0.24++* Ed25519: generateSecret & Documentation updates+* Repair tutorial+* RSA: Allow signing digest directly+* IV add: fix overflow behavior+* P256: validate point when decoding+* Compilation fix with deepseq disabled+* Improve Curve448 and use decaf for Ed448+* Compilation flag blake2 sse merged in sse support+* Process unaligned data better in hashes and AES, on architecture needing alignment+* Drop support for ghc 7.6+* Add ability to create random generator Seed from binary data and+ loosen constraint on ChaChaDRG seed from ByteArray to ByteArrayAccess.+* Add 3 associated types with the HashAlgorithm class, to get+ access to the constant for BlockSize, DigestSize and ContextSize at the type level.+ the related function that this replaced will be deprecated in later release, and+ eventually removed.++API CHANGES:++* Improve ECDH safety to return failure for bad inputs (e.g. public point in small order subgroup).+ To go back to previous behavior you can replace `ecdh` by `ecdhRaw`. It's recommended to+ use `ecdh` and handle the error appropriately.+* Users defining their own HashAlgorithm needs to define the+ HashBlockSize, HashDigest, HashInternalContextSize associated types++## 0.23++* Digest memory usage improvement by using unpinned memory+* Fix generateBetween to generate within the right bounds+* Add pure Twofish implementation+* Fix memory allocation in P256 when using a temp point+* Consolidate hash benchmark code+* Add Nat-length Blake2 support (GHC > 8.0)+* Update tutorial++## 0.22++* Add Argon2 (Password Hashing Competition winner) hash function+* Update blake2 to latest upstream version+* Add extra blake2 hashing size+* Add faster PBKDF2 functions for SHA1/SHA256/SHA512+* Add SHAKE128 and SHAKE256+* Cleanup prime generation, and add tests+* Add Time-based One Time Password (TOTP) and HMAC-based One Time Password (HOTP)+* Rename Ed448 module name to Curve448, old module name still valid for now++## 0.21++* Drop automated tests with GHC 7.0, GHC 7.4, GHC 7.6. support dropped, but probably still working.+* Improve non-aligned support in C sources, ChaCha and SHA3 now probably work on arch without support for unaligned access. not complete or tested.+* Add another ECC framework that is more flexible, allowing different implementations to work instead of+ the existing Pure haskell NIST implementation.+* Add ECIES basic primitives+* Add XSalsa20 stream cipher+* Process partial buffer correctly with Poly1305++## 0.20++* Fixed hash truncation used in ECDSA signature & verification (Olivier Chéron)+* Fix ECDH when scalar and coordinate bit sizes differ (Olivier Chéron)+* Speed up ECDSA verification using Shamir's trick (Olivier Chéron)+* Fix rdrand on windows++## 0.19++* Add tutorial (Yann Esposito)+* Derive Show instance for better interaction with Show pretty printer (Eric Mertens)++## 0.18++* Re-used standard rdrand instructions instead of bytedump of rdrand instruction+* Improvement to F2m, including lots of tests (Andrew Lelechenko)+* Add error check on salt length in bcrypt++## 0.17++* Add Miyaguchi-Preneel construction (Kei Hibino)+* Fix buffer length in scrypt (Luke Taylor)+* build fixes for i686 and arm related to rdrand++## 0.16++* Fix basepoint for Ed448++* Enable 64-bit Curve25519 implementation++## 0.15++* Fix serialization of DH and ECDH++## 0.14++* Reduce size of SHA3 context instead of allocating all-size fit memory. save+ up to 72 bytes of memory per context for SHA3-512.+* Add a Seed capability to the main DRG, to be able to debug/reproduce randomized program+ where you would want to disable the randomness.+* Add support for Cipher-based Message Authentication Code (CMAC) (Kei Hibino)+* *CHANGE* Change the `SharedKey` for `Crypto.PubKey.DH` and `Crypto.PubKey.ECC.DH`,+ from an Integer newtype to a ScrubbedBytes newtype. Prevent mistake where the+ bytes representation is generated without the right padding (when needed).+* *CHANGE* Keep The field size in bits, in the `Params` in `Crypto.PubKey.DH`,+ moving from 2 elements to 3 elements in the structure.++## 0.13++* *SECURITY* Fix buffer overflow issue in SHA384, copying 16 extra bytes from+ the SHA512 context to the destination memory pointer leading to memory+ corruption, segfault. (Mikael Bung)++## 0.12++* Fix compilation issue with Ed448 on 32 bits machine.++## 0.11++* Truncate hashing correctly for DSA+* Add support for HKDF (RFC 5869)+* Add support for Ed448+* Extends support for Blake2s to 224 bits version.+* Compilation workaround for old distribution (RHEL 4.1)+* Compilation fix for AIX+* Compilation fix with AESNI and ghci compiling C source in a weird order.+* Fix example compilation, typo, and warning++## 0.10++* Add reference implementation of blake2 for non-SSE2 platform+* Add support\_blake2\_sse flag++## 0.9++* Quiet down unused module imports+* Move Curve25519 over to Crypto.Error instead of using Either String.+* Add documentation for ChaChaPoly1305+* Add missing documentation for various modules+* Add a way to create Poly1305 Auth tag.+* Added support for the BLAKE2 family of hash algorithms+* Fix endianness of incrementNonce function for ChaChaPoly1305++## 0.8++* Add support for ChaChaPoly1305 Nonce Increment (John Galt)+* Move repository to the haskell-crypto organisation++## 0.7++* Add PKCS5 / PKCS7 padding and unpadding methods+* Fix ChaChaPoly1305 Decryption+* Add support for BCrypt (Luke Taylor)++## 0.6++* Add ChaChaPoly1305 AE cipher+* Add instructions in README for building on old OSX+* Fix blocking /dev/random Andrey Sverdlichenko++## 0.5++* Fix all strays exports to all be under the cryptonite prefix.++## 0.4++* Add a System DRG that represent a referentially transparent of evaluated bytes+ while using lazy evaluation for future entropy values.++## 0.3++* Allow drgNew to run in any MonadRandom, providing cascading initialization+* Remove Crypto.PubKey.HashDescr in favor of just having the algorithm+ specified in PKCS15 RSA function.+* Fix documentation in cipher sub section (Luke Taylor)+* Cleanup AES dead functions (Luke Taylor)+* Fix Show instance of Digest to display without quotes similar to cryptohash+* Use scrubbed bytes instead of bytes for P256 scalar++## 0.2++* Fix P256 compilation and exactness, + add tests+* Add a raw memory number serialization capability (i2osp, os2ip)+* Improve tests for number serialization+* Improve tests for ECC arithmetics+* Add Ord instance for Digest (Nicolas Di Prima)+* Fix entropy compilation on windows 64 bits.++## 0.1++* Initial release+
Crypto/Cipher/Twofish/Primitive.hs view
@@ -15,9 +15,8 @@ import Crypto.Internal.WordArray import Crypto.Internal.Words (Word128 (..)) import Data.Bits-import Data.List (foldl')+import qualified Data.List as L import Data.Word-import Prelude hiding (foldl') -- Based on the Golang referance implementation -- https://github.com/golang/crypto/blob/master/twofish/twofish.go@@ -113,7 +112,7 @@ b' = b `xor` arrayRead32 ks 1 c' = c `xor` arrayRead32 ks 2 d' = d `xor` arrayRead32 ks 3- (!a'', !b'', !c'', !d'') = foldl' shuffle (a', b', c', d') [0 .. 7]+ (!a'', !b'', !c'', !d'') = L.foldl' shuffle (a', b', c', d') [0 .. 7] ts = ( c'' `xor` arrayRead32 ks 4 , d'' `xor` arrayRead32 ks 5@@ -180,7 +179,7 @@ b' = d `xor` arrayRead32 ks 7 c' = a `xor` arrayRead32 ks 4 d' = b `xor` arrayRead32 ks 5- (!a'', !b'', !c'', !d'') = foldl' unshuffle (a', b', c', d') [8, 7 .. 1]+ (!a'', !b'', !c'', !d'') = L.foldl' unshuffle (a', b', c', d') [8, 7 .. 1] ixs = ( a'' `xor` arrayRead32 ks 0 , b'' `xor` arrayRead32 ks 1@@ -284,7 +283,7 @@ ( \wordIndex -> map ( \rsRow ->- foldl'+ L.foldl' ( \acc (!rsVal, !colIndex) -> acc `xor` gfMult rsPolynomial (B.index key $ 8 * wordIndex + colIndex) rsVal )@@ -443,7 +442,7 @@ b' = rotateL b 8 h :: ByteArray ba => [Word8] -> KeyPackage ba -> Int -> Word32-h input keyPackage offset = foldl' xorMdsColMult 0 $ zip [y0f, y1f, y2f, y3f] $ enumFrom Zero+h input keyPackage offset = L.foldl' xorMdsColMult 0 $ zip [y0f, y1f, y2f, y3f] $ enumFrom Zero where key = rawKeyBytes keyPackage [y0, y1, y2, y3] = take 4 input
Crypto/ConstructHash/MiyaguchiPreneel.hs view
@@ -15,8 +15,7 @@ MiyaguchiPreneel, ) where -import Data.List (foldl')-import Prelude hiding (foldl')+import qualified Data.List as L import Crypto.Cipher.Types import Crypto.Cipher.Types.Utils (chunk)@@ -41,7 +40,7 @@ -> MiyaguchiPreneel cipher -- ^ output tag compute' g =- MP . foldl' (step $ g) (B.replicate bsz 0) . chunks . pad (ZERO bsz) . B.convert+ MP . L.foldl' (step $ g) (B.replicate bsz 0) . chunks . pad (ZERO bsz) . B.convert where bsz = blockSize (g B.empty {- dummy to get block size -}) -- 'chunk' slices rather than splitting the message, which copied whatever
Crypto/ECC.hs view
@@ -49,13 +49,12 @@ import qualified Crypto.ECC.Simple.Prim as Simple import qualified Crypto.ECC.Simple.Types as Simple import Crypto.Error+import Crypto.KEM (SharedSecret (..)) import Crypto.Internal.ByteArray ( ByteArray, ByteArrayAccess,- ScrubbedBytes, ) import qualified Crypto.Internal.ByteArray as B-import Crypto.Internal.Imports import Crypto.Number.Basic (numBits) import Crypto.Number.Serialize (i2ospOf_, os2ip) import qualified Crypto.Number.Serialize.LE as LE@@ -74,16 +73,6 @@ { keypairGetPublic :: !(Point curve) , keypairGetPrivate :: !(Scalar curve) }---- | Secret shared via key exchange-newtype SharedSecret = SharedSecret ScrubbedBytes- deriving (Eq, ByteArrayAccess, NFData)--instance Semigroup SharedSecret where- SharedSecret x <> SharedSecret y = SharedSecret (x <> y)--instance Monoid SharedSecret where- mempty = SharedSecret mempty class EllipticCurve curve where -- | Point on an Elliptic Curve
Crypto/Error/Types.hs view
@@ -57,6 +57,10 @@ -- the base point generates, so multiplying it would answer modulo a -- small order. Appended for the same reason as the constructor above. CryptoError_PointSubgroupInvalid+ | -- | A public key is the right length but is not a well-formed encoding+ -- of one, so no honest party produced it. Appended for the same+ -- reason as the two constructors above.+ CryptoError_PublicKeyStructureInvalid deriving (Show, Eq, Enum, Data) instance E.Exception CryptoError
+ Crypto/KEM.hs view
@@ -0,0 +1,121 @@+-- |+-- Module : Crypto.KEM+-- License : BSD-style+-- Maintainer : Kazu Yamamoto <kazu@iij.ad.jp>+-- Stability : experimental+-- Portability : unknown+--+-- Key encapsulation: one side publishes a key, the other draws a secret and+-- returns something only the first side can turn back into it.+--+-- The shape is not ML-KEM's alone, which is why this is a class in a module+-- of its own rather than part of any one algorithm: 'Crypto.PubKey.MLKEM'+-- is one instance of it, and a caller that does not care which it has can+-- be written against this.+--+-- A bare Diffie-Hellman exchange is deliberately __not__ an instance, even+-- though the shapes line up. The KEM a Diffie-Hellman group gives is+-- DHKEM, of RFC 9180 section 4.1, and that is not the raw exchange: its+-- shared secret is the exchange's output run through HKDF with the+-- ephemeral and the recipient public keys as context, under labels that+-- name the ciphersuite. The binding to those two keys, and the separation+-- between suites, are what the KEM security argument rests on; the raw+-- value has neither. A protocol can supply the binding at its own level --+-- TLS 1.3 does, in the key schedule -- but then the binding belongs to the+-- protocol, not to this class, and offering the raw exchange here would+-- invite its use somewhere that supplies nothing.+--+-- DHKEM proper is an instance, and it is not here either: the labels it+-- derives under carry the HPKE ciphersuite identifier, which is an IANA+-- registry value rather than anything a primitive knows, so the instances+-- live beside the registry in the @hpke@ package.+{-# LANGUAGE GeneralizedNewtypeDeriving #-}+{-# LANGUAGE TypeFamilies #-}++module Crypto.KEM (+ KEM (..),+ SharedSecret (..),+) where++import Data.Kind (Type)++import Crypto.Error (CryptoFailable)+import Crypto.Internal.ByteArray (ByteArrayAccess, ScrubbedBytes)+import Crypto.Internal.Imports+import Crypto.Random (MonadRandom)++-- | Secret shared via key exchange.+newtype SharedSecret = SharedSecret ScrubbedBytes+ deriving (Eq, ByteArrayAccess, NFData)++instance Show SharedSecret where+ show _ = "SharedSecret <redacted>"++instance Semigroup SharedSecret where+ SharedSecret x <> SharedSecret y = SharedSecret (x <> y)++instance Monoid SharedSecret where+ mempty = SharedSecret mempty++-- | A key encapsulation mechanism.+--+-- The three values are named for what they do rather than for what they are+-- in any one algorithm. In ML-KEM the encapsulation key and the ciphertext+-- are what the names say; in a Diffie-Hellman exchange both are public+-- values of the group, and the \"ciphertext\" is the ephemeral one the+-- encapsulating side generates. Keeping them apart in the types is what+-- stops one being passed where the other belongs, which they are not+-- interchangeable for even where they have the same representation.+class KEM kem where+ -- | What the encapsulating side is given.+ type EncapsulationKey kem :: Type++ -- | What the decapsulating side keeps.+ type DecapsulationKey kem :: Type++ -- | What travels back, and is decapsulated.+ type Ciphertext kem :: Type++ -- | The randomness 'encapsulate' draws, for the instances that let a+ -- caller supply it. In ML-KEM it is @m@ of FIPS 203, a string of+ -- bytes; in DHKEM it is the ephemeral secret key, a scalar of the+ -- group. Both are secret, and both determine the shared secret+ -- completely.+ type Coins kem :: Type++ -- | Generate a key pair for the decapsulating side.+ generateKeyPair+ :: MonadRandom m+ => proxy kem -> m (EncapsulationKey kem, DecapsulationKey kem)++ -- | Draw a secret and encapsulate it against the key.+ --+ -- This can fail, and does for some instances: a Diffie-Hellman exchange+ -- refuses a peer value that would make the secret degenerate, where+ -- ML-KEM has nothing to refuse.+ encapsulate+ :: MonadRandom m+ => proxy kem+ -> EncapsulationKey kem+ -> m (CryptoFailable (Ciphertext kem, SharedSecret))++ -- | Encapsulate with the randomness supplied rather than drawn.+ --+ -- The secret this produces is a deterministic function of the key and+ -- these coins, so they must come from a source no other party can+ -- predict or repeat, and must not be used twice. 'encapsulate' is the+ -- entry point for ordinary use; this one is for test vectors, and for a+ -- protocol that has to name the ephemeral value it used -- HPKE lets a+ -- sender supply its own, and the RFC 9180 vectors are written that way.+ encapsulateWith+ :: proxy kem+ -> EncapsulationKey kem+ -> Coins kem+ -> CryptoFailable (Ciphertext kem, SharedSecret)++ -- | Recover the secret.+ decapsulate+ :: proxy kem+ -> DecapsulationKey kem+ -> Ciphertext kem+ -> CryptoFailable SharedSecret
Crypto/Number/F2m.hs view
@@ -36,9 +36,8 @@ (.&.), (.|.), )-import Data.List (foldl')+import qualified Data.List as L import Data.Word (Word32)-import Prelude hiding (foldl') -- | Binary Polynomial represented by an integer type BinaryPolynomial = Integer@@ -89,7 +88,7 @@ | otherwise = fold es- (foldl' (\acc e -> acc `xor` (hi `shiftL` e)) (n .&. mask) es)+ (L.foldl' (\acc e -> acc `xor` (hi `shiftL` e)) (n .&. mask) es) where hi = n `shiftR` lfx {-# INLINE modF2m #-}@@ -211,7 +210,7 @@ where spread :: Int -> Word32 spread b =- foldl' (\acc i -> if testBit b i then setBit acc (2 * i) else acc) 0 [0 .. 7]+ L.foldl' (\acc i -> if testBit b i then setBit acc (2 * i) else acc) 0 [0 .. 7] {-# NOINLINE spreadTable #-} {-# INLINE squareF2m' #-}
Crypto/OTP.hs view
@@ -49,9 +49,8 @@ import Crypto.MAC.HMAC import Data.Bits (complement, shiftL, shiftR, xor, (.&.), (.|.)) import Data.ByteArray.Mapping (fromW64BE)-import Data.List (foldl')+import qualified Data.List as L import Data.Word-import Prelude hiding (foldl') -- | A one-time password which is a sequence of 4 to 9 digits. type OTP = Word32@@ -150,7 +149,7 @@ range = map (hotp h d k) [c .. c + fromIntegral s] -- the offset of the first match, accumulated without stopping there- (matched, offset) = foldl' pick (0, 0) (zip [0 ..] range)+ (matched, offset) = L.foldl' pick (0, 0) (zip [0 ..] range) pick (!m, !off) (i, candidate) = (m .|. hit, off .|. (hit .&. i)) where -- zero once something has matched, so only the first match counts@@ -162,7 +161,7 @@ -- the counters continue past the window, and wrap where the old -- 'checkExtraOtps' wrapped extrasMatched =- foldl' step (complement 0) (zip (iterate (+ 1) afterFirst) extras)+ L.foldl' step (complement 0) (zip (iterate (+ 1) afterFirst) extras) step acc (ctr, p) = acc .&. eqMask (hotp h d k ctr) p -- | All ones when the two values are equal, zero otherwise, without branching@@ -250,7 +249,7 @@ -- every candidate is compared, and none of the comparisons stops early, so -- neither which step matched nor how far a mismatch got is visible in how -- long this takes- matched = foldl' step 0 (map (hotp h d k) (range window []))+ matched = L.foldl' step 0 (map (hotp h d k) (range window [])) step acc candidate = acc .|. eqMask candidate otp timeToCounter :: Word64 -> Word64 -> Word16 -> Word64
Crypto/PubKey/Internal.hs view
@@ -12,8 +12,7 @@ ) where import Data.Bits (shiftR)-import Data.List (foldl')-import Prelude hiding (foldl')+import qualified Data.List as L import Crypto.Hash import Crypto.Internal.ByteArray (ByteArrayAccess)@@ -22,7 +21,7 @@ -- | This is a strict version of and and' :: [Bool] -> Bool-and' l = foldl' (&&!) True l+and' l = L.foldl' (&&!) True l -- | This is a strict version of &&. (&&!) :: Bool -> Bool -> Bool
+ Crypto/PubKey/MLDSA.hs view
@@ -0,0 +1,656 @@+-- |+-- Module : Crypto.PubKey.MLDSA+-- License : BSD-style+-- Maintainer : Kazu Yamamoto <kazu@iij.ad.jp>+-- Stability : experimental+-- Portability : unknown+--+-- ML-DSA, the Module-Lattice-Based Digital Signature Algorithm of+-- <https://csrc.nist.gov/pubs/fips/204/final FIPS 204>, in all three+-- parameter sets.+--+-- > (vk, sk) <- generateKeyPair MLDSA65+-- > sig <- sign sk emptyContext message+-- > verify vk emptyContext message sig+--+-- What 'generateKeyPair' and 'sign' draw their randomness from is the+-- 'Crypto.Random.MonadRandom' instance in use. Its documentation says what+-- an instance of your own has to be.+--+-- The parameter set is a type, so an ML-DSA-65 key cannot be passed where+-- an ML-DSA-87 one is expected. The three are fixed by FIPS 204 and the+-- class has no other instances.+--+-- This is pure ML-DSA: the message goes in whole. The pre-hash variant+-- (HashML-DSA) is a different algorithm with a different domain separator+-- and is not offered here.+{-# LANGUAGE DataKinds #-}+{-# LANGUAGE GeneralizedNewtypeDeriving #-}+{-# LANGUAGE ScopedTypeVariables #-}++module Crypto.PubKey.MLDSA (+ -- * Parameter sets+ MLDSA44 (..),+ MLDSA65 (..),+ MLDSA87 (..),+ MLDSA (verificationKeySize, signingKeySize, signatureSize),++ -- * Keys and signatures+ VerificationKey,+ SigningKey,+ Signature,++ -- * Smart constructors+ verificationKey,+ signingKey,+ signature,++ -- * Generating a key pair+ generateKeyPair,+ generateKeyPairAndSeed,+ keyPairFromSeed,+ toPublic,++ -- * The context string+ Context,+ context,+ emptyContext,++ -- * The message representative+ Mu,+ mu,+ messageRepresentative,++ -- ** A message that does not arrive in one piece+ MuContext,+ muInit,+ muUpdate,+ muUpdates,+ muFinalize,++ -- * Signing and verifying+ sign,+ signWith,+ signDeterministic,+ verify,++ -- * Signing and verifying a message representative+ signExternalMu,+ signExternalMuWith,+ signExternalMuDeterministic,+ verifyExternalMu,++ -- * Sizes+ seedSize,+ signingRandomnessSize,+ maxContextLength,+ muSize,+) where++import Data.Proxy (Proxy (..))+import Foreign.C.Types (CInt (..), CSize (..))+import Foreign.Ptr (Ptr, nullPtr)++import Crypto.Debug (DebugShow (..), debugShowBytes)+import Crypto.Hash (Digest, hash, hashFinalize, hashInit, hashUpdate, hashUpdates)+import qualified Crypto.Hash as Hash (Context)+import Crypto.Hash.Algorithms (SHAKE256 (..))+import Crypto.Error+import Crypto.Internal.ByteArray (+ ByteArrayAccess,+ Bytes,+ ScrubbedBytes,+ withByteArray,+ )+import qualified Crypto.Internal.ByteArray as B+import Crypto.Internal.Compat (unsafeDoIO)+import Crypto.Internal.Imports+import Crypto.Random (MonadRandom, getRandomBytes)++-- | ML-DSA-44.+data MLDSA44 = MLDSA44 deriving (Show, Eq)++-- | ML-DSA-65.+data MLDSA65 = MLDSA65 deriving (Show, Eq)++-- | ML-DSA-87.+data MLDSA87 = MLDSA87 deriving (Show, Eq)++-- | The three parameter sets of FIPS 204.+--+-- Named for the algorithm rather than \"DSA\", which is a different one that+-- crypton also has, in "Crypto.PubKey.DSA". It is the three sets FIPS 204+-- defines, closed, carrying their sizes and the calls into the+-- implementation; only the sizes are exported.+class MLDSA p where+ -- | Size in bytes of a 'VerificationKey' of this parameter set.+ verificationKeySize :: proxy p -> Int++ -- | Size in bytes of a 'SigningKey' of this parameter set.+ signingKeySize :: proxy p -> Int++ -- | Size in bytes of a 'Signature' of this parameter set.+ signatureSize :: proxy p -> Int++ c_keypair :: proxy p -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+ c_sign+ :: proxy p+ -> Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> Ptr Word8 -> CInt -> IO CInt+ c_verify+ :: proxy p+ -> Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> CInt -> IO CInt+ c_pkFromSk :: proxy p -> Ptr Word8 -> Ptr Word8 -> IO CInt++-- | A public verification key.+newtype VerificationKey p = VerificationKey Bytes+ deriving (Show, Eq, ByteArrayAccess, NFData)++-- | A private signing key.+newtype SigningKey p = SigningKey ScrubbedBytes+ deriving (Eq, ByteArrayAccess, NFData)++instance Show (SigningKey p) where+ show _ = "SigningKey <redacted>"++instance DebugShow (SigningKey p) where+ debugShow = debugShowBytes "SigningKey"++-- | A signature.+newtype Signature p = Signature Bytes+ deriving (Show, Eq, ByteArrayAccess, NFData)++-- | The context string a signature is bound to, at most+-- 'maxContextLength' bytes.+--+-- FIPS 204 mixes it into what is signed, so a signature made under one+-- context does not verify under another. Two uses of one key that don't+-- agree on a context string cannot be made to accept each other's+-- signatures. Use 'emptyContext' where there is nothing to separate -- TLS,+-- for one, signs with an empty context.+newtype Context = Context Bytes+ deriving (Show, Eq, ByteArrayAccess, NFData)++-- | The context string of length zero, which is what to sign under when+-- there is nothing to separate.+--+-- It is a context and not the absence of one: 'sign' always takes one, and+-- what this separates from is every non-empty context there is.+emptyContext :: Context+emptyContext = Context B.empty++-- | Try to build a context string.+context :: ByteArrayAccess ba => ba -> CryptoFailable Context+context bs+ | B.length bs <= maxContextLength =+ CryptoPassed $ Context $ B.copyAndFreeze bs (\_ -> return ())+ | otherwise = CryptoFailed CryptoError_ParameterInvalid++-- | The longest context string FIPS 204 allows, which is 255 bytes because+-- its length is encoded in one byte.+maxContextLength :: Int+maxContextLength = 255++-- | The message representative, @mu@ in FIPS 204: a 64-byte commitment to+-- the verification key, the context string and the message, and the only+-- part of them that signing and verification actually read.+--+-- Signing it directly is the "external mu" interface. It is for a caller+-- that has the representative without having the message in one piece: a+-- message arriving as a stream, or hashed on another machine, or by a+-- device that holds the key and is handed only this. TLS does not need it.+newtype Mu = Mu Bytes+ deriving (Show, Eq, ByteArrayAccess, NFData)++-- | Size in bytes of a 'Mu'.+muSize :: Int+muSize = 64++-- | Try to read a message representative.+mu :: ByteArrayAccess ba => ba -> CryptoFailable Mu+mu bs+ | B.length bs == muSize = CryptoPassed $ Mu $ B.copyAndFreeze bs (\_ -> return ())+ | otherwise = CryptoFailed CryptoError_ParameterInvalid++-- | Compute the message representative, for a caller that wants to make it+-- here and sign it later, or sign it elsewhere.+--+-- @'signExternalMuDeterministic' sk ('messageRepresentative' ('toPublic' sk) ctx msg)@+-- and @'signDeterministic' sk ctx msg@ are the same signature.+messageRepresentative+ :: (MLDSA p, ByteArrayAccess msg)+ => VerificationKey p -> Context -> msg -> Mu+messageRepresentative vk ctx msg = muFinalize (muUpdate (muInit vk ctx) msg)++-- | A 'Mu' being computed, with the message going in a piece at a time.+--+-- The name is not 'Context': that is ML-DSA's context string, which this+-- is built from and is not.+newtype MuContext = MuContext (Hash.Context (SHAKE256 512))++-- | Begin a message representative. The key and the context string are+-- what it is bound to, and they are all that is needed before the message.+--+-- > muFinalize (muUpdates (muInit vk ctx) chunks)+--+-- is 'messageRepresentative' of the chunks joined, so a message too large+-- to hold at once never has to be.+muInit :: MLDSA p => VerificationKey p -> Context -> MuContext+muInit vk ctx =+ -- FIPS 204: tr <- H(pk, 64) at key generation, and mu <- H(tr || M', 64)+ -- when signing, with M' the domain-separated message. Everything up to+ -- the message itself is absorbed here.+ MuContext $ hashUpdates hashInit [tr, domainPrefix ctx]+ where+ tr = B.convert (shake64 (B.convert vk :: Bytes)) :: Bytes++-- | Absorb a piece of the message.+muUpdate :: ByteArrayAccess msg => MuContext -> msg -> MuContext+muUpdate (MuContext c) msg = MuContext (hashUpdate c msg)++-- | Absorb several pieces, which is 'muUpdate' one after the other.+muUpdates :: ByteArrayAccess msg => MuContext -> [msg] -> MuContext+muUpdates (MuContext c) msgs = MuContext (hashUpdates c msgs)++-- | The message representative of everything absorbed so far.+muFinalize :: MuContext -> Mu+muFinalize (MuContext c) = Mu (B.convert (hashFinalize c :: Digest (SHAKE256 512)))++shake64 :: ByteArrayAccess ba => ba -> Digest (SHAKE256 512)+shake64 = hash++-- | Size in bytes of the seed 'keyPairFromSeed' takes, @xi@ in FIPS 204.+seedSize :: Int+seedSize = 32++-- | Size in bytes of the randomness 'signWith' takes.+signingRandomnessSize :: Int+signingRandomnessSize = 32++-- | Try to read a verification key. Only the length is checked: a+-- verification key is a packed encoding with no redundancy to test, and one+-- that is not a real key simply verifies nothing.+verificationKey+ :: forall p ba+ . (MLDSA p, ByteArrayAccess ba)+ => ba -> CryptoFailable (VerificationKey p)+verificationKey bs+ | B.length bs == verificationKeySize (Proxy :: Proxy p) =+ CryptoPassed $ VerificationKey $ B.copyAndFreeze bs (\_ -> return ())+ | otherwise = CryptoFailed CryptoError_PublicKeySizeInvalid++-- | Try to read a signing key.+--+-- Beyond the length this runs the validity checks of the implementation:+-- the secret polynomials must have coefficients in range, and the+-- commitment and the public-key hash the key carries must match what is+-- recomputed from the rest of it. A key that fails has been damaged or was+-- never a key, and signing with it would produce signatures nothing+-- verifies.+signingKey+ :: forall p ba+ . (MLDSA p, ByteArrayAccess ba)+ => ba -> CryptoFailable (SigningKey p)+signingKey bs+ | B.length bs /= signingKeySize p = CryptoFailed CryptoError_SecretKeySizeInvalid+ | otherwise = unsafeDoIO $ do+ (r, _ :: Bytes) <- B.allocRet (verificationKeySize p) $ \ppk ->+ withByteArray bs $ \psk -> c_pkFromSk p ppk psk+ return $+ if r == 0+ then CryptoPassed $ SigningKey $ B.copyAndFreeze bs (\_ -> return ())+ else CryptoFailed CryptoError_SecretKeyStructureInvalid+ where+ p = Proxy :: Proxy p+{-# NOINLINE signingKey #-}++-- | Try to read a signature. Only the length is checked; whether it is a+-- signature of anything is what 'verify' answers.+signature+ :: forall p ba+ . (MLDSA p, ByteArrayAccess ba)+ => ba -> CryptoFailable (Signature p)+signature bs+ | B.length bs == signatureSize (Proxy :: Proxy p) =+ CryptoPassed $ Signature $ B.copyAndFreeze bs (\_ -> return ())+ | otherwise = CryptoFailed CryptoError_ParameterInvalid++-- | Recover the verification key a signing key was made with.+toPublic :: forall p. MLDSA p => SigningKey p -> VerificationKey p+toPublic sk = VerificationKey $ unsafeDoIO $ do+ (_ :: CInt, pk) <- B.allocRet (verificationKeySize p) $ \ppk ->+ withByteArray sk $ \psk -> c_pkFromSk p ppk psk+ return pk+ where+ p = Proxy :: Proxy p+{-# NOINLINE toPublic #-}++-- | Generate a key pair.+--+-- The seed it is derived from is drawn here and thrown away. Use+-- 'generateKeyPairAndSeed' where it has to be kept.+generateKeyPair+ :: forall p proxy m+ . (MLDSA p, MonadRandom m)+ => proxy p -> m (VerificationKey p, SigningKey p)+generateKeyPair p = do+ (vk, sk, _) <- generateKeyPairAndSeed p+ return (vk, sk)++-- | Generate a key pair and hand back the seed it was derived from, @xi@+-- in FIPS 204.+--+-- A 'SigningKey' is the expanded key and nothing else, so the seed cannot+-- be recovered from a pair afterwards. An application that has to write+-- the key out in a form that keeps the seed -- RFC 9881 lets an ML-DSA+-- private key be the seed, the expanded key, or both -- has to generate it+-- here:+--+-- > (vk, sk, seed) <- generateKeyPairAndSeed MLDSA65+--+-- The seed is as secret as the signing key: 'keyPairFromSeed' turns it+-- back into the same pair.+generateKeyPairAndSeed+ :: forall p proxy m+ . (MLDSA p, MonadRandom m)+ => proxy p -> m (VerificationKey p, SigningKey p, ScrubbedBytes)+generateKeyPairAndSeed p = do+ seed <- getRandomBytes seedSize :: m ScrubbedBytes+ case keyPairFromSeed p seed of+ CryptoPassed (vk, sk) -> return (vk, sk, seed)+ CryptoFailed e ->+ error ("Crypto.PubKey.MLDSA.generateKeyPairAndSeed: " ++ show e)++-- | Derive a key pair from a seed, @xi@ in FIPS 204, which must be+-- 'seedSize' bytes.+keyPairFromSeed+ :: forall p proxy ba+ . (MLDSA p, ByteArrayAccess ba)+ => proxy p -> ba -> CryptoFailable (VerificationKey p, SigningKey p)+keyPairFromSeed p seed+ | B.length seed /= seedSize = CryptoFailed CryptoError_SeedSizeInvalid+ | otherwise = unsafeDoIO $ do+ -- Not zeroed, and does not need to be: the C writes the whole+ -- buffer, and on a non-zero return the result is discarded without+ -- being read. Anything that is *read* before being written has to+ -- use B.zero instead -- see signInternal in Crypto.PubKey.MLDSA.+ sk <- B.alloc (signingKeySize p) (\_ -> return ()) :: IO ScrubbedBytes+ (r, vk) <- B.allocRet (verificationKeySize p) $ \pvk ->+ withByteArray sk $ \psk ->+ withByteArray seed $ \pseed ->+ c_keypair p pvk psk pseed+ return $+ if r == 0+ then CryptoPassed (VerificationKey vk, SigningKey sk)+ else CryptoFailed CryptoError_ParameterInvalid+{-# NOINLINE keyPairFromSeed #-}++-- | Sign a message.+--+-- This is the hedged signing FIPS 204 recommends: fresh randomness goes in+-- alongside the key and the message, so two signatures of one message+-- differ and a fault in one reveals less. Verification does not care which+-- of the three entry points made the signature.+sign+ :: forall p m msg+ . (MLDSA p, MonadRandom m, ByteArrayAccess msg)+ => SigningKey p -> Context -> msg -> m (Signature p)+sign sk ctx msg = do+ rnd <- getRandomBytes signingRandomnessSize :: m ScrubbedBytes+ case signWith sk ctx msg rnd of+ CryptoPassed s -> return s+ CryptoFailed e -> error ("Crypto.PubKey.MLDSA.sign: " ++ show e)++-- | Sign with the randomness supplied, which must be+-- 'signingRandomnessSize' bytes.+--+-- For test vectors, and for callers who draw their own randomness. Ordinary+-- use wants 'sign'.+signWith+ :: forall p msg rnd+ . (MLDSA p, ByteArrayAccess msg, ByteArrayAccess rnd)+ => SigningKey p -> Context -> msg -> rnd -> CryptoFailable (Signature p)+signWith sk ctx msg rnd+ | B.length rnd /= signingRandomnessSize = CryptoFailed CryptoError_SeedSizeInvalid+ | otherwise = signInternal sk ctx msg (Just rnd)++-- | Sign deterministically, as FIPS 204 section 3.4 allows: the randomness+-- is replaced by zeroes, so one key and one message always give one+-- signature.+--+-- This is what test vectors are written against, and what to use where the+-- signature must be reproducible. It gives up what hedging buys, so where+-- there is a usable random source 'sign' is the better default.+signDeterministic+ :: forall p msg+ . (MLDSA p, ByteArrayAccess msg)+ => SigningKey p -> Context -> msg -> Signature p+signDeterministic sk ctx msg =+ case signInternal sk ctx msg (Nothing :: Maybe Bytes) of+ CryptoPassed s -> s+ CryptoFailed e -> error ("Crypto.PubKey.MLDSA.signDeterministic: " ++ show e)++signInternal+ :: forall p msg rnd+ . (MLDSA p, ByteArrayAccess msg, ByteArrayAccess rnd)+ => SigningKey p -> Context -> msg -> Maybe rnd -> CryptoFailable (Signature p)+signInternal sk ctx msg mrnd = unsafeDoIO $ do+ -- B.zero, not B.alloc with an empty action: alloc hands back whatever+ -- was in the memory. That made signDeterministic sign with the last+ -- caller's bytes and produce a different signature every time, which the+ -- ACVP vectors caught only once the whole suite ran and the allocator+ -- stopped handing out fresh zeroed pages.+ let zeroes = B.zero signingRandomnessSize :: ScrubbedBytes+ withRnd f = case mrnd of+ Just r -> withByteArray r f+ Nothing -> withByteArray zeroes f+ (r, sig) <- B.allocRet (signatureSize p) $ \psig ->+ withByteArray msg $ \pmsg ->+ withByteArray pre $ \ppre ->+ withRnd $ \prnd ->+ withByteArray sk $ \psk ->+ c_sign+ p+ psig+ pmsg+ (fromIntegral (B.length msg))+ ppre+ (fromIntegral (B.length pre))+ prnd+ psk+ 0+ return $+ if r == 0+ then CryptoPassed (Signature sig)+ else CryptoFailed CryptoError_ParameterInvalid+ where+ p = Proxy :: Proxy p+ pre = domainPrefix ctx+{-# NOINLINE signInternal #-}++-- | Sign a message representative, drawing the randomness.+--+-- The context string is already inside the representative, which is why+-- this does not take one.+signExternalMu+ :: forall p m+ . (MLDSA p, MonadRandom m)+ => SigningKey p -> Mu -> m (Signature p)+signExternalMu sk m = do+ rnd <- getRandomBytes signingRandomnessSize :: m ScrubbedBytes+ case signExternalMuWith sk m rnd of+ CryptoPassed s -> return s+ CryptoFailed e -> error ("Crypto.PubKey.MLDSA.signExternalMu: " ++ show e)++-- | Sign a message representative with the randomness supplied.+signExternalMuWith+ :: (MLDSA p, ByteArrayAccess rnd)+ => SigningKey p -> Mu -> rnd -> CryptoFailable (Signature p)+signExternalMuWith sk m rnd+ | B.length rnd /= signingRandomnessSize = CryptoFailed CryptoError_SeedSizeInvalid+ | otherwise = signMu sk m (Just rnd)++-- | Sign a message representative deterministically.+signExternalMuDeterministic+ :: MLDSA p => SigningKey p -> Mu -> Signature p+signExternalMuDeterministic sk m =+ case signMu sk m (Nothing :: Maybe Bytes) of+ CryptoPassed s -> s+ CryptoFailed e ->+ error ("Crypto.PubKey.MLDSA.signExternalMuDeterministic: " ++ show e)++-- | Verify a signature of a message representative.+verifyExternalMu+ :: forall p. MLDSA p => VerificationKey p -> Mu -> Signature p -> Bool+verifyExternalMu vk m sig+ | B.length sig /= signatureSize p = False+ | otherwise = unsafeDoIO $+ withByteArray sig $ \psig ->+ withByteArray m $ \pmu ->+ withByteArray vk $ \pvk -> do+ r <-+ c_verify+ p+ psig+ pmu+ (fromIntegral muSize)+ nullPtr+ 0+ pvk+ 1+ return (r == 0)+ where+ p = Proxy :: Proxy p+{-# NOINLINE verifyExternalMu #-}++-- The external-mu entry points are the ordinary ones with the last argument+-- set: the representative goes in where the message would, there is no+-- domain separation prefix to prepend because it is already inside, and the+-- implementation is told so.+signMu+ :: forall p rnd+ . (MLDSA p, ByteArrayAccess rnd)+ => SigningKey p -> Mu -> Maybe rnd -> CryptoFailable (Signature p)+signMu sk m mrnd = unsafeDoIO $ do+ let zeroes = B.zero signingRandomnessSize :: ScrubbedBytes+ withRnd f = case mrnd of+ Just r -> withByteArray r f+ Nothing -> withByteArray zeroes f+ (r, sig) <- B.allocRet (signatureSize p) $ \psig ->+ withByteArray m $ \pmu ->+ withRnd $ \prnd ->+ withByteArray sk $ \psk ->+ c_sign p psig pmu (fromIntegral muSize) nullPtr 0 prnd psk 1+ return $+ if r == 0+ then CryptoPassed (Signature sig)+ else CryptoFailed CryptoError_ParameterInvalid+ where+ p = Proxy :: Proxy p+{-# NOINLINE signMu #-}++-- | Verify a signature.+--+-- The context must be the one it was signed under; anything else is a+-- rejection, which is what the context is for.+verify+ :: forall p msg+ . (MLDSA p, ByteArrayAccess msg)+ => VerificationKey p -> Context -> msg -> Signature p -> Bool+verify vk ctx msg sig+ | B.length sig /= signatureSize p = False+ | otherwise = unsafeDoIO $+ withByteArray sig $ \psig ->+ withByteArray msg $ \pmsg ->+ withByteArray pre $ \ppre ->+ withByteArray vk $ \pvk -> do+ r <-+ c_verify+ p+ psig+ pmsg+ (fromIntegral (B.length msg))+ ppre+ (fromIntegral (B.length pre))+ pvk+ 0+ return (r == 0)+ where+ p = Proxy :: Proxy p+ pre = domainPrefix ctx+{-# NOINLINE verify #-}++-- | The domain separation prefix of FIPS 204 for pure ML-DSA, which is a+-- zero byte, the context's length and the context itself. It is built here+-- rather than taken from the implementation because it is three bytes of+-- concatenation and doing it here keeps one fewer foreign call.+domainPrefix :: Context -> Bytes+domainPrefix (Context ctx) =+ B.concat [B.pack [0, fromIntegral (B.length ctx)] :: Bytes, B.convert ctx]++instance MLDSA MLDSA44 where+ verificationKeySize _ = 1312+ signingKeySize _ = 2560+ signatureSize _ = 2420+ c_keypair _ = c_mldsa44_keypair+ c_sign _ = c_mldsa44_sign+ c_verify _ = c_mldsa44_verify+ c_pkFromSk _ = c_mldsa44_pk_from_sk++instance MLDSA MLDSA65 where+ verificationKeySize _ = 1952+ signingKeySize _ = 4032+ signatureSize _ = 3309+ c_keypair _ = c_mldsa65_keypair+ c_sign _ = c_mldsa65_sign+ c_verify _ = c_mldsa65_verify+ c_pkFromSk _ = c_mldsa65_pk_from_sk++instance MLDSA MLDSA87 where+ verificationKeySize _ = 2592+ signingKeySize _ = 4896+ signatureSize _ = 4627+ c_keypair _ = c_mldsa87_keypair+ c_sign _ = c_mldsa87_sign+ c_verify _ = c_mldsa87_verify+ c_pkFromSk _ = c_mldsa87_pk_from_sk++foreign import ccall unsafe "crypton_mldsa44_keypair_internal"+ c_mldsa44_keypair :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mldsa44_signature_internal"+ c_mldsa44_sign+ :: Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> Ptr Word8 -> CInt -> IO CInt+foreign import ccall unsafe "crypton_mldsa44_verify_internal"+ c_mldsa44_verify+ :: Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> CInt -> IO CInt+foreign import ccall unsafe "crypton_mldsa44_pk_from_sk"+ c_mldsa44_pk_from_sk :: Ptr Word8 -> Ptr Word8 -> IO CInt++foreign import ccall unsafe "crypton_mldsa65_keypair_internal"+ c_mldsa65_keypair :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mldsa65_signature_internal"+ c_mldsa65_sign+ :: Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> Ptr Word8 -> CInt -> IO CInt+foreign import ccall unsafe "crypton_mldsa65_verify_internal"+ c_mldsa65_verify+ :: Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> CInt -> IO CInt+foreign import ccall unsafe "crypton_mldsa65_pk_from_sk"+ c_mldsa65_pk_from_sk :: Ptr Word8 -> Ptr Word8 -> IO CInt++foreign import ccall unsafe "crypton_mldsa87_keypair_internal"+ c_mldsa87_keypair :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mldsa87_signature_internal"+ c_mldsa87_sign+ :: Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> Ptr Word8 -> CInt -> IO CInt+foreign import ccall unsafe "crypton_mldsa87_verify_internal"+ c_mldsa87_verify+ :: Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> CInt -> IO CInt+foreign import ccall unsafe "crypton_mldsa87_pk_from_sk"+ c_mldsa87_pk_from_sk :: Ptr Word8 -> Ptr Word8 -> IO CInt
+ Crypto/PubKey/MLKEM.hs view
@@ -0,0 +1,448 @@+-- |+-- Module : Crypto.PubKey.MLKEM+-- License : BSD-style+-- Maintainer : Kazu Yamamoto <kazu@iij.ad.jp>+-- Stability : experimental+-- Portability : unknown+--+-- ML-KEM, the Module-Lattice-Based Key-Encapsulation Mechanism of+-- <https://csrc.nist.gov/pubs/fips/203/final FIPS 203>, in all three+-- parameter sets.+--+-- A key encapsulation mechanism is not a Diffie-Hellman: there is no shared+-- secret to be computed from two key pairs. One side publishes an+-- 'EncapsulationKey'; the other calls 'encapsulate' on it, which draws a+-- fresh secret and returns it along with a 'Ciphertext' that only the holder+-- of the matching 'DecapsulationKey' can turn back into that secret.+--+-- > (ek, dk) <- generateKeyPair MLKEM768 -- the receiver+-- > (ct, ss) <- encapsulate ek -- the sender+-- > let ss' = decapsulate dk ct -- the receiver, again+-- > ss == ss'+--+-- What 'generateKeyPair' and 'encapsulate' draw their randomness from is+-- the 'Crypto.Random.MonadRandom' instance in use. Its documentation says+-- what an instance of your own has to be.+--+-- The parameter set is a type, so an ML-KEM-768 key cannot be passed where+-- an ML-KEM-1024 one is expected. The three are fixed by FIPS 203 and the+-- class has no other instances.+{-# LANGUAGE GeneralizedNewtypeDeriving #-}+{-# LANGUAGE TypeFamilies #-}+{-# LANGUAGE TypeOperators #-}+{-# LANGUAGE ScopedTypeVariables #-}++module Crypto.PubKey.MLKEM (+ -- * Parameter sets+ MLKEM512 (..),+ MLKEM768 (..),+ MLKEM1024 (..),+ MLKEM (encapsulationKeySize, decapsulationKeySize, ciphertextSize),++ -- * Keys, ciphertexts and shared secrets+ --+ -- | These are the associated types of 'KEM', re-exported so that a+ -- caller of this module alone has them.+ KEM (..),+ SharedSecret (..),++ -- * Smart constructors+ encapsulationKey,+ decapsulationKey,+ ciphertext,++ -- * What ML-KEM has beyond the class+ generateKeyPairAndSeed,+ keyPairFromSeed,++ -- * Sizes+ seedSize,+ encapsulationCoinsSize,+ sharedSecretSize,+) where++import Data.Proxy (Proxy (..))+import Foreign.C.Types (CInt (..))+import Foreign.Ptr (Ptr)++import Crypto.Debug (DebugShow (..), debugShowBytes)+import Crypto.Error+import Crypto.KEM+import Crypto.Internal.ByteArray (+ ByteArrayAccess,+ Bytes,+ ScrubbedBytes,+ withByteArray,+ )+import qualified Crypto.Internal.ByteArray as B+import Crypto.Internal.Compat (unsafeDoIO)+import Crypto.Internal.Imports+import Crypto.Random (MonadRandom, getRandomBytes)++-- | ML-KEM-512.+data MLKEM512 = MLKEM512 deriving (Show, Eq)++-- | ML-KEM-768. This is the set TLS uses, on its own and as the+-- lattice half of the hybrid groups.+data MLKEM768 = MLKEM768 deriving (Show, Eq)++-- | ML-KEM-1024.+data MLKEM1024 = MLKEM1024 deriving (Show, Eq)++-- | The three parameter sets of FIPS 203.+--+-- This is not an abstract KEM interface and does not try to be: it is the+-- three sets FIPS 203 defines, closed, carrying their sizes and the calls+-- into the implementation. Only the sizes are exported. If crypton grows+-- a second KEM and an interface common to both is wanted, that belongs in+-- a module of its own, with this as one of its instances.+class+ ( KEM p+ , EncapsulationKey p ~ MLKEMEncapsulationKey p+ , DecapsulationKey p ~ MLKEMDecapsulationKey p+ , Ciphertext p ~ MLKEMCiphertext p+ , Coins p ~ ScrubbedBytes+ ) =>+ MLKEM p+ where+ -- | Size in bytes of an 'EncapsulationKey' of this parameter set.+ encapsulationKeySize :: proxy p -> Int++ -- | Size in bytes of a 'DecapsulationKey' of this parameter set.+ decapsulationKeySize :: proxy p -> Int++ -- | Size in bytes of a 'Ciphertext' of this parameter set.+ ciphertextSize :: proxy p -> Int++ c_keypair :: proxy p -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+ c_enc :: proxy p -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+ c_dec :: proxy p -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+ c_checkPk :: proxy p -> Ptr Word8 -> IO CInt+ c_checkSk :: proxy p -> Ptr Word8 -> IO CInt++-- | A public encapsulation key, @ek@ in FIPS 203.+newtype MLKEMEncapsulationKey p = MLKEMEncapsulationKey Bytes+ deriving (Show, Eq, ByteArrayAccess, NFData)++-- | A private decapsulation key, @dk@ in FIPS 203. It embeds the matching+-- encapsulation key, which is why it is the larger of the two.+newtype MLKEMDecapsulationKey p = MLKEMDecapsulationKey ScrubbedBytes+ deriving (Eq, ByteArrayAccess, NFData)++instance Show (MLKEMDecapsulationKey p) where+ show _ = "DecapsulationKey <redacted>"++instance DebugShow (MLKEMDecapsulationKey p) where+ debugShow = debugShowBytes "DecapsulationKey"++-- | The value 'encapsulate' produces and 'decapsulate' consumes.+newtype MLKEMCiphertext p = MLKEMCiphertext Bytes+ deriving (Show, Eq, ByteArrayAccess, NFData)++-- | Size in bytes of the seed 'keyPairFromSeed' takes, which is @d@ and @z@+-- of FIPS 203 one after the other.+seedSize :: Int+seedSize = 64++-- | Size in bytes of the randomness 'encapsulateWith' takes, @m@ in+-- FIPS 203.+encapsulationCoinsSize :: Int+encapsulationCoinsSize = 32++-- | Size in bytes of a 'SharedSecret'.+sharedSecretSize :: Int+sharedSecretSize = 32++-- | Try to read an encapsulation key.+--+-- Beyond the length this runs the check of FIPS 203 section 7.2: the key+-- must be the encoding of coefficients that are all in range, which is to+-- say it must survive a decode and re-encode unchanged. A key that fails+-- it is not one any honest party produced.+encapsulationKey+ :: forall p ba+ . (MLKEM p, ByteArrayAccess ba)+ => ba -> CryptoFailable (EncapsulationKey p)+encapsulationKey bs+ | B.length bs /= encapsulationKeySize p = CryptoFailed CryptoError_PublicKeySizeInvalid+ | otherwise = unsafeDoIO $ withByteArray bs $ \inp -> do+ r <- c_checkPk p inp+ return $+ if r == 0+ then CryptoPassed $ MLKEMEncapsulationKey $ B.copyAndFreeze bs (\_ -> return ())+ else CryptoFailed CryptoError_PublicKeyStructureInvalid+ where+ p = Proxy :: Proxy p+{-# NOINLINE encapsulationKey #-}++-- | Try to read a decapsulation key.+--+-- Beyond the length this runs the check of FIPS 203 section 7.3: the hash+-- of the encapsulation key the private key embeds must match the copy of+-- that hash it also embeds. The two disagreeing means the key was not+-- produced as a pair, and decapsulating with it would silently answer with+-- the implicit rejection every time.+decapsulationKey+ :: forall p ba+ . (MLKEM p, ByteArrayAccess ba)+ => ba -> CryptoFailable (DecapsulationKey p)+decapsulationKey bs+ | B.length bs /= decapsulationKeySize p = CryptoFailed CryptoError_SecretKeySizeInvalid+ | otherwise = unsafeDoIO $ withByteArray bs $ \inp -> do+ r <- c_checkSk p inp+ return $+ if r == 0+ then CryptoPassed $ MLKEMDecapsulationKey $ B.copyAndFreeze bs (\_ -> return ())+ else CryptoFailed CryptoError_SecretKeyStructureInvalid+ where+ p = Proxy :: Proxy p+{-# NOINLINE decapsulationKey #-}++-- | Try to read a ciphertext. Only the length is checked; every string of+-- the right length is a ciphertext that 'decapsulate' will answer.+ciphertext+ :: forall p ba+ . (MLKEM p, ByteArrayAccess ba)+ => ba -> CryptoFailable (Ciphertext p)+ciphertext bs+ | B.length bs == ciphertextSize (Proxy :: Proxy p) =+ CryptoPassed $ MLKEMCiphertext $ B.copyAndFreeze bs (\_ -> return ())+ | otherwise = CryptoFailed CryptoError_PointSizeInvalid++-- | Generate a key pair.+--+-- The seed it is derived from is drawn here and thrown away. Use+-- 'generateKeyPairAndSeed' where it has to be kept.+mlkemGenerateKeyPair+ :: forall p proxy m+ . (MLKEM p, MonadRandom m)+ => proxy p -> m (MLKEMEncapsulationKey p, DecapsulationKey p)+mlkemGenerateKeyPair p = do+ (ek, dk, _) <- generateKeyPairAndSeed p+ return (ek, dk)++-- | Generate a key pair and hand back the seed it was derived from, @d@+-- and @z@ of FIPS 203 one after the other.+--+-- A 'DecapsulationKey' is the expanded key and nothing else, so the seed+-- cannot be recovered from a pair afterwards. An application that has to+-- write the key out in a form that keeps the seed has to generate it here:+--+-- > (ek, dk, seed) <- generateKeyPairAndSeed MLKEM768+--+-- The seed is as secret as the decapsulation key: 'keyPairFromSeed' turns+-- it back into the same pair.+generateKeyPairAndSeed+ :: forall p proxy m+ . (MLKEM p, MonadRandom m)+ => proxy p+ -> m (MLKEMEncapsulationKey p, DecapsulationKey p, ScrubbedBytes)+generateKeyPairAndSeed p = do+ seed <- getRandomBytes seedSize :: m ScrubbedBytes+ case keyPairFromSeed p seed of+ CryptoPassed (ek, dk) -> return (ek, dk, seed)+ CryptoFailed e ->+ error ("Crypto.PubKey.MLKEM.generateKeyPairAndSeed: " ++ show e)++-- | Derive a key pair from a seed, which is @d@ and @z@ of FIPS 203 one+-- after the other and must be 'seedSize' bytes.+--+-- This is the entry point to use when the seed comes from somewhere+-- particular -- a test vector, or a store that keeps seeds rather than+-- expanded keys. For an ordinary key, 'generateKeyPair' draws the seed+-- itself.+keyPairFromSeed+ :: forall p proxy ba+ . (MLKEM p, ByteArrayAccess ba)+ => proxy p+ -> ba+ -> CryptoFailable (MLKEMEncapsulationKey p, DecapsulationKey p)+keyPairFromSeed p seed+ | B.length seed /= seedSize = CryptoFailed CryptoError_SeedSizeInvalid+ | otherwise = unsafeDoIO $ do+ -- Not zeroed, and does not need to be: the C writes the whole+ -- buffer, and on a non-zero return the result is discarded without+ -- being read. Anything that is *read* before being written has to+ -- use B.zero instead -- see signInternal in Crypto.PubKey.MLDSA.+ dk <- B.alloc (decapsulationKeySize p) (\_ -> return ()) :: IO ScrubbedBytes+ (r, ek) <- B.allocRet (encapsulationKeySize p) $ \pek ->+ withByteArray dk $ \pdk ->+ withByteArray seed $ \pseed ->+ c_keypair p pek pdk pseed+ return $+ if r == 0+ then CryptoPassed (MLKEMEncapsulationKey ek, MLKEMDecapsulationKey dk)+ else CryptoFailed CryptoError_ParameterInvalid+{-# NOINLINE keyPairFromSeed #-}++-- | Encapsulate against a public key, drawing the randomness.+mlkemEncapsulate+ :: forall p m+ . (MLKEM p, MonadRandom m)+ => MLKEMEncapsulationKey p+ -> m (CryptoFailable (Ciphertext p, SharedSecret))+mlkemEncapsulate ek = do+ coins <- getRandomBytes encapsulationCoinsSize :: m ScrubbedBytes+ return (mlkemEncapsulateWith ek coins)++-- The class's 'encapsulateWith' for ML-KEM, where the coins are @m@ of+-- FIPS 203 and must be 'encapsulationCoinsSize' bytes.+mlkemEncapsulateWith+ :: forall p+ . MLKEM p+ => MLKEMEncapsulationKey p+ -> ScrubbedBytes+ -> CryptoFailable (Ciphertext p, SharedSecret)+mlkemEncapsulateWith ek coins+ | B.length coins /= encapsulationCoinsSize = CryptoFailed CryptoError_SeedSizeInvalid+ | otherwise = unsafeDoIO $ do+ ss <- B.alloc sharedSecretSize (\_ -> return ()) :: IO ScrubbedBytes+ (r, ct) <- B.allocRet (ciphertextSize p) $ \pct ->+ withByteArray ss $ \pss ->+ withByteArray ek $ \pek ->+ withByteArray coins $ \pcoins ->+ c_enc p pct pss pek pcoins+ return $+ if r == 0+ then CryptoPassed (MLKEMCiphertext ct, SharedSecret ss)+ else CryptoFailed CryptoError_ParameterInvalid+ where+ p = Proxy :: Proxy p+{-# NOINLINE mlkemEncapsulateWith #-}++-- | Recover the shared secret from a ciphertext.+--+-- A ciphertext that was not produced by encapsulating against the matching+-- key is not an error. ML-KEM rejects implicitly: it yields a secret+-- derived from the private key and the ciphertext, and the caller cannot+-- tell that case from the other one, which is the point -- telling them+-- apart is what a chosen-ciphertext attack needs. A ciphertext that does+-- not belong here shows up later, as the two sides failing to agree on+-- anything.+--+-- The checks FIPS 203 does require are at the point where bytes become a+-- value of these types, which is where they can be reported:+--+-- * The ciphertext type check of section 7.3 is its length, and+-- 'ciphertext' is the only way to build a 'Ciphertext' from bytes. There+-- is nothing else to check: a ciphertext's coefficients are compressed to+-- fewer than twelve bits, so every bit pattern decodes to a value in+-- range.+-- * The hash check of section 7.3 is on the decapsulation key, and+-- 'decapsulationKey' runs it; a key from 'generateKeyPair' or+-- 'keyPairFromSeed' satisfies it by construction.+--+-- So the result is 'CryptoPassed' for every key and ciphertext this module+-- can produce. It is 'CryptoFailable' rather than a bare 'SharedSecret'+-- because the implementation checks the key again on its way through, and+-- what it finds is better reported than turned into an exception.+mlkemDecapsulate+ :: forall p+ . MLKEM p+ => DecapsulationKey p -> Ciphertext p -> CryptoFailable SharedSecret+mlkemDecapsulate dk ct = unsafeDoIO $ do+ (r, ss) <- B.allocRet sharedSecretSize $ \pss ->+ withByteArray ct $ \pct ->+ withByteArray dk $ \pdk ->+ c_dec (Proxy :: Proxy p) pss pct pdk+ return $+ if r == (0 :: CInt)+ then CryptoPassed (SharedSecret ss)+ else CryptoFailed CryptoError_SecretKeyStructureInvalid+{-# NOINLINE mlkemDecapsulate #-}++-- The class's view of the three sets. The operations are the ones above;+-- only the shape of the arguments differs, because the class takes the+-- mechanism as a proxy.+instance KEM MLKEM512 where+ type EncapsulationKey MLKEM512 = MLKEMEncapsulationKey MLKEM512+ type DecapsulationKey MLKEM512 = MLKEMDecapsulationKey MLKEM512+ type Ciphertext MLKEM512 = MLKEMCiphertext MLKEM512+ type Coins MLKEM512 = ScrubbedBytes+ generateKeyPair = mlkemGenerateKeyPair+ encapsulate _ = mlkemEncapsulate+ encapsulateWith _ = mlkemEncapsulateWith+ decapsulate _ = mlkemDecapsulate++instance KEM MLKEM768 where+ type EncapsulationKey MLKEM768 = MLKEMEncapsulationKey MLKEM768+ type DecapsulationKey MLKEM768 = MLKEMDecapsulationKey MLKEM768+ type Ciphertext MLKEM768 = MLKEMCiphertext MLKEM768+ type Coins MLKEM768 = ScrubbedBytes+ generateKeyPair = mlkemGenerateKeyPair+ encapsulate _ = mlkemEncapsulate+ encapsulateWith _ = mlkemEncapsulateWith+ decapsulate _ = mlkemDecapsulate++instance KEM MLKEM1024 where+ type EncapsulationKey MLKEM1024 = MLKEMEncapsulationKey MLKEM1024+ type DecapsulationKey MLKEM1024 = MLKEMDecapsulationKey MLKEM1024+ type Ciphertext MLKEM1024 = MLKEMCiphertext MLKEM1024+ type Coins MLKEM1024 = ScrubbedBytes+ generateKeyPair = mlkemGenerateKeyPair+ encapsulate _ = mlkemEncapsulate+ encapsulateWith _ = mlkemEncapsulateWith+ decapsulate _ = mlkemDecapsulate++instance MLKEM MLKEM512 where+ encapsulationKeySize _ = 800+ decapsulationKeySize _ = 1632+ ciphertextSize _ = 768+ c_keypair _ = c_mlkem512_keypair+ c_enc _ = c_mlkem512_enc+ c_dec _ = c_mlkem512_dec+ c_checkPk _ = c_mlkem512_check_pk+ c_checkSk _ = c_mlkem512_check_sk++instance MLKEM MLKEM768 where+ encapsulationKeySize _ = 1184+ decapsulationKeySize _ = 2400+ ciphertextSize _ = 1088+ c_keypair _ = c_mlkem768_keypair+ c_enc _ = c_mlkem768_enc+ c_dec _ = c_mlkem768_dec+ c_checkPk _ = c_mlkem768_check_pk+ c_checkSk _ = c_mlkem768_check_sk++instance MLKEM MLKEM1024 where+ encapsulationKeySize _ = 1568+ decapsulationKeySize _ = 3168+ ciphertextSize _ = 1568+ c_keypair _ = c_mlkem1024_keypair+ c_enc _ = c_mlkem1024_enc+ c_dec _ = c_mlkem1024_dec+ c_checkPk _ = c_mlkem1024_check_pk+ c_checkSk _ = c_mlkem1024_check_sk++foreign import ccall unsafe "crypton_mlkem512_keypair_derand"+ c_mlkem512_keypair :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem512_enc_derand"+ c_mlkem512_enc :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem512_dec"+ c_mlkem512_dec :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem512_check_pk"+ c_mlkem512_check_pk :: Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem512_check_sk"+ c_mlkem512_check_sk :: Ptr Word8 -> IO CInt++foreign import ccall unsafe "crypton_mlkem768_keypair_derand"+ c_mlkem768_keypair :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem768_enc_derand"+ c_mlkem768_enc :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem768_dec"+ c_mlkem768_dec :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem768_check_pk"+ c_mlkem768_check_pk :: Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem768_check_sk"+ c_mlkem768_check_sk :: Ptr Word8 -> IO CInt++foreign import ccall unsafe "crypton_mlkem1024_keypair_derand"+ c_mlkem1024_keypair :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem1024_enc_derand"+ c_mlkem1024_enc :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem1024_dec"+ c_mlkem1024_dec :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem1024_check_pk"+ c_mlkem1024_check_pk :: Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem1024_check_sk"+ c_mlkem1024_check_sk :: Ptr Word8 -> IO CInt
Crypto/PubKey/RSA/OAEP.hs view
@@ -31,9 +31,8 @@ import Data.Bits (complement, shiftR, xor, (.&.), (.|.)) import Data.ByteString (ByteString) import qualified Data.ByteString as B-import Data.List (foldl')+import qualified Data.List as L import Data.Word (Word32)-import Prelude hiding (foldl') import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess) import qualified Crypto.Internal.ByteArray as B (constEq, convert)@@ -156,7 +155,7 @@ -- is zero; all of them are looked at either way oneIndex = fst $- foldl'+ L.foldl' step (fromIntegral (B.length db1) :: Word32, 1 :: Word32) (zip [0 ..] (B.unpack db1))
Crypto/PubKey/RSA/PKCS15.hs view
@@ -38,8 +38,7 @@ import Crypto.Internal.ByteArray (ByteArray, Bytes) import qualified Crypto.Internal.ByteArray as B-import Data.List (foldl')-import Prelude hiding (foldl')+import qualified Data.List as L -- | A specialized class for hash algorithm that can product -- a ASN1 wrapped description the algorithm plus the content@@ -431,7 +430,7 @@ -- index of the first zero octet in ps0m, counted from the start of packed, -- or len when there is none; every octet is looked at either way- zeroIndex = fst $ foldl' step (fromIntegral len :: Word32, 1 :: Word32) indexed+ zeroIndex = fst $ L.foldl' step (fromIntegral len :: Word32, 1 :: Word32) indexed indexed = zip [2 ..] (B.unpack ps0m) step (idx, unseen) (i, b) = (select found i idx, unseen .&. complement found) where
Crypto/PubKey/Rabin/OAEP.hs view
@@ -17,9 +17,8 @@ import Data.Bits (complement, shiftR, xor, (.&.), (.|.)) import Data.ByteString (ByteString) import qualified Data.ByteString as B-import Data.List (foldl')+import qualified Data.List as L import Data.Word (Word32)-import Prelude hiding (foldl') import Crypto.Hash import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess)@@ -127,7 +126,7 @@ -- is zero; all of them are looked at either way oneIndex = fst $- foldl'+ L.foldl' step (fromIntegral (B.length db1) :: Word32, 1 :: Word32) (zip [0 ..] (B.unpack db1))
Crypto/Random/Types.hs view
@@ -15,6 +15,18 @@ import Crypto.Random.Entropy -- | A monad constraint that allows to generate random bytes+--+-- Everything in this library that draws a key, a nonce or a signature's+-- randomness draws it through this class, and cannot tell a strong source+-- from a weak one. The two instances below are the ones to reach for:+-- @IO@ reads the system entropy source, and @MonadPseudoRandom@ runs a+-- 'DRG' seeded from it.+--+-- An instance of your own is held to the same standard. A generator that+-- another party can predict, or that repeats, yields keys and signatures+-- that give away what they are meant to keep -- so the deliberately+-- repeatable instance that makes a test reproducible is not one to ship+-- with. class Monad m => MonadRandom m where getRandomBytes :: ByteArray byteArray => Int -> m byteArray
cbits/crypton_cpu.c view
@@ -32,6 +32,23 @@ #include <stdint.h> /*+ * PE has no way to say "hidden": every symbol in an object is local to the+ * image unless something exports it, which is what hidden asks for+ * elsewhere, so nothing is lost by dropping the attribute here. Saying it+ * anyway is not harmless -- the gcc that GHC 9.2 ships for Windows parses+ * the attribute, discards it and warns, and it is the only warning crypton's+ * own code produces anywhere in the CI matrix. Measured on mingw gcc 13.2.0+ * and clang 14.0.6 (the compiler GHC 9.4 and later ship): gcc warns for+ * "hidden" and is silent for "default", clang is silent for both, which is+ * why the vendored decaf and argon2 headers ask for "default" unnoticed.+ */+#if defined(_WIN32) || defined(__CYGWIN__)+#define CRYPTON_HIDDEN+#else+#define CRYPTON_HIDDEN __attribute__((visibility("hidden")))+#endif++/* * The word the assembly reads; crypton_cpu.h says what is in it. Hidden, * so that the reference to it from the assembly resolves at link time in a * shared object as well as a static one. The SHA-256 bit is set by@@ -39,7 +56,7 @@ * instructions. */ #ifdef CRYPTON_ARM_ASM-__attribute__((visibility("hidden"))) unsigned int crypton_armcap_P =+CRYPTON_HIDDEN unsigned int crypton_armcap_P = CRYPTON_ARMCAP_NEON; #endif @@ -101,7 +118,7 @@ } #ifdef CRYPTON_X86_ASM-__attribute__((visibility("hidden"))) unsigned int crypton_ia32cap_P[4];+CRYPTON_HIDDEN unsigned int crypton_ia32cap_P[4]; /* * The AVX-512 bits of leaf 7 EBX -- F, DQ, IFMA, PF, ER, CD, BW and VL,
+ cbits/mldsa/COMMIT view
@@ -0,0 +1,2 @@+834a90d5e846ffa1e1611bd24e160bb2e9b86d35+v2.0.0
+ cbits/mldsa/LICENSE view
@@ -0,0 +1,305 @@+mldsa-native is a fork of the public domain Dilithium reference implementation,+available on https://github.com/pq-crystals/dilithium.++All new files and all files derived from the Dilithium reference+implementation are made available under the Apache-2.0 license OR+the ISC license OR the MIT license. These licenses are+reproduced at the bottom of this file.++Files outside the library itself may carry different terms. Every file+states its own SPDX-License-Identifier, which determines the terms that+apply to it. In particular:++The code in test/notrandombytes/*, and its copies in+examples/*/test_only_rng/* and scripts/notrandombytes, is derived from+https://cr.yp.to/papers.html#surf and licensed under+LicenseRef-PD-hp OR CC0-1.0 OR 0BSD OR MIT-0 OR MIT.+It is only used for testing purposes.++The benchmarking code in test/hal/* carries the+MIT license. It is only used for testing purposes.++The tiny_sha3 code in+examples/bring_your_own_fips202/custom_fips202/tiny_sha3/* and+examples/custom_backend/mldsa_native/src/fips202/native/custom/src/*+carries the MIT license. It is only used to demonstrate custom+FIPS-202 implementations.++The proofs in proofs/* are in part derived from Amazon Web Services+verification infrastructure and licensed under+Apache-2.0 OR ISC OR MIT-0, MIT-0, or MIT-0 AND Apache-2.0. The IACR+document class proofs/isabelle/neon_ntt/document/iacrtrans.cls is+licensed under CC0-1.0. None of this is part of the library.++Documentation is licensed under CC-BY-4.0.++```+Copyright (c) The mldsa-native project authors+Copyright (c) The mlkem-native project authors+Copyright (c) 2020 Dougall Johnson+Copyright (c) 2022 Arm Limited+SPDX-License-Identifier: MIT++Permission is hereby granted, free of charge, to any person obtaining a copy+of this software and associated documentation files (the "Software"), to deal+in the Software without restriction, including without limitation the rights+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell+copies of the Software, and to permit persons to whom the Software is+furnished to do so, subject to the following conditions:++The above copyright notice and this permission notice shall be included in+all copies or substantial portions of the Software.++THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE+SOFTWARE.+```++ISC license for mldsa-native content+------------------------------------++Copyright (c) The mldsa-native project authors++Permission to use, copy, modify, and/or distribute this software for any purpose+with or without fee is hereby granted, provided that the above copyright notice+and this permission notice appear in all copies.++THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH+REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY AND+FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY SPECIAL, DIRECT,+INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS+OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER+TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF+THIS SOFTWARE.+++MIT license for mldsa-native content+------------------------------------++Copyright (c) The mldsa-native project authors++Permission is hereby granted, free of charge, to any person obtaining a copy of+this software and associated documentation files (the “Software”), to deal in+the Software without restriction, including without limitation the rights to+use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of+the Software, and to permit persons to whom the Software is furnished to do so,+subject to the following conditions:++The above copyright notice and this permission notice shall be included in all+copies or substantial portions of the Software.++THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS+FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR+COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER+IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN+CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.++Apache-2.0 license for mldsa-native content+-------------------------------------------++ Apache License+ Version 2.0, January 2004+ http://www.apache.org/licenses/++ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION++ 1. Definitions.++ "License" shall mean the terms and conditions for use, reproduction,+ and distribution as defined by Sections 1 through 9 of this document.++ "Licensor" shall mean the copyright owner or entity authorized by+ the copyright owner that is granting the License.++ "Legal Entity" shall mean the union of the acting entity and all+ other entities that control, are controlled by, or are under common+ control with that entity. For the purposes of this definition,+ "control" means (i) the power, direct or indirect, to cause the+ direction or management of such entity, whether by contract or+ otherwise, or (ii) ownership of fifty percent (50%) or more of the+ outstanding shares, or (iii) beneficial ownership of such entity.++ "You" (or "Your") shall mean an individual or Legal Entity+ exercising permissions granted by this License.++ "Source" form shall mean the preferred form for making modifications,+ including but not limited to software source code, documentation+ source, and configuration files.++ "Object" form shall mean any form resulting from mechanical+ transformation or translation of a Source form, including but+ not limited to compiled object code, generated documentation,+ and conversions to other media types.++ "Work" shall mean the work of authorship, whether in Source or+ Object form, made available under the License, as indicated by a+ copyright notice that is included in or attached to the work+ (an example is provided in the Appendix below).++ "Derivative Works" shall mean any work, whether in Source or Object+ form, that is based on (or derived from) the Work and for which the+ editorial revisions, annotations, elaborations, or other modifications+ represent, as a whole, an original work of authorship. For the purposes+ of this License, Derivative Works shall not include works that remain+ separable from, or merely link (or bind by name) to the interfaces of,+ the Work and Derivative Works thereof.++ "Contribution" shall mean any work of authorship, including+ the original version of the Work and any modifications or additions+ to that Work or Derivative Works thereof, that is intentionally+ submitted to Licensor for inclusion in the Work by the copyright owner+ or by an individual or Legal Entity authorized to submit on behalf of+ the copyright owner. For the purposes of this definition, "submitted"+ means any form of electronic, verbal, or written communication sent+ to the Licensor or its representatives, including but not limited to+ communication on electronic mailing lists, source code control systems,+ and issue tracking systems that are managed by, or on behalf of, the+ Licensor for the purpose of discussing and improving the Work, but+ excluding communication that is conspicuously marked or otherwise+ designated in writing by the copyright owner as "Not a Contribution."++ "Contributor" shall mean Licensor and any individual or Legal Entity+ on behalf of whom a Contribution has been received by Licensor and+ subsequently incorporated within the Work.++ 2. Grant of Copyright License. Subject to the terms and conditions of+ this License, each Contributor hereby grants to You a perpetual,+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable+ copyright license to reproduce, prepare Derivative Works of,+ publicly display, publicly perform, sublicense, and distribute the+ Work and such Derivative Works in Source or Object form.++ 3. Grant of Patent License. Subject to the terms and conditions of+ this License, each Contributor hereby grants to You a perpetual,+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable+ (except as stated in this section) patent license to make, have made,+ use, offer to sell, sell, import, and otherwise transfer the Work,+ where such license applies only to those patent claims licensable+ by such Contributor that are necessarily infringed by their+ Contribution(s) alone or by combination of their Contribution(s)+ with the Work to which such Contribution(s) was submitted. If You+ institute patent litigation against any entity (including a+ cross-claim or counterclaim in a lawsuit) alleging that the Work+ or a Contribution incorporated within the Work constitutes direct+ or contributory patent infringement, then any patent licenses+ granted to You under this License for that Work shall terminate+ as of the date such litigation is filed.++ 4. Redistribution. You may reproduce and distribute copies of the+ Work or Derivative Works thereof in any medium, with or without+ modifications, and in Source or Object form, provided that You+ meet the following conditions:++ (a) You must give any other recipients of the Work or+ Derivative Works a copy of this License; and++ (b) You must cause any modified files to carry prominent notices+ stating that You changed the files; and++ (c) You must retain, in the Source form of any Derivative Works+ that You distribute, all copyright, patent, trademark, and+ attribution notices from the Source form of the Work,+ excluding those notices that do not pertain to any part of+ the Derivative Works; and++ (d) If the Work includes a "NOTICE" text file as part of its+ distribution, then any Derivative Works that You distribute must+ include a readable copy of the attribution notices contained+ within such NOTICE file, excluding those notices that do not+ pertain to any part of the Derivative Works, in at least one+ of the following places: within a NOTICE text file distributed+ as part of the Derivative Works; within the Source form or+ documentation, if provided along with the Derivative Works; or,+ within a display generated by the Derivative Works, if and+ wherever such third-party notices normally appear. The contents+ of the NOTICE file are for informational purposes only and+ do not modify the License. You may add Your own attribution+ notices within Derivative Works that You distribute, alongside+ or as an addendum to the NOTICE text from the Work, provided+ that such additional attribution notices cannot be construed+ as modifying the License.++ You may add Your own copyright statement to Your modifications and+ may provide additional or different license terms and conditions+ for use, reproduction, or distribution of Your modifications, or+ for any such Derivative Works as a whole, provided Your use,+ reproduction, and distribution of the Work otherwise complies with+ the conditions stated in this License.++ 5. Submission of Contributions. Unless You explicitly state otherwise,+ any Contribution intentionally submitted for inclusion in the Work+ by You to the Licensor shall be under the terms and conditions of+ this License, without any additional terms or conditions.+ Notwithstanding the above, nothing herein shall supersede or modify+ the terms of any separate license agreement you may have executed+ with Licensor regarding such Contributions.++ 6. Trademarks. This License does not grant permission to use the trade+ names, trademarks, service marks, or product names of the Licensor,+ except as required for reasonable and customary use in describing the+ origin of the Work and reproducing the content of the NOTICE file.++ 7. Disclaimer of Warranty. Unless required by applicable law or+ agreed to in writing, Licensor provides the Work (and each+ Contributor provides its Contributions) on an "AS IS" BASIS,+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or+ implied, including, without limitation, any warranties or conditions+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A+ PARTICULAR PURPOSE. You are solely responsible for determining the+ appropriateness of using or redistributing the Work and assume any+ risks associated with Your exercise of permissions under this License.++ 8. Limitation of Liability. In no event and under no legal theory,+ whether in tort (including negligence), contract, or otherwise,+ unless required by applicable law (such as deliberate and grossly+ negligent acts) or agreed to in writing, shall any Contributor be+ liable to You for damages, including any direct, indirect, special,+ incidental, or consequential damages of any character arising as a+ result of this License or out of the use or inability to use the+ Work (including but not limited to damages for loss of goodwill,+ work stoppage, computer failure or malfunction, or any and all+ other commercial damages or losses), even if such Contributor+ has been advised of the possibility of such damages.++ 9. Accepting Warranty or Additional Liability. While redistributing+ the Work or Derivative Works thereof, You may choose to offer,+ and charge a fee for, acceptance of support, warranty, indemnity,+ or other liability obligations and/or rights consistent with this+ License. However, in accepting such obligations, You may act only+ on Your own behalf and on Your sole responsibility, not on behalf+ of any other Contributor, and only if You agree to indemnify,+ defend, and hold each Contributor harmless for any liability+ incurred by, or claims asserted against, such Contributor by reason+ of your accepting any such warranty or additional liability.++ END OF TERMS AND CONDITIONS++ APPENDIX: How to apply the Apache License to your work.++ To apply the Apache License to your work, attach the following+ boilerplate notice, with the fields enclosed by brackets "[]"+ replaced with your own identifying information. (Don't include+ the brackets!) The text should be enclosed in the appropriate+ comment syntax for the file format. We also recommend that a+ file or class name and description of purpose be included on the+ same "printed page" as the copyright notice for easier+ identification within third-party archives.++ Copyright (c) The mldsa-native project authors++ Licensed under the Apache License, Version 2.0 (the "License");+ you may not use this file except in compliance with the License.+ You may obtain a copy of the License at++ http://www.apache.org/licenses/LICENSE-2.0++ Unless required by applicable law or agreed to in writing, software+ distributed under the License is distributed on an "AS IS" BASIS,+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.+ See the License for the specific language governing permissions and+ limitations under the License.
+ cbits/mldsa/README.md view
@@ -0,0 +1,60 @@+# mldsa-native++ML-DSA (FIPS 204) from the PQ Code Package, vendored here and built for all+three parameter sets.++## Where it comes from++<https://github.com/pq-code-package/mldsa-native>. `COMMIT` holds the+revision this tree is at; `import.sh` puts it there and is how the tree is+refreshed. Nothing here is edited by hand -- every choice crypton makes is+made in `crypton_mldsa.h`, `crypton_mldsa.c` and `crypton.cabal`, so that a+re-import is a straight overwrite.++## Licence++`Apache-2.0 OR ISC OR MIT`, the same three-way form as `cbits/s2n` and by+some of the same authors. crypton takes it under ISC, which is already in+the package's `license:` field. `LICENSE` is the upstream file and is+listed in `license-files:`.++## What is taken and what is not++The whole of `mldsa/`, less the backends for architectures crypton does not+build for: the 32-bit+`src/fips202/native/armv81m`. (Unlike mlkem-native, this one ships only the+AArch64 and x86-64 arithmetic backends, so there is nothing else to drop.) Every reference to those is behind an+`MLD_SYS_` guard that cannot be true on the architectures crypton does+build for, so dropping them changes no build and keeps a few dozen files of+unreachable assembly out of the release tarball. To take one back, delete+its line from `import.sh` and add its directory to `extra-source-files:`.++## How it is built++Upstream builds for one parameter set at a time. `crypton_mldsa.c`+includes the amalgamation once per set -- the level-independent half kept by+exactly one of them -- which is how a single crypton offers ML-DSA-44, 65+and 87. `crypton_mldsa_asm.S` does the same for the assembly, which is+level-independent and so is included once.++This is the shape `cbits/aes/armv8.c` already uses for the three AES key+sizes, and it has the same hazard: **cabal does not know that the wrapper+depends on the tree it includes.** After changing anything under+`cbits/mldsa`, touch `crypton_mldsa.c`, or the build keeps the object it+already has and the change is not tested.++The hand-written backends are selected by `CRYPTON_MLDSA_NATIVE_BACKEND`,+which `crypton.cabal` defines on x86-64 and AArch64 other than Windows --+the same exclusion, and for the same reasons, as `cbits/s2n`. Everywhere+else the portable C is built, which is the same code and passes the same+tests.++There is no randomised API: `MLD_CONFIG_NO_RANDOMIZED_API` is set, no+`randombytes()` is needed, and randomness is drawn in Haskell through+`MonadRandom` as it is for every other key crypton generates.++The symbols are `crypton_mldsa44_*`, `crypton_mldsa65_*` and+`crypton_mldsa87_*` rather than upstream's defaults, so that an+application linking another copy of mldsa-native -- through some other+library, or its own -- does not present the linker with two sets of+functions answering to one set of names.
+ cbits/mldsa/crypton_mldsa.c view
@@ -0,0 +1,31 @@+/*+ * All three ML-DSA parameter sets in one translation unit.+ *+ * mldsa-native is built for one parameter set at a time; a build wanting+ * several includes the amalgamation once per set, with the level-independent+ * half kept by exactly one of them. This is the same shape as+ * cbits/aes/armv8.c, which includes cbits/aes/armv8_impl.c three times for+ * the three AES key sizes.+ *+ * NOTE, as there: cabal does not know that this file depends on the tree it+ * includes. After changing anything under cbits/mldsa, touch this file, or+ * the build will quietly keep the object it already has.+ */+#include "crypton_mldsa.h"++#define MLD_CONFIG_MULTILEVEL_WITH_SHARED 1+#define MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+#define MLD_CONFIG_PARAMETER_SET 44+#include "mldsa_native.c"+#undef MLD_CONFIG_MULTILEVEL_WITH_SHARED+#undef MLD_CONFIG_PARAMETER_SET++#define MLD_CONFIG_MULTILEVEL_NO_SHARED+#define MLD_CONFIG_PARAMETER_SET 65+#include "mldsa_native.c"+#undef MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+#undef MLD_CONFIG_PARAMETER_SET++#define MLD_CONFIG_PARAMETER_SET 87+#include "mldsa_native.c"+#undef MLD_CONFIG_PARAMETER_SET
+ cbits/mldsa/crypton_mldsa.h view
@@ -0,0 +1,41 @@+/*+ * What crypton asks of the vendored mldsa-native, in one place. The tree+ * under cbits/mldsa is upstream's and is overwritten by import.sh, so every+ * choice crypton makes is made here instead of by editing it.+ *+ * Included first by both crypton_mldsa.c and crypton_mldsa_asm.S, so it must+ * hold nothing but preprocessor directives.+ */+#ifndef CRYPTON_MLDSA_H+#define CRYPTON_MLDSA_H++/*+ * The symbols are crypton's own, not the default PQCP_MLDSA_NATIVE_*. An+ * application is free to link another copy of mldsa-native -- through some+ * other library, or its own -- and two copies answering to one set of names+ * is a problem the linker resolves silently and in nobody's favour. With+ * MLD_CONFIG_MULTILEVEL_BUILD the level is appended, so the entry points+ * are crypton_mldsa44_*, crypton_mldsa65_* and crypton_mldsa87_*.+ */+#define MLD_CONFIG_NAMESPACE_PREFIX crypton_mldsa+#define MLD_CONFIG_MULTILEVEL_BUILD++/*+ * No randomised API, so no randombytes() to provide. Randomness is drawn+ * in Haskell through MonadRandom, the way every other key in crypton is+ * generated, and the deterministic entry points are what the FFI calls.+ * That also keeps the C free of any opinion about where entropy comes from.+ */+#define MLD_CONFIG_NO_RANDOMIZED_API++/*+ * The hand-written backends, where crypton.cabal says the architecture has+ * them. Without this the portable C is built, which is correct everywhere+ * and is what every other architecture gets.+ */+#ifdef CRYPTON_MLDSA_NATIVE_BACKEND+#define MLD_CONFIG_USE_NATIVE_BACKEND_ARITH+#define MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202+#endif++#endif /* CRYPTON_MLDSA_H */
+ cbits/mldsa/crypton_mldsa_asm.S view
@@ -0,0 +1,14 @@+/*+ * The assembly half, which is level-independent: it is included once, with+ * the shared directives kept, and covers all three parameter sets. See the+ * comment at the top of mldsa_native_asm.S.+ *+ * Built only where crypton.cabal defines CRYPTON_MLDSA_NATIVE_BACKEND; on+ * every other architecture this file is not in asm-sources at all.+ */+#include "crypton_mldsa.h"++#define MLD_CONFIG_MULTILEVEL_WITH_SHARED 1+#define MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+#define MLD_CONFIG_PARAMETER_SET 65+#include "mldsa_native_asm.S"
+ cbits/mldsa/import.sh view
@@ -0,0 +1,40 @@+#!/bin/sh+# Re-import the vendored parts of the PQ Code Package's mldsa-native.+#+# The files are kept unmodified. Everything crypton decides -- which+# parameter sets exist, what the symbols are called, that there is no+# randomised API -- is decided in cbits/mldsa/crypton_mldsa.c and in+# crypton.cabal, not by editing anything here. Run this from cbits/mldsa:+#+# ./import.sh [tag-or-commit]+#+# and commit the result together with the COMMIT line it writes, so that the+# tree always says which upstream revision it holds.+set -eu++REPO=https://github.com/pq-code-package/mldsa-native+REV=${1:-v2.0.0}+HERE=$(cd "$(dirname "$0")" && pwd)+TMP=$(mktemp -d)+trap 'rm -rf "$TMP"' EXIT++git clone -q "$REPO" "$TMP/u"+git -C "$TMP/u" checkout -q "$REV"++rm -rf "$HERE/src"+cp "$TMP/u/mldsa/mldsa_native.c" "$HERE/"+cp "$TMP/u/mldsa/mldsa_native.h" "$HERE/"+cp "$TMP/u/mldsa/mldsa_native_asm.S" "$HERE/"+cp "$TMP/u/mldsa/mldsa_native_config.h" "$HERE/"+cp -R "$TMP/u/mldsa/src" "$HERE/src"+cp "$TMP/u/LICENSE" "$HERE/LICENSE"++# The 32-bit Armv8.1-M Keccak, for an architecture crypton does not build+# for. Every reference to it is behind an MLD_SYS_ guard that cannot be true+# on the ones it does, so dropping it changes no build and keeps unreachable+# assembly out of the release tarball.+rm -rf "$HERE/src/fips202/native/armv81m"++git -C "$TMP/u" rev-parse HEAD > "$HERE/COMMIT"+git -C "$TMP/u" describe --tags --exact-match 2>/dev/null >> "$HERE/COMMIT" || true+echo "imported $(head -1 "$HERE/COMMIT")"
+ cbits/mldsa/mldsa_native.c view
@@ -0,0 +1,803 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++/******************************************************************************+ *+ * Single compilation unit (SCU) for fixed-level build of mldsa-native+ *+ * This compilation unit bundles together all source files for a build+ * of mldsa-native for a fixed security level (MLDSA-44/65/87).+ *+ * # API+ *+ * The API exposed by this file is described in mldsa_native.h.+ *+ * # Multi-level build+ *+ * If you want an SCU build of mldsa-native with support for multiple security+ * levels, you need to include this file multiple times, and set+ * MLD_CONFIG_MULTILEVEL_WITH_SHARED and MLD_CONFIG_MULTILEVEL_NO_SHARED+ * appropriately. This is exemplified in examples/monolithic_build_multilevel+ * and examples/monolithic_build_multilevel_native.+ *+ * # Configuration+ *+ * The following options from the mldsa-native configuration are relevant:+ *+ * - MLD_CONFIG_FIPS202_CUSTOM_HEADER+ * Set this option if you use a custom FIPS202 implementation.+ *+ * - MLD_CONFIG_USE_NATIVE_BACKEND_ARITH+ * Set this option if you want to include the native arithmetic backends+ * in your build.+ *+ * - MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202+ * Set this option if you want to include the native FIPS202 backends+ * in your build.+ *+ * - MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+ * Set this option if you want to keep the directives defined in+ * level-independent headers. This is needed for a multi-level build.+ */++/* If parts of the mldsa-native source tree are not used,+ * consider reducing this header via `unifdef`.+ *+ * Example:+ * ```bash+ * unifdef -UMLD_CONFIG_USE_NATIVE_BACKEND_ARITH mldsa_native.c+ * ```+ */++#include "src/common.h"++#include "src/ct.c"+#include "src/debug.c"+#include "src/packing.c"+#include "src/poly.c"+#include "src/poly_kl.c"+#include "src/polyvec.c"+#include "src/polyvec_lazy.c"+#include "src/sign.c"++#if !defined(MLD_CONFIG_FIPS202_CUSTOM_HEADER)+#include "src/fips202/fips202.c"+#include "src/fips202/fips202x4.c"+#include "src/fips202/keccakf1600.c"+#endif++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH)+#if defined(MLD_SYS_AARCH64)+#include "src/native/aarch64/src/aarch64_zetas.c"+#include "src/native/aarch64/src/polyz_unpack_table.c"+#include "src/native/aarch64/src/rej_uniform_eta_table.c"+#include "src/native/aarch64/src/rej_uniform_table.c"+#endif /* MLD_SYS_AARCH64 */+#if defined(MLD_SYS_X86_64)+#include "src/native/x86_64/src/consts.c"+#include "src/native/x86_64/src/rej_uniform_table.c"+#endif+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_ARITH */++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202)+#if defined(MLD_SYS_AARCH64)+#include "src/fips202/native/aarch64/src/keccakf1600_round_constants.c"+#endif+#if defined(MLD_SYS_X86_64)+#include "src/fips202/native/x86_64/src/keccakf1600_constants.c"+#endif+#if defined(MLD_SYS_ARMV81M_MVE)+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_mve.c"+#include "src/fips202/native/armv81m/src/keccakf1600_round_constants.c"+#endif+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 */++/* Macro #undef's+ *+ * The following undefines macros from headers+ * included by the source files imported above.+ *+ * This is to allow building and linking multiple builds+ * of mldsa-native for varying parameter sets through concatenation+ * of this file, as if the files had been compiled separately.+ * If this is not relevant to you, you may remove the following.+ */++/*+ * Undefine macros from MLD_CONFIG_PARAMETER_SET-specific files+ */+/* mldsa/mldsa_native.h */+#undef MLDSA44_BYTES+#undef MLDSA44_CRHBYTES+#undef MLDSA44_PUBLICKEYBYTES+#undef MLDSA44_RNDBYTES+#undef MLDSA44_SECRETKEYBYTES+#undef MLDSA44_SEEDBYTES+#undef MLDSA44_TRBYTES+#undef MLDSA65_BYTES+#undef MLDSA65_CRHBYTES+#undef MLDSA65_PUBLICKEYBYTES+#undef MLDSA65_RNDBYTES+#undef MLDSA65_SECRETKEYBYTES+#undef MLDSA65_SEEDBYTES+#undef MLDSA65_TRBYTES+#undef MLDSA87_BYTES+#undef MLDSA87_CRHBYTES+#undef MLDSA87_PUBLICKEYBYTES+#undef MLDSA87_RNDBYTES+#undef MLDSA87_SECRETKEYBYTES+#undef MLDSA87_SEEDBYTES+#undef MLDSA87_TRBYTES+#undef MLDSA_BYTES+#undef MLDSA_BYTES_+#undef MLDSA_CRHBYTES+#undef MLDSA_PUBLICKEYBYTES+#undef MLDSA_PUBLICKEYBYTES_+#undef MLDSA_RNDBYTES+#undef MLDSA_SECRETKEYBYTES+#undef MLDSA_SECRETKEYBYTES_+#undef MLDSA_SEEDBYTES+#undef MLDSA_TRBYTES+#undef MLD_API_CONCAT+#undef MLD_API_CONCAT_+#undef MLD_API_CONCAT_UNDERSCORE+#undef MLD_API_MUST_CHECK_RETURN_VALUE+#undef MLD_API_NAMESPACE+#undef MLD_API_NAMESPACE_PREFIX+#undef MLD_API_QUALIFIER+#undef MLD_DOMAIN_SEPARATION_MAX_BYTES+#undef MLD_ERR_FAIL+#undef MLD_ERR_INVALID_ARG+#undef MLD_ERR_INVALID_KEY+#undef MLD_ERR_INVALID_SIGNATURE+#undef MLD_ERR_OUT_OF_MEMORY+#undef MLD_ERR_PCT_FAIL+#undef MLD_ERR_RNG_FAIL+#undef MLD_ERR_SIGNING_PAUSED+#undef MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED+#undef MLD_H+#undef MLD_MAX3_+#undef MLD_MAX4_+#undef MLD_PREHASH_NONE+#undef MLD_PREHASH_SHA2_224+#undef MLD_PREHASH_SHA2_256+#undef MLD_PREHASH_SHA2_384+#undef MLD_PREHASH_SHA2_512+#undef MLD_PREHASH_SHA2_512_224+#undef MLD_PREHASH_SHA2_512_256+#undef MLD_PREHASH_SHA3_224+#undef MLD_PREHASH_SHA3_256+#undef MLD_PREHASH_SHA3_384+#undef MLD_PREHASH_SHA3_512+#undef MLD_PREHASH_SHAKE_128+#undef MLD_PREHASH_SHAKE_256+#undef MLD_TOTAL_ALLOC_44+#undef MLD_TOTAL_ALLOC_44_KEYPAIR+#undef MLD_TOTAL_ALLOC_44_KEYPAIR_NO_PCT+#undef MLD_TOTAL_ALLOC_44_KEYPAIR_PCT+#undef MLD_TOTAL_ALLOC_44_PK_FROM_SK+#undef MLD_TOTAL_ALLOC_44_SIGN+#undef MLD_TOTAL_ALLOC_44_VERIFY+#undef MLD_TOTAL_ALLOC_65+#undef MLD_TOTAL_ALLOC_65_KEYPAIR+#undef MLD_TOTAL_ALLOC_65_KEYPAIR_NO_PCT+#undef MLD_TOTAL_ALLOC_65_KEYPAIR_PCT+#undef MLD_TOTAL_ALLOC_65_PK_FROM_SK+#undef MLD_TOTAL_ALLOC_65_SIGN+#undef MLD_TOTAL_ALLOC_65_VERIFY+#undef MLD_TOTAL_ALLOC_87+#undef MLD_TOTAL_ALLOC_87_KEYPAIR+#undef MLD_TOTAL_ALLOC_87_KEYPAIR_NO_PCT+#undef MLD_TOTAL_ALLOC_87_KEYPAIR_PCT+#undef MLD_TOTAL_ALLOC_87_PK_FROM_SK+#undef MLD_TOTAL_ALLOC_87_SIGN+#undef MLD_TOTAL_ALLOC_87_VERIFY+/* mldsa/src/common.h */+#undef MLD_ADD_PARAM_SET+#undef MLD_ALLOC+#undef MLD_APPLY+#undef MLD_ASM_FN_SIZE+#undef MLD_ASM_FN_SYMBOL+#undef MLD_ASM_NAMESPACE+#undef MLD_BUILD_INTERNAL+#undef MLD_COMMON_H+#undef MLD_CONCAT+#undef MLD_CONCAT_+#undef MLD_EMPTY_CU+#undef MLD_ERR_FAIL+#undef MLD_ERR_INVALID_ARG+#undef MLD_ERR_INVALID_KEY+#undef MLD_ERR_INVALID_SIGNATURE+#undef MLD_ERR_OUT_OF_MEMORY+#undef MLD_ERR_PCT_FAIL+#undef MLD_ERR_RNG_FAIL+#undef MLD_ERR_SIGNING_PAUSED+#undef MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED+#undef MLD_EXTERNAL_API+#undef MLD_FIPS202X4_HEADER_FILE+#undef MLD_FIPS202_HEADER_FILE+#undef MLD_FREE+#undef MLD_INTERNAL_API+#undef MLD_INTERNAL_DATA_DECLARATION+#undef MLD_INTERNAL_DATA_DEFINITION+#undef MLD_MULTILEVEL_BUILD+#undef MLD_NAMESPACE+#undef MLD_NAMESPACE_KL+#undef MLD_NAMESPACE_PREFIX+#undef MLD_NAMESPACE_PREFIX_KL+#undef mld_memcpy+#undef mld_memset+/* mldsa/src/packing.h */+#undef MLD_PACKING_H+#undef mld_pack_sig_c+#undef mld_pack_sig_h+#undef mld_pack_sig_z+#undef mld_pack_sk_rho_key_tr_s2+#undef mld_pack_sk_s1+#undef mld_sig_unpack_hints+#undef mld_unpack_pk_t1+#undef mld_unpack_sk+/* mldsa/src/params.h */+#undef MLDSA_BETA+#undef MLDSA_CRHBYTES+#undef MLDSA_CRYPTO_BYTES+#undef MLDSA_CRYPTO_PUBLICKEYBYTES+#undef MLDSA_CRYPTO_SECRETKEYBYTES+#undef MLDSA_CTILDEBYTES+#undef MLDSA_D+#undef MLDSA_ETA+#undef MLDSA_GAMMA1+#undef MLDSA_GAMMA2+#undef MLDSA_GAMMA2_32+#undef MLDSA_GAMMA2_88+#undef MLDSA_K+#undef MLDSA_L+#undef MLDSA_N+#undef MLDSA_OMEGA+#undef MLDSA_PK_END+#undef MLDSA_PK_RHO_BYTES+#undef MLDSA_PK_RHO_OFFSET+#undef MLDSA_PK_T1_BYTES+#undef MLDSA_PK_T1_OFFSET+#undef MLDSA_POLYETA_PACKEDBYTES+#undef MLDSA_POLYT0_PACKEDBYTES+#undef MLDSA_POLYT1_PACKEDBYTES+#undef MLDSA_POLYVECH_PACKEDBYTES+#undef MLDSA_POLYW1_PACKEDBYTES+#undef MLDSA_POLYW1_PACKEDBYTES_32+#undef MLDSA_POLYW1_PACKEDBYTES_88+#undef MLDSA_POLYZ_PACKEDBYTES+#undef MLDSA_Q+#undef MLDSA_Q_HALF+#undef MLDSA_RNDBYTES+#undef MLDSA_SEEDBYTES+#undef MLDSA_SIG_C_BYTES+#undef MLDSA_SIG_C_OFFSET+#undef MLDSA_SIG_END+#undef MLDSA_SIG_H_BYTES+#undef MLDSA_SIG_H_OFFSET+#undef MLDSA_SIG_Z_BYTES+#undef MLDSA_SIG_Z_OFFSET+#undef MLDSA_SK_END+#undef MLDSA_SK_KEY_BYTES+#undef MLDSA_SK_KEY_OFFSET+#undef MLDSA_SK_RHO_BYTES+#undef MLDSA_SK_RHO_OFFSET+#undef MLDSA_SK_S1_BYTES+#undef MLDSA_SK_S1_OFFSET+#undef MLDSA_SK_S2_BYTES+#undef MLDSA_SK_S2_OFFSET+#undef MLDSA_SK_T0_BYTES+#undef MLDSA_SK_T0_OFFSET+#undef MLDSA_SK_TR_BYTES+#undef MLDSA_SK_TR_OFFSET+#undef MLDSA_TAU+#undef MLDSA_TRBYTES+#undef MLD_MAX_KAPPA+#undef MLD_PARAMS_H+/* mldsa/src/poly_kl.h */+#undef MLD_POLYETA_UNPACK_LOWER_BOUND+#undef MLD_POLY_KL_H+#undef mld_poly_challenge+#undef mld_poly_decompose+#undef mld_poly_uniform_eta+#undef mld_poly_uniform_eta_4x+#undef mld_poly_uniform_gamma1+#undef mld_poly_uniform_gamma1_4x+#undef mld_poly_use_hint+#undef mld_polyeta_pack+#undef mld_polyeta_unpack+#undef mld_polyw1_pack+#undef mld_polyz_pack+#undef mld_polyz_unpack+/* mldsa/src/polyvec.h */+#undef MLD_POLYVEC_H+#undef mld_polyveck+#undef mld_polyveck_caddq+#undef mld_polyveck_chknorm+#undef mld_polyveck_decompose+#undef mld_polyveck_invntt_tomont+#undef mld_polyveck_ntt+#undef mld_polyveck_pack_eta+#undef mld_polyveck_pack_w1+#undef mld_polyveck_reduce+#undef mld_polyveck_unpack_eta+#undef mld_polyvecl+#undef mld_polyvecl_chknorm+#undef mld_polyvecl_ntt+#undef mld_polyvecl_pack_eta+#undef mld_polyvecl_pointwise_acc_montgomery+#undef mld_polyvecl_uniform_gamma1+#undef mld_polyvecl_unpack_eta+#undef mld_polyvecl_unpack_z+/* mldsa/src/polyvec_lazy.h */+#undef MLD_POLYVEC_LAZY_H+#undef mld_poly_permute_bitrev_to_custom_optional+#undef mld_polymat+#undef mld_polymat_eager+#undef mld_polymat_lazy+#undef mld_polyvec_matrix_expand+#undef mld_polyvec_matrix_expand_eager+#undef mld_polyvec_matrix_expand_lazy+#undef mld_polyvec_matrix_pointwise_montgomery+#undef mld_polyvec_matrix_pointwise_montgomery_row+#undef mld_polyvec_matrix_pointwise_montgomery_row_eager+#undef mld_polyvec_matrix_pointwise_montgomery_row_lazy+#undef mld_polyvec_matrix_pointwise_montgomery_yvec+#undef mld_polyvec_matrix_pointwise_montgomery_yvec_eager+#undef mld_polyvec_matrix_pointwise_montgomery_yvec_lazy+#undef mld_sk_s1hat+#undef mld_sk_s1hat_eager+#undef mld_sk_s1hat_get_poly+#undef mld_sk_s1hat_get_poly_eager+#undef mld_sk_s1hat_get_poly_lazy+#undef mld_sk_s1hat_lazy+#undef mld_sk_s2hat+#undef mld_sk_s2hat_eager+#undef mld_sk_s2hat_get_poly+#undef mld_sk_s2hat_get_poly_eager+#undef mld_sk_s2hat_get_poly_lazy+#undef mld_sk_s2hat_lazy+#undef mld_sk_t0hat+#undef mld_sk_t0hat_eager+#undef mld_sk_t0hat_get_poly+#undef mld_sk_t0hat_get_poly_eager+#undef mld_sk_t0hat_get_poly_lazy+#undef mld_sk_t0hat_lazy+#undef mld_unpack_sk_s1hat+#undef mld_unpack_sk_s1hat_eager+#undef mld_unpack_sk_s1hat_lazy+#undef mld_unpack_sk_s2hat+#undef mld_unpack_sk_s2hat_eager+#undef mld_unpack_sk_s2hat_lazy+#undef mld_unpack_sk_t0hat+#undef mld_unpack_sk_t0hat_eager+#undef mld_unpack_sk_t0hat_lazy+#undef mld_yvec+#undef mld_yvec_eager+#undef mld_yvec_get_poly+#undef mld_yvec_get_poly_eager+#undef mld_yvec_get_poly_lazy+#undef mld_yvec_init+#undef mld_yvec_init_eager+#undef mld_yvec_init_lazy+#undef mld_yvec_lazy+/* mldsa/src/rounding.h */+#undef MLD_2_POW_D+#undef MLD_ROUNDING_H+#undef mld_decompose+#undef mld_make_hint+#undef mld_power2round+#undef mld_use_hint+/* mldsa/src/sign.h */+#undef MLD_DOMAIN_SEPARATION_MAX_BYTES+#undef MLD_PREHASH_NONE+#undef MLD_PREHASH_SHA2_224+#undef MLD_PREHASH_SHA2_256+#undef MLD_PREHASH_SHA2_384+#undef MLD_PREHASH_SHA2_512+#undef MLD_PREHASH_SHA2_512_224+#undef MLD_PREHASH_SHA2_512_256+#undef MLD_PREHASH_SHA3_224+#undef MLD_PREHASH_SHA3_256+#undef MLD_PREHASH_SHA3_384+#undef MLD_PREHASH_SHA3_512+#undef MLD_PREHASH_SHAKE_128+#undef MLD_PREHASH_SHAKE_256+#undef MLD_SIGN_H+#undef mld_prepare_domain_separation_prefix+#undef mld_sign_keypair+#undef mld_sign_keypair_internal+#undef mld_sign_pk_from_sk+#undef mld_sign_signature+#undef mld_sign_signature_extmu+#undef mld_sign_signature_internal+#undef mld_sign_signature_pre_hash_internal+#undef mld_sign_signature_pre_hash_shake256+#undef mld_sign_verify+#undef mld_sign_verify_extmu+#undef mld_sign_verify_internal+#undef mld_sign_verify_pre_hash_internal+#undef mld_sign_verify_pre_hash_shake256++#if !defined(MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS)+/*+ * Undefine macros from MLD_CONFIG_PARAMETER_SET-generic files+ */+/* mldsa/src/context.h */+#undef MLD_CONTEXT_H+#undef MLD_CONTEXT_PARAMETERS_0+#undef MLD_CONTEXT_PARAMETERS_1+#undef MLD_CONTEXT_PARAMETERS_2+#undef MLD_CONTEXT_PARAMETERS_3+#undef MLD_CONTEXT_PARAMETERS_4+#undef MLD_CONTEXT_PARAMETERS_5+#undef MLD_CONTEXT_PARAMETERS_6+#undef MLD_CONTEXT_PARAMETERS_7+#undef MLD_CONTEXT_PARAMETERS_8+#undef MLD_CONTEXT_PARAMETERS_9+#undef MLD_CONTEXT_UNUSED+#undef mld_sign_attempt+#undef mld_sign_finish+#undef mld_sign_resume+/* mldsa/src/ct.h */+#undef MLD_CT_H+#undef MLD_USE_ASM_VALUE_BARRIER+#undef mld_ct_opt_blocker_u64+/* mldsa/src/debug.h */+#undef MLD_DEBUG_H+#undef mld_assert+#undef mld_assert_abs_bound+#undef mld_assert_abs_bound_2d+#undef mld_assert_bound+#undef mld_assert_bound_2d+#undef mld_debug_check_assert+#undef mld_debug_check_bounds+/* mldsa/src/poly.h */+#undef MLD_FQMUL_BOUND+#undef MLD_INTT_BOUND+#undef MLD_NTT_BOUND+#undef MLD_POLY_H+#undef mld_poly_add+#undef mld_poly_caddq+#undef mld_poly_chknorm+#undef mld_poly_invntt_tomont+#undef mld_poly_ntt+#undef mld_poly_pointwise_montgomery+#undef mld_poly_power2round+#undef mld_poly_reduce+#undef mld_poly_shiftl+#undef mld_poly_sub+#undef mld_poly_uniform+#undef mld_poly_uniform_4x+#undef mld_polyt0_pack+#undef mld_polyt0_unpack+#undef mld_polyt1_pack+#undef mld_polyt1_unpack+#undef mld_polyw1_pack_32+#undef mld_polyw1_pack_88+/* mldsa/src/randombytes.h */+#undef MLD_RANDOMBYTES_H+/* mldsa/src/reduce.h */+#undef MLD_MONT+#undef MLD_REDUCE32_DOMAIN_MAX+#undef MLD_REDUCE32_RANGE_MAX+#undef MLD_REDUCE_H+/* mldsa/src/symmetric.h */+#undef MLD_STREAM128_BLOCKBYTES+#undef MLD_STREAM256_BLOCKBYTES+#undef MLD_SYMMETRIC_H+#undef mld_xof128_absorb_once+#undef mld_xof128_ctx+#undef mld_xof128_init+#undef mld_xof128_release+#undef mld_xof128_squeezeblocks+#undef mld_xof128_x4_absorb+#undef mld_xof128_x4_ctx+#undef mld_xof128_x4_init+#undef mld_xof128_x4_release+#undef mld_xof128_x4_squeezeblocks+#undef mld_xof256_absorb_once+#undef mld_xof256_ctx+#undef mld_xof256_init+#undef mld_xof256_release+#undef mld_xof256_squeezeblocks+#undef mld_xof256_x4_absorb+#undef mld_xof256_x4_ctx+#undef mld_xof256_x4_init+#undef mld_xof256_x4_release+#undef mld_xof256_x4_squeezeblocks+/* mldsa/src/sys.h */+#undef MLD_ALIGN+#undef MLD_ALIGN_UP+#undef MLD_ALWAYS_INLINE+#undef MLD_CET_ENDBR+#undef MLD_CT_TESTING_DECLASSIFY+#undef MLD_CT_TESTING_SECRET+#undef MLD_DEFAULT_ALIGN+#undef MLD_HAVE_INLINE_ASM+#undef MLD_INLINE+#undef MLD_MUST_CHECK_RETURN_VALUE+#undef MLD_NOINLINE+#undef MLD_RESTRICT+#undef MLD_STATIC_TESTABLE+#undef MLD_SYSV_ABI+#undef MLD_SYSV_ABI_SUPPORTED+#undef MLD_SYS_AARCH64+#undef MLD_SYS_AARCH64_EB+#undef MLD_SYS_AARCH64_NEON+#undef MLD_SYS_APPLE+#undef MLD_SYS_ARMV81M_MVE+#undef MLD_SYS_BIG_ENDIAN+#undef MLD_SYS_H+#undef MLD_SYS_LINUX+#undef MLD_SYS_LITTLE_ENDIAN+#undef MLD_SYS_PPC64LE+#undef MLD_SYS_RISCV32+#undef MLD_SYS_RISCV64+#undef MLD_SYS_RISCV64_RVV+#undef MLD_SYS_WINDOWS+#undef MLD_SYS_X86_64+#undef MLD_SYS_X86_64_AVX2+/* mldsa/src/cbmc.h */+#undef MLD_CBMC_H+#undef __contract__+#undef __loop__++#if !defined(MLD_CONFIG_FIPS202_CUSTOM_HEADER)+/*+ * Undefine macros from FIPS-202 files+ */+/* mldsa/src/fips202/fips202.h */+#undef MLD_FIPS202_FIPS202_H+#undef MLD_KECCAK_LANES+#undef SHA3_256_HASHBYTES+#undef SHA3_256_RATE+#undef SHA3_512_HASHBYTES+#undef SHA3_512_RATE+#undef SHAKE128_RATE+#undef SHAKE256_RATE+#undef mld_shake128_absorb+#undef mld_shake128_finalize+#undef mld_shake128_init+#undef mld_shake128_release+#undef mld_shake128_squeeze+#undef mld_shake256+#undef mld_shake256_absorb+#undef mld_shake256_finalize+#undef mld_shake256_init+#undef mld_shake256_release+#undef mld_shake256_squeeze+/* mldsa/src/fips202/fips202x4.h */+#undef MLD_FIPS202_FIPS202X4_H+#undef mld_shake128x4_absorb_once+#undef mld_shake128x4_init+#undef mld_shake128x4_release+#undef mld_shake128x4_squeezeblocks+#undef mld_shake256x4_absorb_once+#undef mld_shake256x4_init+#undef mld_shake256x4_release+#undef mld_shake256x4_squeezeblocks+/* mldsa/src/fips202/keccakf1600.h */+#undef MLD_FIPS202_KECCAKF1600_H+#undef MLD_KECCAK_LANES+#undef MLD_KECCAK_WAY+#undef mld_keccakf1600_extract_bytes+#undef mld_keccakf1600_permute+#undef mld_keccakf1600_xor_bytes+#undef mld_keccakf1600x4_extract_bytes+#undef mld_keccakf1600x4_permute+#undef mld_keccakf1600x4_xor_bytes+#endif /* !MLD_CONFIG_FIPS202_CUSTOM_HEADER */++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202)+/* mldsa/src/fips202/native/api.h */+#undef MLD_FIPS202_NATIVE_API_H+#undef MLD_NATIVE_FUNC_FALLBACK+#undef MLD_NATIVE_FUNC_SUCCESS+/* mldsa/src/fips202/native/auto.h */+#undef MLD_FIPS202_NATIVE_AUTO_H+#if defined(MLD_SYS_AARCH64)+/*+ * Undefine macros from native code (FIPS202, AArch64)+ */+/* mldsa/src/fips202/native/aarch64/auto.h */+#undef MLD_FIPS202_NATIVE_AARCH64_AUTO_H+/* mldsa/src/fips202/native/aarch64/src/fips202_native_aarch64.h */+#undef MLD_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H+#undef mld_keccak_f1600_x1_scalar_aarch64_asm+#undef mld_keccak_f1600_x1_v84a_aarch64_asm+#undef mld_keccak_f1600_x2_v84a_aarch64_asm+#undef mld_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm+#undef mld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm+#undef mld_keccakf1600_round_constants+/* mldsa/src/fips202/native/aarch64/x1_scalar.h */+#undef MLD_FIPS202_AARCH64_NEED_X1_SCALAR+#undef MLD_FIPS202_NATIVE_AARCH64_X1_SCALAR_H+#undef MLD_USE_NATIVE_FIPS202_X1+/* mldsa/src/fips202/native/aarch64/x1_v84a.h */+#undef MLD_FIPS202_AARCH64_NEED_X1_V84A+#undef MLD_FIPS202_NATIVE_AARCH64_X1_V84A_H+#undef MLD_USE_NATIVE_FIPS202_X1+/* mldsa/src/fips202/native/aarch64/x2_v84a.h */+#undef MLD_FIPS202_AARCH64_NEED_X2_V84A+#undef MLD_FIPS202_NATIVE_AARCH64_X2_V84A_H+#undef MLD_USE_NATIVE_FIPS202_X4+/* mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h */+#undef MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID+#undef MLD_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H+#undef MLD_USE_NATIVE_FIPS202_X4+/* mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h */+#undef MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID+#undef MLD_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H+#undef MLD_USE_NATIVE_FIPS202_X4+#endif /* MLD_SYS_AARCH64 */+#if defined(MLD_SYS_X86_64)+/*+ * Undefine macros from native code (FIPS202, x86_64)+ */+/* mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h */+#undef MLD_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H+#undef MLD_FIPS202_X86_64_NEED_X4_AVX2+#undef MLD_USE_NATIVE_FIPS202_X4+/* mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h */+#undef MLD_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H+#undef mld_keccak_f1600_x4_avx2_asm+#undef mld_keccak_rho56+#undef mld_keccak_rho8+#undef mld_keccakf1600_round_constants+#endif /* MLD_SYS_X86_64 */+#if defined(MLD_SYS_ARMV81M_MVE)+/*+ * Undefine macros from native code (FIPS202, Armv8.1-M)+ */+/* mldsa/src/fips202/native/armv81m/mve.h */+#undef MLD_FIPS202_ARMV81M_NEED_X4+#undef MLD_FIPS202_NATIVE_ARMV81M+#undef MLD_FIPS202_NATIVE_ARMV81M_MVE_H+#undef MLD_USE_NATIVE_FIPS202_X4+#undef MLD_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES+#undef MLD_USE_NATIVE_FIPS202_X4_XOR_BYTES+#undef mld_keccak_f1600_x4_native_impl+/* mldsa/src/fips202/native/armv81m/src/fips202_native_armv81m.h */+#undef MLD_FIPS202_NATIVE_ARMV81M_SRC_FIPS202_NATIVE_ARMV81M_H+#undef mld_keccak_f1600_x4_mve_asm+#undef mld_keccak_f1600_x4_state_extract_bytes_asm+#undef mld_keccak_f1600_x4_state_xor_bytes_asm+#undef mld_keccakf1600_round_constants+#endif /* MLD_SYS_ARMV81M_MVE */+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH)+/* mldsa/src/native/api.h */+#undef MLD_FQMUL_BOUND+#undef MLD_INTT_BOUND+#undef MLD_NATIVE_API_H+#undef MLD_NATIVE_FUNC_FALLBACK+#undef MLD_NATIVE_FUNC_SUCCESS+#undef MLD_NTT_BOUND+#undef MLD_REDUCE32_RANGE_MAX+/* mldsa/src/native/meta.h */+#undef MLD_NATIVE_META_H+#if defined(MLD_SYS_AARCH64)+/*+ * Undefine macros from native code (Arith, AArch64)+ */+/* mldsa/src/native/aarch64/meta.h */+#undef MLD_ARITH_BACKEND_AARCH64+#undef MLD_NATIVE_AARCH64_META_H+#undef MLD_USE_NATIVE_INTT+#undef MLD_USE_NATIVE_NTT+#undef MLD_USE_NATIVE_POINTWISE_MONTGOMERY+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7+#undef MLD_USE_NATIVE_POLYZ_UNPACK_17+#undef MLD_USE_NATIVE_POLYZ_UNPACK_19+#undef MLD_USE_NATIVE_POLY_CADDQ+#undef MLD_USE_NATIVE_POLY_CHKNORM+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_32+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_88+#undef MLD_USE_NATIVE_POLY_USE_HINT_32+#undef MLD_USE_NATIVE_POLY_USE_HINT_88+#undef MLD_USE_NATIVE_REJ_UNIFORM+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA2+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA4+/* mldsa/src/native/aarch64/src/arith_native_aarch64.h */+#undef MLD_AARCH64_REJ_UNIFORM_ETA2_BUFLEN+#undef MLD_AARCH64_REJ_UNIFORM_ETA4_BUFLEN+#undef MLD_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H+#undef mld_aarch64_intt_zetas_layer123456+#undef mld_aarch64_intt_zetas_layer78+#undef mld_aarch64_ntt_zetas_layer123456+#undef mld_aarch64_ntt_zetas_layer78+#undef mld_intt_aarch64_asm+#undef mld_ntt_aarch64_asm+#undef mld_poly_caddq_aarch64_asm+#undef mld_poly_chknorm_aarch64_asm+#undef mld_poly_decompose_32_aarch64_asm+#undef mld_poly_decompose_88_aarch64_asm+#undef mld_poly_pointwise_montgomery_aarch64_asm+#undef mld_poly_use_hint_32_aarch64_asm+#undef mld_poly_use_hint_88_aarch64_asm+#undef mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm+#undef mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm+#undef mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm+#undef mld_polyz_unpack_17_aarch64_asm+#undef mld_polyz_unpack_17_indices+#undef mld_polyz_unpack_19_aarch64_asm+#undef mld_polyz_unpack_19_indices+#undef mld_rej_uniform_aarch64_asm+#undef mld_rej_uniform_eta2_aarch64_asm+#undef mld_rej_uniform_eta4_aarch64_asm+#undef mld_rej_uniform_eta_table+#undef mld_rej_uniform_table+#endif /* MLD_SYS_AARCH64 */+#if defined(MLD_SYS_X86_64)+/*+ * Undefine macros from native code (Arith, X86_64)+ */+/* mldsa/src/native/x86_64/meta.h */+#undef MLD_ARITH_BACKEND_X86_64_DEFAULT+#undef MLD_NATIVE_X86_64_META_H+#undef MLD_USE_NATIVE_INTT+#undef MLD_USE_NATIVE_NTT+#undef MLD_USE_NATIVE_NTT_CUSTOM_ORDER+#undef MLD_USE_NATIVE_POINTWISE_MONTGOMERY+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7+#undef MLD_USE_NATIVE_POLYZ_UNPACK_17+#undef MLD_USE_NATIVE_POLYZ_UNPACK_19+#undef MLD_USE_NATIVE_POLY_CADDQ+#undef MLD_USE_NATIVE_POLY_CHKNORM+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_32+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_88+#undef MLD_USE_NATIVE_POLY_USE_HINT_32+#undef MLD_USE_NATIVE_POLY_USE_HINT_88+#undef MLD_USE_NATIVE_REJ_UNIFORM+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA2+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA4+/* mldsa/src/native/x86_64/src/arith_native_x86_64.h */+#undef MLD_AVX2_REJ_UNIFORM_BUFLEN+#undef MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN+#undef MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN+#undef MLD_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H+#undef mld_invntt_avx2_asm+#undef mld_ntt_avx2_asm+#undef mld_nttunpack_avx2_asm+#undef mld_pointwise_acc_l4_avx2_asm+#undef mld_pointwise_acc_l5_avx2_asm+#undef mld_pointwise_acc_l7_avx2_asm+#undef mld_pointwise_avx2_asm+#undef mld_poly_caddq_avx2_asm+#undef mld_poly_chknorm_avx2_asm+#undef mld_poly_decompose_32_avx2_asm+#undef mld_poly_decompose_88_avx2_asm+#undef mld_poly_use_hint_32_avx2_asm+#undef mld_poly_use_hint_88_avx2_asm+#undef mld_polyz_unpack_17_avx2_asm+#undef mld_polyz_unpack_19_avx2_asm+#undef mld_rej_uniform_avx2_asm+#undef mld_rej_uniform_eta2_avx2_asm+#undef mld_rej_uniform_eta4_avx2_asm+#undef mld_rej_uniform_table+/* mldsa/src/native/x86_64/src/consts.h */+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XDIV+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XDIV_QINV+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XQ+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XQINV+#undef MLD_AVX2_BACKEND_DATA_OFFSET_ZETAS+#undef MLD_AVX2_BACKEND_DATA_OFFSET_ZETAS_QINV+#undef MLD_NATIVE_X86_64_SRC_CONSTS_H+#undef mld_qdata+#endif /* MLD_SYS_X86_64 */+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_ARITH */+#endif /* !MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS */
+ cbits/mldsa/mldsa_native.h view
@@ -0,0 +1,956 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_H+#define MLD_H++/*+ * Public API for mldsa-native+ *+ * This header defines the public API of a single build of mldsa-native.+ *+ * Make sure the configuration file is in the include path+ * (this is "mldsa_native_config.h" by default, or MLD_CONFIG_FILE if defined).+ *+ * # API conventions+ *+ * Conventions shared by all functions below (return values, pointer validity,+ * output buffers on error) are documented in API-CONVENTIONS.md.+ *+ * # Multi-level builds+ *+ * This header specifies a build of mldsa-native for a fixed security level.+ * If you need multiple security levels, leave the security level unspecified+ * in the configuration file and include this header multiple times, setting+ * MLD_CONFIG_PARAMETER_SET accordingly for each, and #undef'ing the MLD_H+ * guard to allow multiple inclusions.+ */++/******************************* Key sizes ************************************/++/* Sizes of cryptographic material, per parameter set */+/* See mldsa/src/params.h for the arithmetic expressions giving rise to these */+/* check-magic: off */+#define MLDSA44_SECRETKEYBYTES 2560+#define MLDSA44_PUBLICKEYBYTES 1312+#define MLDSA44_BYTES 2420++#define MLDSA65_SECRETKEYBYTES 4032+#define MLDSA65_PUBLICKEYBYTES 1952+#define MLDSA65_BYTES 3309++#define MLDSA87_SECRETKEYBYTES 4896+#define MLDSA87_PUBLICKEYBYTES 2592+#define MLDSA87_BYTES 4627+/* check-magic: on */++/* Size of seed and randomness in bytes (level-independent) */+#define MLDSA_SEEDBYTES 32+#define MLDSA44_SEEDBYTES MLDSA_SEEDBYTES+#define MLDSA65_SEEDBYTES MLDSA_SEEDBYTES+#define MLDSA87_SEEDBYTES MLDSA_SEEDBYTES++/* Size of CRH output in bytes (level-independent) */+#define MLDSA_CRHBYTES 64+#define MLDSA44_CRHBYTES MLDSA_CRHBYTES+#define MLDSA65_CRHBYTES MLDSA_CRHBYTES+#define MLDSA87_CRHBYTES MLDSA_CRHBYTES++/* Size of TR in bytes (level-independent)+ *+ * TR = SHAKE256(pk, 64) is the hash of the public key. Callers of the+ * external-mu API (signature_extmu / verify_extmu) that compute the message+ * representative mu = SHAKE256(TR || M', MLDSA_CRHBYTES) themselves -- e.g. to+ * sign or verify a message that is too large or streamed to hold in memory --+ * need this constant to size the TR buffer. */+#define MLDSA_TRBYTES 64+#define MLDSA44_TRBYTES MLDSA_TRBYTES+#define MLDSA65_TRBYTES MLDSA_TRBYTES+#define MLDSA87_TRBYTES MLDSA_TRBYTES++/* Size of randomness for signing in bytes (level-independent) */+#define MLDSA_RNDBYTES 32+#define MLDSA44_RNDBYTES MLDSA_RNDBYTES+#define MLDSA65_RNDBYTES MLDSA_RNDBYTES+#define MLDSA87_RNDBYTES MLDSA_RNDBYTES++/* Sizes of cryptographic material, as a function of LVL=44,65,87 */+#define MLDSA_SECRETKEYBYTES_(LVL) MLDSA##LVL##_SECRETKEYBYTES+#define MLDSA_PUBLICKEYBYTES_(LVL) MLDSA##LVL##_PUBLICKEYBYTES+#define MLDSA_BYTES_(LVL) MLDSA##LVL##_BYTES+#define MLDSA_SECRETKEYBYTES(LVL) MLDSA_SECRETKEYBYTES_(LVL)+#define MLDSA_PUBLICKEYBYTES(LVL) MLDSA_PUBLICKEYBYTES_(LVL)+#define MLDSA_BYTES(LVL) MLDSA_BYTES_(LVL)++/****************************** Error codes ***********************************/++/* Generic failure condition, reserved for failures not covered by a more+ * specific error code. */+#define MLD_ERR_FAIL (-1)+/* An allocation failed. This can only happen if MLD_CONFIG_CUSTOM_ALLOC_FREE+ * is defined and the provided MLD_CUSTOM_ALLOC can fail. */+#define MLD_ERR_OUT_OF_MEMORY (-2)+/* An RNG failure occurred. Might be due to insufficient entropy or+ * system misconfiguration. */+#define MLD_ERR_RNG_FAIL (-3)+/* The signing rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS iterations without producing a valid+ * signature. With a FIPS 204 Appendix C compliant bound (>= 821) this+ * has probability < 2^-256. */+#define MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED (-4)+/* Signing was paused before completing, at the request of a caller-provided+ * MLD_CONFIG_SIGN_HOOK_ATTEMPT hook (see mldsa_native_config.h). The caller+ * resumes by re-invoking signing with the same inputs; the attempt hook,+ * together with MLD_CONFIG_SIGN_HOOK_RESUME, decides where to continue. */+#define MLD_ERR_SIGNING_PAUSED (-5)+/* Signature verification failed: the signature is not valid for the given+ * message and public key. Returned by the verification API. */+#define MLD_ERR_INVALID_SIGNATURE (-6)+/* Secret key validation failed: the secret key is malformed or internally+ * inconsistent. Returned by pk_from_sk. */+#define MLD_ERR_INVALID_KEY (-7)+/* The Pairwise Consistency Test failed. Only possible when+ * MLD_CONFIG_KEYGEN_PCT is enabled; signals that the freshly generated key+ * pair failed its sign/verify self-test. */+#define MLD_ERR_PCT_FAIL (-8)+/* An argument was invalid, e.g. an unsupported pre-hash algorithm or a context+ * string longer than 255 bytes. */+#define MLD_ERR_INVALID_ARG (-9)++/********************* Namespacing and Qualifiers *****************************/++#define MLD_API_CONCAT_(x, y) x##y+#define MLD_API_CONCAT(x, y) MLD_API_CONCAT_(x, y)+#define MLD_API_CONCAT_UNDERSCORE(x, y) MLD_API_CONCAT(MLD_API_CONCAT(x, _), y)++/* You need to make sure the config file is in the include path. */+#if defined(MLD_CONFIG_FILE)+#include MLD_CONFIG_FILE+#else+#include "mldsa_native_config.h"+#endif++/* Namespace prefix for the public API symbols. For multi-level builds, the+ * parameter set is appended to disambiguate the security levels. */+#if defined(MLD_CONFIG_MULTILEVEL_BUILD)+#define MLD_API_NAMESPACE_PREFIX \+ MLD_API_CONCAT(MLD_CONFIG_NAMESPACE_PREFIX, MLD_CONFIG_PARAMETER_SET)+#else+#define MLD_API_NAMESPACE_PREFIX MLD_CONFIG_NAMESPACE_PREFIX+#endif++#define MLD_API_NAMESPACE(sym) \+ MLD_API_CONCAT_UNDERSCORE(MLD_API_NAMESPACE_PREFIX, sym)++#if defined(__GNUC__) || defined(__clang__)+#define MLD_API_MUST_CHECK_RETURN_VALUE __attribute__((warn_unused_result))+#else+#define MLD_API_MUST_CHECK_RETURN_VALUE+#endif++#if defined(MLD_CONFIG_EXTERNAL_API_QUALIFIER)+#define MLD_API_QUALIFIER MLD_CONFIG_EXTERNAL_API_QUALIFIER+#else+#define MLD_API_QUALIFIER+#endif++/* Hash algorithm constants for domain separation */+#define MLD_PREHASH_NONE 0+#define MLD_PREHASH_SHA2_224 1+#define MLD_PREHASH_SHA2_256 2+#define MLD_PREHASH_SHA2_384 3+#define MLD_PREHASH_SHA2_512 4+#define MLD_PREHASH_SHA2_512_224 5+#define MLD_PREHASH_SHA2_512_256 6+#define MLD_PREHASH_SHA3_224 7+#define MLD_PREHASH_SHA3_256 8+#define MLD_PREHASH_SHA3_384 9+#define MLD_PREHASH_SHA3_512 10+#define MLD_PREHASH_SHAKE_128 11+#define MLD_PREHASH_SHAKE_256 12++/* Maximum formatted domain separation message length */+#define MLD_DOMAIN_SEPARATION_MAX_BYTES (2 + 255 + 11 + 64)++/****************************** Function API **********************************/++#if !defined(MLD_CONFIG_CONSTANTS_ONLY)++#include <stddef.h>+#include <stdint.h>++#ifdef __cplusplus+extern "C"+{+#endif++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+/**+ * Generate a public-private key pair from a seed.+ *+ * When MLD_CONFIG_KEYGEN_PCT is set, performs a Pairwise Consistency Test+ * (PCT) as required by FIPS 140-3 IG.+ *+ * @warning The seed must be generated by a cryptographically secure random+ * number generator.+ *+ * @spec{Implements @[FIPS204, Algorithm 6, ML-DSA.KeyGen_internal].}+ *+ * @param[out] pk Output public key.+ * @param[out] sk Output private key.+ * @param[in] seed Input random seed.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed+ * during the PCT. Only possible when+ * MLD_CONFIG_KEYGEN_PCT is enabled.+ * @retval MLD_ERR_SIGNING_PAUSED The PCT's signing step was paused by+ * a MLD_CONFIG_SIGN_HOOK_ATTEMPT hook.+ * This should currently never happen:+ * signing hooks require+ * MLD_CONFIG_NO_RANDOMIZED_API, which+ * is incompatible with+ * MLD_CONFIG_KEYGEN_PCT, so the two+ * cannot be enabled simultaneously.+ * @retval MLD_ERR_PCT_FAIL MLD_CONFIG_KEYGEN_PCT is enabled and+ * the PCT check failed.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(keypair_internal)(+ uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ const uint8_t seed[MLDSA_SEEDBYTES]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+/**+ * Generate a public-private key pair.+ *+ * When MLD_CONFIG_KEYGEN_PCT is set, performs a Pairwise Consistency Test+ * (PCT) as required by FIPS 140-3 IG.+ *+ * @spec{Implements @[FIPS204, Algorithm 1, ML-DSA.KeyGen].}+ *+ * @param[out] pk Output public key.+ * @param[out] sk Output private key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed.+ * @retval MLD_ERR_SIGNING_PAUSED The PCT's signing step was paused by+ * a MLD_CONFIG_SIGN_HOOK_ATTEMPT hook.+ * This should currently never happen:+ * signing hooks require+ * MLD_CONFIG_NO_RANDOMIZED_API, which+ * is incompatible with+ * MLD_CONFIG_KEYGEN_PCT, so the two+ * cannot be enabled simultaneously.+ * @retval MLD_ERR_PCT_FAIL MLD_CONFIG_KEYGEN_PCT is enabled and+ * the PCT check failed.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(keypair)(+ uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute signature using a caller-supplied random seed and prefix.+ *+ * On error (non-zero return value), the signature buffer sig is zeroized.+ *+ * @spec{Implements @[FIPS204, Algorithm 7, ML-DSA.Sign_internal].}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Pointer to buffer to hold the generated signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] m Pointer to message to be signed (when+ * externalmu == 0), or to a precomputed+ * message representative mu (when externalmu != 0).+ * @param mlen Length of m. Must equal MLDSA_CRHBYTES when+ * externalmu != 0.+ * @param[in] pre Pointer to prefix string. Ignored when+ * externalmu != 0.+ * @param prelen Length of prefix string. Ignored when+ * externalmu != 0.+ * @param[in] rnd Random seed.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param externalmu 0: m/mlen is the raw message; mu = H(tr, pre, m) is+ * computed internally.+ * non-zero: m points to a precomputed mu of+ * MLDSA_CRHBYTES bytes; pre/prelen unused.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(signature_internal)(+ uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *m,+ size_t mlen, const uint8_t *pre, size_t prelen,+ const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ int externalmu+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+/**+ * Compute signature. This function implements the randomized variant of+ * ML-DSA. If you require the deterministic variant, use+ * signature_internal directly.+ *+ * @spec{Implements @[FIPS204, Algorithm 2, ML-DSA.Sign].}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Pointer to buffer to hold the generated signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] m Pointer to message to be signed. May be NULL if+ * mlen == 0.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string. Should be <= 255.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255+ * bytes.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(signature)(+ uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++/**+ * Compute signature in "external mu" mode: the caller has already computed+ * the message representative mu = SHAKE256(tr || M', 64), where+ * tr = SHAKE256(pk, 64) and M' is the FIPS 204 formatted message (e.g.+ * 0x00 || ctxlen || ctx || msg for pure ML-DSA). This is useful when the+ * message is large or streamed and cannot be held in memory.+ *+ * @spec{Implements @[FIPS204, Algorithm 2, ML-DSA.Sign external mu variant].}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Pointer to buffer to hold the generated signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] mu Precomputed message representative.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(signature_extmu)(+ uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)],+ const uint8_t mu[MLDSA_CRHBYTES],+ const uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Verify signature. Internal API.+ *+ * @spec{Implements @[FIPS204, Algorithm 8, ML-DSA.Verify_internal].}+ *+ * @param[in] sig Pointer to input signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] m Pointer to message (when externalmu == 0), or to a+ * precomputed message representative mu (when+ * externalmu != 0).+ * @param mlen Length of m. Must equal MLDSA_CRHBYTES when+ * externalmu != 0.+ * @param[in] pre Pointer to prefix string. Ignored when externalmu != 0.+ * @param prelen Length of prefix string. Ignored when externalmu != 0.+ * @param[in] pk Bit-packed public key.+ * @param externalmu 0: m/mlen is the raw message; mu = H(H(pk), pre, m) is+ * computed internally.+ * non-zero: m points to a precomputed mu of+ * MLDSA_CRHBYTES bytes; pre/prelen unused.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(verify_internal)(+ const uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *m,+ size_t mlen, const uint8_t *pre, size_t prelen,+ const uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ int externalmu+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++#if !defined(MLD_CONFIG_CORE_API_ONLY)+/**+ * Verify signature.+ *+ * @spec{Implements @[FIPS204, Algorithm 3, ML-DSA.Verify].}+ *+ * @param[in] sig Pointer to input signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] m Pointer to message. May be NULL if mlen == 0.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] pk Bit-packed public key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255 bytes.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(verify)(+ const uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++/**+ * Verify signature in "external mu" mode: the caller has already computed+ * the message representative mu = SHAKE256(tr || M', 64), where+ * tr = SHAKE256(pk, 64) and M' is the FIPS 204 formatted message (e.g.+ * 0x00 || ctxlen || ctx || msg for pure ML-DSA). The same mu must have+ * been used at signing time.+ *+ * @spec{Implements @[FIPS204, Algorithm 3, ML-DSA.Verify external mu variant].}+ *+ * @param[in] sig Pointer to input signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] mu Precomputed message representative.+ * @param[in] pk Bit-packed public key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(verify_extmu)(+ const uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)],+ const uint8_t mu[MLDSA_CRHBYTES],+ const uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute signature with pre-hashed message.+ *+ * @spec{Implements @[FIPS204, Algorithm 4, HashML-DSA.Sign].}+ *+ * Supported hash algorithm constants:+ * MLD_PREHASH_SHA2_224, MLD_PREHASH_SHA2_256, MLD_PREHASH_SHA2_384,+ * MLD_PREHASH_SHA2_512, MLD_PREHASH_SHA2_512_224, MLD_PREHASH_SHA2_512_256,+ * MLD_PREHASH_SHA3_224, MLD_PREHASH_SHA3_256, MLD_PREHASH_SHA3_384,+ * MLD_PREHASH_SHA3_512, MLD_PREHASH_SHAKE_128, MLD_PREHASH_SHAKE_256.+ *+ * MLD_PREHASH_NONE is rejected by this API.+ *+ * @warning This is an unstable API that may change in the future. If you need+ * a stable API use signature_pre_hash_shake256.+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Pointer to buffer to hold the generated signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] ph Pointer to pre-hashed message.+ * @param phlen Length of pre-hashed message. Must match the output+ * length of hashalg (the digest size for SHA-2/SHA-3,+ * 32 bytes for MLD_PREHASH_SHAKE_128, 64 bytes for+ * MLD_PREHASH_SHAKE_256).+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] rnd Random seed.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param hashalg Hash algorithm constant (one of MLD_PREHASH_*).+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ * @retval MLD_ERR_INVALID_ARG The pre-hash algorithm was+ * MLD_PREHASH_NONE or unsupported,+ * phlen did not match the output+ * length of hashalg, or the context+ * string exceeded 255 bytes.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(signature_pre_hash_internal)(+ uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *ph,+ size_t phlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ int hashalg+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Verifies signature with pre-hashed message.+ *+ * @spec{Implements @[FIPS204, Algorithm 5, HashML-DSA.Verify].}+ *+ * Supported hash algorithm constants:+ * MLD_PREHASH_SHA2_224, MLD_PREHASH_SHA2_256, MLD_PREHASH_SHA2_384,+ * MLD_PREHASH_SHA2_512, MLD_PREHASH_SHA2_512_224, MLD_PREHASH_SHA2_512_256,+ * MLD_PREHASH_SHA3_224, MLD_PREHASH_SHA3_256, MLD_PREHASH_SHA3_384,+ * MLD_PREHASH_SHA3_512, MLD_PREHASH_SHAKE_128, MLD_PREHASH_SHAKE_256.+ *+ * MLD_PREHASH_NONE is rejected by this API.+ *+ * @warning This is an unstable API that may change in the future. If you need+ * a stable API use verify_pre_hash_shake256.+ *+ * @param[in] sig Pointer to input signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] ph Pointer to pre-hashed message.+ * @param phlen Length of pre-hashed message. Must match the output+ * length of hashalg (the digest size for SHA-2/SHA-3,+ * 32 bytes for MLD_PREHASH_SHAKE_128, 64 bytes for+ * MLD_PREHASH_SHAKE_256).+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] pk Bit-packed public key.+ * @param hashalg Hash algorithm constant (one of MLD_PREHASH_*).+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ * @retval MLD_ERR_INVALID_ARG The pre-hash algorithm was+ * MLD_PREHASH_NONE or unsupported, phlen+ * did not match the output length of+ * hashalg, or the context string exceeded+ * 255 bytes.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(verify_pre_hash_internal)(+ const uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *ph,+ size_t phlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ int hashalg+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute signature with pre-hashed message using SHAKE256. This function+ * computes the SHAKE256 hash of the message internally.+ *+ * @spec{Implements @[FIPS204, Algorithm 4, HashML-DSA.Sign] with SHAKE256 as+ * the pre-hash.}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Pointer to buffer to hold the generated signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] m Pointer to message to be hashed and signed. May be+ * NULL if mlen == 0.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] rnd Random seed.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255+ * bytes.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(signature_pre_hash_shake256)(+ uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Verify signature with pre-hashed message using SHAKE256. This function+ * computes the SHAKE256 hash of the message internally.+ *+ * @spec{Implements @[FIPS204, Algorithm 5, HashML-DSA.Verify] with SHAKE256 as+ * the pre-hash.}+ *+ * @param[in] sig Pointer to input signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] m Pointer to message to be hashed and verified. May be+ * NULL if mlen == 0.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] pk Bit-packed public key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255 bytes.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(verify_pre_hash_shake256)(+ const uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */++#if !defined(MLD_CONFIG_CORE_API_ONLY)+/**+ * Prepare domain separation prefix for ML-DSA signing.+ *+ * For pure ML-DSA (hashalg == MLD_PREHASH_NONE):+ * Format: 0x00 || ctxlen (1 byte) || ctx.+ *+ * For HashML-DSA (hashalg != MLD_PREHASH_NONE):+ * Format: 0x01 || ctxlen (1 byte) || ctx || oid (11 bytes) || ph.+ *+ * This function is useful for building incremental signing APIs.+ *+ * @spec{For HashML-DSA (hashalg != MLD_PREHASH_NONE), implements+ * @[FIPS204, Algorithm 4, line 23]. For Pure ML-DSA+ * (hashalg == MLD_PREHASH_NONE), implements+ * ```+ * M' <- BytesToBits(IntegerToBytes(0, 1)+ * || IntegerToBytes(|ctx|, 1)+ * || ctx+ * ```+ * which is part of @[FIPS204, Algorithm 2, ML-DSA.Sign, line 10] and+ * @[FIPS204, Algorithm 3, ML-DSA.Verify, line 5].}+ *+ * @param[out] prefix Output domain separation prefix buffer.+ * @param[in] ph Pointer to pre-hashed message (ignored for pure+ * ML-DSA).+ * @param phlen Length of pre-hashed message; must match the output+ * length of hashalg (ignored for pure ML-DSA).+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param hashalg Hash algorithm constant (MLD_PREHASH_NONE for pure+ * ML-DSA, or MLD_PREHASH_* for HashML-DSA).+ *+ * @return The total length of the formatted prefix, or 0 on error.+ * Errors are:+ * - The context string exceeded 255 bytes.+ * - For HashML-DSA: hashalg was unsupported, ph was NULL, or phlen+ * did not match the output length of hashalg.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+size_t MLD_API_NAMESPACE(prepare_domain_separation_prefix)(+ uint8_t prefix[MLD_DOMAIN_SEPARATION_MAX_BYTES], const uint8_t *ph,+ size_t phlen, const uint8_t *ctx, size_t ctxlen, int hashalg);++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+/**+ * Perform basic validity checks on secret key, and derive public key.+ *+ * Referring to the decoding of the secret key `sk=(rho, K, tr, s1, s2, t0)`+ * (cf. @[FIPS204, Algorithm 25, skDecode]), the following checks are+ * performed:+ * - Check that s1 and s2 have coefficients in [-MLDSA_ETA, MLDSA_ETA].+ * - Check that t0 and tr stored in sk match recomputed values.+ *+ * @note This function leaks whether the secret key is valid or invalid+ * through its return value and timing.+ *+ * @param[out] pk Output public key.+ * @param[in] sk Input secret key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was used and an+ * allocation via MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_KEY Secret key validation failed.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(pk_from_sk)(+ uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ const uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */++#ifdef __cplusplus+}+#endif++#undef MLD_API_NAMESPACE_PREFIX++#endif /* !MLD_CONFIG_CONSTANTS_ONLY */++/***************************** Memory Usage **********************************/++/*+ * By default mldsa-native performs all memory allocations on the stack.+ * Alternatively, mldsa-native supports custom allocation of large structures+ * through the `MLD_CONFIG_CUSTOM_ALLOC_FREE` configuration option.+ * See mldsa_native_config.h for details.+ *+ * `MLD_TOTAL_ALLOC_{44,65,87}_{KEYPAIR,SIGN,VERIFY}` indicates the maximum+ * (accumulative) allocation via MLD_ALLOC for each parameter set and operation.+ * Note that some stack allocation remains even+ * when using custom allocators, so these values are lower than total stack+ * usage with the default stack-only allocation.+ *+ * These constants may be used to implement custom allocations using a+ * fixed-sized buffer and a simple allocator (e.g., bump allocator).+ */+/* check-magic: off */+#if !defined(MLD_CONFIG_REDUCE_RAM)+#define MLD_TOTAL_ALLOC_44_KEYPAIR_NO_PCT 26912+#define MLD_TOTAL_ALLOC_44_KEYPAIR_PCT 48480+#define MLD_TOTAL_ALLOC_44_PK_FROM_SK 28480+#define MLD_TOTAL_ALLOC_44_SIGN 44704+#define MLD_TOTAL_ALLOC_44_VERIFY 24448+#define MLD_TOTAL_ALLOC_65_KEYPAIR_NO_PCT 44320+#define MLD_TOTAL_ALLOC_65_KEYPAIR_PCT 74624+#define MLD_TOTAL_ALLOC_65_PK_FROM_SK 46720+#define MLD_TOTAL_ALLOC_65_SIGN 69312+#define MLD_TOTAL_ALLOC_65_VERIFY 39872+#define MLD_TOTAL_ALLOC_87_KEYPAIR_NO_PCT 75040+#define MLD_TOTAL_ALLOC_87_KEYPAIR_PCT 115488+#define MLD_TOTAL_ALLOC_87_PK_FROM_SK 78272+#define MLD_TOTAL_ALLOC_87_SIGN 108224+#define MLD_TOTAL_ALLOC_87_VERIFY 68800+#else /* !MLD_CONFIG_REDUCE_RAM */+#define MLD_TOTAL_ALLOC_44_KEYPAIR_NO_PCT 11584+#define MLD_TOTAL_ALLOC_44_KEYPAIR_PCT 16896+#define MLD_TOTAL_ALLOC_44_PK_FROM_SK 13152+#define MLD_TOTAL_ALLOC_44_SIGN 13120+#define MLD_TOTAL_ALLOC_44_VERIFY 9120+#define MLD_TOTAL_ALLOC_65_KEYPAIR_NO_PCT 14656+#define MLD_TOTAL_ALLOC_65_KEYPAIR_PCT 22560+#define MLD_TOTAL_ALLOC_65_PK_FROM_SK 17056+#define MLD_TOTAL_ALLOC_65_SIGN 17248+#define MLD_TOTAL_ALLOC_65_VERIFY 10208+#define MLD_TOTAL_ALLOC_87_KEYPAIR_NO_PCT 18752+#define MLD_TOTAL_ALLOC_87_KEYPAIR_PCT 28608+#define MLD_TOTAL_ALLOC_87_PK_FROM_SK 21984+#define MLD_TOTAL_ALLOC_87_SIGN 21344+#define MLD_TOTAL_ALLOC_87_VERIFY 12512+#endif /* MLD_CONFIG_REDUCE_RAM */+/* check-magic: on */++/*+ * MLD_TOTAL_ALLOC_*_KEYPAIR adapts based on MLD_CONFIG_KEYGEN_PCT.+ */+#if defined(MLD_CONFIG_KEYGEN_PCT)+#define MLD_TOTAL_ALLOC_44_KEYPAIR MLD_TOTAL_ALLOC_44_KEYPAIR_PCT+#define MLD_TOTAL_ALLOC_65_KEYPAIR MLD_TOTAL_ALLOC_65_KEYPAIR_PCT+#define MLD_TOTAL_ALLOC_87_KEYPAIR MLD_TOTAL_ALLOC_87_KEYPAIR_PCT+#else+#define MLD_TOTAL_ALLOC_44_KEYPAIR MLD_TOTAL_ALLOC_44_KEYPAIR_NO_PCT+#define MLD_TOTAL_ALLOC_65_KEYPAIR MLD_TOTAL_ALLOC_65_KEYPAIR_NO_PCT+#define MLD_TOTAL_ALLOC_87_KEYPAIR MLD_TOTAL_ALLOC_87_KEYPAIR_NO_PCT+#endif++#define MLD_MAX3_(a, b, c) \+ ((a) > (b) ? ((a) > (c) ? (a) : (c)) : ((b) > (c) ? (b) : (c)))+#define MLD_MAX4_(a, b, c, d) MLD_MAX3_((a), (b), MLD_MAX3_((c), (d), (d)))++/*+ * `MLD_TOTAL_ALLOC_{44,65,87}` is the maximum across standard API operations+ * (keygen, sign, verify) for each parameter set.+ */+#define MLD_TOTAL_ALLOC_44 \+ MLD_MAX4_(MLD_TOTAL_ALLOC_44_KEYPAIR, MLD_TOTAL_ALLOC_44_PK_FROM_SK, \+ MLD_TOTAL_ALLOC_44_SIGN, MLD_TOTAL_ALLOC_44_VERIFY)+#define MLD_TOTAL_ALLOC_65 \+ MLD_MAX4_(MLD_TOTAL_ALLOC_65_KEYPAIR, MLD_TOTAL_ALLOC_65_PK_FROM_SK, \+ MLD_TOTAL_ALLOC_65_SIGN, MLD_TOTAL_ALLOC_65_VERIFY)+#define MLD_TOTAL_ALLOC_87 \+ MLD_MAX4_(MLD_TOTAL_ALLOC_87_KEYPAIR, MLD_TOTAL_ALLOC_87_PK_FROM_SK, \+ MLD_TOTAL_ALLOC_87_SIGN, MLD_TOTAL_ALLOC_87_VERIFY)++#endif /* !MLD_H */
+ cbits/mldsa/mldsa_native_asm.S view
@@ -0,0 +1,830 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++/******************************************************************************+ *+ * Single assembly unit for fixed-level build of mldsa-native+ *+ * This assembly unit bundles together all assembly files for a build+ * of mldsa-native for a fixed security level (MLDSA-44/65/87).+ *+ * # Multi-level build+ *+ * If you want an SCU build of mldsa-native with support for multiple security+ * levels, you should include this file once with+ * MLD_CONFIG_MULTILEVEL_WITH_SHARED set.+ *+ * (You could also follow the same pattern as for mldsa_native.c+ * and include it for every level, setting MLD_CONFIG_MULTILEVEL_NO_SHARED+ * for all but one. For builds with MLD_CONFIG_MULTILEVEL_NO_SHARED, this+ * file will then be ignored.)+ *+ * # Configuration+ *+ * The following options from the mldsa-native configuration are relevant:+ *+ * - MLD_CONFIG_FIPS202_CUSTOM_HEADER+ * Set this option if you use a custom FIPS202 implementation.+ *+ * - MLD_CONFIG_USE_NATIVE_BACKEND_ARITH+ * Set this option if you want to include the native arithmetic backends+ * in your build.+ *+ * - MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202+ * Set this option if you want to include the native FIPS202 backends+ * in your build.+ *+ * - MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+ * Set this option if you want to keep the directives defined in+ * level-independent headers. This is needed for a multi-level build.+ */++/* If parts of the mldsa-native source tree are not used,+ * consider reducing this header via `unifdef`.+ *+ * Example:+ * ```bash+ * unifdef -UMLD_CONFIG_USE_NATIVE_BACKEND_ARITH mldsa_native_asm.S+ * ```+ */++#include "src/common.h"++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH)+#if defined(MLD_SYS_AARCH64)+#include "src/native/aarch64/src/mldsa_intt_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_ntt_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_pointwise_montgomery_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_poly_caddq_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_poly_chknorm_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_poly_decompose_32_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_poly_decompose_88_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_poly_use_hint_32_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_poly_use_hint_88_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_polyz_unpack_17_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_polyz_unpack_19_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_rej_uniform_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_rej_uniform_eta2_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_rej_uniform_eta4_aarch64_asm.S"+#endif /* MLD_SYS_AARCH64 */+#if defined(MLD_SYS_X86_64)+#include "src/native/x86_64/src/mldsa_intt_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_ntt_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_nttunpack_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_pointwise_acc_l4_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_pointwise_acc_l5_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_pointwise_acc_l7_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_pointwise_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_poly_caddq_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_rej_uniform_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S"+#endif /* MLD_SYS_X86_64 */+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_ARITH */++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202)+#if defined(MLD_SYS_AARCH64)+#include "src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S"+#endif /* MLD_SYS_AARCH64 */+#if defined(MLD_SYS_X86_64)+#include "src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S"+#endif+#if defined(MLD_SYS_ARMV81M_MVE)+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S"+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_state_extract_bytes_mve.S"+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_state_xor_bytes_mve.S"+#endif+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+++/* Macro #undef's+ *+ * The following undefines macros from headers+ * included by the source files imported above.+ *+ * This is to allow building and linking multiple builds+ * of mldsa-native for varying parameter sets through concatenation+ * of this file, as if the files had been compiled separately.+ * If this is not relevant to you, you may remove the following.+ *+ * NOTE: This is not needed for the assembly SCU since, at present,+ * there is no need to include it multiple times.+ * We keep it for uniformity with mldsa_native.c only.+ *+ * NOTE: To avoid having to distinguish between which headers are included+ * from the assembly files, we #undef the same set of directives+ * as in mldsa_native.c+ */++/*+ * Undefine macros from MLD_CONFIG_PARAMETER_SET-specific files+ */+/* mldsa/mldsa_native.h */+#undef MLDSA44_BYTES+#undef MLDSA44_CRHBYTES+#undef MLDSA44_PUBLICKEYBYTES+#undef MLDSA44_RNDBYTES+#undef MLDSA44_SECRETKEYBYTES+#undef MLDSA44_SEEDBYTES+#undef MLDSA44_TRBYTES+#undef MLDSA65_BYTES+#undef MLDSA65_CRHBYTES+#undef MLDSA65_PUBLICKEYBYTES+#undef MLDSA65_RNDBYTES+#undef MLDSA65_SECRETKEYBYTES+#undef MLDSA65_SEEDBYTES+#undef MLDSA65_TRBYTES+#undef MLDSA87_BYTES+#undef MLDSA87_CRHBYTES+#undef MLDSA87_PUBLICKEYBYTES+#undef MLDSA87_RNDBYTES+#undef MLDSA87_SECRETKEYBYTES+#undef MLDSA87_SEEDBYTES+#undef MLDSA87_TRBYTES+#undef MLDSA_BYTES+#undef MLDSA_BYTES_+#undef MLDSA_CRHBYTES+#undef MLDSA_PUBLICKEYBYTES+#undef MLDSA_PUBLICKEYBYTES_+#undef MLDSA_RNDBYTES+#undef MLDSA_SECRETKEYBYTES+#undef MLDSA_SECRETKEYBYTES_+#undef MLDSA_SEEDBYTES+#undef MLDSA_TRBYTES+#undef MLD_API_CONCAT+#undef MLD_API_CONCAT_+#undef MLD_API_CONCAT_UNDERSCORE+#undef MLD_API_MUST_CHECK_RETURN_VALUE+#undef MLD_API_NAMESPACE+#undef MLD_API_NAMESPACE_PREFIX+#undef MLD_API_QUALIFIER+#undef MLD_DOMAIN_SEPARATION_MAX_BYTES+#undef MLD_ERR_FAIL+#undef MLD_ERR_INVALID_ARG+#undef MLD_ERR_INVALID_KEY+#undef MLD_ERR_INVALID_SIGNATURE+#undef MLD_ERR_OUT_OF_MEMORY+#undef MLD_ERR_PCT_FAIL+#undef MLD_ERR_RNG_FAIL+#undef MLD_ERR_SIGNING_PAUSED+#undef MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED+#undef MLD_H+#undef MLD_MAX3_+#undef MLD_MAX4_+#undef MLD_PREHASH_NONE+#undef MLD_PREHASH_SHA2_224+#undef MLD_PREHASH_SHA2_256+#undef MLD_PREHASH_SHA2_384+#undef MLD_PREHASH_SHA2_512+#undef MLD_PREHASH_SHA2_512_224+#undef MLD_PREHASH_SHA2_512_256+#undef MLD_PREHASH_SHA3_224+#undef MLD_PREHASH_SHA3_256+#undef MLD_PREHASH_SHA3_384+#undef MLD_PREHASH_SHA3_512+#undef MLD_PREHASH_SHAKE_128+#undef MLD_PREHASH_SHAKE_256+#undef MLD_TOTAL_ALLOC_44+#undef MLD_TOTAL_ALLOC_44_KEYPAIR+#undef MLD_TOTAL_ALLOC_44_KEYPAIR_NO_PCT+#undef MLD_TOTAL_ALLOC_44_KEYPAIR_PCT+#undef MLD_TOTAL_ALLOC_44_PK_FROM_SK+#undef MLD_TOTAL_ALLOC_44_SIGN+#undef MLD_TOTAL_ALLOC_44_VERIFY+#undef MLD_TOTAL_ALLOC_65+#undef MLD_TOTAL_ALLOC_65_KEYPAIR+#undef MLD_TOTAL_ALLOC_65_KEYPAIR_NO_PCT+#undef MLD_TOTAL_ALLOC_65_KEYPAIR_PCT+#undef MLD_TOTAL_ALLOC_65_PK_FROM_SK+#undef MLD_TOTAL_ALLOC_65_SIGN+#undef MLD_TOTAL_ALLOC_65_VERIFY+#undef MLD_TOTAL_ALLOC_87+#undef MLD_TOTAL_ALLOC_87_KEYPAIR+#undef MLD_TOTAL_ALLOC_87_KEYPAIR_NO_PCT+#undef MLD_TOTAL_ALLOC_87_KEYPAIR_PCT+#undef MLD_TOTAL_ALLOC_87_PK_FROM_SK+#undef MLD_TOTAL_ALLOC_87_SIGN+#undef MLD_TOTAL_ALLOC_87_VERIFY+/* mldsa/src/common.h */+#undef MLD_ADD_PARAM_SET+#undef MLD_ALLOC+#undef MLD_APPLY+#undef MLD_ASM_FN_SIZE+#undef MLD_ASM_FN_SYMBOL+#undef MLD_ASM_NAMESPACE+#undef MLD_BUILD_INTERNAL+#undef MLD_COMMON_H+#undef MLD_CONCAT+#undef MLD_CONCAT_+#undef MLD_EMPTY_CU+#undef MLD_ERR_FAIL+#undef MLD_ERR_INVALID_ARG+#undef MLD_ERR_INVALID_KEY+#undef MLD_ERR_INVALID_SIGNATURE+#undef MLD_ERR_OUT_OF_MEMORY+#undef MLD_ERR_PCT_FAIL+#undef MLD_ERR_RNG_FAIL+#undef MLD_ERR_SIGNING_PAUSED+#undef MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED+#undef MLD_EXTERNAL_API+#undef MLD_FIPS202X4_HEADER_FILE+#undef MLD_FIPS202_HEADER_FILE+#undef MLD_FREE+#undef MLD_INTERNAL_API+#undef MLD_INTERNAL_DATA_DECLARATION+#undef MLD_INTERNAL_DATA_DEFINITION+#undef MLD_MULTILEVEL_BUILD+#undef MLD_NAMESPACE+#undef MLD_NAMESPACE_KL+#undef MLD_NAMESPACE_PREFIX+#undef MLD_NAMESPACE_PREFIX_KL+#undef mld_memcpy+#undef mld_memset+/* mldsa/src/packing.h */+#undef MLD_PACKING_H+#undef mld_pack_sig_c+#undef mld_pack_sig_h+#undef mld_pack_sig_z+#undef mld_pack_sk_rho_key_tr_s2+#undef mld_pack_sk_s1+#undef mld_sig_unpack_hints+#undef mld_unpack_pk_t1+#undef mld_unpack_sk+/* mldsa/src/params.h */+#undef MLDSA_BETA+#undef MLDSA_CRHBYTES+#undef MLDSA_CRYPTO_BYTES+#undef MLDSA_CRYPTO_PUBLICKEYBYTES+#undef MLDSA_CRYPTO_SECRETKEYBYTES+#undef MLDSA_CTILDEBYTES+#undef MLDSA_D+#undef MLDSA_ETA+#undef MLDSA_GAMMA1+#undef MLDSA_GAMMA2+#undef MLDSA_GAMMA2_32+#undef MLDSA_GAMMA2_88+#undef MLDSA_K+#undef MLDSA_L+#undef MLDSA_N+#undef MLDSA_OMEGA+#undef MLDSA_PK_END+#undef MLDSA_PK_RHO_BYTES+#undef MLDSA_PK_RHO_OFFSET+#undef MLDSA_PK_T1_BYTES+#undef MLDSA_PK_T1_OFFSET+#undef MLDSA_POLYETA_PACKEDBYTES+#undef MLDSA_POLYT0_PACKEDBYTES+#undef MLDSA_POLYT1_PACKEDBYTES+#undef MLDSA_POLYVECH_PACKEDBYTES+#undef MLDSA_POLYW1_PACKEDBYTES+#undef MLDSA_POLYW1_PACKEDBYTES_32+#undef MLDSA_POLYW1_PACKEDBYTES_88+#undef MLDSA_POLYZ_PACKEDBYTES+#undef MLDSA_Q+#undef MLDSA_Q_HALF+#undef MLDSA_RNDBYTES+#undef MLDSA_SEEDBYTES+#undef MLDSA_SIG_C_BYTES+#undef MLDSA_SIG_C_OFFSET+#undef MLDSA_SIG_END+#undef MLDSA_SIG_H_BYTES+#undef MLDSA_SIG_H_OFFSET+#undef MLDSA_SIG_Z_BYTES+#undef MLDSA_SIG_Z_OFFSET+#undef MLDSA_SK_END+#undef MLDSA_SK_KEY_BYTES+#undef MLDSA_SK_KEY_OFFSET+#undef MLDSA_SK_RHO_BYTES+#undef MLDSA_SK_RHO_OFFSET+#undef MLDSA_SK_S1_BYTES+#undef MLDSA_SK_S1_OFFSET+#undef MLDSA_SK_S2_BYTES+#undef MLDSA_SK_S2_OFFSET+#undef MLDSA_SK_T0_BYTES+#undef MLDSA_SK_T0_OFFSET+#undef MLDSA_SK_TR_BYTES+#undef MLDSA_SK_TR_OFFSET+#undef MLDSA_TAU+#undef MLDSA_TRBYTES+#undef MLD_MAX_KAPPA+#undef MLD_PARAMS_H+/* mldsa/src/poly_kl.h */+#undef MLD_POLYETA_UNPACK_LOWER_BOUND+#undef MLD_POLY_KL_H+#undef mld_poly_challenge+#undef mld_poly_decompose+#undef mld_poly_uniform_eta+#undef mld_poly_uniform_eta_4x+#undef mld_poly_uniform_gamma1+#undef mld_poly_uniform_gamma1_4x+#undef mld_poly_use_hint+#undef mld_polyeta_pack+#undef mld_polyeta_unpack+#undef mld_polyw1_pack+#undef mld_polyz_pack+#undef mld_polyz_unpack+/* mldsa/src/polyvec.h */+#undef MLD_POLYVEC_H+#undef mld_polyveck+#undef mld_polyveck_caddq+#undef mld_polyveck_chknorm+#undef mld_polyveck_decompose+#undef mld_polyveck_invntt_tomont+#undef mld_polyveck_ntt+#undef mld_polyveck_pack_eta+#undef mld_polyveck_pack_w1+#undef mld_polyveck_reduce+#undef mld_polyveck_unpack_eta+#undef mld_polyvecl+#undef mld_polyvecl_chknorm+#undef mld_polyvecl_ntt+#undef mld_polyvecl_pack_eta+#undef mld_polyvecl_pointwise_acc_montgomery+#undef mld_polyvecl_uniform_gamma1+#undef mld_polyvecl_unpack_eta+#undef mld_polyvecl_unpack_z+/* mldsa/src/polyvec_lazy.h */+#undef MLD_POLYVEC_LAZY_H+#undef mld_poly_permute_bitrev_to_custom_optional+#undef mld_polymat+#undef mld_polymat_eager+#undef mld_polymat_lazy+#undef mld_polyvec_matrix_expand+#undef mld_polyvec_matrix_expand_eager+#undef mld_polyvec_matrix_expand_lazy+#undef mld_polyvec_matrix_pointwise_montgomery+#undef mld_polyvec_matrix_pointwise_montgomery_row+#undef mld_polyvec_matrix_pointwise_montgomery_row_eager+#undef mld_polyvec_matrix_pointwise_montgomery_row_lazy+#undef mld_polyvec_matrix_pointwise_montgomery_yvec+#undef mld_polyvec_matrix_pointwise_montgomery_yvec_eager+#undef mld_polyvec_matrix_pointwise_montgomery_yvec_lazy+#undef mld_sk_s1hat+#undef mld_sk_s1hat_eager+#undef mld_sk_s1hat_get_poly+#undef mld_sk_s1hat_get_poly_eager+#undef mld_sk_s1hat_get_poly_lazy+#undef mld_sk_s1hat_lazy+#undef mld_sk_s2hat+#undef mld_sk_s2hat_eager+#undef mld_sk_s2hat_get_poly+#undef mld_sk_s2hat_get_poly_eager+#undef mld_sk_s2hat_get_poly_lazy+#undef mld_sk_s2hat_lazy+#undef mld_sk_t0hat+#undef mld_sk_t0hat_eager+#undef mld_sk_t0hat_get_poly+#undef mld_sk_t0hat_get_poly_eager+#undef mld_sk_t0hat_get_poly_lazy+#undef mld_sk_t0hat_lazy+#undef mld_unpack_sk_s1hat+#undef mld_unpack_sk_s1hat_eager+#undef mld_unpack_sk_s1hat_lazy+#undef mld_unpack_sk_s2hat+#undef mld_unpack_sk_s2hat_eager+#undef mld_unpack_sk_s2hat_lazy+#undef mld_unpack_sk_t0hat+#undef mld_unpack_sk_t0hat_eager+#undef mld_unpack_sk_t0hat_lazy+#undef mld_yvec+#undef mld_yvec_eager+#undef mld_yvec_get_poly+#undef mld_yvec_get_poly_eager+#undef mld_yvec_get_poly_lazy+#undef mld_yvec_init+#undef mld_yvec_init_eager+#undef mld_yvec_init_lazy+#undef mld_yvec_lazy+/* mldsa/src/rounding.h */+#undef MLD_2_POW_D+#undef MLD_ROUNDING_H+#undef mld_decompose+#undef mld_make_hint+#undef mld_power2round+#undef mld_use_hint+/* mldsa/src/sign.h */+#undef MLD_DOMAIN_SEPARATION_MAX_BYTES+#undef MLD_PREHASH_NONE+#undef MLD_PREHASH_SHA2_224+#undef MLD_PREHASH_SHA2_256+#undef MLD_PREHASH_SHA2_384+#undef MLD_PREHASH_SHA2_512+#undef MLD_PREHASH_SHA2_512_224+#undef MLD_PREHASH_SHA2_512_256+#undef MLD_PREHASH_SHA3_224+#undef MLD_PREHASH_SHA3_256+#undef MLD_PREHASH_SHA3_384+#undef MLD_PREHASH_SHA3_512+#undef MLD_PREHASH_SHAKE_128+#undef MLD_PREHASH_SHAKE_256+#undef MLD_SIGN_H+#undef mld_prepare_domain_separation_prefix+#undef mld_sign_keypair+#undef mld_sign_keypair_internal+#undef mld_sign_pk_from_sk+#undef mld_sign_signature+#undef mld_sign_signature_extmu+#undef mld_sign_signature_internal+#undef mld_sign_signature_pre_hash_internal+#undef mld_sign_signature_pre_hash_shake256+#undef mld_sign_verify+#undef mld_sign_verify_extmu+#undef mld_sign_verify_internal+#undef mld_sign_verify_pre_hash_internal+#undef mld_sign_verify_pre_hash_shake256++#if !defined(MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS)+/*+ * Undefine macros from MLD_CONFIG_PARAMETER_SET-generic files+ */+/* mldsa/src/context.h */+#undef MLD_CONTEXT_H+#undef MLD_CONTEXT_PARAMETERS_0+#undef MLD_CONTEXT_PARAMETERS_1+#undef MLD_CONTEXT_PARAMETERS_2+#undef MLD_CONTEXT_PARAMETERS_3+#undef MLD_CONTEXT_PARAMETERS_4+#undef MLD_CONTEXT_PARAMETERS_5+#undef MLD_CONTEXT_PARAMETERS_6+#undef MLD_CONTEXT_PARAMETERS_7+#undef MLD_CONTEXT_PARAMETERS_8+#undef MLD_CONTEXT_PARAMETERS_9+#undef MLD_CONTEXT_UNUSED+#undef mld_sign_attempt+#undef mld_sign_finish+#undef mld_sign_resume+/* mldsa/src/ct.h */+#undef MLD_CT_H+#undef MLD_USE_ASM_VALUE_BARRIER+#undef mld_ct_opt_blocker_u64+/* mldsa/src/debug.h */+#undef MLD_DEBUG_H+#undef mld_assert+#undef mld_assert_abs_bound+#undef mld_assert_abs_bound_2d+#undef mld_assert_bound+#undef mld_assert_bound_2d+#undef mld_debug_check_assert+#undef mld_debug_check_bounds+/* mldsa/src/poly.h */+#undef MLD_FQMUL_BOUND+#undef MLD_INTT_BOUND+#undef MLD_NTT_BOUND+#undef MLD_POLY_H+#undef mld_poly_add+#undef mld_poly_caddq+#undef mld_poly_chknorm+#undef mld_poly_invntt_tomont+#undef mld_poly_ntt+#undef mld_poly_pointwise_montgomery+#undef mld_poly_power2round+#undef mld_poly_reduce+#undef mld_poly_shiftl+#undef mld_poly_sub+#undef mld_poly_uniform+#undef mld_poly_uniform_4x+#undef mld_polyt0_pack+#undef mld_polyt0_unpack+#undef mld_polyt1_pack+#undef mld_polyt1_unpack+#undef mld_polyw1_pack_32+#undef mld_polyw1_pack_88+/* mldsa/src/randombytes.h */+#undef MLD_RANDOMBYTES_H+/* mldsa/src/reduce.h */+#undef MLD_MONT+#undef MLD_REDUCE32_DOMAIN_MAX+#undef MLD_REDUCE32_RANGE_MAX+#undef MLD_REDUCE_H+/* mldsa/src/symmetric.h */+#undef MLD_STREAM128_BLOCKBYTES+#undef MLD_STREAM256_BLOCKBYTES+#undef MLD_SYMMETRIC_H+#undef mld_xof128_absorb_once+#undef mld_xof128_ctx+#undef mld_xof128_init+#undef mld_xof128_release+#undef mld_xof128_squeezeblocks+#undef mld_xof128_x4_absorb+#undef mld_xof128_x4_ctx+#undef mld_xof128_x4_init+#undef mld_xof128_x4_release+#undef mld_xof128_x4_squeezeblocks+#undef mld_xof256_absorb_once+#undef mld_xof256_ctx+#undef mld_xof256_init+#undef mld_xof256_release+#undef mld_xof256_squeezeblocks+#undef mld_xof256_x4_absorb+#undef mld_xof256_x4_ctx+#undef mld_xof256_x4_init+#undef mld_xof256_x4_release+#undef mld_xof256_x4_squeezeblocks+/* mldsa/src/sys.h */+#undef MLD_ALIGN+#undef MLD_ALIGN_UP+#undef MLD_ALWAYS_INLINE+#undef MLD_CET_ENDBR+#undef MLD_CT_TESTING_DECLASSIFY+#undef MLD_CT_TESTING_SECRET+#undef MLD_DEFAULT_ALIGN+#undef MLD_HAVE_INLINE_ASM+#undef MLD_INLINE+#undef MLD_MUST_CHECK_RETURN_VALUE+#undef MLD_NOINLINE+#undef MLD_RESTRICT+#undef MLD_STATIC_TESTABLE+#undef MLD_SYSV_ABI+#undef MLD_SYSV_ABI_SUPPORTED+#undef MLD_SYS_AARCH64+#undef MLD_SYS_AARCH64_EB+#undef MLD_SYS_AARCH64_NEON+#undef MLD_SYS_APPLE+#undef MLD_SYS_ARMV81M_MVE+#undef MLD_SYS_BIG_ENDIAN+#undef MLD_SYS_H+#undef MLD_SYS_LINUX+#undef MLD_SYS_LITTLE_ENDIAN+#undef MLD_SYS_PPC64LE+#undef MLD_SYS_RISCV32+#undef MLD_SYS_RISCV64+#undef MLD_SYS_RISCV64_RVV+#undef MLD_SYS_WINDOWS+#undef MLD_SYS_X86_64+#undef MLD_SYS_X86_64_AVX2+/* mldsa/src/cbmc.h */+#undef MLD_CBMC_H+#undef __contract__+#undef __loop__++#if !defined(MLD_CONFIG_FIPS202_CUSTOM_HEADER)+/*+ * Undefine macros from FIPS-202 files+ */+/* mldsa/src/fips202/fips202.h */+#undef MLD_FIPS202_FIPS202_H+#undef MLD_KECCAK_LANES+#undef SHA3_256_HASHBYTES+#undef SHA3_256_RATE+#undef SHA3_512_HASHBYTES+#undef SHA3_512_RATE+#undef SHAKE128_RATE+#undef SHAKE256_RATE+#undef mld_shake128_absorb+#undef mld_shake128_finalize+#undef mld_shake128_init+#undef mld_shake128_release+#undef mld_shake128_squeeze+#undef mld_shake256+#undef mld_shake256_absorb+#undef mld_shake256_finalize+#undef mld_shake256_init+#undef mld_shake256_release+#undef mld_shake256_squeeze+/* mldsa/src/fips202/fips202x4.h */+#undef MLD_FIPS202_FIPS202X4_H+#undef mld_shake128x4_absorb_once+#undef mld_shake128x4_init+#undef mld_shake128x4_release+#undef mld_shake128x4_squeezeblocks+#undef mld_shake256x4_absorb_once+#undef mld_shake256x4_init+#undef mld_shake256x4_release+#undef mld_shake256x4_squeezeblocks+/* mldsa/src/fips202/keccakf1600.h */+#undef MLD_FIPS202_KECCAKF1600_H+#undef MLD_KECCAK_LANES+#undef MLD_KECCAK_WAY+#undef mld_keccakf1600_extract_bytes+#undef mld_keccakf1600_permute+#undef mld_keccakf1600_xor_bytes+#undef mld_keccakf1600x4_extract_bytes+#undef mld_keccakf1600x4_permute+#undef mld_keccakf1600x4_xor_bytes+#endif /* !MLD_CONFIG_FIPS202_CUSTOM_HEADER */++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202)+/* mldsa/src/fips202/native/api.h */+#undef MLD_FIPS202_NATIVE_API_H+#undef MLD_NATIVE_FUNC_FALLBACK+#undef MLD_NATIVE_FUNC_SUCCESS+/* mldsa/src/fips202/native/auto.h */+#undef MLD_FIPS202_NATIVE_AUTO_H+#if defined(MLD_SYS_AARCH64)+/*+ * Undefine macros from native code (FIPS202, AArch64)+ */+/* mldsa/src/fips202/native/aarch64/auto.h */+#undef MLD_FIPS202_NATIVE_AARCH64_AUTO_H+/* mldsa/src/fips202/native/aarch64/src/fips202_native_aarch64.h */+#undef MLD_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H+#undef mld_keccak_f1600_x1_scalar_aarch64_asm+#undef mld_keccak_f1600_x1_v84a_aarch64_asm+#undef mld_keccak_f1600_x2_v84a_aarch64_asm+#undef mld_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm+#undef mld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm+#undef mld_keccakf1600_round_constants+/* mldsa/src/fips202/native/aarch64/x1_scalar.h */+#undef MLD_FIPS202_AARCH64_NEED_X1_SCALAR+#undef MLD_FIPS202_NATIVE_AARCH64_X1_SCALAR_H+#undef MLD_USE_NATIVE_FIPS202_X1+/* mldsa/src/fips202/native/aarch64/x1_v84a.h */+#undef MLD_FIPS202_AARCH64_NEED_X1_V84A+#undef MLD_FIPS202_NATIVE_AARCH64_X1_V84A_H+#undef MLD_USE_NATIVE_FIPS202_X1+/* mldsa/src/fips202/native/aarch64/x2_v84a.h */+#undef MLD_FIPS202_AARCH64_NEED_X2_V84A+#undef MLD_FIPS202_NATIVE_AARCH64_X2_V84A_H+#undef MLD_USE_NATIVE_FIPS202_X4+/* mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h */+#undef MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID+#undef MLD_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H+#undef MLD_USE_NATIVE_FIPS202_X4+/* mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h */+#undef MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID+#undef MLD_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H+#undef MLD_USE_NATIVE_FIPS202_X4+#endif /* MLD_SYS_AARCH64 */+#if defined(MLD_SYS_X86_64)+/*+ * Undefine macros from native code (FIPS202, x86_64)+ */+/* mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h */+#undef MLD_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H+#undef MLD_FIPS202_X86_64_NEED_X4_AVX2+#undef MLD_USE_NATIVE_FIPS202_X4+/* mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h */+#undef MLD_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H+#undef mld_keccak_f1600_x4_avx2_asm+#undef mld_keccak_rho56+#undef mld_keccak_rho8+#undef mld_keccakf1600_round_constants+#endif /* MLD_SYS_X86_64 */+#if defined(MLD_SYS_ARMV81M_MVE)+/*+ * Undefine macros from native code (FIPS202, Armv8.1-M)+ */+/* mldsa/src/fips202/native/armv81m/mve.h */+#undef MLD_FIPS202_ARMV81M_NEED_X4+#undef MLD_FIPS202_NATIVE_ARMV81M+#undef MLD_FIPS202_NATIVE_ARMV81M_MVE_H+#undef MLD_USE_NATIVE_FIPS202_X4+#undef MLD_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES+#undef MLD_USE_NATIVE_FIPS202_X4_XOR_BYTES+#undef mld_keccak_f1600_x4_native_impl+/* mldsa/src/fips202/native/armv81m/src/fips202_native_armv81m.h */+#undef MLD_FIPS202_NATIVE_ARMV81M_SRC_FIPS202_NATIVE_ARMV81M_H+#undef mld_keccak_f1600_x4_mve_asm+#undef mld_keccak_f1600_x4_state_extract_bytes_asm+#undef mld_keccak_f1600_x4_state_xor_bytes_asm+#undef mld_keccakf1600_round_constants+#endif /* MLD_SYS_ARMV81M_MVE */+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH)+/* mldsa/src/native/api.h */+#undef MLD_FQMUL_BOUND+#undef MLD_INTT_BOUND+#undef MLD_NATIVE_API_H+#undef MLD_NATIVE_FUNC_FALLBACK+#undef MLD_NATIVE_FUNC_SUCCESS+#undef MLD_NTT_BOUND+#undef MLD_REDUCE32_RANGE_MAX+/* mldsa/src/native/meta.h */+#undef MLD_NATIVE_META_H+#if defined(MLD_SYS_AARCH64)+/*+ * Undefine macros from native code (Arith, AArch64)+ */+/* mldsa/src/native/aarch64/meta.h */+#undef MLD_ARITH_BACKEND_AARCH64+#undef MLD_NATIVE_AARCH64_META_H+#undef MLD_USE_NATIVE_INTT+#undef MLD_USE_NATIVE_NTT+#undef MLD_USE_NATIVE_POINTWISE_MONTGOMERY+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7+#undef MLD_USE_NATIVE_POLYZ_UNPACK_17+#undef MLD_USE_NATIVE_POLYZ_UNPACK_19+#undef MLD_USE_NATIVE_POLY_CADDQ+#undef MLD_USE_NATIVE_POLY_CHKNORM+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_32+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_88+#undef MLD_USE_NATIVE_POLY_USE_HINT_32+#undef MLD_USE_NATIVE_POLY_USE_HINT_88+#undef MLD_USE_NATIVE_REJ_UNIFORM+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA2+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA4+/* mldsa/src/native/aarch64/src/arith_native_aarch64.h */+#undef MLD_AARCH64_REJ_UNIFORM_ETA2_BUFLEN+#undef MLD_AARCH64_REJ_UNIFORM_ETA4_BUFLEN+#undef MLD_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H+#undef mld_aarch64_intt_zetas_layer123456+#undef mld_aarch64_intt_zetas_layer78+#undef mld_aarch64_ntt_zetas_layer123456+#undef mld_aarch64_ntt_zetas_layer78+#undef mld_intt_aarch64_asm+#undef mld_ntt_aarch64_asm+#undef mld_poly_caddq_aarch64_asm+#undef mld_poly_chknorm_aarch64_asm+#undef mld_poly_decompose_32_aarch64_asm+#undef mld_poly_decompose_88_aarch64_asm+#undef mld_poly_pointwise_montgomery_aarch64_asm+#undef mld_poly_use_hint_32_aarch64_asm+#undef mld_poly_use_hint_88_aarch64_asm+#undef mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm+#undef mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm+#undef mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm+#undef mld_polyz_unpack_17_aarch64_asm+#undef mld_polyz_unpack_17_indices+#undef mld_polyz_unpack_19_aarch64_asm+#undef mld_polyz_unpack_19_indices+#undef mld_rej_uniform_aarch64_asm+#undef mld_rej_uniform_eta2_aarch64_asm+#undef mld_rej_uniform_eta4_aarch64_asm+#undef mld_rej_uniform_eta_table+#undef mld_rej_uniform_table+#endif /* MLD_SYS_AARCH64 */+#if defined(MLD_SYS_X86_64)+/*+ * Undefine macros from native code (Arith, X86_64)+ */+/* mldsa/src/native/x86_64/meta.h */+#undef MLD_ARITH_BACKEND_X86_64_DEFAULT+#undef MLD_NATIVE_X86_64_META_H+#undef MLD_USE_NATIVE_INTT+#undef MLD_USE_NATIVE_NTT+#undef MLD_USE_NATIVE_NTT_CUSTOM_ORDER+#undef MLD_USE_NATIVE_POINTWISE_MONTGOMERY+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7+#undef MLD_USE_NATIVE_POLYZ_UNPACK_17+#undef MLD_USE_NATIVE_POLYZ_UNPACK_19+#undef MLD_USE_NATIVE_POLY_CADDQ+#undef MLD_USE_NATIVE_POLY_CHKNORM+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_32+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_88+#undef MLD_USE_NATIVE_POLY_USE_HINT_32+#undef MLD_USE_NATIVE_POLY_USE_HINT_88+#undef MLD_USE_NATIVE_REJ_UNIFORM+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA2+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA4+/* mldsa/src/native/x86_64/src/arith_native_x86_64.h */+#undef MLD_AVX2_REJ_UNIFORM_BUFLEN+#undef MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN+#undef MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN+#undef MLD_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H+#undef mld_invntt_avx2_asm+#undef mld_ntt_avx2_asm+#undef mld_nttunpack_avx2_asm+#undef mld_pointwise_acc_l4_avx2_asm+#undef mld_pointwise_acc_l5_avx2_asm+#undef mld_pointwise_acc_l7_avx2_asm+#undef mld_pointwise_avx2_asm+#undef mld_poly_caddq_avx2_asm+#undef mld_poly_chknorm_avx2_asm+#undef mld_poly_decompose_32_avx2_asm+#undef mld_poly_decompose_88_avx2_asm+#undef mld_poly_use_hint_32_avx2_asm+#undef mld_poly_use_hint_88_avx2_asm+#undef mld_polyz_unpack_17_avx2_asm+#undef mld_polyz_unpack_19_avx2_asm+#undef mld_rej_uniform_avx2_asm+#undef mld_rej_uniform_eta2_avx2_asm+#undef mld_rej_uniform_eta4_avx2_asm+#undef mld_rej_uniform_table+/* mldsa/src/native/x86_64/src/consts.h */+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XDIV+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XDIV_QINV+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XQ+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XQINV+#undef MLD_AVX2_BACKEND_DATA_OFFSET_ZETAS+#undef MLD_AVX2_BACKEND_DATA_OFFSET_ZETAS_QINV+#undef MLD_NATIVE_X86_64_SRC_CONSTS_H+#undef mld_qdata+#endif /* MLD_SYS_X86_64 */+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_ARITH */+#endif /* !MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS */
+ cbits/mldsa/mldsa_native_config.h view
@@ -0,0 +1,855 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS140_3_IG]+ * Implementation Guidance for FIPS 140-3 and the Cryptographic Module+ * Validation Program+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/projects/cryptographic-module-validation-program/fips-140-3-ig-announcements+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ *+ * - [FIPS204_UPDATES]+ * FIPS 204 Potential Updates (Errata)+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/files/pubs/fips/204/final/docs/fips-204-potential-updates.xlsx+ */++#ifndef MLD_CONFIG_H+#define MLD_CONFIG_H++/**+ * MLD_CONFIG_PARAMETER_SET+ *+ * Specifies the parameter set for ML-DSA+ * - MLD_CONFIG_PARAMETER_SET=44 corresponds to ML-DSA-44+ * - MLD_CONFIG_PARAMETER_SET=65 corresponds to ML-DSA-65+ * - MLD_CONFIG_PARAMETER_SET=87 corresponds to ML-DSA-87+ *+ * If you want to support multiple parameter sets, build the+ * library multiple times and set MLD_CONFIG_MULTILEVEL_BUILD.+ * See MLD_CONFIG_MULTILEVEL_BUILD for how to do this while+ * minimizing code duplication.+ *+ * This can also be set using CFLAGS.+ */+#ifndef MLD_CONFIG_PARAMETER_SET+#define MLD_CONFIG_PARAMETER_SET \+ 44 /* Change this for different security strengths */+#endif++/**+ * MLD_CONFIG_FILE+ *+ * If defined, this is a header that will be included instead+ * of the default configuration file mldsa/mldsa_native_config.h.+ *+ * When you need to build mldsa-native in multiple configurations,+ * using varying MLD_CONFIG_FILE can be more convenient+ * than configuring everything through CFLAGS.+ *+ * To use, MLD_CONFIG_FILE _must_ be defined prior+ * to the inclusion of any mldsa-native headers. For example,+ * it can be set by passing `-DMLD_CONFIG_FILE="..."`+ * on the command line.+ */+/* #define MLD_CONFIG_FILE "mldsa_native_config.h" */++/**+ * MLD_CONFIG_NAMESPACE_PREFIX+ *+ * The prefix to use to namespace global symbols from mldsa/.+ *+ * In a multi-level build, level-dependent symbols will+ * additionally be prefixed with the parameter set (44/65/87).+ *+ * This can also be set using CFLAGS.+ */+#if !defined(MLD_CONFIG_NAMESPACE_PREFIX)+#define MLD_CONFIG_NAMESPACE_PREFIX MLD_DEFAULT_NAMESPACE_PREFIX+#endif++/**+ * MLD_CONFIG_MULTILEVEL_BUILD+ *+ * Set this if the build is part of a multi-level build supporting+ * multiple parameter sets.+ *+ * If you need only a single parameter set, keep this unset.+ *+ * To build mldsa-native with support for all parameter sets,+ * build it three times -- once per parameter set -- and set the+ * option MLD_CONFIG_MULTILEVEL_WITH_SHARED for exactly one of+ * them, and MLD_CONFIG_MULTILEVEL_NO_SHARED for the others.+ * MLD_CONFIG_MULTILEVEL_BUILD should be set for all of them.+ *+ * See examples/multilevel_build for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLD_CONFIG_MULTILEVEL_BUILD */++/**+ * MLD_CONFIG_EXTERNAL_API_QUALIFIER+ *+ * If set, this option provides an additional function+ * qualifier to be added to declarations of mldsa-native's+ * public API.+ *+ * The primary use case for this option are single-CU builds+ * where the public API exposed by mldsa-native is wrapped by+ * another API in the consuming application. In this case,+ * even mldsa-native's public API can be marked `static`.+ */+/* #define MLD_CONFIG_EXTERNAL_API_QUALIFIER */++/**+ * MLD_CONFIG_NO_KEYPAIR_API+ *+ * By default, mldsa-native includes support for generating key+ * pairs. If you don't need this, set MLD_CONFIG_NO_KEYPAIR_API+ * to exclude keypair, keypair_internal,+ * pk_from_sk, and all internal APIs only needed by+ * those functions.+ */+/* #define MLD_CONFIG_NO_KEYPAIR_API */++/**+ * MLD_CONFIG_NO_SIGN_API+ *+ * By default, mldsa-native includes support for creating+ * signatures. If you don't need this, set MLD_CONFIG_NO_SIGN_API+ * to exclude signature,+ * signature_extmu, signature_internal,+ * signature_pre_hash_internal,+ * signature_pre_hash_shake256, and all internal APIs+ * only needed by those functions.+ */+/* #define MLD_CONFIG_NO_SIGN_API */++/**+ * MLD_CONFIG_NO_VERIFY_API+ *+ * By default, mldsa-native includes support for verifying+ * signatures. If you don't need this, set+ * MLD_CONFIG_NO_VERIFY_API to exclude verify,+ * verify_extmu, verify_internal,+ * verify_pre_hash_internal,+ * verify_pre_hash_shake256, and all internal APIs+ * only needed by those functions.+ */+/* #define MLD_CONFIG_NO_VERIFY_API */++/**+ * MLD_CONFIG_CORE_API_ONLY+ *+ * Set this to remove all public APIs except+ * keypair_internal, signature_internal,+ * and verify_internal.+ */+/* #define MLD_CONFIG_CORE_API_ONLY */++/**+ * MLD_CONFIG_NO_RANDOMIZED_API+ *+ * If this option is set, mldsa-native will be built without the+ * randomized API functions (keypair,+ * signature, and signature_extmu).+ * This allows users to build mldsa-native without providing a+ * randombytes() implementation if they only need the+ * internal deterministic API+ * (keypair_internal, signature_internal).+ *+ * @note This option is incompatible with MLD_CONFIG_KEYGEN_PCT+ * as the current PCT implementation requires+ * signature().+ */+/* #define MLD_CONFIG_NO_RANDOMIZED_API */++/**+ * MLD_CONFIG_CONSTANTS_ONLY+ *+ * If you only need the size constants (MLDSA_PUBLICKEYBYTES, etc.)+ * but no function declarations, set MLD_CONFIG_CONSTANTS_ONLY.+ *+ * This only affects the public header mldsa_native.h, not+ * the implementation.+ */+/* #define MLD_CONFIG_CONSTANTS_ONLY */+/******************************************************************************+ *+ * Build-only configuration options+ *+ * The remaining configurations are build-options only.+ * They do not affect the API described in mldsa_native.h.+ *+ *****************************************************************************/+#if defined(MLD_BUILD_INTERNAL)++/**+ * MLD_CONFIG_MULTILEVEL_WITH_SHARED+ *+ * This is for multi-level builds of mldsa-native only. If you+ * need only a single parameter set, keep this unset.+ *+ * If this is set, all MLD_CONFIG_PARAMETER_SET-independent+ * code will be included in the build, including code needed only+ * for other parameter sets.+ *+ * Example: mld_polyw1_pack_88 is only needed for+ * MLD_CONFIG_PARAMETER_SET == 44. Yet, if this option is set for a+ * build with MLD_CONFIG_PARAMETER_SET == 65/87, it would be included.+ *+ * To build mldsa-native with support for all parameter sets,+ * build it three times -- once per parameter set -- and set the+ * option MLD_CONFIG_MULTILEVEL_WITH_SHARED for exactly one of+ * them, and MLD_CONFIG_MULTILEVEL_NO_SHARED for the others.+ * MLD_CONFIG_MULTILEVEL_BUILD should be set for all of them.+ *+ * See examples/multilevel_build for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLD_CONFIG_MULTILEVEL_WITH_SHARED */++/**+ * MLD_CONFIG_MULTILEVEL_NO_SHARED+ *+ * This is for multi-level builds of mldsa-native only. If you+ * need only a single parameter set, keep this unset.+ *+ * If this is set, no MLD_CONFIG_PARAMETER_SET-independent code+ * will be included in the build.+ *+ * To build mldsa-native with support for all parameter sets,+ * build it three times -- once per parameter set -- and set the+ * option MLD_CONFIG_MULTILEVEL_WITH_SHARED for exactly one of+ * them, and MLD_CONFIG_MULTILEVEL_NO_SHARED for the others.+ * MLD_CONFIG_MULTILEVEL_BUILD should be set for all of them.+ *+ * See examples/multilevel_build for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLD_CONFIG_MULTILEVEL_NO_SHARED */++/**+ * MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+ *+ * This is only relevant for single compilation unit (SCU)+ * builds of mldsa-native. In this case, it determines whether+ * directives defined in parameter-set-independent headers should+ * be #undef'ined or not at the end of the SCU file. This is+ * needed in multilevel builds.+ *+ * See examples/multilevel_build_native for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS */++/**+ * MLD_CONFIG_USE_NATIVE_BACKEND_ARITH+ *+ * Determines whether a native arithmetic backend should be used.+ *+ * The arithmetic backend covers performance-critical functions+ * such as the number-theoretic transform (NTT).+ *+ * If this option is unset, the C backend will be used.+ *+ * If this option is set, the arithmetic backend to be used is+ * determined by MLD_CONFIG_ARITH_BACKEND_FILE: If the latter is+ * unset, the default backend for your target architecture+ * will be used. If set, it must be the name of a backend metadata+ * file.+ *+ * This can also be set using CFLAGS.+ */+#if !defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH)+/* #define MLD_CONFIG_USE_NATIVE_BACKEND_ARITH */+#endif++/**+ * MLD_CONFIG_ARITH_BACKEND_FILE+ *+ * The arithmetic backend to use.+ *+ * If MLD_CONFIG_USE_NATIVE_BACKEND_ARITH is unset, this option+ * is ignored.+ *+ * If MLD_CONFIG_USE_NATIVE_BACKEND_ARITH is set, this option must+ * either be undefined or the filename of an arithmetic backend.+ * If unset, the default backend will be used.+ *+ * This can be set using CFLAGS.+ */+#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH) && \+ !defined(MLD_CONFIG_ARITH_BACKEND_FILE)+#define MLD_CONFIG_ARITH_BACKEND_FILE "native/meta.h"+#endif++/**+ * MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202+ *+ * Determines whether a native FIPS202 backend should be used.+ *+ * The FIPS202 backend covers 1x/2x/4x-fold Keccak-f1600, which is+ * the performance bottleneck of SHA3 and SHAKE.+ *+ * If this option is unset, the C backend will be used.+ *+ * If this option is set, the FIPS202 backend to be used is+ * determined by MLD_CONFIG_FIPS202_BACKEND_FILE: If the latter is+ * unset, the default backend for your target architecture+ * will be used. If set, it must be the name of a backend metadata+ * file.+ *+ * This can also be set using CFLAGS.+ */+#if !defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202)+/* #define MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+#endif++/**+ * MLD_CONFIG_FIPS202_BACKEND_FILE+ *+ * The FIPS-202 backend to use.+ *+ * If MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 is set, this option+ * must either be undefined or the filename of a FIPS202 backend.+ * If unset, the default backend will be used.+ *+ * This can be set using CFLAGS.+ */+#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202) && \+ !defined(MLD_CONFIG_FIPS202_BACKEND_FILE)+#define MLD_CONFIG_FIPS202_BACKEND_FILE "fips202/native/auto.h"+#endif++/**+ * MLD_CONFIG_FIPS202_CUSTOM_HEADER+ *+ * Custom header to use for FIPS-202+ *+ * This should only be set if you intend to use a custom+ * FIPS-202 implementation, different from the one shipped+ * with mldsa-native.+ *+ * If set, it must be the name of a file serving as the+ * replacement for mldsa/src/fips202/fips202.h, and exposing+ * the same API (see FIPS202.md).+ */+/* #define MLD_CONFIG_FIPS202_CUSTOM_HEADER "SOME_FILE.h" */++/**+ * MLD_CONFIG_FIPS202X4_CUSTOM_HEADER+ *+ * Custom header to use for FIPS-202-X4+ *+ * This should only be set if you intend to use a custom+ * FIPS-202 implementation, different from the one shipped+ * with mldsa-native.+ *+ * If set, it must be the name of a file serving as the+ * replacement for mldsa/src/fips202/fips202x4.h, and exposing+ * the same API (see FIPS202.md).+ */+/* #define MLD_CONFIG_FIPS202X4_CUSTOM_HEADER "SOME_FILE.h" */++/**+ * MLD_CONFIG_CUSTOM_ZEROIZE+ *+ * In compliance with @[FIPS204, Section 3.6.3], mldsa-native zeroizes+ * intermediate buffers before returning from function calls. By default,+ * those buffers are allocated from the stack; if MLD_CONFIG_CUSTOM_ALLOC_FREE+ * is set, they are (mostly -- few exceptions remain at present) allocated from+ * the configured custom allocator.+ *+ * mldsa-native also zeroizes caller-owned output buffers as needed to uphold+ * the API convention that outputs be either unmodified or zeroized upon+ * failure.+ *+ * Set this option and define `mld_zeroize` if you want to use a custom+ * method to zeroize intermediate and output buffers.+ *+ * The default implementation uses SecureZeroMemory on Windows and a+ * memset + compiler barrier otherwise. If neither of those is available on+ * the target platform, compilation will fail, and you will need to use+ * MLD_CONFIG_CUSTOM_ZEROIZE to provide a custom implementation of+ * `mld_zeroize()`.+ *+ * @warning+ * The zeroization conducted by mldsa-native reduces the likelihood of data+ * leaking on the stack or custom allocators, but it does not eliminate it.+ * For example, the C standard makes no guarantee about where a compiler+ * allocates local structures and whether/where it makes copies of them.+ * Also, in addition to entire structures, there may also be potentially+ * exploitable leakage of individual values on the stack. If you need+ * bullet-proof zeroization of the stack, you need to consider additional+ * measures instead of what this feature provides. In this case, you can+ * set mld_zeroize to a no-op. Note that in this case you are also responsible+ * for zeroizing output buffers upon failure.+ */+/* #define MLD_CONFIG_CUSTOM_ZEROIZE+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/src.h"+ static MLD_INLINE void mld_zeroize(void *ptr, size_t len)+ {+ ... your implementation ...+ }+ #endif+*/++/**+ * MLD_CONFIG_CUSTOM_RANDOMBYTES+ *+ * mldsa-native does not provide a secure randombytes+ * implementation. Such an implementation has to be provided by+ * the consumer.+ *+ * If this option is not set, mldsa-native expects a function+ * int randombytes(uint8_t *out, size_t outlen).+ *+ * Set this option and define `mld_randombytes` if you want to+ * use a custom method to sample randombytes with a different name+ * or signature.+ */+/* #define MLD_CONFIG_CUSTOM_RANDOMBYTES+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/src.h"+ static MLD_INLINE int mld_randombytes(uint8_t *ptr, size_t len)+ {+ ... your implementation ...+ return 0;+ }+ #endif+*/++/**+ * MLD_CONFIG_CUSTOM_CAPABILITY_FUNC+ *+ * mldsa-native backends may rely on specific hardware features.+ * Those backends will only be included in an mldsa-native build+ * if support for the respective features is enabled at+ * compile-time. However, when building for a heterogeneous set+ * of CPUs to run the resulting binary/library on, feature+ * detection at _runtime_ is needed to decide whether a backend+ * can be used or not.+ *+ * Set this option and define `mld_sys_check_capability` if you+ * want to use a custom method to dispatch between implementations.+ *+ * Return value 1 indicates that a capability is supported.+ * Return value 0 indicates that a capability is not supported.+ *+ * If this option is not set, mldsa-native uses compile-time+ * feature detection only to decide which backend to use.+ *+ * If you compile mldsa-native on a system with different+ * capabilities than the system that the resulting binary/library+ * will be run on, you must use this option.+ */+/* #define MLD_CONFIG_CUSTOM_CAPABILITY_FUNC+ static MLD_INLINE int mld_sys_check_capability(mld_sys_cap cap)+ {+ ... your implementation ...+ }+*/++/**+ * MLD_CONFIG_CUSTOM_ALLOC_FREE+ *+ * Set this option and define `MLD_CUSTOM_ALLOC` and+ * `MLD_CUSTOM_FREE` if you want to use custom allocation for+ * large local structures or buffers.+ *+ * By default, all buffers/structures are allocated on the stack.+ * If this option is set, most of them will be allocated via+ * MLD_CUSTOM_ALLOC.+ *+ * Parameters to MLD_CUSTOM_ALLOC:+ * - T* v: Target pointer to declare.+ * - T: Type of structure to be allocated+ * - N: Number of elements to be allocated.+ *+ * Parameters to MLD_CUSTOM_FREE:+ * - T* v: Target pointer to free. May be NULL.+ * - T: Type of structure to be freed.+ * - N: Number of elements to be freed.+ *+ * @warning This option is experimental. Its scope, configuration and+ * function/macro signatures may change at any time. We expect a+ * stable API in a future version.+ *+ * @note Even if this option is set, some allocations further down+ * the call stack will still be made from the stack. Those will+ * likely be added to the scope of this option in the future.+ *+ * @note MLD_CUSTOM_ALLOC need not guarantee a successful+ * allocation nor include error handling. Upon failure, the+ * target pointer should simply be set to NULL. The calling+ * code will handle this case and invoke MLD_CUSTOM_FREE.+ */+/* #define MLD_CONFIG_CUSTOM_ALLOC_FREE+ #if !defined(__ASSEMBLER__)+ #include <stdlib.h>+ #define MLD_CUSTOM_ALLOC(v, T, N) \+ T* (v) = (T *)aligned_alloc(MLD_DEFAULT_ALIGN, \+ MLD_ALIGN_UP(sizeof(T) * (N)))+ #define MLD_CUSTOM_FREE(v, T, N) free(v)+ #endif+*/++/**+ * MLD_CONFIG_CUSTOM_MEMCPY+ *+ * Set this option and define `mld_memcpy` if you want to+ * use a custom method to copy memory instead of the standard+ * library memcpy function.+ *+ * The custom implementation must have the same signature and+ * behavior as the standard memcpy function:+ * void *mld_memcpy(void *dest, const void *src, size_t n)+ */+/* #define MLD_CONFIG_CUSTOM_MEMCPY+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/src.h"+ static MLD_INLINE void *mld_memcpy(void *dest, const void *src, size_t n)+ {+ ... your implementation ...+ }+ #endif+*/++/**+ * MLD_CONFIG_CUSTOM_MEMSET+ *+ * Set this option and define `mld_memset` if you want to+ * use a custom method to set memory instead of the standard+ * library memset function.+ *+ * The custom implementation must have the same signature and+ * behavior as the standard memset function:+ * void *mld_memset(void *s, int c, size_t n)+ */+/* #define MLD_CONFIG_CUSTOM_MEMSET+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/src.h"+ static MLD_INLINE void *mld_memset(void *s, int c, size_t n)+ {+ ... your implementation ...+ }+ #endif+*/++/**+ * MLD_CONFIG_INTERNAL_API_QUALIFIER+ *+ * If set, this option provides an additional qualifier+ * to be added to declarations of internal API functions and data.+ *+ * The primary use case for this option are single-CU builds,+ * in which case this option can be set to `static`.+ */+/* #define MLD_CONFIG_INTERNAL_API_QUALIFIER */++/**+ * MLD_CONFIG_CT_TESTING_ENABLED+ *+ * If set, mldsa-native annotates data as secret / public using+ * valgrind's annotations VALGRIND_MAKE_MEM_UNDEFINED and+ * VALGRIND_MAKE_MEM_DEFINED, enabling various checks for secret-+ * dependent control flow of variable time execution (depending+ * on the exact version of valgrind installed).+ */+/* #define MLD_CONFIG_CT_TESTING_ENABLED */++/**+ * MLD_CONFIG_NO_ASM+ *+ * If this option is set, mldsa-native will be built without+ * use of native code or inline assembly.+ *+ * By default, inline assembly is used to implement value barriers.+ * Without inline assembly, mldsa-native will use a global volatile+ * 'opt blocker' instead; see ct.h.+ *+ * Inline assembly is also used to implement a secure zeroization+ * function on non-Windows platforms. If this option is set and+ * the target platform is not Windows, you MUST set+ * MLD_CONFIG_CUSTOM_ZEROIZE and provide a custom zeroization+ * function.+ *+ * If this option is set, MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 and+ * MLD_CONFIG_USE_NATIVE_BACKEND_ARITH will be ignored, and no+ * native backends will be used.+ */+/* #define MLD_CONFIG_NO_ASM */++/**+ * MLD_CONFIG_NO_ASM_VALUE_BARRIER+ *+ * If this option is set, mldsa-native will be built without+ * use of native code or inline assembly for value barriers.+ *+ * By default, inline assembly (if available) is used to implement+ * value barriers.+ * Without inline assembly, mldsa-native will use a global volatile+ * 'opt blocker' instead; see ct.h.+ */+/* #define MLD_CONFIG_NO_ASM_VALUE_BARRIER */++/**+ * MLD_CONFIG_KEYGEN_PCT+ *+ * Compliance with @[FIPS140_3_IG, p.87] requires a+ * Pairwise Consistency Test (PCT) to be carried out on a freshly+ * generated keypair before it can be exported.+ *+ * Set this option if such a check should be implemented.+ * In this case, keypair_internal and+ * keypair will return MLD_ERR_PCT_FAIL if the+ * PCT failed.+ *+ * @note This feature will drastically lower the performance of+ * key generation.+ *+ * @note This option is incompatible with MLD_CONFIG_NO_SIGN_API+ * and MLD_CONFIG_NO_VERIFY_API as the current PCT implementation+ * requires signature() and verify().+ */+/* #define MLD_CONFIG_KEYGEN_PCT */++/**+ * MLD_CONFIG_KEYGEN_PCT_BREAKAGE_TEST+ *+ * If this option is set, the user must provide a runtime+ * function `static inline int mld_break_pct() { ... }` to+ * indicate whether the PCT should be made fail.+ *+ * This option only has an effect if MLD_CONFIG_KEYGEN_PCT is set.+ */+/* #define MLD_CONFIG_KEYGEN_PCT_BREAKAGE_TEST+ #if !defined(__ASSEMBLER__)+ #include "src/src.h"+ static MLD_INLINE int mld_break_pct(void)+ {+ ... return 0/1 depending on whether PCT should be broken ...+ }+ #endif+*/++/**+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ *+ * Upper bound on the number of rejection-sampling iterations+ * performed by ML-DSA signing (@[FIPS204, Algorithm 7]).+ *+ * If a valid signature is not produced within this many+ * attempts, signing returns MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED.+ * This is useful in timing-sensitive environments that+ * require a deterministic worst-case bound on signing time.+ *+ * For FIPS 204 compliance, this value MUST be at least 821,+ * cf. @[FIPS204, Appendix C] and @[FIPS204_UPDATES], which is+ * chosen so that the signing failure rate is < 2^{-256}.+ *+ * Default: Largest possible value before internal counters+ * would overflow. This is larger than the FIPS204 bound.+ *+ * In particular, in the default configuration, the signing+ * failure rate is < 2^{-256}.+ */+/* #define MLD_CONFIG_MAX_SIGNING_ATTEMPTS 821 */++/**+ * MLD_CONFIG_SERIAL_FIPS202_ONLY+ *+ * Set this to use a FIPS202 implementation with global state+ * that supports only one active Keccak computation at a time+ * (e.g. some hardware accelerators).+ *+ * If this option is set, ML-DSA will use FIPS202 operations+ * serially, ensuring that only one SHAKE context is active+ * at any given time.+ *+ * This allows offloading Keccak computations to a hardware+ * accelerator that holds only a single Keccak state locally,+ * rather than requiring support for multiple concurrent+ * Keccak states.+ *+ * @note Depending on the target CPU, this may reduce+ * performance when using software FIPS202 implementations.+ * Only enable this when you have to.+ */+/* #define MLD_CONFIG_SERIAL_FIPS202_ONLY */++/**+ * MLD_CONFIG_CONTEXT_PARAMETER+ *+ * Set this to add a caller-supplied context parameter to the public API+ * functions, which is then forwarded unchanged to the custom callbacks+ * (allocation, and signing hooks below).+ *+ * When this option is set, every public API function gains a trailing+ * parameter+ *+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+ *+ * as its last argument; its type is configured via+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE (see below). mldsa-native treats this+ * value as opaque: it never dereferences it and only passes it on to the+ * configurable hook macros. It is meant to carry per-caller state -- e.g. a+ * pointer to a memory pool for the allocation hooks, or the resume state for+ * the signing hooks -- into those hooks.+ *+ * When this option is unset (the default), no extra parameter is added and+ * the hook macros never receive a context argument.+ *+ * The hooks that receive the context are the allocation hooks (see+ * MLD_CONFIG_CUSTOM_ALLOC_FREE) and the signing hooks (see+ * MLD_CONFIG_SIGN_HOOK_RESUME / _ATTEMPT / _FINISH); each is documented with+ * its own option below.+ */+/* #define MLD_CONFIG_CONTEXT_PARAMETER */++/**+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE+ *+ * Set this to define the type of the context parameter added by+ * MLD_CONFIG_CONTEXT_PARAMETER. It can be any C type usable as a function+ * parameter, e.g. `void *` or a pointer to a caller-defined struct such as+ * `struct my_ctx *`.+ *+ * This option must be defined if and only if MLD_CONFIG_CONTEXT_PARAMETER is+ * defined; defining one without the other is a compile-time error.+ */+/* #define MLD_CONFIG_CONTEXT_PARAMETER_TYPE void* */++/**+ * Signing hooks: MLD_CONFIG_SIGN_HOOK_RESUME / _ATTEMPT / _FINISH+ *+ * Three optional, independent hooks into the ML-DSA signing rejection-sampling+ * loop. Each is enabled by defining the matching option, in which case the+ * integration must provide the corresponding function. If a hook needs+ * per-operation state, enable MLD_CONFIG_CONTEXT_PARAMETER; the context is then+ * appended as the last argument.+ *+ * @warning This feature is experimental. Its scope, configuration and+ * function signatures may change at any time, including after v2.+ *+ * Enabling any of the hooks requires MLD_CONFIG_NO_RANDOMIZED_API (restricting+ * the public API to deterministic operations). This is because the restartable+ * signing as enabled by the signing hooks only produces the uninterrupted+ * signature when the randomness is fixed across calls. A logging-only use+ * (attempt always returns 0; resume/finish merely observe) would be safe with+ * the randomized API too, but for now the requirement is imposed uniformly on+ * all three hooks.+ *+ * Note: Randomized signing is a shim wrapper around deterministic signing, and+ * all helper functions you need to build it are exposed publicly. Thus, if you+ * need a restartable, randomized signing operation, you can build your own by+ * replicating the logic and adding the RNG seed to the restart context. In this+ * case, please also consider letting the mldsa-native maintainers know of your+ * need for randomized, restartable signing, so the feature can be appropriately+ * prioritized.+ *+ * - MLD_CONFIG_SIGN_HOOK_ATTEMPT: int mld_sign_hook_attempt(attempt[, ctxt])+ * Called before each attempt. Returns 0 to proceed, or non-zero to pause:+ * signing then returns MLD_ERR_SIGNING_PAUSED with `attempt` as the resume+ * point (needs MLD_CONFIG_SIGN_HOOK_RESUME to resume; otherwise just aborts).+ * Always returning 0 makes it a logging/benchmarking hook.+ *+ * - MLD_CONFIG_SIGN_HOOK_RESUME: uint16_t mld_sign_hook_resume([ctxt])+ * Returns the attempt to resume from (0 for a fresh operation), i.e. the one+ * recorded when a previous call paused.+ *+ * - MLD_CONFIG_SIGN_HOOK_FINISH: void mld_sign_hook_finish(attempt[, ctxt])+ * Called on success with the succeeding attempt. Observe-only.+ *+ * When an option is unset, the hook is a no-op (resume to 0, attempt proceeds),+ * i.e. ordinary one-shot signing.+ *+ * Independent of MLD_CONFIG_MAX_SIGNING_ATTEMPTS, which is a static upper bound+ * on the number of signing attempts.+ *+ * See test/src/test_sign_hook.c for a worked example using all three.+ */+/* #define MLD_CONFIG_SIGN_HOOK_RESUME+ #define MLD_CONFIG_SIGN_HOOK_ATTEMPT+ #define MLD_CONFIG_SIGN_HOOK_FINISH+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/sys.h"+ static MLD_INLINE uint16_t mld_sign_hook_resume(void)+ {+ ... return the attempt to resume from ...+ }+ static MLD_INLINE int mld_sign_hook_attempt(uint16_t attempt)+ {+ ... return non-zero to pause here; for resume, store attempt ...+ return 0;+ }+ static MLD_INLINE void mld_sign_hook_finish(uint16_t attempt)+ {+ ... mark the operation complete (attempt = successful attempt) ...+ }+ #endif+*/++/**+ * MLD_CONFIG_REDUCE_RAM+ *+ * Set this to reduce RAM usage. This trades memory for performance.+ *+ * For expected memory usage, see the MLD_TOTAL_ALLOC_* constants defined in+ * mldsa_native.h.+ *+ * This option is useful for embedded systems with tight RAM constraints but+ * relaxed performance requirements.+ *+ */+/* #define MLD_CONFIG_REDUCE_RAM */++/************************* Config internals ********************************/++#endif /* MLD_BUILD_INTERNAL */++/* Default namespace+ *+ * Don't change this. If you need a different namespace, re-define+ * MLD_CONFIG_NAMESPACE_PREFIX above instead, and remove the following.+ *+ * The default MLDSA namespace is+ *+ * PQCP_MLDSA_NATIVE_MLDSA<LEVEL>_+ *+ * e.g., PQCP_MLDSA_NATIVE_MLDSA44_+ */++#if defined(MLD_CONFIG_MULTILEVEL_BUILD)+/* In a multi-level build the parameter set is appended by the namespacing+ * machinery, so the default prefix must not embed it. */+#define MLD_DEFAULT_NAMESPACE_PREFIX PQCP_MLDSA_NATIVE_MLDSA+#elif MLD_CONFIG_PARAMETER_SET == 44+#define MLD_DEFAULT_NAMESPACE_PREFIX PQCP_MLDSA_NATIVE_MLDSA44+#elif MLD_CONFIG_PARAMETER_SET == 65+#define MLD_DEFAULT_NAMESPACE_PREFIX PQCP_MLDSA_NATIVE_MLDSA65+#elif MLD_CONFIG_PARAMETER_SET == 87+#define MLD_DEFAULT_NAMESPACE_PREFIX PQCP_MLDSA_NATIVE_MLDSA87+#endif++#endif /* !MLD_CONFIG_H */
+ cbits/mldsa/src/cbmc.h view
@@ -0,0 +1,233 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_CBMC_H+#define MLD_CBMC_H++/***************************************************+ * Basic replacements for __CPROVER_XXX contracts+ ***************************************************/+/*+ * The `__contract__` / `__loop__` annotation macros use a+ * leading-double-underscore spelling in line with other CBMC macros.+ * clang-tidy flags these as reserved identifiers; we suppress the diagnostic+ * at each definition site (NOLINT) rather than disabling the check globally,+ * so it stays active for the rest of the tree.+ */+#ifndef CBMC++/* clang-format off */+#define __contract__(x) /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+#define __loop__(x) /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+/* clang-format on */+#define cassert(x)++#else /* !CBMC */+++/* clang-format off */+#define __contract__(x) x /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+#define __loop__(x) x /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+/* clang-format on */++/* Conditionally expand to __VA_ARGS__ depending on MLD_CONFIG_REDUCE_RAM. */+#if defined(MLD_CONFIG_REDUCE_RAM)+#define MLD_IF_REDUCE_RAM(...) __VA_ARGS__+#define MLD_IF_NOT_REDUCE_RAM(...)+#else+#define MLD_IF_REDUCE_RAM(...)+#define MLD_IF_NOT_REDUCE_RAM(...) __VA_ARGS__+#endif++/* https://diffblue.github.io/cbmc/contracts-assigns.html */+#define assigns(...) __CPROVER_assigns(__VA_ARGS__)++/* https://diffblue.github.io/cbmc/contracts-requires-ensures.html */+#define requires(...) __CPROVER_requires(__VA_ARGS__)+#define ensures(...) __CPROVER_ensures(__VA_ARGS__)+/* https://diffblue.github.io/cbmc/contracts-loops.html */+#define invariant(...) __CPROVER_loop_invariant(__VA_ARGS__)+#define decreases(...) __CPROVER_decreases(__VA_ARGS__)+/* cassert to avoid confusion with in-built assert */+#define cassert(x) __CPROVER_assert(x, "cbmc assertion failed")+#define assume(...) __CPROVER_assume(__VA_ARGS__)++/***************************************************+ * Macros for "expression" forms that may appear+ * _inside_ top-level contracts.+ ***************************************************/++/*+ * function return value - useful inside ensures+ * https://diffblue.github.io/cbmc/contracts-functions.html+ */+#define return_value (__CPROVER_return_value)++/*+ * assigns l-value targets+ * https://diffblue.github.io/cbmc/contracts-assigns.html+ */+#define object_whole(...) __CPROVER_object_whole(__VA_ARGS__)+#define memory_slice(...) __CPROVER_object_upto(__VA_ARGS__)+#define same_object(...) __CPROVER_same_object(__VA_ARGS__)++/*+ * Pointer-related predicates+ * https://diffblue.github.io/cbmc/contracts-memory-predicates.html+ */+#define memory_no_alias(...) __CPROVER_is_fresh(__VA_ARGS__)+#define readable(...) __CPROVER_r_ok(__VA_ARGS__)+#define writeable(...) __CPROVER_w_ok(__VA_ARGS__)++/* Maximum supported buffer size+ *+ * Larger buffers may be supported, but due to internal modeling constraints+ * in CBMC, the proofs of memory- and type-safety won't be able to run.+ *+ * If you find yourself in need for a buffer size larger than this,+ * please contact the maintainers, so we can prioritize work to relax+ * this somewhat artificial bound.+ */+#define MLD_MAX_BUFFER_SIZE (SIZE_MAX >> 12)+++/*+ * History variables+ * https://diffblue.github.io/cbmc/contracts-history-variables.html+ */+#define old(...) __CPROVER_old(__VA_ARGS__)+#define loop_entry(...) __CPROVER_loop_entry(__VA_ARGS__)++/*+ * Quantifiers+ * Note that the range on qvar is _exclusive_ between qvar_lb .. qvar_ub+ * https://diffblue.github.io/cbmc/contracts-quantifiers.html+ *+ * The quantified variable is declared as uint32_t, so these macros+ * quantify only over indices in [0, UINT32_MAX). Bounds larger than+ * UINT32_MAX (4 GiB) are NOT supported: the explicit (uint32_t) casts+ * on the bounds will trigger CBMC's conversion check if a wider bound+ * (e.g. a size_t > UINT32_MAX) is passed.+ *+ * Quantifying over size_t (64-bit) was found to blow up SMT proof+ * times, so we deliberately keep the index width at 32 bits. Callers+ * dealing with size_t-typed buffers must add an explicit+ * requires(len <= UINT32_MAX)+ * precondition.+ */++/*+ * Prevent clang-format from corrupting CBMC's special ==> operator+ */+/* clang-format off */+#define forall(qvar, qvar_lb, qvar_ub, predicate) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> (predicate) \+ }++#define exists(qvar, qvar_lb, qvar_ub, predicate) \+ __CPROVER_exists \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) && (predicate) \+ }+/* clang-format on */++/***************************************************+ * Convenience macros for common contract patterns+ ***************************************************/+/*+ * Prevent clang-format from corrupting CBMC's special ==> operator+ */+/* clang-format off */+#define CBMC_CONCAT_(left, right) left##right+#define CBMC_CONCAT(left, right) CBMC_CONCAT_(left, right)++#define array_bound_core(qvar, qvar_lb, qvar_ub, array_var, \+ value_lb, value_ub) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ (((int)(value_lb) <= ((array_var)[(qvar)])) && \+ (((array_var)[(qvar)]) < (int)(value_ub))) \+ }++#define array_bound(array_var, qvar_lb, qvar_ub, value_lb, value_ub) \+ array_bound_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), (qvar_lb), \+ (qvar_ub), (array_var), (value_lb), (value_ub))++#define array_unchanged_core(qvar, qvar_lb, qvar_ub, array_var) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ ((array_var)[(qvar)]) == (old(* (int32_t (*)[(qvar_ub)])(array_var)))[(qvar)] \+ }++#define array_unchanged(array_var, N) \+ array_unchanged_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), 0, (N), (array_var))++#define array_unchanged_u64_core(qvar, qvar_lb, qvar_ub, array_var) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ ((array_var)[(qvar)]) == (old(* (uint64_t (*)[(qvar_ub)])(array_var)))[(qvar)] \+ }++#define array_unchanged_u64(array_var, N) \+ array_unchanged_u64_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), 0, (N), (array_var))++#define array_unchanged_u8_core(qvar, qvar_lb, qvar_ub, array_var) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ ((array_var)[(qvar)]) == (old(* (uint8_t (*)[(qvar_ub)])(array_var)))[(qvar)] \+ }++#define array_unchanged_u8(array_var, N) \+ array_unchanged_u8_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), 0, (N), (array_var))++#define array_zeroized_u8_core(qvar, qvar_lb, qvar_ub, array_var) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ ((array_var)[(qvar)]) == 0 \+ }++#define array_zeroized_u8(array_var, N) \+ array_zeroized_u8_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), 0, (N), (array_var))+/* clang-format on */++/*+ * Output-buffer discipline on failure, as documented in API-CONVENTIONS.md:+ * when a function fails, each caller-owned output buffer is left either+ * fully unchanged or fully zeroized -- never holding partially computed or+ * stale data that could be mistaken for a valid result.+ *+ * Note the disjunction is over the buffer as a whole: it is not enough for+ * each byte to be individually either unchanged or zero.+ */+#define array_unchanged_or_zeroized_u8(array_var, N) \+ (array_unchanged_u8((array_var), (N)) || array_zeroized_u8((array_var), (N)))++/* Wrapper around array_bound operating on absolute values.+ *+ * The absolute value bound `k` is exclusive.+ *+ * Note that since the lower bound in array_bound is inclusive, we have to+ * raise it by 1 here.+ */+#define array_abs_bound(arr, lb, ub, k) \+ array_bound((arr), (lb), (ub), -((int)(k)) + 1, (k))++#endif /* CBMC */++#endif /* !MLD_CBMC_H */
+ cbits/mldsa/src/common.h view
@@ -0,0 +1,301 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_COMMON_H+#define MLD_COMMON_H++#ifndef __ASSEMBLER__+#include <stdint.h>+#endif+++#define MLD_BUILD_INTERNAL++#if defined(MLD_CONFIG_FILE)+#include MLD_CONFIG_FILE+#else+#include "mldsa_native_config.h"+#endif++#include "params.h"+#include "sys.h"++/* Internal and public API have external linkage by default, but+ * this can be overwritten by the user, e.g. for single-CU builds. */+#if !defined(MLD_CONFIG_INTERNAL_API_QUALIFIER)+#define MLD_INTERNAL_API+#define MLD_INTERNAL_DATA_DECLARATION extern+#define MLD_INTERNAL_DATA_DEFINITION+#else+#define MLD_INTERNAL_API MLD_CONFIG_INTERNAL_API_QUALIFIER+#define MLD_INTERNAL_DATA_DECLARATION MLD_CONFIG_INTERNAL_API_QUALIFIER+#define MLD_INTERNAL_DATA_DEFINITION MLD_CONFIG_INTERNAL_API_QUALIFIER+#endif++#if !defined(MLD_CONFIG_EXTERNAL_API_QUALIFIER)+#define MLD_EXTERNAL_API+#else+#define MLD_EXTERNAL_API MLD_CONFIG_EXTERNAL_API_QUALIFIER+#endif++#if defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) || \+ defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED)+#define MLD_MULTILEVEL_BUILD+#endif++#define MLD_CONCAT_(x1, x2) x1##x2+#define MLD_CONCAT(x1, x2) MLD_CONCAT_(x1, x2)++#if defined(MLD_MULTILEVEL_BUILD)+#define MLD_ADD_PARAM_SET(s) MLD_CONCAT(s, MLD_CONFIG_PARAMETER_SET)+#else+#define MLD_ADD_PARAM_SET(s) s+#endif++#define MLD_NAMESPACE_PREFIX MLD_CONCAT(MLD_CONFIG_NAMESPACE_PREFIX, _)+#define MLD_NAMESPACE_PREFIX_KL \+ MLD_CONCAT(MLD_ADD_PARAM_SET(MLD_CONFIG_NAMESPACE_PREFIX), _)++/* Functions are prefixed by MLD_CONFIG_NAMESPACE_PREFIX.+ *+ * If multiple parameter sets are used, functions depending on the parameter+ * set are additionally prefixed with 44/65/87. See mldsa_native_config.h.+ *+ * Example: If MLD_CONFIG_NAMESPACE_PREFIX is PQCP_MLDSA_NATIVE, then+ * MLD_NAMESPACE_KL(keypair) becomes PQCP_MLDSA_NATIVE44_keypair/+ * PQCP_MLDSA_NATIVE65_keypair/PQCP_MLDSA_NATIVE87_keypair.+ */+#define MLD_NAMESPACE(s) MLD_CONCAT(MLD_NAMESPACE_PREFIX, s)+#define MLD_NAMESPACE_KL(s) MLD_CONCAT(MLD_NAMESPACE_PREFIX_KL, s)++/* On Apple platforms, we need to emit leading underscore+ * in front of assembly symbols. We thus introduce a separate+ * namespace wrapper for ASM symbols. */+#if !defined(__APPLE__)+#define MLD_ASM_NAMESPACE(sym) MLD_NAMESPACE(sym)+#else+#define MLD_ASM_NAMESPACE(sym) MLD_CONCAT(_, MLD_NAMESPACE(sym))+#endif++/*+ * On X86_64 if control-flow protections (CET) are enabled (through+ * -fcf-protection=), we add an endbr64 instruction at every global function+ * label. See sys.h for more details+ */+#if defined(MLD_SYS_X86_64)+#define MLD_ASM_FN_SYMBOL(sym) MLD_ASM_NAMESPACE(sym) : MLD_CET_ENDBR+#elif defined(MLD_SYS_ARMV81M_MVE)+/* clang-format off */+#define MLD_ASM_FN_SYMBOL(sym) \+ .type MLD_ASM_NAMESPACE(sym), %function; \+ MLD_ASM_NAMESPACE(sym) :+/* clang-format on */+#else /* !MLD_SYS_X86_64 && MLD_SYS_ARMV81M_MVE */+#define MLD_ASM_FN_SYMBOL(sym) MLD_ASM_NAMESPACE(sym) :+#endif /* !MLD_SYS_X86_64 && !MLD_SYS_ARMV81M_MVE */++/*+ * Output the size of an assembly function.+ */+#if defined(__ELF__)+#define MLD_ASM_FN_SIZE(sym) \+ .size MLD_ASM_NAMESPACE(sym), .- MLD_ASM_NAMESPACE(sym)+#else+#define MLD_ASM_FN_SIZE(sym)+#endif++/* We aim to simplify the user's life by supporting builds where+ * all source files are included, even those that are not needed.+ * Those files are appropriately guarded and will be empty when unneeded.+ * The following is to avoid compilers complaining about this. */+#define MLD_EMPTY_CU(s) extern int MLD_NAMESPACE_KL(empty_cu_##s);++/* MLD_CONFIG_NO_ASM takes precedence over MLD_USE_NATIVE_XXX */+#if defined(MLD_CONFIG_NO_ASM)+#undef MLD_CONFIG_USE_NATIVE_BACKEND_ARITH+#undef MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202+#endif++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH) && \+ !defined(MLD_CONFIG_ARITH_BACKEND_FILE)+#error Bad configuration: MLD_CONFIG_USE_NATIVE_BACKEND_ARITH is set, but MLD_CONFIG_ARITH_BACKEND_FILE is not.+#endif++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202) && \+ !defined(MLD_CONFIG_FIPS202_BACKEND_FILE)+#error Bad configuration: MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 is set, but MLD_CONFIG_FIPS202_BACKEND_FILE is not.+#endif++#if defined(MLD_CONFIG_NO_RANDOMIZED_API) && defined(MLD_CONFIG_KEYGEN_PCT)+#error Bad configuration: MLD_CONFIG_NO_RANDOMIZED_API is incompatible with MLD_CONFIG_KEYGEN_PCT as the current PCT implementation requires signature()+#endif++#if defined(MLD_CONFIG_NO_SIGN_API) && defined(MLD_CONFIG_KEYGEN_PCT)+#error Bad configuration: MLD_CONFIG_NO_SIGN_API is incompatible with MLD_CONFIG_KEYGEN_PCT as the current PCT implementation requires signature()+#endif++#if defined(MLD_CONFIG_NO_VERIFY_API) && defined(MLD_CONFIG_KEYGEN_PCT)+#error Bad configuration: MLD_CONFIG_NO_VERIFY_API is incompatible with MLD_CONFIG_KEYGEN_PCT as the current PCT implementation requires verify()+#endif++#if defined(MLD_CONFIG_CORE_API_ONLY) && defined(MLD_CONFIG_KEYGEN_PCT)+#error Bad configuration: MLD_CONFIG_CORE_API_ONLY is incompatible with MLD_CONFIG_KEYGEN_PCT as the current PCT implementation requires signature() and verify()+#endif++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH)+#include MLD_CONFIG_ARITH_BACKEND_FILE+/* Include to enforce consistency of API and implementation,+ * and conduct sanity checks on the backend.+ *+ * Keep this _after_ the inclusion of the backend; otherwise,+ * the sanity checks won't have an effect. */+#if defined(MLD_CHECK_APIS) && !defined(__ASSEMBLER__)+#include "native/api.h"+#endif+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_ARITH */++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202)+#include MLD_CONFIG_FIPS202_BACKEND_FILE+/* Include to enforce consistency of API and implementation,+ * and conduct sanity checks on the backend.+ *+ * Keep this _after_ the inclusion of the backend; otherwise,+ * the sanity checks won't have an effect. */+#if defined(MLD_CHECK_APIS) && !defined(__ASSEMBLER__)+#include "fips202/native/api.h"+#endif+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 */++#if !defined(MLD_CONFIG_FIPS202_CUSTOM_HEADER)+#define MLD_FIPS202_HEADER_FILE "fips202/fips202.h"+#else+#define MLD_FIPS202_HEADER_FILE MLD_CONFIG_FIPS202_CUSTOM_HEADER+#endif++#if !defined(MLD_CONFIG_FIPS202X4_CUSTOM_HEADER)+#define MLD_FIPS202X4_HEADER_FILE "fips202/fips202x4.h"+#else+#define MLD_FIPS202X4_HEADER_FILE MLD_CONFIG_FIPS202X4_CUSTOM_HEADER+#endif++/* Standard library function replacements */+#if !defined(__ASSEMBLER__)+#if !defined(MLD_CONFIG_CUSTOM_MEMCPY)+#include <string.h>+#define mld_memcpy memcpy+#endif++#if !defined(MLD_CONFIG_CUSTOM_MEMSET)+#include <string.h>+#define mld_memset memset+#endif++/* Allocation macros for large local structures+ *+ * MLD_ALLOC(v, T, N) declares T *v and attempts to point it to an T[N]+ * MLD_FREE(v, T, N) zeroizes and frees the allocation+ *+ * Default implementation uses stack allocation.+ * Can be overridden by setting the config option MLD_CONFIG_CUSTOM_ALLOC_FREE+ * and defining MLD_CUSTOM_ALLOC and MLD_CUSTOM_FREE.+ */+#if defined(MLD_CONFIG_CUSTOM_ALLOC_FREE) != \+ (defined(MLD_CUSTOM_ALLOC) && defined(MLD_CUSTOM_FREE))+#error Bad configuration: MLD_CONFIG_CUSTOM_ALLOC_FREE must be set together with MLD_CUSTOM_ALLOC and MLD_CUSTOM_FREE+#endif++/* Context-parameter machinery (MLD_CONTEXT_PARAMETERS_n and related config+ * checks). Kept in a separate, level-generic header for readability; included+ * here so it is available to the allocation macros below and to all consumers+ * of common.h. */+#include "context.h"++#if !defined(MLD_CONFIG_CUSTOM_ALLOC_FREE)+/* Default: stack allocation */++/* This is a declaration macro, not an expression macro: T is a type and v is+ * a declarator, neither of which can be wrapped in parentheses. The+ * bugprone-macro-parentheses diagnostic is therefore a false positive here. */+#define MLD_ALLOC(v, T, N, context) \+ MLD_ALIGN T mld_alloc_##v[N]; \+ T *v = mld_alloc_##v /* NOLINT(bugprone-macro-parentheses) */++/* The MLD_FREE macro body references mld_zeroize(), which is declared in+ * ct.h. We deliberately do NOT include ct.h here: doing so would create a+ * circular dependency (ct.h includes common.h), and common.h itself never+ * calls mld_zeroize() -- only the macro expansion does. Each translation+ * unit that uses MLD_FREE therefore includes ct.h directly. */+#define MLD_FREE(v, T, N, context) \+ do \+ { \+ MLD_CONTEXT_UNUSED(context); \+ mld_zeroize(mld_alloc_##v, sizeof(mld_alloc_##v)); \+ (v) = NULL; \+ } while (0)++#else /* !MLD_CONFIG_CUSTOM_ALLOC_FREE */++/* Custom allocation */++/*+ * The indirection here is necessary to use MLD_CONTEXT_PARAMETERS_3 here.+ */+#define MLD_APPLY(f, args) f args++#define MLD_ALLOC(v, T, N, context) \+ MLD_APPLY(MLD_CUSTOM_ALLOC, MLD_CONTEXT_PARAMETERS_3(v, T, N, context))++#define MLD_FREE(v, T, N, context) \+ do \+ { \+ if (v != NULL) \+ { \+ mld_zeroize(v, sizeof(T) * (N)); \+ MLD_APPLY(MLD_CUSTOM_FREE, MLD_CONTEXT_PARAMETERS_3(v, T, N, context)); \+ v = NULL; \+ } \+ } while (0)++#endif /* MLD_CONFIG_CUSTOM_ALLOC_FREE */++/****************************** Error codes ***********************************/++/* Generic failure condition, reserved for failures not covered by a more+ * specific error code. */+#define MLD_ERR_FAIL (-1)+/* An allocation failed. This can only happen if MLD_CONFIG_CUSTOM_ALLOC_FREE+ * is defined and the provided MLD_CUSTOM_ALLOC can fail. */+#define MLD_ERR_OUT_OF_MEMORY (-2)+/* An RNG failure occurred. Might be due to insufficient entropy or+ * system misconfiguration. */+#define MLD_ERR_RNG_FAIL (-3)+/* The signing rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS iterations without producing a valid+ * signature. With a FIPS 204 Appendix C compliant bound (>= 821) this+ * has probability < 2^-256. */+#define MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED (-4)+/* Signing was paused before completing, at the request of a caller-provided+ * MLD_CONFIG_SIGN_HOOK_ATTEMPT hook (see mldsa_native_config.h). The caller+ * resumes by re-invoking signing with the same inputs; the attempt hook,+ * together with MLD_CONFIG_SIGN_HOOK_RESUME, decides where to continue. */+#define MLD_ERR_SIGNING_PAUSED (-5)+/* Signature verification failed: the signature is not valid for the given+ * message and public key. Returned by the verification API. */+#define MLD_ERR_INVALID_SIGNATURE (-6)+/* Secret key validation failed: the secret key is malformed or internally+ * inconsistent. Returned by pk_from_sk. */+#define MLD_ERR_INVALID_KEY (-7)+/* The Pairwise Consistency Test failed. Only possible when+ * MLD_CONFIG_KEYGEN_PCT is enabled; signals that the freshly generated key+ * pair failed its sign/verify self-test. */+#define MLD_ERR_PCT_FAIL (-8)+/* An argument was invalid, e.g. an unsupported pre-hash algorithm or a context+ * string longer than 255 bytes. */+#define MLD_ERR_INVALID_ARG (-9)+++#endif /* !__ASSEMBLER__ */++#endif /* !MLD_COMMON_H */
+ cbits/mldsa/src/context.h view
@@ -0,0 +1,152 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_CONTEXT_H+#define MLD_CONTEXT_H++/* This header is included by common.h once the configuration has been pulled+ * in; it is not meant to be included directly. */+#if !defined(__ASSEMBLER__)++#include <stdint.h>+#include "cbmc.h"+#include "sys.h"++/*+ * If the integration wants to provide a context parameter for use in+ * platform-specific hooks, then it should define this parameter.+ *+ * The MLD_CONTEXT_PARAMETERS_n macros are intended to be used with macros+ * defining the function names and expand to either pass or discard the context+ * argument as required by the current build. If there is no context parameter+ * requested then these are removed from the prototypes and from all calls.+ */+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+#define MLD_CONTEXT_PARAMETERS_0(context) (context)+#define MLD_CONTEXT_PARAMETERS_1(arg0, context) (arg0, context)+#define MLD_CONTEXT_PARAMETERS_2(arg0, arg1, context) (arg0, arg1, context)+#define MLD_CONTEXT_PARAMETERS_3(arg0, arg1, arg2, context) \+ (arg0, arg1, arg2, context)+#define MLD_CONTEXT_PARAMETERS_4(arg0, arg1, arg2, arg3, context) \+ (arg0, arg1, arg2, arg3, context)+#define MLD_CONTEXT_PARAMETERS_5(arg0, arg1, arg2, arg3, arg4, context) \+ (arg0, arg1, arg2, arg3, arg4, context)+#define MLD_CONTEXT_PARAMETERS_6(arg0, arg1, arg2, arg3, arg4, arg5, context) \+ (arg0, arg1, arg2, arg3, arg4, arg5, context)+#define MLD_CONTEXT_PARAMETERS_7(arg0, arg1, arg2, arg3, arg4, arg5, arg6, \+ context) \+ (arg0, arg1, arg2, arg3, arg4, arg5, arg6, context)+#define MLD_CONTEXT_PARAMETERS_8(arg0, arg1, arg2, arg3, arg4, arg5, arg6, \+ arg7, context) \+ (arg0, arg1, arg2, arg3, arg4, arg5, arg6, arg7, context)+#define MLD_CONTEXT_PARAMETERS_9(arg0, arg1, arg2, arg3, arg4, arg5, arg6, \+ arg7, arg8, context) \+ (arg0, arg1, arg2, arg3, arg4, arg5, arg6, arg7, arg8, context)+#else /* MLD_CONFIG_CONTEXT_PARAMETER */+#define MLD_CONTEXT_PARAMETERS_0(context) ()+#define MLD_CONTEXT_PARAMETERS_1(arg0, context) (arg0)+#define MLD_CONTEXT_PARAMETERS_2(arg0, arg1, context) (arg0, arg1)+#define MLD_CONTEXT_PARAMETERS_3(arg0, arg1, arg2, context) (arg0, arg1, arg2)+#define MLD_CONTEXT_PARAMETERS_4(arg0, arg1, arg2, arg3, context) \+ (arg0, arg1, arg2, arg3)+#define MLD_CONTEXT_PARAMETERS_5(arg0, arg1, arg2, arg3, arg4, context) \+ (arg0, arg1, arg2, arg3, arg4)+#define MLD_CONTEXT_PARAMETERS_6(arg0, arg1, arg2, arg3, arg4, arg5, context) \+ (arg0, arg1, arg2, arg3, arg4, arg5)+#define MLD_CONTEXT_PARAMETERS_7(arg0, arg1, arg2, arg3, arg4, arg5, arg6, \+ context) \+ (arg0, arg1, arg2, arg3, arg4, arg5, arg6)+#define MLD_CONTEXT_PARAMETERS_8(arg0, arg1, arg2, arg3, arg4, arg5, arg6, \+ arg7, context) \+ (arg0, arg1, arg2, arg3, arg4, arg5, arg6, arg7)+#define MLD_CONTEXT_PARAMETERS_9(arg0, arg1, arg2, arg3, arg4, arg5, arg6, \+ arg7, arg8, context) \+ (arg0, arg1, arg2, arg3, arg4, arg5, arg6, arg7, arg8)+#endif /* !MLD_CONFIG_CONTEXT_PARAMETER */++/* Consume a context parameter carried only for the integration's benefit,+ * avoiding -Wunused-parameter; expands to nothing when no context is+ * configured. */+#if defined(MLD_CONFIG_CONTEXT_PARAMETER)+#define MLD_CONTEXT_UNUSED(context) ((void)(context))+#else+#define MLD_CONTEXT_UNUSED(context) ((void)0)+#endif++#if defined(MLD_CONFIG_CONTEXT_PARAMETER_TYPE) != \+ defined(MLD_CONFIG_CONTEXT_PARAMETER)+#error MLD_CONFIG_CONTEXT_PARAMETER_TYPE must be defined if and only if MLD_CONFIG_CONTEXT_PARAMETER is defined+#endif++/* The signing hooks tie into the rejection-sampling loop. A pausing attempt+ * hook only reproduces the uninterrupted signature if the randomness is fixed+ * across calls, and thus requires the deterministic API.+ * For now we impose that requirement on all three hooks uniformly: enabling any+ * of them requires MLD_CONFIG_NO_RANDOMIZED_API. This also rules out+ * MLD_CONFIG_KEYGEN_PCT (whose PCT needs the randomized signature(), see+ * common.h).+ *+ * A logging-only use (attempt always returns 0; resume/finish merely observe)+ * would be safe with the randomized API too, but the restriction is applied+ * uniformly for now. */+#if (defined(MLD_CONFIG_SIGN_HOOK_RESUME) || \+ defined(MLD_CONFIG_SIGN_HOOK_ATTEMPT) || \+ defined(MLD_CONFIG_SIGN_HOOK_FINISH)) && \+ !defined(MLD_CONFIG_NO_RANDOMIZED_API)+#error Signing hooks (MLD_CONFIG_SIGN_HOOK_RESUME / _ATTEMPT / _FINISH) require MLD_CONFIG_NO_RANDOMIZED_API+#endif /* (MLD_CONFIG_SIGN_HOOK_RESUME || MLD_CONFIG_SIGN_HOOK_ATTEMPT || \+ MLD_CONFIG_SIGN_HOOK_FINISH) && !MLD_CONFIG_NO_RANDOMIZED_API */++/* Signing hooks (MLD_CONFIG_SIGN_HOOK_RESUME / _ATTEMPT / _FINISH; documented+ * in mldsa_native_config.h). The following macros route the call sites to+ * mld_sign_hook_*, appending or dropping the context argument; each unset hook+ * uses the dummy below. */+#define mld_sign_resume mld_sign_hook_resume MLD_CONTEXT_PARAMETERS_0+#define mld_sign_attempt mld_sign_hook_attempt MLD_CONTEXT_PARAMETERS_1+#define mld_sign_finish mld_sign_hook_finish MLD_CONTEXT_PARAMETERS_1++/* We don't use mld_sign_resume here because MLD_CONTEXT_PARAMETERS_0 is+ * unsuitable for function declarations: it misses `void` as the placeholder+ * argument. */+#if !defined(MLD_CONFIG_SIGN_HOOK_RESUME)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint16_t mld_sign_hook_resume(+#if defined(MLD_CONFIG_CONTEXT_PARAMETER)+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#else+ void+#endif+)+__contract__(assigns() ensures(1))+{+ MLD_CONTEXT_UNUSED(context);+ return 0;+}+#endif /* !MLD_CONFIG_SIGN_HOOK_RESUME */++#if !defined(MLD_CONFIG_SIGN_HOOK_ATTEMPT)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_sign_attempt(+ uint16_t attempt, MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(assigns() ensures(1))+{+ ((void)attempt);+ MLD_CONTEXT_UNUSED(context);+ return 0;+}+#endif /* !MLD_CONFIG_SIGN_HOOK_ATTEMPT */++#if !defined(MLD_CONFIG_SIGN_HOOK_FINISH)+static MLD_INLINE void mld_sign_finish(+ uint16_t attempt, MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(assigns() ensures(1))+{+ ((void)attempt);+ MLD_CONTEXT_UNUSED(context);+}+#endif /* !MLD_CONFIG_SIGN_HOOK_FINISH */++#endif /* !__ASSEMBLER__ */++#endif /* !MLD_CONTEXT_H */
+ cbits/mldsa/src/ct.c view
@@ -0,0 +1,21 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#include "ct.h"++#if !defined(MLD_USE_ASM_VALUE_BARRIER) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)+/*+ * Masking value used in constant-time functions from+ * ct.h to block the compiler's range analysis and+ * thereby reduce the risk of compiler-introduced branches.+ */+volatile uint64_t mld_ct_opt_blocker_u64 = 0;++#else /* !MLD_USE_ASM_VALUE_BARRIER && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(ct)++#endif /* !(!MLD_USE_ASM_VALUE_BARRIER && !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mldsa/src/ct.h view
@@ -0,0 +1,373 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ *+ * - [libmceliece]+ * libmceliece implementation of Classic McEliece+ * Bernstein, Chou+ * https://lib.mceliece.org/+ *+ * - [optblocker]+ * PQC forum post on opt-blockers using volatile globals+ * Daniel J. Bernstein+ * https://groups.google.com/a/list.nist.gov/g/pqc-forum/c/hqbtIGFKIpU/m/H14H0wOlBgAJ+ */++#ifndef MLD_CT_H+#define MLD_CT_H++#include "cbmc.h"+#include "common.h"++/* Constant-time comparisons and conditional operations++ We reduce the risk for compilation into variable-time code+ through the use of 'value barriers'.++ Functionally, a value barrier is a no-op. To the compiler, however,+ it constitutes an arbitrary modification of its input, and therefore+ harden's value propagation and range analysis.++ We consider two approaches to implement a value barrier:+ - An empty inline asm block which marks the target value as clobbered.+ - XOR'ing with the value of a volatile global that's set to 0;+ see @[optblocker] for a discussion of this idea, and+ @[libmceliece, inttypes/crypto_intN.h] for an implementation.++ The first approach is cheap because it only prevents the compiler+ from reasoning about the value of the variable past the barrier,+ but does not directly generate additional instructions.++ The second approach generates redundant loads and XOR operations+ and therefore comes at a higher runtime cost. However, it appears+ more robust towards optimization, as compilers should never drop+ a volatile load.++ We use the empty-ASM value barrier for GCC and clang, and fall+ back to the global volatile barrier otherwise.++ The global value barrier can be forced by setting+ MLD_CONFIG_NO_ASM_VALUE_BARRIER.++*/++#if defined(MLD_HAVE_INLINE_ASM) && !defined(MLD_CONFIG_NO_ASM_VALUE_BARRIER)+#define MLD_USE_ASM_VALUE_BARRIER+#endif+++#if !defined(MLD_USE_ASM_VALUE_BARRIER)+/*+ * Declaration of global volatile that the global value barrier+ * is loading from and masking with.+ */+#define mld_ct_opt_blocker_u64 MLD_NAMESPACE(ct_opt_blocker_u64)+extern volatile uint64_t mld_ct_opt_blocker_u64;+++/* Helper functions for obtaining global masks of various sizes */++/* This contract is not proved but treated as an axiom.+ *+ * Its validity relies on the assumption that the global opt-blocker+ * constant mld_ct_opt_blocker_u64 is not modified.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint64_t mld_ct_get_optblocker_u64(void)+__contract__(ensures(return_value == 0)) { return mld_ct_opt_blocker_u64; }++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int64_t mld_ct_get_optblocker_i64(void)+__contract__(ensures(return_value == 0)) { return (int64_t)mld_ct_get_optblocker_u64(); }++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint32_t mld_ct_get_optblocker_u32(void)+__contract__(ensures(return_value == 0)) { return (uint32_t)mld_ct_get_optblocker_u64(); }++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint8_t mld_ct_get_optblocker_u8(void)+__contract__(ensures(return_value == 0)) { return (uint8_t)mld_ct_get_optblocker_u64(); }++/* Opt-blocker based implementation of value barriers */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int64_t mld_value_barrier_i64(int64_t b)+__contract__(ensures(return_value == b)) { return (b ^ mld_ct_get_optblocker_i64()); }++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint32_t mld_value_barrier_u32(uint32_t b)+__contract__(ensures(return_value == b)) { return (b ^ mld_ct_get_optblocker_u32()); }++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint8_t mld_value_barrier_u8(uint8_t b)+__contract__(ensures(return_value == b)) { return (b ^ mld_ct_get_optblocker_u8()); }+++#else /* !MLD_USE_ASM_VALUE_BARRIER */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int64_t mld_value_barrier_i64(int64_t b)+__contract__(ensures(return_value == b))+{+ __asm__ volatile("" : "+r"(b));+ return b;+}++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint32_t mld_value_barrier_u32(uint32_t b)+__contract__(ensures(return_value == b))+{+ __asm__ volatile("" : "+r"(b));+ return b;+}++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint8_t mld_value_barrier_u8(uint8_t b)+__contract__(ensures(return_value == b))+{+ __asm__ volatile("" : "+r"(b));+ return b;+}+#endif /* MLD_USE_ASM_VALUE_BARRIER */++#ifdef CBMC+#pragma CPROVER check push+#pragma CPROVER check disable "conversion"+#endif++/**+ * Cast uint32 value to int32.+ *+ * @param x Input value.+ *+ * @return For uint32_t x, the unique y in int32_t so that x == y mod 2^32.+ * Concretely:+ * - x < 2^31: returns x+ * - x >= 2^31: returns x - 2^32+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_ALWAYS_INLINE int32_t mld_cast_uint32_to_int32(uint32_t x)+{+ /*+ * PORTABILITY: This relies on uint32_t -> int32_t+ * being implemented as the inverse of int32_t -> uint32_t,+ * which is implementation-defined (C99 6.3.1.3 (3))+ * CBMC (correctly) fails to prove this conversion is OK,+ * so we have to suppress that check here+ */+ return (int32_t)x;+}++#ifdef CBMC+#pragma CPROVER check pop+#endif+++/**+ * Cast int64 value to uint32 as per C standard.+ *+ * @param x Input value.+ *+ * @return For int64_t x, the unique y in uint32_t so that x == y mod 2^32.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_ALWAYS_INLINE uint32_t mld_cast_int64_to_uint32(int64_t x)+{+ return (uint32_t)(x & (int64_t)UINT32_MAX);+}++/**+ * Cast int32 value to uint32 as per C standard.+ *+ * @param x Input value.+ *+ * @return For int32_t x, the unique y in uint32_t so that x == y mod 2^32.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_ALWAYS_INLINE uint32_t mld_cast_int32_to_uint32(int32_t x)+{+ return mld_cast_int64_to_uint32((int64_t)x);+}++/**+ * Functionally equivalent to cond ? a : b, but implemented with guards against+ * compiler-introduced branches.+ *+ * @param a First alternative.+ * @param b Second alternative.+ * @param cond Condition variable.+ *+ * @return a if cond is 0xFFFFFFFF, b if cond is 0.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int32_t mld_ct_sel_int32(int32_t a, int32_t b, uint32_t cond)+__contract__(+ requires(cond == 0x0 || cond == 0xFFFFFFFF)+ ensures(return_value == (cond ? a : b))+)+{+ uint32_t au = mld_cast_int32_to_uint32(a);+ uint32_t bu = mld_cast_int32_to_uint32(b);+ uint32_t res = bu ^ (mld_value_barrier_u32(cond) & (au ^ bu));+ return mld_cast_uint32_to_int32(res);+}++/**+ * Return 0 if input is zero, and -1 otherwise.+ *+ * @param x Value to be converted into a mask.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint32_t mld_ct_cmask_nonzero_u32(uint32_t x)+__contract__(ensures(return_value == ((x == 0) ? 0 : 0xFFFFFFFF)))+{+ int64_t tmp = mld_value_barrier_i64(-((int64_t)x));+ /*+ * PORTABILITY: Right-shift on a signed integer is+ * implementation-defined for negative left argument.+ * Here, we assume it's sign-preserving "arithmetic" shift right.+ * See (C99 6.5.7 (5))+ */+ tmp >>= 32;+ return mld_cast_int64_to_uint32(tmp);+}++/**+ * Return 0 if input is zero, and -1 otherwise.+ *+ * @param x Value to be converted into a mask.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint8_t mld_ct_cmask_nonzero_u8(uint8_t x)+__contract__(ensures(return_value == ((x == 0) ? 0 : 0xFF)))+{+ uint32_t mask = mld_ct_cmask_nonzero_u32((uint32_t)x);+ return (uint8_t)(mask & 0xFF);+}++/**+ * Return 0 if input is non-negative, and -1 otherwise.+ *+ * @param x Value to be converted into a mask.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint32_t mld_ct_cmask_neg_i32(int32_t x)+__contract__(+ ensures(return_value == ((x < 0) ? 0xFFFFFFFF : 0))+)+{+ int64_t tmp = mld_value_barrier_i64((int64_t)x);+ /*+ * PORTABILITY: Right-shift on a signed integer is+ * implementation-defined for negative left argument.+ * Here, we assume it's sign-preserving "arithmetic" shift right.+ * See (C99 6.5.7 (5))+ */+ tmp >>= 31;+ return mld_cast_int64_to_uint32(tmp);+}++/**+ * Return -x if x<0, x otherwise.+ *+ * @param x Input value.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int32_t mld_ct_abs_i32(int32_t x)+__contract__(+ requires(x >= -INT32_MAX)+ ensures(return_value == ((x < 0) ? -x : x))+)+{+ return mld_ct_sel_int32(-x, x, mld_ct_cmask_neg_i32(x));+}++/**+ * Compare two arrays for equality in constant time.+ *+ * @param[in] a Pointer to first byte array.+ * @param[in] b Pointer to second byte array.+ * @param len Length of the byte arrays, upper-bounded to UINT16_MAX to+ * control proof complexity only.+ *+ * @return 0 if the byte arrays are equal, 0xFF otherwise.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint8_t mld_ct_memcmp(const uint8_t *a, const uint8_t *b,+ const size_t len)+__contract__(+ requires(len <= UINT16_MAX)+ requires(memory_no_alias(a, len))+ requires(memory_no_alias(b, len))+ ensures((return_value == 0) || (return_value == 0xFF))+ ensures((return_value == 0) == forall(i, 0, len, (a[i] == b[i]))))+{+ uint8_t r = 0, s = 0;+ unsigned i;++ for (i = 0; i < len; i++)+ __loop__(+ invariant(i <= len)+ invariant((r == 0) == (forall(k, 0, i, (a[k] == b[k]))))+ decreases(len - i))+ {+ r |= a[i] ^ b[i];+ /* s is useless, but prevents the loop from being aborted once r=0xff. */+ s ^= a[i] ^ b[i];+ }++ /*+ * - Convert r into a mask; this may not be necessary, but is an additional+ * safeguard+ * towards leaking information about a and b.+ * - XOR twice with s, separated by a value barrier, to prevent the compile+ * from dropping the s computation in the loop.+ */+ return (mld_value_barrier_u8(mld_ct_cmask_nonzero_u8(r) ^ s) ^ s);+}++/**+ * Force-zeroize a buffer.+ *+ * @[FIPS204, Section 3.6.3] Destruction of intermediate values.+ *+ * @param[out] ptr Pointer to buffer to be zeroed.+ * @param len Amount of bytes to be zeroed.+ */+#if !defined(MLD_CONFIG_CUSTOM_ZEROIZE)+#if defined(MLD_SYS_WINDOWS)+#include <windows.h>+#elif !defined(MLD_HAVE_INLINE_ASM)+#error No plausibly-secure implementation of mld_zeroize available. Please provide your own using MLD_CONFIG_CUSTOM_ZEROIZE.+#endif++static MLD_INLINE void mld_zeroize(void *ptr, size_t len)+__contract__(+ requires(len <= UINT32_MAX)+ requires(memory_no_alias(ptr, len))+ assigns(memory_slice(ptr, len))+ ensures(array_zeroized_u8((uint8_t *)ptr, len)))+{+#if defined(MLD_SYS_WINDOWS)+ SecureZeroMemory(ptr, len);+#else+ mld_memset(ptr, 0, len);+ /* This follows OpenSSL and seems sufficient to prevent the compiler+ * from optimizing away the memset.+ *+ * If there was a reliable way to detect availability of memset_s(),+ * that would be preferred. */+ __asm__ __volatile__("" : : "r"(ptr) : "memory");+#endif /* !MLD_SYS_WINDOWS */+}+#endif /* !MLD_CONFIG_CUSTOM_ZEROIZE */++#endif /* !MLD_CT_H */
+ cbits/mldsa/src/debug.c view
@@ -0,0 +1,75 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* NOTE: You can remove this file unless you compile with MLDSA_DEBUG. */++#include "common.h"++#if !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#if defined(MLDSA_DEBUG)++#include <inttypes.h>+#include <stdio.h>+#include <stdlib.h>+#include "debug.h"++#define MLD_DEBUG_ERROR_HEADER "[ERROR:%s:%04d] "++MLD_INTERNAL_API+void mld_debug_check_assert(const char *file, int line, const int val)+{+ if (val == 0)+ {+ fprintf(stderr, MLD_DEBUG_ERROR_HEADER "Assertion failed (value %d)\n",+ file, line, val);+ exit(1);+ }+}++MLD_INTERNAL_API+void mld_debug_check_bounds(const char *file, int line, const int32_t *ptr,+ unsigned len, int64_t lower_bound_exclusive,+ int64_t upper_bound_exclusive)+{+ int err = 0;+ unsigned i;+ for (i = 0; i < len; i++)+ {+ int32_t val = ptr[i];+ if (!(val > lower_bound_exclusive && val < upper_bound_exclusive))+ {+ fprintf(stderr,+ MLD_DEBUG_ERROR_HEADER+ "Bounds assertion failed: Index %u, value %d out of bounds "+ "(%" PRId64 ",%" PRId64 ")\n",+ file, line, i, (int)val, lower_bound_exclusive,+ upper_bound_exclusive);+ err = 1;+ }+ }++ if (err == 1)+ {+ exit(1);+ }+}++#else /* MLDSA_DEBUG */++MLD_EMPTY_CU(debug)++#endif /* !MLDSA_DEBUG */++#else /* !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(debug)++#endif /* MLD_CONFIG_MULTILEVEL_NO_SHARED */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef MLD_DEBUG_ERROR_HEADER
+ cbits/mldsa/src/debug.h view
@@ -0,0 +1,125 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_DEBUG_H+#define MLD_DEBUG_H+#include "common.h"++#if defined(MLDSA_DEBUG)++/**+ * Check debug assertion.+ *+ * Prints an error message to stderr and calls exit(1) if not.+ *+ * @param file Filename.+ * @param line Line number.+ * @param val Value asserted to be non-zero.+ */+#define mld_debug_check_assert MLD_NAMESPACE(mldsa_debug_assert)+MLD_INTERNAL_API+void mld_debug_check_assert(const char *file, int line, const int val);++/**+ * Check whether values in an array of int32_t are within specified bounds.+ *+ * Prints an error message to stderr and calls exit(1) if not.+ *+ * @param file Filename.+ * @param line Line number.+ * @param[in] ptr Base of array to be checked.+ * @param len Number of int32_t in ptr.+ * @param lower_bound_exclusive Exclusive lower bound.+ * @param upper_bound_exclusive Exclusive upper bound.+ */+#define mld_debug_check_bounds MLD_NAMESPACE(mldsa_debug_check_bounds)+MLD_INTERNAL_API+void mld_debug_check_bounds(const char *file, int line, const int32_t *ptr,+ unsigned len, int64_t lower_bound_exclusive,+ int64_t upper_bound_exclusive);++/* Check assertion, calling exit() upon failure+ *+ * val: Value that's asserted to be non-zero+ */+#define mld_assert(val) mld_debug_check_assert(__FILE__, __LINE__, (val))++/* Check bounds in array of int32_t's+ * ptr: Base of int32_t array; will be explicitly cast to int32_t*,+ * so you may pass a byte-compatible type such as mld_poly or mld_polyvec.+ * len: Number of int32_t in array+ * value_lb: Inclusive lower value bound+ * value_ub: Exclusive upper value bound */+#define mld_assert_bound(ptr, len, value_lb, value_ub) \+ mld_debug_check_bounds(__FILE__, __LINE__, (const int32_t *)(ptr), (len), \+ ((int64_t)(value_lb)) - 1, (value_ub))++/* Check absolute bounds in array of int32_t's+ * ptr: Base of array, expression of type int32_t*+ * len: Number of int32_t in array+ * value_abs_bd: Exclusive absolute upper bound */+#define mld_assert_abs_bound(ptr, len, value_abs_bd) \+ mld_assert_bound((ptr), (len), (-((int64_t)(value_abs_bd)) + 1), \+ (value_abs_bd))++/* Version of bounds assertions for 2-dimensional arrays */+#define mld_assert_bound_2d(ptr, len0, len1, value_lb, value_ub) \+ mld_assert_bound((ptr), ((len0) * (len1)), (value_lb), (value_ub))++#define mld_assert_abs_bound_2d(ptr, len0, len1, value_abs_bd) \+ mld_assert_abs_bound((ptr), ((len0) * (len1)), (value_abs_bd))++/* When running CBMC, convert debug assertions into proof obligations */+#elif defined(CBMC)+#include "cbmc.h"++#define mld_assert(val) cassert(val)++#define mld_assert_bound(ptr, len, value_lb, value_ub) \+ cassert(array_bound(((int32_t *)(ptr)), 0, (len), (value_lb), (value_ub)))++#define mld_assert_abs_bound(ptr, len, value_abs_bd) \+ cassert(array_abs_bound(((int32_t *)(ptr)), 0, (len), (value_abs_bd)))++/* Because of https://github.com/diffblue/cbmc/issues/8570, we can't+ * just use a single flattened array_bound(...) here. */+#define mld_assert_bound_2d(ptr, M, N, value_lb, value_ub) \+ cassert(forall(kN, 0, (M), \+ array_bound(&((int32_t (*)[(N)])(ptr))[kN][0], 0, (N), \+ (value_lb), (value_ub))))++#define mld_assert_abs_bound_2d(ptr, M, N, value_abs_bd) \+ cassert(forall(kN, 0, (M), \+ array_abs_bound(&((int32_t (*)[(N)])(ptr))[kN][0], 0, (N), \+ (value_abs_bd))))++#else /* !MLDSA_DEBUG && CBMC */++#define mld_assert(val) \+ do \+ { \+ } while (0)+#define mld_assert_bound(ptr, len, value_lb, value_ub) \+ do \+ { \+ } while (0)+#define mld_assert_abs_bound(ptr, len, value_abs_bd) \+ do \+ { \+ } while (0)++#define mld_assert_bound_2d(ptr, len0, len1, value_lb, value_ub) \+ do \+ { \+ } while (0)++#define mld_assert_abs_bound_2d(ptr, len0, len1, value_abs_bd) \+ do \+ { \+ } while (0)+++#endif /* !MLDSA_DEBUG && !CBMC */+#endif /* !MLD_DEBUG_H */
+ cbits/mldsa/src/fips202/fips202.c view
@@ -0,0 +1,270 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ *+ * - [mupq]+ * Common files for pqm4, pqm3, pqriscv+ * Kannwischer, Petri, Rijneveld, Schwabe, Stoffelen+ * https://github.com/mupq/mupq+ *+ * - [supercop]+ * SUPERCOP benchmarking framework+ * Daniel J. Bernstein+ * http://bench.cr.yp.to/supercop.html+ *+ * - [tweetfips]+ * 'tweetfips202' FIPS202 implementation+ * Van Assche, Bernstein, Schwabe+ * https://keccak.team/2015/tweetfips202.html+ */++/* Based on the CC0 implementation from @[mupq] and the public domain+ * implementation @[supercop, crypto_hash/keccakc512/simple/]+ * by Ronny Van Keer, and the public domain @[tweetfips] implementation. */++#include <stddef.h>++#include "../common.h"+#include "../ct.h"+#include "fips202.h"+#include "keccakf1600.h"+#if !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/**+ * Initializes the Keccak state.+ *+ * @param[out] s Pointer to Keccak state.+ */+static void keccak_init(uint64_t s[MLD_KECCAK_LANES])+__contract__(+ requires(memory_no_alias(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+ assigns(memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+)+{+ mld_memset(s, 0, sizeof(uint64_t) * MLD_KECCAK_LANES);+}++/**+ * Absorb step of Keccak; incremental.+ *+ * @param[in,out] s Pointer to Keccak state.+ * @param pos Position in current block to be absorbed.+ * @param r Rate in bytes (e.g., 168 for SHAKE128).+ * @param[in] in Pointer to input to be absorbed into s.+ * @param inlen Length of input in bytes.+ *+ * @return New position pos in current block.+ */+static unsigned int keccak_absorb(uint64_t s[MLD_KECCAK_LANES],+ unsigned int pos, unsigned int r,+ const uint8_t *in, size_t inlen)+__contract__(+ requires(inlen <= MLD_MAX_BUFFER_SIZE)+ requires(r > 0)+ requires(r < sizeof(uint64_t) * MLD_KECCAK_LANES)+ requires(pos <= r)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+ requires(memory_no_alias(in, inlen))+ assigns(memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+ ensures(return_value < r))+{+ while (inlen >= r - pos)+ __loop__(+ assigns(pos, in, inlen,+ memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+ invariant(inlen <= loop_entry(inlen))+ invariant(pos <= r)+ invariant(in == loop_entry(in) + (loop_entry(inlen) - inlen))+ decreases(inlen + pos))+ {+ mld_keccakf1600_xor_bytes(s, in, pos, r - pos);+ inlen -= r - pos;+ in += r - pos;+ mld_keccakf1600_permute(s);+ pos = 0;+ }+ /* Safety: At this point, inlen < r, so the truncation to unsigned is safe. */+ mld_keccakf1600_xor_bytes(s, in, pos, (unsigned)inlen);++ /* Safety: At this point, inlen < r and pos <= r so the truncation to unsigned+ * is safe. */+ return (unsigned)(pos + inlen);+}++/**+ * Finalize absorb step.+ *+ * @param[in,out] s Pointer to Keccak state.+ * @param pos Position in current block to be absorbed.+ * @param r Rate in bytes (e.g., 168 for SHAKE128).+ * @param p Domain separation byte.+ */+static void keccak_finalize(uint64_t s[MLD_KECCAK_LANES], unsigned int pos,+ unsigned int r, uint8_t p)+__contract__(+ requires(pos <= r && r < sizeof(uint64_t) * MLD_KECCAK_LANES)+ requires((r / 8) >= 1)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+ assigns(memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+)+{+ uint8_t b = 0x80;+ mld_keccakf1600_xor_bytes(s, &p, pos, 1);+ mld_keccakf1600_xor_bytes(s, &b, r - 1, 1);+}++/**+ * Squeeze step of Keccak. Squeezes arbitrarily many bytes. Modifies the+ * state. Can be called multiple times to keep squeezing, i.e., is+ * incremental.+ *+ * @param[out] out Pointer to output data.+ * @param outlen Number of bytes to be squeezed (written to out).+ * @param[in,out] s Pointer to input/output Keccak state.+ * @param pos Number of bytes in current block already squeezed.+ * @param r Rate in bytes (e.g., 168 for SHAKE128).+ *+ * @return New position pos in current block.+ */+static unsigned int keccak_squeeze(uint8_t *out, size_t outlen,+ uint64_t s[MLD_KECCAK_LANES],+ unsigned int pos, unsigned int r)+__contract__(+ requires((r == SHAKE128_RATE && pos <= SHAKE128_RATE) ||+ (r == SHAKE256_RATE && pos <= SHAKE256_RATE) ||+ (r == SHA3_512_RATE && pos <= SHA3_512_RATE))+ requires(outlen <= 8 * r /* somewhat arbitrary bound */)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+ requires(memory_no_alias(out, outlen))+ assigns(memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+ assigns(memory_slice(out, outlen))+ ensures(return_value <= r))+{+ unsigned int i;+ size_t out_offset = 0;++ /* Reference: This code is re-factored from the reference implementation+ * to facilitate proof with CBMC and to improve readability.+ *+ * Take a mutable copy of outlen to count down the number of bytes+ * still to squeeze. The initial value of outlen is needed for the CBMC+ * assigns() clauses. */+ size_t bytes_to_go = outlen;++ while (bytes_to_go > 0)+ __loop__(+ assigns(i, bytes_to_go, pos, out_offset, memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES), memory_slice(out, outlen))+ invariant(bytes_to_go <= outlen)+ invariant(out_offset == outlen - bytes_to_go)+ invariant(pos <= r)+ decreases(bytes_to_go)+ )+ {+ if (pos == r)+ {+ mld_keccakf1600_permute(s);+ pos = 0;+ }+ /* Safety: If bytes_to_go < r - pos, truncation to unsigned is safe. */+ i = bytes_to_go < r - pos ? (unsigned)bytes_to_go : r - pos;+ mld_keccakf1600_extract_bytes(s, out + out_offset, pos, i);+ bytes_to_go -= i;+ pos += i;+ out_offset += i;+ }++ return pos;+}++MLD_INTERNAL_API+void mld_shake128_init(mld_shake128ctx *state)+{+ keccak_init(state->s);+ state->pos = 0;+}++MLD_INTERNAL_API+void mld_shake128_absorb(mld_shake128ctx *state, const uint8_t *in,+ size_t inlen)+{+ state->pos = keccak_absorb(state->s, state->pos, SHAKE128_RATE, in, inlen);+}++MLD_INTERNAL_API+void mld_shake128_finalize(mld_shake128ctx *state)+{+ keccak_finalize(state->s, state->pos, SHAKE128_RATE, 0x1F);+ state->pos = SHAKE128_RATE;+}++MLD_INTERNAL_API+void mld_shake128_squeeze(uint8_t *out, size_t outlen, mld_shake128ctx *state)+{+ state->pos = keccak_squeeze(out, outlen, state->s, state->pos, SHAKE128_RATE);+}++MLD_INTERNAL_API+void mld_shake128_release(mld_shake128ctx *state)+{+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(state, sizeof(mld_shake128ctx));+}++MLD_INTERNAL_API+void mld_shake256_init(mld_shake256ctx *state)+{+ keccak_init(state->s);+ state->pos = 0;+}++MLD_INTERNAL_API+void mld_shake256_absorb(mld_shake256ctx *state, const uint8_t *in,+ size_t inlen)+{+ state->pos = keccak_absorb(state->s, state->pos, SHAKE256_RATE, in, inlen);+}++MLD_INTERNAL_API+void mld_shake256_finalize(mld_shake256ctx *state)+{+ keccak_finalize(state->s, state->pos, SHAKE256_RATE, 0x1F);+ state->pos = SHAKE256_RATE;+}++MLD_INTERNAL_API+void mld_shake256_squeeze(uint8_t *out, size_t outlen, mld_shake256ctx *state)+{+ state->pos = keccak_squeeze(out, outlen, state->s, state->pos, SHAKE256_RATE);+}++MLD_INTERNAL_API+void mld_shake256_release(mld_shake256ctx *state)+{+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(state, sizeof(mld_shake256ctx));+}++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_CORE_API_ONLY)+MLD_INTERNAL_API+void mld_shake256(uint8_t *out, size_t outlen, const uint8_t *in, size_t inlen)+{+ mld_shake256ctx state;++ mld_shake256_init(&state);+ mld_shake256_absorb(&state, in, inlen);+ mld_shake256_finalize(&state);+ mld_shake256_squeeze(out, outlen, &state);+ mld_shake256_release(&state);+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_CORE_API_ONLY */++#endif /* !MLD_CONFIG_MULTILEVEL_NO_SHARED */
+ cbits/mldsa/src/fips202/fips202.h view
@@ -0,0 +1,224 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_FIPS202_FIPS202_H+#define MLD_FIPS202_FIPS202_H++#include <stddef.h>+#include "../cbmc.h"+#include "../common.h"++#define SHAKE128_RATE 168+#define SHAKE256_RATE 136+#define SHA3_256_RATE 136+#define SHA3_512_RATE 72+#define MLD_KECCAK_LANES 25+#define SHA3_256_HASHBYTES 32+#define SHA3_512_HASHBYTES 64+++/** Context for the incremental SHAKE128 XOF. */+typedef struct+{+ uint64_t s[MLD_KECCAK_LANES]; /**< Keccak state. */+ unsigned int pos; /**< Byte position within the current Keccak block. */+} mld_shake128ctx;++/** Context for the incremental SHAKE256 XOF. */+typedef struct+{+ uint64_t s[MLD_KECCAK_LANES]; /**< Keccak state. */+ unsigned int pos; /**< Byte position within the current Keccak block. */+} mld_shake256ctx;++#define mld_shake128_init MLD_NAMESPACE(shake128_init)+/**+ * Initializes state for use as SHAKE128 XOF.+ *+ * @param[out] state Pointer to (uninitialized) state.+ */+MLD_INTERNAL_API+void mld_shake128_init(mld_shake128ctx *state)+__contract__(+ requires(memory_no_alias(state, sizeof(mld_shake128ctx)))+ assigns(memory_slice(state, sizeof(mld_shake128ctx)))+ ensures(state->pos == 0)+);++#define mld_shake128_absorb MLD_NAMESPACE(shake128_absorb)+/**+ * Absorb step of the SHAKE128 XOF. Absorbs arbitrarily many bytes. Can be+ * called multiple times to absorb multiple chunks of data.+ *+ * @param[in,out] state Pointer to (initialized) output state.+ * @param[in] in Pointer to input to be absorbed into s.+ * @param inlen Length of input in bytes.+ */+MLD_INTERNAL_API+void mld_shake128_absorb(mld_shake128ctx *state, const uint8_t *in,+ size_t inlen)+__contract__(+ requires(inlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(state, sizeof(mld_shake128ctx)))+ requires(memory_no_alias(in, inlen))+ requires(state->pos <= SHAKE128_RATE)+ assigns(memory_slice(state, sizeof(mld_shake128ctx)))+ ensures(state->pos <= SHAKE128_RATE)+);++#define mld_shake128_finalize MLD_NAMESPACE(shake128_finalize)+/**+ * Concludes the absorb phase of the SHAKE128 XOF.+ *+ * @param[in,out] state Pointer to state.+ */+MLD_INTERNAL_API+void mld_shake128_finalize(mld_shake128ctx *state)+__contract__(+ requires(memory_no_alias(state, sizeof(mld_shake128ctx)))+ requires(state->pos <= SHAKE128_RATE)+ assigns(memory_slice(state, sizeof(mld_shake128ctx)))+ ensures(state->pos <= SHAKE128_RATE)+);++#define mld_shake128_squeeze MLD_NAMESPACE(shake128_squeeze)+/**+ * Squeeze step of SHAKE128 XOF. Squeezes arbitrarily many bytes. Can be+ * called multiple times to keep squeezing.+ *+ * @param[out] out Pointer to output blocks.+ * @param outlen Number of bytes to be squeezed (written to output).+ * @param[in,out] state Pointer to input/output state.+ */+MLD_INTERNAL_API+void mld_shake128_squeeze(uint8_t *out, size_t outlen, mld_shake128ctx *state)+__contract__(+ requires(outlen <= 8 * SHAKE128_RATE /* somewhat arbitrary bound */)+ requires(memory_no_alias(state, sizeof(mld_shake128ctx)))+ requires(memory_no_alias(out, outlen))+ requires(state->pos <= SHAKE128_RATE)+ assigns(memory_slice(state, sizeof(mld_shake128ctx)))+ assigns(memory_slice(out, outlen))+ ensures(state->pos <= SHAKE128_RATE)+);++#define mld_shake128_release MLD_NAMESPACE(shake128_release)+/**+ * Release and securely zero the SHAKE128 state.+ *+ * @param[in,out] state Pointer to state.+ */+MLD_INTERNAL_API+void mld_shake128_release(mld_shake128ctx *state)+__contract__(+ requires(memory_no_alias(state, sizeof(mld_shake128ctx)))+ assigns(memory_slice(state, sizeof(mld_shake128ctx)))+);++#define mld_shake256_init MLD_NAMESPACE(shake256_init)+/**+ * Initializes state for use as SHAKE256 XOF.+ *+ * @param[out] state Pointer to (uninitialized) state.+ */+MLD_INTERNAL_API+void mld_shake256_init(mld_shake256ctx *state)+__contract__(+ requires(memory_no_alias(state, sizeof(mld_shake256ctx)))+ assigns(memory_slice(state, sizeof(mld_shake256ctx)))+ ensures(state->pos == 0)+);++#define mld_shake256_absorb MLD_NAMESPACE(shake256_absorb)+/**+ * Absorb step of the SHAKE256 XOF. Absorbs arbitrarily many bytes. Can be+ * called multiple times to absorb multiple chunks of data.+ *+ * @param[in,out] state Pointer to (initialized) output state.+ * @param[in] in Pointer to input to be absorbed into s.+ * @param inlen Length of input in bytes.+ */+MLD_INTERNAL_API+void mld_shake256_absorb(mld_shake256ctx *state, const uint8_t *in,+ size_t inlen)+__contract__(+ requires(inlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(state, sizeof(mld_shake256ctx)))+ requires(memory_no_alias(in, inlen))+ requires(state->pos <= SHAKE256_RATE)+ assigns(memory_slice(state, sizeof(mld_shake256ctx)))+ ensures(state->pos <= SHAKE256_RATE)+);++#define mld_shake256_finalize MLD_NAMESPACE(shake256_finalize)+/**+ * Concludes the absorb phase of the SHAKE256 XOF.+ *+ * @param[in,out] state Pointer to state.+ */+MLD_INTERNAL_API+void mld_shake256_finalize(mld_shake256ctx *state)+__contract__(+ requires(memory_no_alias(state, sizeof(mld_shake256ctx)))+ requires(state->pos <= SHAKE256_RATE)+ assigns(memory_slice(state, sizeof(mld_shake256ctx)))+ ensures(state->pos <= SHAKE256_RATE)+);++#define mld_shake256_squeeze MLD_NAMESPACE(shake256_squeeze)+/**+ * Squeeze step of SHAKE256 XOF. Squeezes arbitrarily many bytes. Can be+ * called multiple times to keep squeezing.+ *+ * @param[out] out Pointer to output blocks.+ * @param outlen Number of bytes to be squeezed (written to output).+ * @param[in,out] state Pointer to input/output state.+ */+MLD_INTERNAL_API+void mld_shake256_squeeze(uint8_t *out, size_t outlen, mld_shake256ctx *state)+__contract__(+ requires(outlen <= 8 * SHAKE256_RATE /* somewhat arbitrary bound */)+ requires(memory_no_alias(state, sizeof(mld_shake256ctx)))+ requires(memory_no_alias(out, outlen))+ requires(state->pos <= SHAKE256_RATE)+ assigns(memory_slice(state, sizeof(mld_shake256ctx)))+ assigns(memory_slice(out, outlen))+ ensures(state->pos <= SHAKE256_RATE)+);++#define mld_shake256_release MLD_NAMESPACE(shake256_release)+/**+ * Release and securely zero the SHAKE256 state.+ *+ * @param[in,out] state Pointer to state.+ */+MLD_INTERNAL_API+void mld_shake256_release(mld_shake256ctx *state)+__contract__(+ requires(memory_no_alias(state, sizeof(mld_shake256ctx)))+ assigns(memory_slice(state, sizeof(mld_shake256ctx)))+);++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_CORE_API_ONLY)+#define mld_shake256 MLD_NAMESPACE(shake256)+/**+ * SHAKE256 XOF with non-incremental API.+ *+ * @param[out] out Pointer to output.+ * @param outlen Requested output length in bytes.+ * @param[in] in Pointer to input.+ * @param inlen Length of input in bytes.+ */+MLD_INTERNAL_API+void mld_shake256(uint8_t *out, size_t outlen, const uint8_t *in, size_t inlen)+__contract__(+ requires(inlen <= MLD_MAX_BUFFER_SIZE)+ requires(outlen <= 8 * SHAKE256_RATE /* somewhat arbitrary bound */)+ requires(memory_no_alias(in, inlen))+ requires(memory_no_alias(out, outlen))+ assigns(memory_slice(out, outlen))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_CORE_API_ONLY */++#endif /* !MLD_FIPS202_FIPS202_H */
+ cbits/mldsa/src/fips202/fips202x4.c view
@@ -0,0 +1,187 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#include "../common.h"+#if !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)++#include "../ct.h"+#include "fips202.h"+#include "fips202x4.h"+#include "keccakf1600.h"++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_REDUCE_RAM) || \+ defined(MLD_UNIT_TEST)+static void mld_keccak_absorb_once_x4(uint64_t *s, unsigned r,+ const uint8_t *in0, const uint8_t *in1,+ const uint8_t *in2, const uint8_t *in3,+ size_t inlen, uint8_t p)+__contract__(+ requires(inlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ requires(r > 0)+ requires(r <= sizeof(uint64_t) * MLD_KECCAK_LANES)+ requires(memory_no_alias(in0, inlen))+ requires(memory_no_alias(in1, inlen))+ requires(memory_no_alias(in2, inlen))+ requires(memory_no_alias(in3, inlen))+ assigns(memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY)))+{+ while (inlen >= r)+ __loop__(+ assigns(inlen, in0, in1, in2, in3, memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ invariant(inlen <= loop_entry(inlen))+ invariant(in0 == loop_entry(in0) + (loop_entry(inlen) - inlen))+ invariant(in1 == loop_entry(in1) + (loop_entry(inlen) - inlen))+ invariant(in2 == loop_entry(in2) + (loop_entry(inlen) - inlen))+ invariant(in3 == loop_entry(in3) + (loop_entry(inlen) - inlen))+ decreases(inlen))+ {+ mld_keccakf1600x4_xor_bytes(s, in0, in1, in2, in3, 0, r);+ mld_keccakf1600x4_permute(s);++ in0 += r;+ in1 += r;+ in2 += r;+ in3 += r;+ inlen -= r;+ }++ /* Safety: At this point, inlen < r, so the truncations to unsigned are safe+ * below. */+ if (inlen > 0)+ {+ mld_keccakf1600x4_xor_bytes(s, in0, in1, in2, in3, 0, (unsigned)inlen);+ }++ if (inlen == r - 1)+ {+ p |= 128;+ mld_keccakf1600x4_xor_bytes(s, &p, &p, &p, &p, (unsigned)inlen, 1);+ }+ else+ {+ mld_keccakf1600x4_xor_bytes(s, &p, &p, &p, &p, (unsigned)inlen, 1);+ p = 128;+ mld_keccakf1600x4_xor_bytes(s, &p, &p, &p, &p, r - 1, 1);+ }+}++static void mld_keccak_squeezeblocks_x4(uint8_t *out0, uint8_t *out1,+ uint8_t *out2, uint8_t *out3,+ size_t nblocks, uint64_t *s, unsigned r)+__contract__(+ requires(r <= sizeof(uint64_t) * MLD_KECCAK_LANES)+ requires(nblocks <= 8 /* somewhat arbitrary bound */)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ requires(memory_no_alias(out0, nblocks * r))+ requires(memory_no_alias(out1, nblocks * r))+ requires(memory_no_alias(out2, nblocks * r))+ requires(memory_no_alias(out3, nblocks * r))+ assigns(memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ assigns(memory_slice(out0, nblocks * r))+ assigns(memory_slice(out1, nblocks * r))+ assigns(memory_slice(out2, nblocks * r))+ assigns(memory_slice(out3, nblocks * r)))+{+ size_t current_offset = 0;+ while (nblocks > 0)+ __loop__(+ assigns(nblocks, current_offset,+ memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY),+ memory_slice(out0, nblocks * r),+ memory_slice(out1, nblocks * r),+ memory_slice(out2, nblocks * r),+ memory_slice(out3, nblocks * r))+ invariant(nblocks <= loop_entry(nblocks))+ invariant(current_offset == (loop_entry(nblocks) - nblocks) * r)+ decreases(nblocks))+ {+ mld_keccakf1600x4_permute(s);+ mld_keccakf1600x4_extract_bytes(+ s, &out0[current_offset], &out1[current_offset], &out2[current_offset],+ &out3[current_offset], 0, r);+ current_offset += r;+ nblocks--;+ }+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+void mld_shake128x4_absorb_once(mld_shake128x4ctx *state, const uint8_t *in0,+ const uint8_t *in1, const uint8_t *in2,+ const uint8_t *in3, size_t inlen)+{+ mld_memset(state, 0, sizeof(mld_shake128x4ctx));+ mld_keccak_absorb_once_x4(state->ctx, SHAKE128_RATE, in0, in1, in2, in3,+ inlen, 0x1F);+}++MLD_INTERNAL_API+void mld_shake128x4_squeezeblocks(uint8_t *out0, uint8_t *out1, uint8_t *out2,+ uint8_t *out3, size_t nblocks,+ mld_shake128x4ctx *state)+{+ mld_keccak_squeezeblocks_x4(out0, out1, out2, out3, nblocks, state->ctx,+ SHAKE128_RATE);+}++MLD_INTERNAL_API+void mld_shake128x4_init(mld_shake128x4ctx *state) { (void)state; }+MLD_INTERNAL_API+void mld_shake128x4_release(mld_shake128x4ctx *state)+{+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(state, sizeof(mld_shake128x4ctx));+}+#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+MLD_INTERNAL_API+void mld_shake256x4_absorb_once(mld_shake256x4ctx *state, const uint8_t *in0,+ const uint8_t *in1, const uint8_t *in2,+ const uint8_t *in3, size_t inlen)+{+ mld_memset(state, 0, sizeof(mld_shake256x4ctx));+ mld_keccak_absorb_once_x4(state->ctx, SHAKE256_RATE, in0, in1, in2, in3,+ inlen, 0x1F);+}++MLD_INTERNAL_API+void mld_shake256x4_squeezeblocks(uint8_t *out0, uint8_t *out1, uint8_t *out2,+ uint8_t *out3, size_t nblocks,+ mld_shake256x4ctx *state)+{+ mld_keccak_squeezeblocks_x4(out0, out1, out2, out3, nblocks, state->ctx,+ SHAKE256_RATE);+}++MLD_INTERNAL_API+void mld_shake256x4_init(mld_shake256x4ctx *state) { (void)state; }+MLD_INTERNAL_API+void mld_shake256x4_release(mld_shake256x4ctx *state)+{+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(state, sizeof(mld_shake256x4ctx));+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || (!MLD_CONFIG_NO_SIGN_API && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST)) */++#endif /* !MLD_CONFIG_MULTILEVEL_NO_SHARED && !MLD_CONFIG_SERIAL_FIPS202_ONLY \+ */
+ cbits/mldsa/src/fips202/fips202x4.h view
@@ -0,0 +1,125 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_FIPS202_FIPS202X4_H+#define MLD_FIPS202_FIPS202X4_H++#include "../common.h"++#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)++#include <stddef.h>++#include "../cbmc.h"+#include "fips202.h"+#include "keccakf1600.h"++/** Context for the non-incremental 4-way SHAKE128 API. */+typedef struct+{+ uint64_t ctx[MLD_KECCAK_LANES *+ MLD_KECCAK_WAY]; /**< 4-way Keccak state, stored sequentially. */+} mld_shake128x4ctx;++/** Context for the 4-way batched SHAKE256 XOF. */+typedef struct+{+ uint64_t ctx[MLD_KECCAK_LANES *+ MLD_KECCAK_WAY]; /**< Interleaved 4-way Keccak state. */+} mld_shake256x4ctx;++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+#define mld_shake128x4_absorb_once MLD_NAMESPACE(shake128x4_absorb_once)+MLD_INTERNAL_API+void mld_shake128x4_absorb_once(mld_shake128x4ctx *state, const uint8_t *in0,+ const uint8_t *in1, const uint8_t *in2,+ const uint8_t *in3, size_t inlen)+__contract__(+ requires(inlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(state, sizeof(mld_shake128x4ctx)))+ requires(memory_no_alias(in0, inlen))+ requires(memory_no_alias(in1, inlen))+ requires(memory_no_alias(in2, inlen))+ requires(memory_no_alias(in3, inlen))+ assigns(memory_slice(state, sizeof(mld_shake128x4ctx)))+);++#define mld_shake128x4_squeezeblocks MLD_NAMESPACE(shake128x4_squeezeblocks)+MLD_INTERNAL_API+void mld_shake128x4_squeezeblocks(uint8_t *out0, uint8_t *out1, uint8_t *out2,+ uint8_t *out3, size_t nblocks,+ mld_shake128x4ctx *state)+__contract__(+ requires(nblocks <= 8 /* somewhat arbitrary bound */)+ requires(memory_no_alias(state, sizeof(mld_shake128x4ctx)))+ requires(memory_no_alias(out0, nblocks * SHAKE128_RATE))+ requires(memory_no_alias(out1, nblocks * SHAKE128_RATE))+ requires(memory_no_alias(out2, nblocks * SHAKE128_RATE))+ requires(memory_no_alias(out3, nblocks * SHAKE128_RATE))+ assigns(memory_slice(out0, nblocks * SHAKE128_RATE),+ memory_slice(out1, nblocks * SHAKE128_RATE),+ memory_slice(out2, nblocks * SHAKE128_RATE),+ memory_slice(out3, nblocks * SHAKE128_RATE),+ memory_slice(state, sizeof(mld_shake128x4ctx)))+);++#define mld_shake128x4_init MLD_NAMESPACE(shake128x4_init)+MLD_INTERNAL_API+void mld_shake128x4_init(mld_shake128x4ctx *state);++#define mld_shake128x4_release MLD_NAMESPACE(shake128x4_release)+MLD_INTERNAL_API+void mld_shake128x4_release(mld_shake128x4ctx *state);+#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+#define mld_shake256x4_absorb_once MLD_NAMESPACE(shake256x4_absorb_once)+MLD_INTERNAL_API+void mld_shake256x4_absorb_once(mld_shake256x4ctx *state, const uint8_t *in0,+ const uint8_t *in1, const uint8_t *in2,+ const uint8_t *in3, size_t inlen)+__contract__(+ requires(inlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(state, sizeof(mld_shake256x4ctx)))+ requires(memory_no_alias(in0, inlen))+ requires(memory_no_alias(in1, inlen))+ requires(memory_no_alias(in2, inlen))+ requires(memory_no_alias(in3, inlen))+ assigns(memory_slice(state, sizeof(mld_shake256x4ctx)))+);++#define mld_shake256x4_squeezeblocks MLD_NAMESPACE(shake256x4_squeezeblocks)+MLD_INTERNAL_API+void mld_shake256x4_squeezeblocks(uint8_t *out0, uint8_t *out1, uint8_t *out2,+ uint8_t *out3, size_t nblocks,+ mld_shake256x4ctx *state)+__contract__(+ requires(nblocks <= 8 /* somewhat arbitrary bound */)+ requires(memory_no_alias(state, sizeof(mld_shake256x4ctx)))+ requires(memory_no_alias(out0, nblocks * SHAKE256_RATE))+ requires(memory_no_alias(out1, nblocks * SHAKE256_RATE))+ requires(memory_no_alias(out2, nblocks * SHAKE256_RATE))+ requires(memory_no_alias(out3, nblocks * SHAKE256_RATE))+ assigns(memory_slice(out0, nblocks * SHAKE256_RATE),+ memory_slice(out1, nblocks * SHAKE256_RATE),+ memory_slice(out2, nblocks * SHAKE256_RATE),+ memory_slice(out3, nblocks * SHAKE256_RATE),+ memory_slice(state, sizeof(mld_shake256x4ctx)))+);++#define mld_shake256x4_init MLD_NAMESPACE(shake256x4_init)+MLD_INTERNAL_API+void mld_shake256x4_init(mld_shake256x4ctx *state);++#define mld_shake256x4_release MLD_NAMESPACE(shake256x4_release)+MLD_INTERNAL_API+void mld_shake256x4_release(mld_shake256x4ctx *state);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || (!MLD_CONFIG_NO_SIGN_API && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST)) */++#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY */+#endif /* !MLD_FIPS202_FIPS202X4_H */
+ cbits/mldsa/src/fips202/keccakf1600.c view
@@ -0,0 +1,510 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [mupq]+ * Common files for pqm4, pqm3, pqriscv+ * Kannwischer, Petri, Rijneveld, Schwabe, Stoffelen+ * https://github.com/mupq/mupq+ *+ * - [supercop]+ * SUPERCOP benchmarking framework+ * Daniel J. Bernstein+ * http://bench.cr.yp.to/supercop.html+ *+ * - [tweetfips]+ * 'tweetfips202' FIPS202 implementation+ * Van Assche, Bernstein, Schwabe+ * https://keccak.team/2015/tweetfips202.html+ */++/* Based on the CC0 implementation from @[mupq] and the public domain+ * implementation @[supercop, crypto_hash/keccakc512/simple/]+ * by Ronny Van Keer, and the public domain @[tweetfips] implementation. */++#include "keccakf1600.h"+#if !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#define MLD_KECCAK_NROUNDS 24+#define MLD_KECCAK_ROL(a, offset) (((a) << (offset)) ^ ((a) >> (64 - (offset))))++MLD_INTERNAL_API+void mld_keccakf1600_extract_bytes(uint64_t *state, unsigned char *data,+ unsigned offset, unsigned length)+{+ unsigned i;+#if defined(MLD_SYS_LITTLE_ENDIAN)+ uint8_t *state_ptr = (uint8_t *)state + offset;+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ data[i] = state_ptr[i];+ }+#else /* MLD_SYS_LITTLE_ENDIAN */+ /* Portable version */+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ data[i] = (state[(offset + i) >> 3] >> (8 * ((offset + i) & 0x07))) & 0xFF;+ }+#endif /* !MLD_SYS_LITTLE_ENDIAN */+}++MLD_INTERNAL_API+void mld_keccakf1600_xor_bytes(uint64_t *state, const unsigned char *data,+ unsigned offset, unsigned length)+{+ unsigned i;+#if defined(MLD_SYS_LITTLE_ENDIAN)+ uint8_t *state_ptr = (uint8_t *)state + offset;+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ state_ptr[i] ^= data[i];+ }+#else /* MLD_SYS_LITTLE_ENDIAN */+ /* Portable version */+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ state[(offset + i) >> 3] ^= (uint64_t)data[i]+ << (8 * ((offset + i) & 0x07));+ }+#endif /* !MLD_SYS_LITTLE_ENDIAN */+}++#if (!defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_REDUCE_RAM) || \+ defined(MLD_UNIT_TEST)) && \+ !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+static void mld_keccakf1600x4_extract_bytes_c(uint64_t *state,+ unsigned char *data0,+ unsigned char *data1,+ unsigned char *data2,+ unsigned char *data3,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLD_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLD_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ requires(memory_no_alias(data1, length))+ requires(memory_no_alias(data2, length))+ requires(memory_no_alias(data3, length))+ assigns(memory_slice(data0, length))+ assigns(memory_slice(data1, length))+ assigns(memory_slice(data2, length))+ assigns(memory_slice(data3, length))+)+{+ mld_keccakf1600_extract_bytes(state + MLD_KECCAK_LANES * 0, data0, offset,+ length);+ mld_keccakf1600_extract_bytes(state + MLD_KECCAK_LANES * 1, data1, offset,+ length);+ mld_keccakf1600_extract_bytes(state + MLD_KECCAK_LANES * 2, data2, offset,+ length);+ mld_keccakf1600_extract_bytes(state + MLD_KECCAK_LANES * 3, data3, offset,+ length);+}++MLD_INTERNAL_API+void mld_keccakf1600x4_extract_bytes(uint64_t *state, unsigned char *data0,+ unsigned char *data1, unsigned char *data2,+ unsigned char *data3, unsigned offset,+ unsigned length)+{+#if defined(MLD_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES)+ if (mld_keccakf1600_extract_bytes_x4_native(state, data0, data1, data2, data3,+ offset, length) ==+ MLD_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLD_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES */+ mld_keccakf1600x4_extract_bytes_c(state, data0, data1, data2, data3, offset,+ length);+}++static void mld_keccakf1600x4_xor_bytes_c(uint64_t *state,+ const unsigned char *data0,+ const unsigned char *data1,+ const unsigned char *data2,+ const unsigned char *data3,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLD_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLD_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ /* Case 1: all input buffers are distinct; Case 2: All input buffers are the same */+ requires((data0 == data1 &&+ data0 == data2 &&+ data0 == data3) ||+ (memory_no_alias(data1, length) &&+ memory_no_alias(data2, length) &&+ memory_no_alias(data3, length)))+ assigns(memory_slice(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+)+{+ mld_keccakf1600_xor_bytes(state + MLD_KECCAK_LANES * 0, data0, offset,+ length);+ mld_keccakf1600_xor_bytes(state + MLD_KECCAK_LANES * 1, data1, offset,+ length);+ mld_keccakf1600_xor_bytes(state + MLD_KECCAK_LANES * 2, data2, offset,+ length);+ mld_keccakf1600_xor_bytes(state + MLD_KECCAK_LANES * 3, data3, offset,+ length);+}++MLD_INTERNAL_API+void mld_keccakf1600x4_xor_bytes(uint64_t *state, const unsigned char *data0,+ const unsigned char *data1,+ const unsigned char *data2,+ const unsigned char *data3, unsigned offset,+ unsigned length)+{+#if defined(MLD_USE_NATIVE_FIPS202_X4_XOR_BYTES)+ if (mld_keccakf1600_xor_bytes_x4_native(state, data0, data1, data2, data3,+ offset,+ length) == MLD_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLD_USE_NATIVE_FIPS202_X4_XOR_BYTES */+ mld_keccakf1600x4_xor_bytes_c(state, data0, data1, data2, data3, offset,+ length);+}++MLD_INTERNAL_API+void mld_keccakf1600x4_permute(uint64_t *state)+{+#if defined(MLD_USE_NATIVE_FIPS202_X4)+ if (mld_keccak_f1600_x4_native(state) == MLD_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLD_USE_NATIVE_FIPS202_X4 */+ mld_keccakf1600_permute(state + MLD_KECCAK_LANES * 0);+ mld_keccakf1600_permute(state + MLD_KECCAK_LANES * 1);+ mld_keccakf1600_permute(state + MLD_KECCAK_LANES * 2);+ mld_keccakf1600_permute(state + MLD_KECCAK_LANES * 3);+}+#endif /* (!MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST) && !MLD_CONFIG_SERIAL_FIPS202_ONLY */++static const uint64_t mld_KeccakF_RoundConstants[MLD_KECCAK_NROUNDS] = {+ (uint64_t)0x0000000000000001ULL, (uint64_t)0x0000000000008082ULL,+ (uint64_t)0x800000000000808aULL, (uint64_t)0x8000000080008000ULL,+ (uint64_t)0x000000000000808bULL, (uint64_t)0x0000000080000001ULL,+ (uint64_t)0x8000000080008081ULL, (uint64_t)0x8000000000008009ULL,+ (uint64_t)0x000000000000008aULL, (uint64_t)0x0000000000000088ULL,+ (uint64_t)0x0000000080008009ULL, (uint64_t)0x000000008000000aULL,+ (uint64_t)0x000000008000808bULL, (uint64_t)0x800000000000008bULL,+ (uint64_t)0x8000000000008089ULL, (uint64_t)0x8000000000008003ULL,+ (uint64_t)0x8000000000008002ULL, (uint64_t)0x8000000000000080ULL,+ (uint64_t)0x000000000000800aULL, (uint64_t)0x800000008000000aULL,+ (uint64_t)0x8000000080008081ULL, (uint64_t)0x8000000000008080ULL,+ (uint64_t)0x0000000080000001ULL, (uint64_t)0x8000000080008008ULL};++MLD_STATIC_TESTABLE+void mld_keccakf1600_permute_c(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES))+ assigns(memory_slice(state, sizeof(uint64_t) * MLD_KECCAK_LANES))+)+{+ unsigned round;++ uint64_t Aba, Abe, Abi, Abo, Abu;+ uint64_t Aga, Age, Agi, Ago, Agu;+ uint64_t Aka, Ake, Aki, Ako, Aku;+ uint64_t Ama, Ame, Ami, Amo, Amu;+ uint64_t Asa, Ase, Asi, Aso, Asu;+ uint64_t BCa, BCe, BCi, BCo, BCu;+ uint64_t Da, De, Di, Do, Du;+ uint64_t Eba, Ebe, Ebi, Ebo, Ebu;+ uint64_t Ega, Ege, Egi, Ego, Egu;+ uint64_t Eka, Eke, Eki, Eko, Eku;+ uint64_t Ema, Eme, Emi, Emo, Emu;+ uint64_t Esa, Ese, Esi, Eso, Esu;++ /* copyFromState(A, state) */+ Aba = state[0];+ Abe = state[1];+ Abi = state[2];+ Abo = state[3];+ Abu = state[4];+ Aga = state[5];+ Age = state[6];+ Agi = state[7];+ Ago = state[8];+ Agu = state[9];+ Aka = state[10];+ Ake = state[11];+ Aki = state[12];+ Ako = state[13];+ Aku = state[14];+ Ama = state[15];+ Ame = state[16];+ Ami = state[17];+ Amo = state[18];+ Amu = state[19];+ Asa = state[20];+ Ase = state[21];+ Asi = state[22];+ Aso = state[23];+ Asu = state[24];++ for (round = 0; round < MLD_KECCAK_NROUNDS; round += 2)+ __loop__(invariant(round <= MLD_KECCAK_NROUNDS && round % 2 == 0)+ decreases(MLD_KECCAK_NROUNDS - round))+ {+ /* prepareTheta */+ BCa = Aba ^ Aga ^ Aka ^ Ama ^ Asa;+ BCe = Abe ^ Age ^ Ake ^ Ame ^ Ase;+ BCi = Abi ^ Agi ^ Aki ^ Ami ^ Asi;+ BCo = Abo ^ Ago ^ Ako ^ Amo ^ Aso;+ BCu = Abu ^ Agu ^ Aku ^ Amu ^ Asu;++ /* thetaRhoPiChiIotaPrepareTheta(round, A, E) */+ Da = BCu ^ MLD_KECCAK_ROL(BCe, 1);+ De = BCa ^ MLD_KECCAK_ROL(BCi, 1);+ Di = BCe ^ MLD_KECCAK_ROL(BCo, 1);+ Do = BCi ^ MLD_KECCAK_ROL(BCu, 1);+ Du = BCo ^ MLD_KECCAK_ROL(BCa, 1);++ Aba ^= Da;+ BCa = Aba;+ Age ^= De;+ BCe = MLD_KECCAK_ROL(Age, 44);+ Aki ^= Di;+ BCi = MLD_KECCAK_ROL(Aki, 43);+ Amo ^= Do;+ BCo = MLD_KECCAK_ROL(Amo, 21);+ Asu ^= Du;+ BCu = MLD_KECCAK_ROL(Asu, 14);+ Eba = BCa ^ ((~BCe) & BCi);+ Eba ^= (uint64_t)mld_KeccakF_RoundConstants[round];+ Ebe = BCe ^ ((~BCi) & BCo);+ Ebi = BCi ^ ((~BCo) & BCu);+ Ebo = BCo ^ ((~BCu) & BCa);+ Ebu = BCu ^ ((~BCa) & BCe);++ Abo ^= Do;+ BCa = MLD_KECCAK_ROL(Abo, 28);+ Agu ^= Du;+ BCe = MLD_KECCAK_ROL(Agu, 20);+ Aka ^= Da;+ BCi = MLD_KECCAK_ROL(Aka, 3);+ Ame ^= De;+ BCo = MLD_KECCAK_ROL(Ame, 45);+ Asi ^= Di;+ BCu = MLD_KECCAK_ROL(Asi, 61);+ Ega = BCa ^ ((~BCe) & BCi);+ Ege = BCe ^ ((~BCi) & BCo);+ Egi = BCi ^ ((~BCo) & BCu);+ Ego = BCo ^ ((~BCu) & BCa);+ Egu = BCu ^ ((~BCa) & BCe);++ Abe ^= De;+ BCa = MLD_KECCAK_ROL(Abe, 1);+ Agi ^= Di;+ BCe = MLD_KECCAK_ROL(Agi, 6);+ Ako ^= Do;+ BCi = MLD_KECCAK_ROL(Ako, 25);+ Amu ^= Du;+ BCo = MLD_KECCAK_ROL(Amu, 8);+ Asa ^= Da;+ BCu = MLD_KECCAK_ROL(Asa, 18);+ Eka = BCa ^ ((~BCe) & BCi);+ Eke = BCe ^ ((~BCi) & BCo);+ Eki = BCi ^ ((~BCo) & BCu);+ Eko = BCo ^ ((~BCu) & BCa);+ Eku = BCu ^ ((~BCa) & BCe);++ Abu ^= Du;+ BCa = MLD_KECCAK_ROL(Abu, 27);+ Aga ^= Da;+ BCe = MLD_KECCAK_ROL(Aga, 36);+ Ake ^= De;+ BCi = MLD_KECCAK_ROL(Ake, 10);+ Ami ^= Di;+ BCo = MLD_KECCAK_ROL(Ami, 15);+ Aso ^= Do;+ BCu = MLD_KECCAK_ROL(Aso, 56);+ Ema = BCa ^ ((~BCe) & BCi);+ Eme = BCe ^ ((~BCi) & BCo);+ Emi = BCi ^ ((~BCo) & BCu);+ Emo = BCo ^ ((~BCu) & BCa);+ Emu = BCu ^ ((~BCa) & BCe);++ Abi ^= Di;+ BCa = MLD_KECCAK_ROL(Abi, 62);+ Ago ^= Do;+ BCe = MLD_KECCAK_ROL(Ago, 55);+ Aku ^= Du;+ BCi = MLD_KECCAK_ROL(Aku, 39);+ Ama ^= Da;+ BCo = MLD_KECCAK_ROL(Ama, 41);+ Ase ^= De;+ BCu = MLD_KECCAK_ROL(Ase, 2);+ Esa = BCa ^ ((~BCe) & BCi);+ Ese = BCe ^ ((~BCi) & BCo);+ Esi = BCi ^ ((~BCo) & BCu);+ Eso = BCo ^ ((~BCu) & BCa);+ Esu = BCu ^ ((~BCa) & BCe);++ /* prepareTheta */+ BCa = Eba ^ Ega ^ Eka ^ Ema ^ Esa;+ BCe = Ebe ^ Ege ^ Eke ^ Eme ^ Ese;+ BCi = Ebi ^ Egi ^ Eki ^ Emi ^ Esi;+ BCo = Ebo ^ Ego ^ Eko ^ Emo ^ Eso;+ BCu = Ebu ^ Egu ^ Eku ^ Emu ^ Esu;++ /* thetaRhoPiChiIotaPrepareTheta(round+1, E, A) */+ Da = BCu ^ MLD_KECCAK_ROL(BCe, 1);+ De = BCa ^ MLD_KECCAK_ROL(BCi, 1);+ Di = BCe ^ MLD_KECCAK_ROL(BCo, 1);+ Do = BCi ^ MLD_KECCAK_ROL(BCu, 1);+ Du = BCo ^ MLD_KECCAK_ROL(BCa, 1);++ Eba ^= Da;+ BCa = Eba;+ Ege ^= De;+ BCe = MLD_KECCAK_ROL(Ege, 44);+ Eki ^= Di;+ BCi = MLD_KECCAK_ROL(Eki, 43);+ Emo ^= Do;+ BCo = MLD_KECCAK_ROL(Emo, 21);+ Esu ^= Du;+ BCu = MLD_KECCAK_ROL(Esu, 14);+ Aba = BCa ^ ((~BCe) & BCi);+ Aba ^= (uint64_t)mld_KeccakF_RoundConstants[round + 1];+ Abe = BCe ^ ((~BCi) & BCo);+ Abi = BCi ^ ((~BCo) & BCu);+ Abo = BCo ^ ((~BCu) & BCa);+ Abu = BCu ^ ((~BCa) & BCe);++ Ebo ^= Do;+ BCa = MLD_KECCAK_ROL(Ebo, 28);+ Egu ^= Du;+ BCe = MLD_KECCAK_ROL(Egu, 20);+ Eka ^= Da;+ BCi = MLD_KECCAK_ROL(Eka, 3);+ Eme ^= De;+ BCo = MLD_KECCAK_ROL(Eme, 45);+ Esi ^= Di;+ BCu = MLD_KECCAK_ROL(Esi, 61);+ Aga = BCa ^ ((~BCe) & BCi);+ Age = BCe ^ ((~BCi) & BCo);+ Agi = BCi ^ ((~BCo) & BCu);+ Ago = BCo ^ ((~BCu) & BCa);+ Agu = BCu ^ ((~BCa) & BCe);++ Ebe ^= De;+ BCa = MLD_KECCAK_ROL(Ebe, 1);+ Egi ^= Di;+ BCe = MLD_KECCAK_ROL(Egi, 6);+ Eko ^= Do;+ BCi = MLD_KECCAK_ROL(Eko, 25);+ Emu ^= Du;+ BCo = MLD_KECCAK_ROL(Emu, 8);+ Esa ^= Da;+ BCu = MLD_KECCAK_ROL(Esa, 18);+ Aka = BCa ^ ((~BCe) & BCi);+ Ake = BCe ^ ((~BCi) & BCo);+ Aki = BCi ^ ((~BCo) & BCu);+ Ako = BCo ^ ((~BCu) & BCa);+ Aku = BCu ^ ((~BCa) & BCe);++ Ebu ^= Du;+ BCa = MLD_KECCAK_ROL(Ebu, 27);+ Ega ^= Da;+ BCe = MLD_KECCAK_ROL(Ega, 36);+ Eke ^= De;+ BCi = MLD_KECCAK_ROL(Eke, 10);+ Emi ^= Di;+ BCo = MLD_KECCAK_ROL(Emi, 15);+ Eso ^= Do;+ BCu = MLD_KECCAK_ROL(Eso, 56);+ Ama = BCa ^ ((~BCe) & BCi);+ Ame = BCe ^ ((~BCi) & BCo);+ Ami = BCi ^ ((~BCo) & BCu);+ Amo = BCo ^ ((~BCu) & BCa);+ Amu = BCu ^ ((~BCa) & BCe);++ Ebi ^= Di;+ BCa = MLD_KECCAK_ROL(Ebi, 62);+ Ego ^= Do;+ BCe = MLD_KECCAK_ROL(Ego, 55);+ Eku ^= Du;+ BCi = MLD_KECCAK_ROL(Eku, 39);+ Ema ^= Da;+ BCo = MLD_KECCAK_ROL(Ema, 41);+ Ese ^= De;+ BCu = MLD_KECCAK_ROL(Ese, 2);+ Asa = BCa ^ ((~BCe) & BCi);+ Ase = BCe ^ ((~BCi) & BCo);+ Asi = BCi ^ ((~BCo) & BCu);+ Aso = BCo ^ ((~BCu) & BCa);+ Asu = BCu ^ ((~BCa) & BCe);+ }++ /* copyToState(state, A) */+ state[0] = Aba;+ state[1] = Abe;+ state[2] = Abi;+ state[3] = Abo;+ state[4] = Abu;+ state[5] = Aga;+ state[6] = Age;+ state[7] = Agi;+ state[8] = Ago;+ state[9] = Agu;+ state[10] = Aka;+ state[11] = Ake;+ state[12] = Aki;+ state[13] = Ako;+ state[14] = Aku;+ state[15] = Ama;+ state[16] = Ame;+ state[17] = Ami;+ state[18] = Amo;+ state[19] = Amu;+ state[20] = Asa;+ state[21] = Ase;+ state[22] = Asi;+ state[23] = Aso;+ state[24] = Asu;+}++MLD_INTERNAL_API+void mld_keccakf1600_permute(uint64_t *state)+{+#if defined(MLD_USE_NATIVE_FIPS202_X1)+ if (mld_keccak_f1600_x1_native(state) == MLD_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLD_USE_NATIVE_FIPS202_X1 */+ mld_keccakf1600_permute_c(state);+}++#else /* !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(keccakf1600)++#endif /* MLD_CONFIG_MULTILEVEL_NO_SHARED */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef MLD_KECCAK_NROUNDS+#undef MLD_KECCAK_ROL
+ cbits/mldsa/src/fips202/keccakf1600.h view
@@ -0,0 +1,110 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_FIPS202_KECCAKF1600_H+#define MLD_FIPS202_KECCAKF1600_H+#include "../cbmc.h"+#include "../common.h"+#include "fips202.h"++#define MLD_KECCAK_LANES 25+#define MLD_KECCAK_WAY 4++/*+ * WARNING:+ * The contents of this structure, including the placement+ * and interleaving of Keccak lanes, are IMPLEMENTATION-DEFINED.+ * The struct is only exposed here to allow its construction on the stack.+ */++#define mld_keccakf1600_extract_bytes MLD_NAMESPACE(keccakf1600_extract_bytes)+MLD_INTERNAL_API+void mld_keccakf1600_extract_bytes(uint64_t *state, unsigned char *data,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLD_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLD_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES))+ requires(memory_no_alias(data, length))+ assigns(memory_slice(data, length))+);++#define mld_keccakf1600_xor_bytes MLD_NAMESPACE(keccakf1600_xor_bytes)+MLD_INTERNAL_API+void mld_keccakf1600_xor_bytes(uint64_t *state, const unsigned char *data,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLD_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLD_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES))+ requires(memory_no_alias(data, length))+ assigns(memory_slice(state, sizeof(uint64_t) * MLD_KECCAK_LANES))+);++#if (!defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_REDUCE_RAM) || \+ defined(MLD_UNIT_TEST)) && \+ !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+#define mld_keccakf1600x4_extract_bytes \+ MLD_NAMESPACE(keccakf1600x4_extract_bytes)+MLD_INTERNAL_API+void mld_keccakf1600x4_extract_bytes(uint64_t *state, unsigned char *data0,+ unsigned char *data1, unsigned char *data2,+ unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLD_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLD_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ requires(memory_no_alias(data1, length))+ requires(memory_no_alias(data2, length))+ requires(memory_no_alias(data3, length))+ assigns(memory_slice(data0, length))+ assigns(memory_slice(data1, length))+ assigns(memory_slice(data2, length))+ assigns(memory_slice(data3, length))+);++#define mld_keccakf1600x4_xor_bytes MLD_NAMESPACE(keccakf1600x4_xor_bytes)+MLD_INTERNAL_API+void mld_keccakf1600x4_xor_bytes(uint64_t *state, const unsigned char *data0,+ const unsigned char *data1,+ const unsigned char *data2,+ const unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLD_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLD_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ /* Case 1: all input buffers are distinct; Case 2: All input buffers are the same */+ requires((data0 == data1 &&+ data0 == data2 &&+ data0 == data3) ||+ (memory_no_alias(data1, length) &&+ memory_no_alias(data2, length) &&+ memory_no_alias(data3, length)))+ assigns(memory_slice(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+);++#define mld_keccakf1600x4_permute MLD_NAMESPACE(keccakf1600x4_permute)+MLD_INTERNAL_API+void mld_keccakf1600x4_permute(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ assigns(memory_slice(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+);+#endif /* (!MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST) && !MLD_CONFIG_SERIAL_FIPS202_ONLY */++#define mld_keccakf1600_permute MLD_NAMESPACE(keccakf1600_permute)+MLD_INTERNAL_API+void mld_keccakf1600_permute(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES))+ assigns(memory_slice(state, sizeof(uint64_t) * MLD_KECCAK_LANES))+);++#endif /* !MLD_FIPS202_KECCAKF1600_H */
+ cbits/mldsa/src/fips202/native/aarch64/auto.h view
@@ -0,0 +1,85 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [HYBRID]+ * Hybrid scalar/vector implementations of Keccak and SPHINCS+ on AArch64+ * Becker, Kannwischer+ * https://eprint.iacr.org/2022/1243+ */++#ifndef MLD_FIPS202_NATIVE_AARCH64_AUTO_H+#define MLD_FIPS202_NATIVE_AARCH64_AUTO_H+/* Default FIPS202 assembly profile for AArch64 systems */++/*+ * Default logic to decide which implementation to use.+ *+ */++/*+ * Keccak-f1600+ *+ * - On Arm-based Apple CPUs, or CPUs with MLD_SYS_AARCH64_FAST_SHA3 set,+ * we pick a pure Neon implementation.+ * - Otherwise, unless MLD_SYS_AARCH64_SLOW_BARREL_SHIFTER is set,+ * we use lazy-rotation scalar assembly from @[HYBRID].+ * - Otherwise, if MLD_SYS_AARCH64_SLOW_BARREL_SHIFTER is set, we+ * fall back to the standard C implementation.+ */+#if defined(__ARM_FEATURE_SHA3) && \+ (defined(__APPLE__) || defined(MLD_SYS_AARCH64_FAST_SHA3))+#include "x1_v84a.h"+#elif !defined(MLD_SYS_AARCH64_SLOW_BARREL_SHIFTER)+#include "x1_scalar.h"+#endif++#if (!defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_REDUCE_RAM)) && \+ !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+/* Batched, SIMD-based Keccak-f1600 implementations. */+#if defined(MLD_SYS_AARCH64_NEON)++/*+ * Keccak-f1600x2/x4+ *+ * The optimal implementation is highly CPU-specific; see @[HYBRID].+ *+ * For now, if v8.4-A is not implemented, we fall back to Keccak-f1600.+ * If v8.4-A is implemented and we are on an Apple CPU (or a CPU with+ * MLD_SYS_AARCH64_FAST_SHA3 set), we use a plain Neon-based implementation.+ * If v8.4-A is implemented and we are on neither, we use a+ * scalar/Neon/Neon hybrid.+ * The reason for this distinction is that Apple CPUs (and other CPUs flagged+ * via MLD_SYS_AARCH64_FAST_SHA3) appear to implement the SHA3 instructions on+ * all SIMD units, while Arm CPUs prior to Cortex-X4 don't, and ordinary Neon+ * instructions are still needed.+ */+#if defined(__ARM_FEATURE_SHA3)+/*+ * For Apple-M cores (and other cores flagged via MLD_SYS_AARCH64_FAST_SHA3),+ * we use a plain implementation leveraging SHA3 instructions only.+ */+#if defined(__APPLE__) || defined(MLD_SYS_AARCH64_FAST_SHA3)+#include "x2_v84a.h"+#else+#include "x4_v8a_v84a_scalar.h"+#endif++#else /* __ARM_FEATURE_SHA3 */++#include "x4_v8a_scalar.h"++#endif /* !__ARM_FEATURE_SHA3 */++#endif /* MLD_SYS_AARCH64_NEON */++#endif /* (!MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_REDUCE_RAM) && !MLD_CONFIG_SERIAL_FIPS202_ONLY */++#endif /* !MLD_FIPS202_NATIVE_AARCH64_AUTO_H */
+ cbits/mldsa/src/fips202/native/aarch64/src/fips202_native_aarch64.h view
@@ -0,0 +1,69 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H+#define MLD_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H+++#include "../../../../cbmc.h"+#include "../../../../common.h"+++#define mld_keccakf1600_round_constants \+ MLD_NAMESPACE(keccakf1600_round_constants)+MLD_INTERNAL_DATA_DECLARATION const uint64_t+ mld_keccakf1600_round_constants[24];++#define mld_keccak_f1600_x1_scalar_aarch64_asm \+ MLD_NAMESPACE(keccak_f1600_x1_scalar_aarch64_asm)+void mld_keccak_f1600_x1_scalar_aarch64_asm(uint64_t state[25],+ const uint64_t rc[24])+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 1))+ requires(rc == mld_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 1))+);++#define mld_keccak_f1600_x1_v84a_aarch64_asm \+ MLD_NAMESPACE(keccak_f1600_x1_v84a_aarch64_asm)+void mld_keccak_f1600_x1_v84a_aarch64_asm(uint64_t state[25],+ const uint64_t rc[24])+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 1))+ requires(rc == mld_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 1))+);++#define mld_keccak_f1600_x2_v84a_aarch64_asm \+ MLD_NAMESPACE(keccak_f1600_x2_v84a_aarch64_asm)+void mld_keccak_f1600_x2_v84a_aarch64_asm(uint64_t state[50],+ const uint64_t rc[24])+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 2))+ requires(rc == mld_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 2))+);++#define mld_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm \+ MLD_NAMESPACE(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)+void mld_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(uint64_t state[100],+ const uint64_t rc[24])+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(rc == mld_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+);++#define mld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm \+ MLD_NAMESPACE(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)+void mld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm(+ uint64_t state[100], const uint64_t rc[24])+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(rc == mld_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+);++#endif /* !MLD_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H */
+ cbits/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S view
@@ -0,0 +1,378 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++// Author: Hanno Becker <hanno.becker@arm.com>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>++/*yaml+ Name: keccak_f1600_x1_scalar_aarch64_asm+ Description: AArch64 scalar implementation of Keccak-f[1600] permutation for single state+ Signature: void mld_keccak_f1600_x1_scalar_aarch64_asm(uint64_t state[25], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ x0:+ type: buffer+ size_bytes: 200+ permissions: read/write+ c_parameter: uint64_t state[25]+ description: Keccak state (25 x uint64_t)+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: uint64_t const *rc+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 128+ description: register preservation and temporary storage+*/++#include "../../../../common.h"+#if defined(MLD_FIPS202_AARCH64_NEED_X1_SCALAR) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(keccak_f1600_x1_scalar_aarch64_asm)+MLD_ASM_FN_SYMBOL(keccak_f1600_x1_scalar_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x80+ .cfi_adjust_cfa_offset 0x80+ stp x19, x20, [sp, #0x20]+ .cfi_rel_offset x19, 0x20+ .cfi_rel_offset x20, 0x28+ stp x21, x22, [sp, #0x30]+ .cfi_rel_offset x21, 0x30+ .cfi_rel_offset x22, 0x38+ stp x23, x24, [sp, #0x40]+ .cfi_rel_offset x23, 0x40+ .cfi_rel_offset x24, 0x48+ stp x25, x26, [sp, #0x50]+ .cfi_rel_offset x25, 0x50+ .cfi_rel_offset x26, 0x58+ stp x27, x28, [sp, #0x60]+ .cfi_rel_offset x27, 0x60+ .cfi_rel_offset x28, 0x68+ stp x29, x30, [sp, #0x70]+ .cfi_rel_offset x29, 0x70+ .cfi_rel_offset x30, 0x78++Lmld_keccak_f1600_x1_scalar_initial:+ mov x26, x1+ str x1, [sp, #0x8]+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ str x0, [sp]+ eor x30, x24, x25+ eor x27, x9, x10+ eor x0, x30, x21+ eor x26, x27, x6+ eor x27, x26, x7+ eor x29, x0, x22+ eor x26, x29, x23+ eor x29, x4, x5+ eor x30, x29, x1+ eor x0, x27, x8+ eor x29, x30, x2+ eor x30, x19, x20+ eor x30, x30, x16+ eor x27, x26, x0, ror #63+ eor x4, x4, x27+ eor x30, x30, x17+ eor x30, x30, x28+ eor x29, x29, x3+ eor x0, x0, x30, ror #63+ eor x30, x30, x29, ror #63+ eor x22, x22, x30+ eor x23, x23, x30+ str x23, [sp, #0x18]+ eor x23, x14, x15+ eor x14, x14, x0+ eor x23, x23, x11+ eor x15, x15, x0+ eor x1, x1, x27+ eor x23, x23, x12+ eor x23, x23, x13+ eor x11, x11, x0+ eor x29, x29, x23, ror #63+ eor x23, x23, x26, ror #63+ eor x26, x13, x0+ eor x13, x28, x23+ eor x28, x24, x30+ eor x24, x16, x23+ eor x16, x21, x30+ eor x21, x25, x30+ eor x30, x19, x23+ eor x19, x20, x23+ eor x20, x17, x23+ eor x17, x12, x0+ eor x0, x2, x27+ eor x2, x6, x29+ eor x6, x8, x29+ bic x8, x28, x13, ror #47+ eor x12, x3, x27+ bic x3, x13, x17, ror #19+ eor x5, x5, x27+ ldr x27, [sp, #0x18]+ bic x25, x17, x2, ror #5+ eor x9, x9, x29+ eor x23, x25, x5, ror #52+ eor x3, x3, x2, ror #24+ eor x8, x8, x17, ror #2+ eor x17, x10, x29+ bic x25, x12, x22, ror #47+ eor x29, x7, x29+ bic x10, x4, x27, ror #2+ bic x7, x5, x28, ror #10+ eor x10, x10, x20, ror #50+ eor x13, x7, x13, ror #57+ bic x7, x2, x5, ror #47+ eor x2, x25, x24, ror #39+ bic x25, x20, x11, ror #57+ bic x5, x17, x4, ror #25+ eor x25, x25, x17, ror #53+ bic x17, x11, x17, ror #60+ eor x28, x7, x28, ror #57+ bic x7, x9, x12, ror #42+ eor x7, x7, x22, ror #25+ bic x22, x22, x24, ror #56+ bic x24, x24, x15, ror #31+ eor x22, x22, x15, ror #23+ bic x20, x27, x20, ror #48+ bic x15, x15, x9, ror #16+ eor x12, x15, x12, ror #58+ eor x15, x5, x27, ror #27+ eor x5, x20, x11, ror #41+ ldr x11, [sp, #0x8]+ eor x20, x17, x4, ror #21+ eor x17, x24, x9, ror #47+ mov x24, #0x1 // =1+ bic x9, x0, x16, ror #9+ str x24, [sp, #0x10]+ bic x24, x29, x1, ror #44+ bic x27, x1, x21, ror #50+ bic x4, x26, x29, ror #63+ eor x1, x1, x4, ror #21+ ldr x11, [x11]+ bic x4, x21, x30, ror #57+ eor x21, x24, x21, ror #30+ eor x24, x9, x19, ror #44+ bic x9, x14, x6, ror #5+ eor x9, x9, x0, ror #43+ bic x0, x6, x0, ror #38+ eor x1, x1, x11+ eor x11, x4, x26, ror #35+ eor x4, x0, x16, ror #47+ bic x0, x16, x19, ror #35+ eor x16, x27, x30, ror #43+ bic x27, x30, x26, ror #42+ bic x26, x19, x14, ror #41+ eor x19, x0, x14, ror #12+ eor x14, x26, x6, ror #46+ eor x6, x27, x29, ror #41++Lmld_keccak_f1600_x1_scalar_loop:+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor x26, x8, x9, ror #57+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ eor x30, x23, x22, ror #50+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ eor x26, x30, x21, ror #26+ eor x26, x26, x25, ror #15+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ eor x16, x30, x16+ eor x28, x30, x28, ror #63+ str x28, [sp, #0x18]+ eor x29, x29, x17, ror #36+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ eor x27, x28, x27, ror #61+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ eor x20, x26, x3, ror #39+ eor x11, x0, x11, ror #50+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ eor x1, x30, x17, ror #36+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ eor x17, x27, x7, ror #19+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ eor x4, x26, x4, ror #54+ eor x0, x0, x12, ror #3+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ eor x26, x26, x5, ror #25+ eor x2, x7, x16, ror #39+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ eor x7, x7, x22, ror #25+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ eor x30, x27, x6, ror #43+ eor x22, x20, x15, ror #23+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ bic x5, x13, x17, ror #63+ eor x5, x21, x5, ror #21+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ bic x21, x21, x25, ror #50+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ eor x21, x17, x25, ror #30+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x10]+ eor x17, x10, x9, ror #47+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ bic x20, x4, x28, ror #2+ eor x10, x20, x1, ror #50+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ bic x4, x28, x1, ror #48+ bic x1, x1, x11, ror #57+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0x18]+ add x25, x25, #0x1+ str x25, [sp, #0x10]+ cmp x25, #0x17+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ eor x11, x19, x13, ror #35+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ bic x27, x24, x9, ror #47+ bic x19, x23, x3, ror #9+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic x29, x3, x29, ror #35+ eor x13, x13, x9, ror #57+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ bic x14, x14, x8, ror #5+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ b.le Lmld_keccak_f1600_x1_scalar_loop+ ror x6, x6, #0x2b+ ror x11, x11, #0x32+ ror x21, x21, #0x14+ ror x2, x2, #0x3d+ ror x7, x7, #0x13+ ror x12, x12, #0x3+ ror x17, x17, #0x24+ ror x22, x22, #0x2c+ ror x3, x3, #0x27+ ror x8, x8, #0x38+ ror x13, x13, #0x2e+ ror x28, x28, #0x3f+ ror x23, x23, #0x3a+ ror x4, x4, #0x36+ ror x9, x9, #0x31+ ror x14, x14, #0x8+ ror x19, x19, #0x25+ ror x24, x24, #0x1c+ ror x5, x5, #0x19+ ror x10, x10, #0x17+ ror x15, x15, #0x3e+ ror x20, x20, #0x2+ ror x25, x25, #0x9+ ldr x0, [sp]+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ ldp x19, x20, [sp, #0x20]+ .cfi_restore x19+ .cfi_restore x20+ ldp x21, x22, [sp, #0x30]+ .cfi_restore x21+ .cfi_restore x22+ ldp x23, x24, [sp, #0x40]+ .cfi_restore x23+ .cfi_restore x24+ ldp x25, x26, [sp, #0x50]+ .cfi_restore x25+ .cfi_restore x26+ ldp x27, x28, [sp, #0x60]+ .cfi_restore x27+ .cfi_restore x28+ ldp x29, x30, [sp, #0x70]+ .cfi_restore x29+ .cfi_restore x30+ add sp, sp, #0x80+ .cfi_adjust_cfa_offset -0x80+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(keccak_f1600_x1_scalar_aarch64_asm)++#endif /* MLD_FIPS202_AARCH64_NEED_X1_SCALAR && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S view
@@ -0,0 +1,207 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [HYBRID]+ * Hybrid scalar/vector implementations of Keccak and SPHINCS+ on AArch64+ * Becker, Kannwischer+ * https://eprint.iacr.org/2022/1243+ */++/*yaml+ Name: keccak_f1600_x1_v84a_aarch64_asm+ Description: AArch64 ARMv8.4-A implementation of Keccak-f[1600] permutation for single state+ Signature: void mld_keccak_f1600_x1_v84a_aarch64_asm(uint64_t state[25], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON, SHA3]+ x0:+ type: buffer+ size_bytes: 200+ permissions: read/write+ c_parameter: uint64_t state[25]+ description: Keccak state (25 x uint64_t)+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: uint64_t const *rc+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 64+ description: register preservation+*/++//+// Author: Hanno Becker <hanno.becker@arm.com>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>+//+// This implementation is essentially from the paper @[HYBRID].+// The only difference is interleaving/deinterleaving of Keccak state+// during load and store, so that the caller need not do this.+//++#include "../../../../common.h"+#if defined(MLD_FIPS202_AARCH64_NEED_X1_V84A) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#if defined(__ARM_FEATURE_SHA3)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(keccak_f1600_x1_v84a_aarch64_asm)+MLD_ASM_FN_SYMBOL(keccak_f1600_x1_v84a_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ ldp d0, d1, [x0]+ ldp d2, d3, [x0, #0x10]+ ldp d4, d5, [x0, #0x20]+ ldp d6, d7, [x0, #0x30]+ ldp d8, d9, [x0, #0x40]+ ldp d10, d11, [x0, #0x50]+ ldp d12, d13, [x0, #0x60]+ ldp d14, d15, [x0, #0x70]+ ldp d16, d17, [x0, #0x80]+ ldp d18, d19, [x0, #0x90]+ ldp d20, d21, [x0, #0xa0]+ ldp d22, d23, [x0, #0xb0]+ ldr d24, [x0, #0xc0]+ mov x2, #0x18 // =24++Lmld_keccak_f1600_x1_v84a_loop:+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor3 v30.16b, v30.16b, v15.16b, v20.16b+ eor3 v29.16b, v29.16b, v16.16b, v21.16b+ eor3 v28.16b, v28.16b, v17.16b, v22.16b+ eor3 v27.16b, v27.16b, v18.16b, v23.16b+ eor3 v26.16b, v26.16b, v19.16b, v24.16b+ rax1 v25.2d, v30.2d, v28.2d+ rax1 v28.2d, v28.2d, v26.2d+ rax1 v26.2d, v26.2d, v29.2d+ rax1 v29.2d, v29.2d, v27.2d+ rax1 v27.2d, v27.2d, v30.2d+ eor v30.16b, v0.16b, v26.16b+ xar v0.2d, v2.2d, v29.2d, #0x2+ xar v2.2d, v12.2d, v29.2d, #0x15+ xar v12.2d, v13.2d, v28.2d, #0x27+ xar v13.2d, v19.2d, v27.2d, #0x38+ xar v19.2d, v23.2d, v28.2d, #0x8+ xar v23.2d, v15.2d, v26.2d, #0x17+ xar v15.2d, v1.2d, v25.2d, #0x3f+ xar v1.2d, v8.2d, v28.2d, #0x9+ xar v8.2d, v16.2d, v25.2d, #0x13+ xar v16.2d, v7.2d, v29.2d, #0x3a+ xar v7.2d, v10.2d, v26.2d, #0x3d+ xar v10.2d, v3.2d, v28.2d, #0x24+ xar v3.2d, v18.2d, v28.2d, #0x2b+ xar v18.2d, v17.2d, v29.2d, #0x31+ xar v17.2d, v11.2d, v25.2d, #0x36+ xar v11.2d, v9.2d, v27.2d, #0x2c+ xar v9.2d, v22.2d, v29.2d, #0x3+ xar v22.2d, v14.2d, v27.2d, #0x19+ xar v14.2d, v20.2d, v26.2d, #0x2e+ xar v20.2d, v4.2d, v27.2d, #0x25+ xar v4.2d, v24.2d, v27.2d, #0x32+ xar v24.2d, v21.2d, v25.2d, #0x3e+ xar v21.2d, v5.2d, v26.2d, #0x1c+ xar v27.2d, v6.2d, v25.2d, #0x14+ ld1r { v31.2d }, [x1], #8+ bcax v5.16b, v10.16b, v7.16b, v11.16b+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ bcax v7.16b, v7.16b, v9.16b, v8.16b+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ bcax v9.16b, v9.16b, v11.16b, v10.16b+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ bcax v11.16b, v16.16b, v13.16b, v12.16b+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ bcax v13.16b, v13.16b, v15.16b, v14.16b+ bcax v14.16b, v14.16b, v16.16b, v15.16b+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ bcax v16.16b, v21.16b, v18.16b, v17.16b+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bcax v18.16b, v18.16b, v20.16b, v19.16b+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ bcax v20.16b, v0.16b, v22.16b, v1.16b+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ bcax v22.16b, v22.16b, v24.16b, v23.16b+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ bcax v24.16b, v24.16b, v1.16b, v0.16b+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ bcax v1.16b, v27.16b, v3.16b, v2.16b+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bcax v3.16b, v3.16b, v30.16b, v4.16b+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ eor v0.16b, v0.16b, v31.16b+ sub x2, x2, #0x1+ cbnz x2, Lmld_keccak_f1600_x1_v84a_loop+ stp d0, d1, [x0]+ stp d2, d3, [x0, #0x10]+ stp d4, d5, [x0, #0x20]+ stp d6, d7, [x0, #0x30]+ stp d8, d9, [x0, #0x40]+ stp d10, d11, [x0, #0x50]+ stp d12, d13, [x0, #0x60]+ stp d14, d15, [x0, #0x70]+ stp d16, d17, [x0, #0x80]+ stp d18, d19, [x0, #0x90]+ stp d20, d21, [x0, #0xa0]+ stp d22, d23, [x0, #0xb0]+ str d24, [x0, #0xc0]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(keccak_f1600_x1_v84a_aarch64_asm)++#endif /* __ARM_FEATURE_SHA3 */++#endif /* MLD_FIPS202_AARCH64_NEED_X1_V84A && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S view
@@ -0,0 +1,262 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [HYBRID]+ * Hybrid scalar/vector implementations of Keccak and SPHINCS+ on AArch64+ * Becker, Kannwischer+ * https://eprint.iacr.org/2022/1243+ */++/*yaml+ Name: keccak_f1600_x2_v84a_aarch64_asm+ Description: AArch64 ARMv8.4-A implementation of Keccak-f[1600] permutation for two sequential states+ Signature: void mld_keccak_f1600_x2_v84a_aarch64_asm(uint64_t state[50], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON, SHA3]+ x0:+ type: buffer+ size_bytes: 400+ permissions: read/write+ c_parameter: uint64_t state[50]+ description: Two sequential Keccak states (state0[25], state1[25])+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: uint64_t const *rc+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 64+ description: register preservation+*/++//+// Author: Hanno Becker <hanno.becker@arm.com>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>+//+// This implementation is essentially from the paper @[HYBRID].+// The only difference is interleaving/deinterleaving of Keccak state+// during load and store, so that the caller need not do this.+//++#include "../../../../common.h"+#if defined(MLD_FIPS202_AARCH64_NEED_X2_V84A) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#if defined(__ARM_FEATURE_SHA3)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(keccak_f1600_x2_v84a_aarch64_asm)+MLD_ASM_FN_SYMBOL(keccak_f1600_x2_v84a_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ add x2, x0, #0xc8+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v0.2d, v25.2d, v27.2d+ trn2 v1.2d, v25.2d, v27.2d+ trn1 v2.2d, v26.2d, v28.2d+ trn2 v3.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v4.2d, v25.2d, v27.2d+ trn2 v5.2d, v25.2d, v27.2d+ trn1 v6.2d, v26.2d, v28.2d+ trn2 v7.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v8.2d, v25.2d, v27.2d+ trn2 v9.2d, v25.2d, v27.2d+ trn1 v10.2d, v26.2d, v28.2d+ trn2 v11.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v12.2d, v25.2d, v27.2d+ trn2 v13.2d, v25.2d, v27.2d+ trn1 v14.2d, v26.2d, v28.2d+ trn2 v15.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v16.2d, v25.2d, v27.2d+ trn2 v17.2d, v25.2d, v27.2d+ trn1 v18.2d, v26.2d, v28.2d+ trn2 v19.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v20.2d, v25.2d, v27.2d+ trn2 v21.2d, v25.2d, v27.2d+ trn1 v22.2d, v26.2d, v28.2d+ trn2 v23.2d, v26.2d, v28.2d+ ldr d25, [x0]+ ldr d27, [x2]+ trn1 v24.2d, v25.2d, v27.2d+ mov x2, #0x18 // =24++Lmld_keccak_f1600_x2_v84a_loop:+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor3 v30.16b, v30.16b, v15.16b, v20.16b+ eor3 v29.16b, v29.16b, v16.16b, v21.16b+ eor3 v28.16b, v28.16b, v17.16b, v22.16b+ eor3 v27.16b, v27.16b, v18.16b, v23.16b+ eor3 v26.16b, v26.16b, v19.16b, v24.16b+ rax1 v25.2d, v30.2d, v28.2d+ rax1 v28.2d, v28.2d, v26.2d+ rax1 v26.2d, v26.2d, v29.2d+ rax1 v29.2d, v29.2d, v27.2d+ rax1 v27.2d, v27.2d, v30.2d+ eor v30.16b, v0.16b, v26.16b+ xar v0.2d, v2.2d, v29.2d, #0x2+ xar v2.2d, v12.2d, v29.2d, #0x15+ xar v12.2d, v13.2d, v28.2d, #0x27+ xar v13.2d, v19.2d, v27.2d, #0x38+ xar v19.2d, v23.2d, v28.2d, #0x8+ xar v23.2d, v15.2d, v26.2d, #0x17+ xar v15.2d, v1.2d, v25.2d, #0x3f+ xar v1.2d, v8.2d, v28.2d, #0x9+ xar v8.2d, v16.2d, v25.2d, #0x13+ xar v16.2d, v7.2d, v29.2d, #0x3a+ xar v7.2d, v10.2d, v26.2d, #0x3d+ xar v10.2d, v3.2d, v28.2d, #0x24+ xar v3.2d, v18.2d, v28.2d, #0x2b+ xar v18.2d, v17.2d, v29.2d, #0x31+ xar v17.2d, v11.2d, v25.2d, #0x36+ xar v11.2d, v9.2d, v27.2d, #0x2c+ xar v9.2d, v22.2d, v29.2d, #0x3+ xar v22.2d, v14.2d, v27.2d, #0x19+ xar v14.2d, v20.2d, v26.2d, #0x2e+ xar v20.2d, v4.2d, v27.2d, #0x25+ xar v4.2d, v24.2d, v27.2d, #0x32+ xar v24.2d, v21.2d, v25.2d, #0x3e+ xar v21.2d, v5.2d, v26.2d, #0x1c+ xar v27.2d, v6.2d, v25.2d, #0x14+ ld1r { v31.2d }, [x1], #8+ bcax v5.16b, v10.16b, v7.16b, v11.16b+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ bcax v7.16b, v7.16b, v9.16b, v8.16b+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ bcax v9.16b, v9.16b, v11.16b, v10.16b+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ bcax v11.16b, v16.16b, v13.16b, v12.16b+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ bcax v13.16b, v13.16b, v15.16b, v14.16b+ bcax v14.16b, v14.16b, v16.16b, v15.16b+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ bcax v16.16b, v21.16b, v18.16b, v17.16b+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bcax v18.16b, v18.16b, v20.16b, v19.16b+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ bcax v20.16b, v0.16b, v22.16b, v1.16b+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ bcax v22.16b, v22.16b, v24.16b, v23.16b+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ bcax v24.16b, v24.16b, v1.16b, v0.16b+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ bcax v1.16b, v27.16b, v3.16b, v2.16b+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bcax v3.16b, v3.16b, v30.16b, v4.16b+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ eor v0.16b, v0.16b, v31.16b+ sub x2, x2, #0x1+ cbnz x2, Lmld_keccak_f1600_x2_v84a_loop+ sub x0, x0, #0xc0+ add x2, x0, #0xc8+ trn1 v25.2d, v0.2d, v1.2d+ trn1 v26.2d, v2.2d, v3.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v0.2d, v1.2d+ trn2 v28.2d, v2.2d, v3.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v4.2d, v5.2d+ trn1 v26.2d, v6.2d, v7.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v4.2d, v5.2d+ trn2 v28.2d, v6.2d, v7.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v8.2d, v9.2d+ trn1 v26.2d, v10.2d, v11.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v8.2d, v9.2d+ trn2 v28.2d, v10.2d, v11.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v12.2d, v13.2d+ trn1 v26.2d, v14.2d, v15.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v12.2d, v13.2d+ trn2 v28.2d, v14.2d, v15.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v16.2d, v17.2d+ trn1 v26.2d, v18.2d, v19.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v16.2d, v17.2d+ trn2 v28.2d, v18.2d, v19.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v20.2d, v21.2d+ trn1 v26.2d, v22.2d, v23.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v20.2d, v21.2d+ trn2 v28.2d, v22.2d, v23.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ str d24, [x0]+ trn2 v25.2d, v24.2d, v24.2d+ str d25, [x2]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(keccak_f1600_x2_v84a_aarch64_asm)++#endif /* __ARM_FEATURE_SHA3 */++#endif /* MLD_FIPS202_AARCH64_NEED_X2_V84A && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S view
@@ -0,0 +1,1080 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++// Author: Hanno Becker <hannobecker@posteo.de>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>++/*yaml+ Name: keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm+ Description: AArch64 hybrid scalar/vector implementation of Keccak-f[1600] permutation for four sequential states+ Signature: void mld_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(uint64_t state[100], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 800+ permissions: read/write+ c_parameter: uint64_t state[100]+ description: Four sequential Keccak states (state0[25], state1[25], state2[25], state3[25])+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: uint64_t const *rc+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 224+ description: register preservation and temporary storage+*/++#include "../../../../common.h"+#if defined(MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)+MLD_ASM_FN_SYMBOL(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0xe0+ .cfi_adjust_cfa_offset 0xe0+ stp x19, x20, [sp, #0x30]+ .cfi_rel_offset x19, 0x30+ .cfi_rel_offset x20, 0x38+ stp x21, x22, [sp, #0x40]+ .cfi_rel_offset x21, 0x40+ .cfi_rel_offset x22, 0x48+ stp x23, x24, [sp, #0x50]+ .cfi_rel_offset x23, 0x50+ .cfi_rel_offset x24, 0x58+ stp x25, x26, [sp, #0x60]+ .cfi_rel_offset x25, 0x60+ .cfi_rel_offset x26, 0x68+ stp x27, x28, [sp, #0x70]+ .cfi_rel_offset x27, 0x70+ .cfi_rel_offset x28, 0x78+ stp x29, x30, [sp, #0x80]+ .cfi_rel_offset x29, 0x80+ .cfi_rel_offset x30, 0x88+ stp d8, d9, [sp, #0x90]+ .cfi_rel_offset d8, 0x90+ .cfi_rel_offset d9, 0x98+ stp d10, d11, [sp, #0xa0]+ .cfi_rel_offset d10, 0xa0+ .cfi_rel_offset d11, 0xa8+ stp d12, d13, [sp, #0xb0]+ .cfi_rel_offset d12, 0xb0+ .cfi_rel_offset d13, 0xb8+ stp d14, d15, [sp, #0xc0]+ .cfi_rel_offset d14, 0xc0+ .cfi_rel_offset d15, 0xc8+ mov x29, x1+ mov x30, #0x0 // =0+ str x30, [sp, #0x20]+ str x29, [sp, #0x8]+ str x29, [sp, #0x10]+ str x0, [sp]+ add x4, x0, #0xc8+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v0.2d, v25.2d, v27.2d+ trn2 v1.2d, v25.2d, v27.2d+ trn1 v2.2d, v26.2d, v28.2d+ trn2 v3.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v4.2d, v25.2d, v27.2d+ trn2 v5.2d, v25.2d, v27.2d+ trn1 v6.2d, v26.2d, v28.2d+ trn2 v7.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v8.2d, v25.2d, v27.2d+ trn2 v9.2d, v25.2d, v27.2d+ trn1 v10.2d, v26.2d, v28.2d+ trn2 v11.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v12.2d, v25.2d, v27.2d+ trn2 v13.2d, v25.2d, v27.2d+ trn1 v14.2d, v26.2d, v28.2d+ trn2 v15.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v16.2d, v25.2d, v27.2d+ trn2 v17.2d, v25.2d, v27.2d+ trn1 v18.2d, v26.2d, v28.2d+ trn2 v19.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v20.2d, v25.2d, v27.2d+ trn2 v21.2d, v25.2d, v27.2d+ trn1 v22.2d, v26.2d, v28.2d+ trn2 v23.2d, v26.2d, v28.2d+ ldr d25, [x0]+ ldr d27, [x4]+ trn1 v24.2d, v25.2d, v27.2d+ sub x0, x0, #0xc0+ add x0, x0, #0x190+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x190++Lmld_keccak_f1600_x4_v8a_scalar_hybrid_initial:+ eor x30, x24, x25+ eor x27, x9, x10+ eor v30.16b, v0.16b, v5.16b+ eor v30.16b, v30.16b, v10.16b+ eor x0, x30, x21+ eor v30.16b, v30.16b, v15.16b+ eor x26, x27, x6+ eor x27, x26, x7+ eor v30.16b, v30.16b, v20.16b+ eor x29, x0, x22+ eor v29.16b, v1.16b, v6.16b+ eor x26, x29, x23+ eor v29.16b, v29.16b, v11.16b+ eor x29, x4, x5+ eor x30, x29, x1+ eor v29.16b, v29.16b, v16.16b+ eor x0, x27, x8+ eor v29.16b, v29.16b, v21.16b+ eor x29, x30, x2+ eor v28.16b, v2.16b, v7.16b+ eor x30, x19, x20+ eor x30, x30, x16+ eor v28.16b, v28.16b, v12.16b+ eor x27, x26, x0, ror #63+ eor v28.16b, v28.16b, v17.16b+ eor x4, x4, x27+ eor v28.16b, v28.16b, v22.16b+ eor x30, x30, x17+ eor x30, x30, x28+ eor v27.16b, v3.16b, v8.16b+ eor x29, x29, x3+ eor v27.16b, v27.16b, v13.16b+ eor x0, x0, x30, ror #63+ eor v27.16b, v27.16b, v18.16b+ eor x30, x30, x29, ror #63+ eor x22, x22, x30+ eor v27.16b, v27.16b, v23.16b+ eor x23, x23, x30+ eor v26.16b, v4.16b, v9.16b+ str x23, [sp, #0xd0]+ eor v26.16b, v26.16b, v14.16b+ eor x23, x14, x15+ eor x14, x14, x0+ eor v26.16b, v26.16b, v19.16b+ eor x23, x23, x11+ eor v26.16b, v26.16b, v24.16b+ eor x15, x15, x0+ eor x1, x1, x27+ add v31.2d, v28.2d, v28.2d+ eor x23, x23, x12+ sri v31.2d, v28.2d, #0x3f+ eor x23, x23, x13+ eor v25.16b, v31.16b, v30.16b+ eor x11, x11, x0+ eor x29, x29, x23, ror #63+ add v31.2d, v26.2d, v26.2d+ eor x23, x23, x26, ror #63+ sri v31.2d, v26.2d, #0x3f+ eor x26, x13, x0+ eor v28.16b, v31.16b, v28.16b+ eor x13, x28, x23+ eor x28, x24, x30+ add v31.2d, v29.2d, v29.2d+ eor x24, x16, x23+ sri v31.2d, v29.2d, #0x3f+ eor x16, x21, x30+ eor v26.16b, v31.16b, v26.16b+ eor x21, x25, x30+ eor x30, x19, x23+ add v31.2d, v27.2d, v27.2d+ eor x19, x20, x23+ sri v31.2d, v27.2d, #0x3f+ eor x20, x17, x23+ eor v29.16b, v31.16b, v29.16b+ eor x17, x12, x0+ eor x0, x2, x27+ add v31.2d, v30.2d, v30.2d+ eor x2, x6, x29+ sri v31.2d, v30.2d, #0x3f+ eor x6, x8, x29+ eor v27.16b, v31.16b, v27.16b+ bic x8, x28, x13, ror #47+ eor x12, x3, x27+ eor v30.16b, v0.16b, v26.16b+ bic x3, x13, x17, ror #19+ eor v31.16b, v2.16b, v29.16b+ eor x5, x5, x27+ ldr x27, [sp, #0xd0]+ shl v0.2d, v31.2d, #0x3e+ bic x25, x17, x2, ror #5+ sri v0.2d, v31.2d, #0x2+ eor x9, x9, x29+ eor v31.16b, v12.16b, v29.16b+ eor x23, x25, x5, ror #52+ eor x3, x3, x2, ror #24+ shl v2.2d, v31.2d, #0x2b+ eor x8, x8, x17, ror #2+ sri v2.2d, v31.2d, #0x15+ eor x17, x10, x29+ eor v31.16b, v13.16b, v28.16b+ bic x25, x12, x22, ror #47+ eor x29, x7, x29+ shl v12.2d, v31.2d, #0x19+ bic x10, x4, x27, ror #2+ sri v12.2d, v31.2d, #0x27+ bic x7, x5, x28, ror #10+ eor v31.16b, v19.16b, v27.16b+ eor x10, x10, x20, ror #50+ eor x13, x7, x13, ror #57+ shl v13.2d, v31.2d, #0x8+ bic x7, x2, x5, ror #47+ sri v13.2d, v31.2d, #0x38+ eor x2, x25, x24, ror #39+ eor v31.16b, v23.16b, v28.16b+ bic x25, x20, x11, ror #57+ bic x5, x17, x4, ror #25+ shl v19.2d, v31.2d, #0x38+ eor x25, x25, x17, ror #53+ sri v19.2d, v31.2d, #0x8+ bic x17, x11, x17, ror #60+ eor v31.16b, v15.16b, v26.16b+ eor x28, x7, x28, ror #57+ bic x7, x9, x12, ror #42+ shl v23.2d, v31.2d, #0x29+ eor x7, x7, x22, ror #25+ sri v23.2d, v31.2d, #0x17+ bic x22, x22, x24, ror #56+ bic x24, x24, x15, ror #31+ eor v31.16b, v1.16b, v25.16b+ eor x22, x22, x15, ror #23+ shl v15.2d, v31.2d, #0x1+ bic x20, x27, x20, ror #48+ sri v15.2d, v31.2d, #0x3f+ bic x15, x15, x9, ror #16+ eor x12, x15, x12, ror #58+ eor v31.16b, v8.16b, v28.16b+ eor x15, x5, x27, ror #27+ shl v1.2d, v31.2d, #0x37+ eor x5, x20, x11, ror #41+ sri v1.2d, v31.2d, #0x9+ ldr x11, [sp, #0x8]+ eor x20, x17, x4, ror #21+ eor v31.16b, v16.16b, v25.16b+ eor x17, x24, x9, ror #47+ shl v8.2d, v31.2d, #0x2d+ mov x24, #0x1 // =1+ sri v8.2d, v31.2d, #0x13+ bic x9, x0, x16, ror #9+ str x24, [sp, #0x18]+ eor v31.16b, v7.16b, v29.16b+ bic x24, x29, x1, ror #44+ shl v16.2d, v31.2d, #0x6+ bic x27, x1, x21, ror #50+ sri v16.2d, v31.2d, #0x3a+ bic x4, x26, x29, ror #63+ eor x1, x1, x4, ror #21+ eor v31.16b, v10.16b, v26.16b+ ldr x11, [x11]+ shl v7.2d, v31.2d, #0x3+ bic x4, x21, x30, ror #57+ sri v7.2d, v31.2d, #0x3d+ eor x21, x24, x21, ror #30+ eor x24, x9, x19, ror #44+ eor v31.16b, v3.16b, v28.16b+ bic x9, x14, x6, ror #5+ shl v10.2d, v31.2d, #0x1c+ eor x9, x9, x0, ror #43+ sri v10.2d, v31.2d, #0x24+ bic x0, x6, x0, ror #38+ eor x1, x1, x11+ eor v31.16b, v18.16b, v28.16b+ eor x11, x4, x26, ror #35+ shl v3.2d, v31.2d, #0x15+ eor x4, x0, x16, ror #47+ bic x0, x16, x19, ror #35+ sri v3.2d, v31.2d, #0x2b+ eor x16, x27, x30, ror #43+ eor v31.16b, v17.16b, v29.16b+ bic x27, x30, x26, ror #42+ shl v18.2d, v31.2d, #0xf+ bic x26, x19, x14, ror #41+ eor x19, x0, x14, ror #12+ sri v18.2d, v31.2d, #0x31+ eor x14, x26, x6, ror #46+ eor v31.16b, v11.16b, v25.16b+ eor x6, x27, x29, ror #41+ shl v17.2d, v31.2d, #0xa+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ sri v17.2d, v31.2d, #0x36+ eor x26, x8, x9, ror #57+ eor v31.16b, v9.16b, v27.16b+ eor x27, x0, x14, ror #10+ shl v11.2d, v31.2d, #0x14+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ sri v11.2d, v31.2d, #0x2c+ eor x30, x23, x22, ror #50+ eor v31.16b, v22.16b, v29.16b+ eor x0, x26, x10, ror #31+ shl v9.2d, v31.2d, #0x3d+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ sri v9.2d, v31.2d, #0x3+ eor x30, x30, x24, ror #34+ eor v31.16b, v14.16b, v27.16b+ eor x0, x0, x7, ror #27+ shl v22.2d, v31.2d, #0x27+ eor x26, x30, x21, ror #26+ eor x26, x26, x25, ror #15+ sri v22.2d, v31.2d, #0x19+ ror x30, x27, #0x3e+ eor v31.16b, v20.16b, v26.16b+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ shl v14.2d, v31.2d, #0x12+ eor x16, x30, x16+ sri v14.2d, v31.2d, #0x2e+ eor x28, x30, x28, ror #63+ eor v31.16b, v4.16b, v27.16b+ str x28, [sp, #0xd0]+ eor x29, x29, x17, ror #36+ shl v20.2d, v31.2d, #0x1b+ eor x28, x1, x2, ror #61+ sri v20.2d, v31.2d, #0x25+ eor x19, x30, x19, ror #37+ eor v31.16b, v24.16b, v27.16b+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ shl v4.2d, v31.2d, #0xe+ eor x26, x26, x0, ror #55+ sri v4.2d, v31.2d, #0x32+ eor x28, x28, x3, ror #39+ eor v31.16b, v21.16b, v25.16b+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ shl v24.2d, v31.2d, #0x2+ eor x0, x0, x29, ror #63+ sri v24.2d, v31.2d, #0x3e+ eor x27, x28, x27, ror #61+ eor v31.16b, v5.16b, v26.16b+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ shl v21.2d, v31.2d, #0x24+ eor x29, x30, x20, ror #2+ sri v21.2d, v31.2d, #0x1c+ eor x20, x26, x3, ror #39+ eor v31.16b, v6.16b, v25.16b+ eor x11, x0, x11, ror #50+ eor x25, x28, x25, ror #9+ shl v27.2d, v31.2d, #0x2c+ eor x3, x28, x21, ror #20+ sri v27.2d, v31.2d, #0x14+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ bic v31.16b, v7.16b, v11.16b+ eor x24, x28, x24, ror #28+ eor v5.16b, v31.16b, v10.16b+ eor x1, x30, x17, ror #36+ bic v31.16b, v8.16b, v7.16b+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ eor v6.16b, v31.16b, v11.16b+ eor x8, x27, x8, ror #56+ bic v31.16b, v9.16b, v8.16b+ eor x17, x27, x7, ror #19+ eor v7.16b, v31.16b, v7.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ bic v31.16b, v10.16b, v9.16b+ eor x4, x26, x4, ror #54+ eor v8.16b, v31.16b, v8.16b+ eor x0, x0, x12, ror #3+ bic v31.16b, v11.16b, v10.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ eor v9.16b, v31.16b, v9.16b+ eor x26, x26, x5, ror #25+ bic v31.16b, v12.16b, v16.16b+ eor x2, x7, x16, ror #39+ eor v10.16b, v31.16b, v15.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ bic v31.16b, v13.16b, v12.16b+ eor x7, x7, x22, ror #25+ eor v11.16b, v31.16b, v16.16b+ eor x12, x30, x20, ror #58+ bic v31.16b, v14.16b, v13.16b+ bic x20, x22, x16, ror #56+ eor x30, x27, x6, ror #43+ eor v12.16b, v31.16b, v12.16b+ eor x22, x20, x15, ror #23+ bic v31.16b, v15.16b, v14.16b+ bic x6, x19, x13, ror #42+ eor v13.16b, v31.16b, v13.16b+ eor x6, x6, x17, ror #41+ bic x5, x13, x17, ror #63+ bic v31.16b, v16.16b, v15.16b+ eor x5, x21, x5, ror #21+ eor v14.16b, v31.16b, v14.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ bic v31.16b, v17.16b, v21.16b+ bic x21, x21, x25, ror #50+ eor v15.16b, v31.16b, v20.16b+ bic x20, x27, x4, ror #25+ bic v31.16b, v18.16b, v17.16b+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ eor v16.16b, v31.16b, v21.16b+ eor x21, x17, x25, ror #30+ bic v31.16b, v19.16b, v18.16b+ bic x19, x25, x19, ror #57+ eor v17.16b, v31.16b, v17.16b+ ldr x25, [sp, #0x18]+ eor x17, x10, x9, ror #47+ bic v31.16b, v20.16b, v19.16b+ ldr x9, [sp, #0x8]+ eor v18.16b, v31.16b, v18.16b+ eor x15, x20, x28, ror #27+ bic v31.16b, v21.16b, v20.16b+ bic x20, x4, x28, ror #2+ eor x10, x20, x1, ror #50+ eor v19.16b, v31.16b, v19.16b+ bic x20, x11, x27, ror #60+ bic v31.16b, v22.16b, v1.16b+ eor x20, x20, x4, ror #21+ eor v20.16b, v31.16b, v0.16b+ bic x4, x28, x1, ror #48+ bic x1, x1, x11, ror #57+ bic v31.16b, v23.16b, v22.16b+ ldr x28, [x9, x25, lsl #3]+ eor v21.16b, v31.16b, v1.16b+ ldr x9, [sp, #0xd0]+ bic v31.16b, v24.16b, v23.16b+ add x25, x25, #0x1+ str x25, [sp, #0x18]+ eor v22.16b, v31.16b, v22.16b+ cmp x25, #0x17+ bic v31.16b, v0.16b, v24.16b+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ eor v23.16b, v31.16b, v23.16b+ eor x1, x5, x28+ bic v31.16b, v1.16b, v0.16b+ eor x5, x4, x11, ror #41+ eor v24.16b, v31.16b, v24.16b+ eor x11, x19, x13, ror #35+ bic x13, x26, x24, ror #10+ bic v31.16b, v2.16b, v27.16b+ eor x28, x27, x24, ror #57+ eor v0.16b, v31.16b, v30.16b+ bic x27, x24, x9, ror #47+ bic v31.16b, v3.16b, v2.16b+ bic x19, x23, x3, ror #9+ bic x4, x29, x14, ror #41+ eor v1.16b, v31.16b, v27.16b+ eor x24, x19, x29, ror #44+ bic v31.16b, v4.16b, v3.16b+ bic x29, x3, x29, ror #35+ eor v2.16b, v31.16b, v2.16b+ eor x13, x13, x9, ror #57+ eor x19, x29, x14, ror #12+ bic v31.16b, v30.16b, v4.16b+ bic x29, x9, x0, ror #19+ eor v3.16b, v31.16b, v3.16b+ bic x14, x14, x8, ror #5+ bic v31.16b, v27.16b, v30.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ eor v4.16b, v31.16b, v4.16b+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmld_keccak_f1600_x4_v8a_scalar_hybrid_loop:+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor v30.16b, v0.16b, v5.16b+ eor v30.16b, v30.16b, v10.16b+ eor x26, x8, x9, ror #57+ eor v30.16b, v30.16b, v15.16b+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ eor v30.16b, v30.16b, v20.16b+ eor x26, x26, x6, ror #51+ eor v29.16b, v1.16b, v6.16b+ eor x30, x23, x22, ror #50+ eor v29.16b, v29.16b, v11.16b+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ eor v29.16b, v29.16b, v16.16b+ eor x27, x27, x12, ror #5+ eor v29.16b, v29.16b, v21.16b+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ eor v28.16b, v2.16b, v7.16b+ eor x26, x30, x21, ror #26+ eor v28.16b, v28.16b, v12.16b+ eor x26, x26, x25, ror #15+ eor v28.16b, v28.16b, v17.16b+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ eor v28.16b, v28.16b, v22.16b+ ror x26, x26, #0x3a+ eor v27.16b, v3.16b, v8.16b+ eor x16, x30, x16+ eor v27.16b, v27.16b, v13.16b+ eor x28, x30, x28, ror #63+ str x28, [sp, #0xd0]+ eor v27.16b, v27.16b, v18.16b+ eor x29, x29, x17, ror #36+ eor v27.16b, v27.16b, v23.16b+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ eor v26.16b, v4.16b, v9.16b+ eor x29, x29, x20, ror #2+ eor v26.16b, v26.16b, v14.16b+ eor x28, x28, x4, ror #54+ eor v26.16b, v26.16b, v19.16b+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ eor v26.16b, v26.16b, v24.16b+ eor x28, x28, x5, ror #25+ add v31.2d, v28.2d, v28.2d+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ sri v31.2d, v28.2d, #0x3f+ eor x27, x28, x27, ror #61+ eor v25.16b, v31.16b, v30.16b+ eor x13, x0, x13, ror #46+ add v31.2d, v26.2d, v26.2d+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ sri v31.2d, v26.2d, #0x3f+ eor x20, x26, x3, ror #39+ eor v28.16b, v31.16b, v28.16b+ eor x11, x0, x11, ror #50+ add v31.2d, v29.2d, v29.2d+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ sri v31.2d, v29.2d, #0x3f+ eor x21, x26, x1+ eor v26.16b, v31.16b, v26.16b+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ add v31.2d, v27.2d, v27.2d+ eor x1, x30, x17, ror #36+ sri v31.2d, v27.2d, #0x3f+ eor x14, x0, x14, ror #8+ eor v29.16b, v31.16b, v29.16b+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ add v31.2d, v30.2d, v30.2d+ eor x17, x27, x7, ror #19+ sri v31.2d, v30.2d, #0x3f+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ eor v27.16b, v31.16b, v27.16b+ eor x4, x26, x4, ror #54+ eor v30.16b, v0.16b, v26.16b+ eor x0, x0, x12, ror #3+ eor v31.16b, v2.16b, v29.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ shl v0.2d, v31.2d, #0x3e+ eor x26, x26, x5, ror #25+ sri v0.2d, v31.2d, #0x2+ eor x2, x7, x16, ror #39+ eor v31.16b, v12.16b, v29.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ shl v2.2d, v31.2d, #0x2b+ eor x7, x7, x22, ror #25+ sri v2.2d, v31.2d, #0x15+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ eor v31.16b, v13.16b, v28.16b+ eor x30, x27, x6, ror #43+ shl v12.2d, v31.2d, #0x19+ eor x22, x20, x15, ror #23+ sri v12.2d, v31.2d, #0x27+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ eor v31.16b, v19.16b, v27.16b+ bic x5, x13, x17, ror #63+ shl v13.2d, v31.2d, #0x8+ eor x5, x21, x5, ror #21+ sri v13.2d, v31.2d, #0x38+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ eor v31.16b, v23.16b, v28.16b+ bic x21, x21, x25, ror #50+ shl v19.2d, v31.2d, #0x38+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ sri v19.2d, v31.2d, #0x8+ eor x16, x21, x19, ror #43+ eor v31.16b, v15.16b, v26.16b+ eor x21, x17, x25, ror #30+ shl v23.2d, v31.2d, #0x29+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x18]+ sri v23.2d, v31.2d, #0x17+ eor x17, x10, x9, ror #47+ eor v31.16b, v1.16b, v25.16b+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ shl v15.2d, v31.2d, #0x1+ bic x20, x4, x28, ror #2+ sri v15.2d, v31.2d, #0x3f+ eor x10, x20, x1, ror #50+ eor v31.16b, v8.16b, v28.16b+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ shl v1.2d, v31.2d, #0x37+ bic x4, x28, x1, ror #48+ sri v1.2d, v31.2d, #0x9+ bic x1, x1, x11, ror #57+ eor v31.16b, v16.16b, v25.16b+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0xd0]+ shl v8.2d, v31.2d, #0x2d+ add x25, x25, #0x1+ sri v8.2d, v31.2d, #0x13+ str x25, [sp, #0x18]+ cmp x25, #0x17+ eor v31.16b, v7.16b, v29.16b+ eor x25, x1, x27, ror #53+ shl v16.2d, v31.2d, #0x6+ bic x27, x30, x26, ror #47+ sri v16.2d, v31.2d, #0x3a+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ eor v31.16b, v10.16b, v26.16b+ eor x11, x19, x13, ror #35+ shl v7.2d, v31.2d, #0x3+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ sri v7.2d, v31.2d, #0x3d+ bic x27, x24, x9, ror #47+ eor v31.16b, v3.16b, v28.16b+ bic x19, x23, x3, ror #9+ shl v10.2d, v31.2d, #0x1c+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ sri v10.2d, v31.2d, #0x24+ bic x29, x3, x29, ror #35+ eor v31.16b, v18.16b, v28.16b+ eor x13, x13, x9, ror #57+ shl v3.2d, v31.2d, #0x15+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ sri v3.2d, v31.2d, #0x2b+ bic x14, x14, x8, ror #5+ eor v31.16b, v17.16b, v29.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ shl v18.2d, v31.2d, #0xf+ bic x23, x8, x23, ror #38+ sri v18.2d, v31.2d, #0x31+ eor x8, x27, x0, ror #2+ eor v31.16b, v11.16b, v25.16b+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ shl v17.2d, v31.2d, #0xa+ eor x23, x3, x26, ror #52+ sri v17.2d, v31.2d, #0x36+ eor x3, x29, x30, ror #24+ eor x0, x15, x11, ror #52+ eor v31.16b, v9.16b, v27.16b+ eor x0, x0, x13, ror #48+ shl v11.2d, v31.2d, #0x14+ eor x26, x8, x9, ror #57+ sri v11.2d, v31.2d, #0x2c+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ eor v31.16b, v22.16b, v29.16b+ eor x26, x26, x6, ror #51+ shl v9.2d, v31.2d, #0x3d+ eor x30, x23, x22, ror #50+ sri v9.2d, v31.2d, #0x3+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ eor v31.16b, v14.16b, v27.16b+ eor x27, x27, x12, ror #5+ shl v22.2d, v31.2d, #0x27+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ sri v22.2d, v31.2d, #0x19+ eor x26, x30, x21, ror #26+ eor v31.16b, v20.16b, v26.16b+ eor x26, x26, x25, ror #15+ shl v14.2d, v31.2d, #0x12+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ sri v14.2d, v31.2d, #0x2e+ ror x26, x26, #0x3a+ eor v31.16b, v4.16b, v27.16b+ eor x16, x30, x16+ shl v20.2d, v31.2d, #0x1b+ eor x28, x30, x28, ror #63+ str x28, [sp, #0xd0]+ sri v20.2d, v31.2d, #0x25+ eor x29, x29, x17, ror #36+ eor v31.16b, v24.16b, v27.16b+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ shl v4.2d, v31.2d, #0xe+ eor x29, x29, x20, ror #2+ sri v4.2d, v31.2d, #0x32+ eor x28, x28, x4, ror #54+ eor v31.16b, v21.16b, v25.16b+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ shl v24.2d, v31.2d, #0x2+ eor x28, x28, x5, ror #25+ sri v24.2d, v31.2d, #0x3e+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ eor v31.16b, v5.16b, v26.16b+ eor x27, x28, x27, ror #61+ shl v21.2d, v31.2d, #0x24+ eor x13, x0, x13, ror #46+ sri v21.2d, v31.2d, #0x1c+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ eor v31.16b, v6.16b, v25.16b+ eor x20, x26, x3, ror #39+ shl v27.2d, v31.2d, #0x2c+ eor x11, x0, x11, ror #50+ sri v27.2d, v31.2d, #0x14+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ bic v31.16b, v7.16b, v11.16b+ eor x21, x26, x1+ eor v5.16b, v31.16b, v10.16b+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ bic v31.16b, v8.16b, v7.16b+ eor x1, x30, x17, ror #36+ eor v6.16b, v31.16b, v11.16b+ eor x14, x0, x14, ror #8+ bic v31.16b, v9.16b, v8.16b+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ eor v7.16b, v31.16b, v7.16b+ eor x17, x27, x7, ror #19+ bic v31.16b, v10.16b, v9.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ eor v8.16b, v31.16b, v8.16b+ eor x4, x26, x4, ror #54+ bic v31.16b, v11.16b, v10.16b+ eor x0, x0, x12, ror #3+ eor v9.16b, v31.16b, v9.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ bic v31.16b, v12.16b, v16.16b+ eor x26, x26, x5, ror #25+ eor v10.16b, v31.16b, v15.16b+ eor x2, x7, x16, ror #39+ bic v31.16b, v13.16b, v12.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ eor v11.16b, v31.16b, v16.16b+ eor x7, x7, x22, ror #25+ bic v31.16b, v14.16b, v13.16b+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ eor v12.16b, v31.16b, v12.16b+ eor x30, x27, x6, ror #43+ bic v31.16b, v15.16b, v14.16b+ eor x22, x20, x15, ror #23+ eor v13.16b, v31.16b, v13.16b+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ bic v31.16b, v16.16b, v15.16b+ bic x5, x13, x17, ror #63+ eor v14.16b, v31.16b, v14.16b+ eor x5, x21, x5, ror #21+ bic v31.16b, v17.16b, v21.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ eor v15.16b, v31.16b, v20.16b+ bic x21, x21, x25, ror #50+ bic v31.16b, v18.16b, v17.16b+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ eor v16.16b, v31.16b, v21.16b+ eor x16, x21, x19, ror #43+ bic v31.16b, v19.16b, v18.16b+ eor x21, x17, x25, ror #30+ eor v17.16b, v31.16b, v17.16b+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x18]+ bic v31.16b, v20.16b, v19.16b+ eor x17, x10, x9, ror #47+ eor v18.16b, v31.16b, v18.16b+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ bic v31.16b, v21.16b, v20.16b+ bic x20, x4, x28, ror #2+ eor v19.16b, v31.16b, v19.16b+ eor x10, x20, x1, ror #50+ bic v31.16b, v22.16b, v1.16b+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ eor v20.16b, v31.16b, v0.16b+ bic x4, x28, x1, ror #48+ bic v31.16b, v23.16b, v22.16b+ bic x1, x1, x11, ror #57+ eor v21.16b, v31.16b, v1.16b+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0xd0]+ bic v31.16b, v24.16b, v23.16b+ add x25, x25, #0x1+ eor v22.16b, v31.16b, v22.16b+ str x25, [sp, #0x18]+ cmp x25, #0x17+ bic v31.16b, v0.16b, v24.16b+ eor x25, x1, x27, ror #53+ eor v23.16b, v31.16b, v23.16b+ bic x27, x30, x26, ror #47+ bic v31.16b, v1.16b, v0.16b+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ eor v24.16b, v31.16b, v24.16b+ eor x11, x19, x13, ror #35+ bic v31.16b, v2.16b, v27.16b+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ eor v0.16b, v31.16b, v30.16b+ bic x27, x24, x9, ror #47+ bic v31.16b, v3.16b, v2.16b+ bic x19, x23, x3, ror #9+ eor v1.16b, v31.16b, v27.16b+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic v31.16b, v4.16b, v3.16b+ bic x29, x3, x29, ror #35+ eor v2.16b, v31.16b, v2.16b+ eor x13, x13, x9, ror #57+ bic v31.16b, v30.16b, v4.16b+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ eor v3.16b, v31.16b, v3.16b+ bic x14, x14, x8, ror #5+ bic v31.16b, v27.16b, v30.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ eor v4.16b, v31.16b, v4.16b+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmld_keccak_f1600_x4_v8a_scalar_hybrid_loop_end:+ b.le Lmld_keccak_f1600_x4_v8a_scalar_hybrid_loop+ ror x2, x2, #0x3d+ ror x3, x3, #0x27+ ror x4, x4, #0x36+ ror x5, x5, #0x19+ ror x6, x6, #0x2b+ ror x7, x7, #0x13+ ror x8, x8, #0x38+ ror x9, x9, #0x31+ ror x10, x10, #0x17+ ror x11, x11, #0x32+ ror x12, x12, #0x3+ ror x13, x13, #0x2e+ ror x14, x14, #0x8+ ror x15, x15, #0x3e+ ror x17, x17, #0x24+ ror x28, x28, #0x3f+ ror x19, x19, #0x25+ ror x20, x20, #0x2+ ror x21, x21, #0x14+ ror x22, x22, #0x2c+ ror x23, x23, #0x3a+ ror x24, x24, #0x1c+ ror x25, x25, #0x9+ ldr x30, [sp, #0x20]+ cmp x30, #0x1+ b.eq Lmld_keccak_f1600_x4_v8a_scalar_hybrid_done+ mov x30, #0x1 // =1+ str x30, [sp, #0x20]+ ldr x0, [sp]+ add x0, x0, #0x190+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x190+ add x0, x0, #0x258+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x258+ b Lmld_keccak_f1600_x4_v8a_scalar_hybrid_initial++Lmld_keccak_f1600_x4_v8a_scalar_hybrid_done:+ ldr x0, [sp]+ add x0, x0, #0x258+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x258+ add x4, x0, #0xc8+ trn1 v25.2d, v0.2d, v1.2d+ trn1 v26.2d, v2.2d, v3.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v0.2d, v1.2d+ trn2 v28.2d, v2.2d, v3.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v4.2d, v5.2d+ trn1 v26.2d, v6.2d, v7.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v4.2d, v5.2d+ trn2 v28.2d, v6.2d, v7.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v8.2d, v9.2d+ trn1 v26.2d, v10.2d, v11.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v8.2d, v9.2d+ trn2 v28.2d, v10.2d, v11.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v12.2d, v13.2d+ trn1 v26.2d, v14.2d, v15.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v12.2d, v13.2d+ trn2 v28.2d, v14.2d, v15.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v16.2d, v17.2d+ trn1 v26.2d, v18.2d, v19.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v16.2d, v17.2d+ trn2 v28.2d, v18.2d, v19.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v20.2d, v21.2d+ trn1 v26.2d, v22.2d, v23.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v20.2d, v21.2d+ trn2 v28.2d, v22.2d, v23.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ str d24, [x0]+ trn2 v25.2d, v24.2d, v24.2d+ str d25, [x4]+ ldp d8, d9, [sp, #0x90]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0xa0]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0xb0]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0xc0]+ .cfi_restore d14+ .cfi_restore d15+ ldp x19, x20, [sp, #0x30]+ .cfi_restore x19+ .cfi_restore x20+ ldp x21, x22, [sp, #0x40]+ .cfi_restore x21+ .cfi_restore x22+ ldp x23, x24, [sp, #0x50]+ .cfi_restore x23+ .cfi_restore x24+ ldp x25, x26, [sp, #0x60]+ .cfi_restore x25+ .cfi_restore x26+ ldp x27, x28, [sp, #0x70]+ .cfi_restore x27+ .cfi_restore x28+ ldp x29, x30, [sp, #0x80]+ .cfi_restore x29+ .cfi_restore x30+ add sp, sp, #0xe0+ .cfi_adjust_cfa_offset -0xe0+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)++#endif /* MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S view
@@ -0,0 +1,990 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++// Author: Hanno Becker <hannobecker@posteo.de>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>++/*yaml+ Name: keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm+ Description: AArch64 hybrid scalar/vector implementation of Keccak-f[1600] permutation for four sequential states with ARMv8.4-A optimizations+ Signature: void mld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm(uint64_t state[100], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON, SHA3]+ x0:+ type: buffer+ size_bytes: 800+ permissions: read/write+ c_parameter: uint64_t state[100]+ description: Four sequential Keccak states (state0[25], state1[25], state2[25], state3[25])+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: uint64_t const *rc+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 224+ description: register preservation and temporary storage+*/++#include "../../../../common.h"+#if defined(MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#if defined(__ARM_FEATURE_SHA3)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)+MLD_ASM_FN_SYMBOL(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0xe0+ .cfi_adjust_cfa_offset 0xe0+ stp x19, x20, [sp, #0x30]+ .cfi_rel_offset x19, 0x30+ .cfi_rel_offset x20, 0x38+ stp x21, x22, [sp, #0x40]+ .cfi_rel_offset x21, 0x40+ .cfi_rel_offset x22, 0x48+ stp x23, x24, [sp, #0x50]+ .cfi_rel_offset x23, 0x50+ .cfi_rel_offset x24, 0x58+ stp x25, x26, [sp, #0x60]+ .cfi_rel_offset x25, 0x60+ .cfi_rel_offset x26, 0x68+ stp x27, x28, [sp, #0x70]+ .cfi_rel_offset x27, 0x70+ .cfi_rel_offset x28, 0x78+ stp x29, x30, [sp, #0x80]+ .cfi_rel_offset x29, 0x80+ .cfi_rel_offset x30, 0x88+ stp d8, d9, [sp, #0x90]+ .cfi_rel_offset d8, 0x90+ .cfi_rel_offset d9, 0x98+ stp d10, d11, [sp, #0xa0]+ .cfi_rel_offset d10, 0xa0+ .cfi_rel_offset d11, 0xa8+ stp d12, d13, [sp, #0xb0]+ .cfi_rel_offset d12, 0xb0+ .cfi_rel_offset d13, 0xb8+ stp d14, d15, [sp, #0xc0]+ .cfi_rel_offset d14, 0xc0+ .cfi_rel_offset d15, 0xc8+ mov x29, x1+ mov x30, #0x0 // =0+ str x30, [sp, #0x20]+ str x29, [sp, #0x8]+ str x29, [sp, #0x10]+ str x0, [sp]+ add x4, x0, #0xc8+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v0.2d, v25.2d, v27.2d+ trn2 v1.2d, v25.2d, v27.2d+ trn1 v2.2d, v26.2d, v28.2d+ trn2 v3.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v4.2d, v25.2d, v27.2d+ trn2 v5.2d, v25.2d, v27.2d+ trn1 v6.2d, v26.2d, v28.2d+ trn2 v7.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v8.2d, v25.2d, v27.2d+ trn2 v9.2d, v25.2d, v27.2d+ trn1 v10.2d, v26.2d, v28.2d+ trn2 v11.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v12.2d, v25.2d, v27.2d+ trn2 v13.2d, v25.2d, v27.2d+ trn1 v14.2d, v26.2d, v28.2d+ trn2 v15.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v16.2d, v25.2d, v27.2d+ trn2 v17.2d, v25.2d, v27.2d+ trn1 v18.2d, v26.2d, v28.2d+ trn2 v19.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v20.2d, v25.2d, v27.2d+ trn2 v21.2d, v25.2d, v27.2d+ trn1 v22.2d, v26.2d, v28.2d+ trn2 v23.2d, v26.2d, v28.2d+ ldr d25, [x0]+ ldr d27, [x4]+ trn1 v24.2d, v25.2d, v27.2d+ sub x0, x0, #0xc0+ add x0, x0, #0x190+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x190++Lmld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_initial:+ eor x30, x24, x25+ eor x27, x9, x10+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor v30.16b, v30.16b, v15.16b+ eor x0, x30, x21+ eor x26, x27, x6+ eor v30.16b, v30.16b, v20.16b+ eor x27, x26, x7+ eor x29, x0, x22+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor x26, x29, x23+ eor x29, x4, x5+ eor v29.16b, v29.16b, v16.16b+ eor x30, x29, x1+ eor x0, x27, x8+ eor v29.16b, v29.16b, v21.16b+ eor x29, x30, x2+ eor x30, x19, x20+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor x30, x30, x16+ eor x27, x26, x0, ror #63+ eor v28.16b, v28.16b, v17.16b+ eor x4, x4, x27+ eor x30, x30, x17+ eor v28.16b, v28.16b, v22.16b+ eor x30, x30, x28+ eor x29, x29, x3+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor x0, x0, x30, ror #63+ eor x30, x30, x29, ror #63+ eor v27.16b, v27.16b, v18.16b+ eor x22, x22, x30+ eor v27.16b, v27.16b, v23.16b+ eor x23, x23, x30+ str x23, [sp, #0xd0]+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor x23, x14, x15+ eor x14, x14, x0+ eor v26.16b, v26.16b, v19.16b+ eor x23, x23, x11+ eor x15, x15, x0+ eor v26.16b, v26.16b, v24.16b+ eor x1, x1, x27+ eor x23, x23, x12+ rax1 v25.2d, v30.2d, v28.2d+ eor x23, x23, x13+ eor x11, x11, x0+ add v31.2d, v26.2d, v26.2d+ eor x29, x29, x23, ror #63+ eor x23, x23, x26, ror #63+ sri v31.2d, v26.2d, #0x3f+ eor x26, x13, x0+ eor x13, x28, x23+ eor v28.16b, v31.16b, v28.16b+ eor x28, x24, x30+ eor x24, x16, x23+ rax1 v26.2d, v26.2d, v29.2d+ eor x16, x21, x30+ eor x21, x25, x30+ add v31.2d, v27.2d, v27.2d+ eor x30, x19, x23+ sri v31.2d, v27.2d, #0x3f+ eor x19, x20, x23+ eor x20, x17, x23+ eor v29.16b, v31.16b, v29.16b+ eor x17, x12, x0+ eor x0, x2, x27+ rax1 v27.2d, v27.2d, v30.2d+ eor x2, x6, x29+ eor x6, x8, x29+ eor v30.16b, v0.16b, v26.16b+ bic x8, x28, x13, ror #47+ eor x12, x3, x27+ eor v31.16b, v2.16b, v29.16b+ bic x3, x13, x17, ror #19+ eor x5, x5, x27+ shl v0.2d, v31.2d, #0x3e+ ldr x27, [sp, #0xd0]+ bic x25, x17, x2, ror #5+ sri v0.2d, v31.2d, #0x2+ eor x9, x9, x29+ eor x23, x25, x5, ror #52+ xar v2.2d, v12.2d, v29.2d, #0x15+ eor x3, x3, x2, ror #24+ eor x8, x8, x17, ror #2+ eor v31.16b, v13.16b, v28.16b+ eor x17, x10, x29+ bic x25, x12, x22, ror #47+ shl v12.2d, v31.2d, #0x19+ eor x29, x7, x29+ bic x10, x4, x27, ror #2+ sri v12.2d, v31.2d, #0x27+ bic x7, x5, x28, ror #10+ xar v13.2d, v19.2d, v27.2d, #0x38+ eor x10, x10, x20, ror #50+ eor x13, x7, x13, ror #57+ eor v31.16b, v23.16b, v28.16b+ bic x7, x2, x5, ror #47+ eor x2, x25, x24, ror #39+ shl v19.2d, v31.2d, #0x38+ bic x25, x20, x11, ror #57+ bic x5, x17, x4, ror #25+ sri v19.2d, v31.2d, #0x8+ eor x25, x25, x17, ror #53+ bic x17, x11, x17, ror #60+ xar v23.2d, v15.2d, v26.2d, #0x17+ eor x28, x7, x28, ror #57+ bic x7, x9, x12, ror #42+ eor v31.16b, v1.16b, v25.16b+ eor x7, x7, x22, ror #25+ bic x22, x22, x24, ror #56+ shl v15.2d, v31.2d, #0x1+ bic x24, x24, x15, ror #31+ eor x22, x22, x15, ror #23+ sri v15.2d, v31.2d, #0x3f+ bic x20, x27, x20, ror #48+ bic x15, x15, x9, ror #16+ xar v1.2d, v8.2d, v28.2d, #0x9+ eor x12, x15, x12, ror #58+ eor x15, x5, x27, ror #27+ eor v31.16b, v16.16b, v25.16b+ eor x5, x20, x11, ror #41+ shl v8.2d, v31.2d, #0x2d+ ldr x11, [sp, #0x8]+ eor x20, x17, x4, ror #21+ sri v8.2d, v31.2d, #0x13+ eor x17, x24, x9, ror #47+ mov x24, #0x1 // =1+ xar v16.2d, v7.2d, v29.2d, #0x3a+ bic x9, x0, x16, ror #9+ str x24, [sp, #0x18]+ eor v31.16b, v10.16b, v26.16b+ bic x24, x29, x1, ror #44+ bic x27, x1, x21, ror #50+ shl v7.2d, v31.2d, #0x3+ bic x4, x26, x29, ror #63+ eor x1, x1, x4, ror #21+ sri v7.2d, v31.2d, #0x3d+ ldr x11, [x11]+ bic x4, x21, x30, ror #57+ xar v10.2d, v3.2d, v28.2d, #0x24+ eor x21, x24, x21, ror #30+ eor x24, x9, x19, ror #44+ eor v31.16b, v18.16b, v28.16b+ bic x9, x14, x6, ror #5+ eor x9, x9, x0, ror #43+ shl v3.2d, v31.2d, #0x15+ bic x0, x6, x0, ror #38+ eor x1, x1, x11+ sri v3.2d, v31.2d, #0x2b+ eor x11, x4, x26, ror #35+ eor x4, x0, x16, ror #47+ xar v18.2d, v17.2d, v29.2d, #0x31+ bic x0, x16, x19, ror #35+ eor v31.16b, v11.16b, v25.16b+ eor x16, x27, x30, ror #43+ bic x27, x30, x26, ror #42+ shl v17.2d, v31.2d, #0xa+ bic x26, x19, x14, ror #41+ eor x19, x0, x14, ror #12+ sri v17.2d, v31.2d, #0x36+ eor x14, x26, x6, ror #46+ eor x6, x27, x29, ror #41+ xar v11.2d, v9.2d, v27.2d, #0x2c+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor v31.16b, v22.16b, v29.16b+ eor x26, x8, x9, ror #57+ eor x27, x0, x14, ror #10+ shl v9.2d, v31.2d, #0x3d+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ sri v9.2d, v31.2d, #0x3+ eor x30, x23, x22, ror #50+ eor x0, x26, x10, ror #31+ xar v22.2d, v14.2d, v27.2d, #0x19+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ eor v31.16b, v20.16b, v26.16b+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ shl v14.2d, v31.2d, #0x12+ eor x26, x30, x21, ror #26+ sri v14.2d, v31.2d, #0x2e+ eor x26, x26, x25, ror #15+ ror x30, x27, #0x3e+ xar v20.2d, v4.2d, v27.2d, #0x25+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ eor v31.16b, v24.16b, v27.16b+ eor x16, x30, x16+ eor x28, x30, x28, ror #63+ shl v4.2d, v31.2d, #0xe+ str x28, [sp, #0xd0]+ eor x29, x29, x17, ror #36+ sri v4.2d, v31.2d, #0x32+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ xar v24.2d, v21.2d, v25.2d, #0x3e+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ eor v31.16b, v5.16b, v26.16b+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ shl v21.2d, v31.2d, #0x24+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ sri v21.2d, v31.2d, #0x1c+ eor x0, x0, x29, ror #63+ eor x27, x28, x27, ror #61+ xar v27.2d, v6.2d, v25.2d, #0x14+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ bic v31.16b, v7.16b, v11.16b+ eor x29, x30, x20, ror #2+ eor v5.16b, v31.16b, v10.16b+ eor x20, x26, x3, ror #39+ eor x11, x0, x11, ror #50+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ bic v31.16b, v9.16b, v8.16b+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ eor v7.16b, v31.16b, v7.16b+ eor x24, x28, x24, ror #28+ eor x1, x30, x17, ror #36+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ bic v31.16b, v11.16b, v10.16b+ eor x8, x27, x8, ror #56+ eor x17, x27, x7, ror #19+ eor v9.16b, v31.16b, v9.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ eor x4, x26, x4, ror #54+ eor x0, x0, x12, ror #3+ bic v31.16b, v13.16b, v12.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ eor v11.16b, v31.16b, v16.16b+ eor x26, x26, x5, ror #25+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ eor x2, x7, x16, ror #39+ bic x7, x9, x20, ror #42+ bic v31.16b, v15.16b, v14.16b+ bic x30, x15, x9, ror #16+ eor x7, x7, x22, ror #25+ eor v13.16b, v31.16b, v13.16b+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ bic v31.16b, v16.16b, v15.16b+ eor x30, x27, x6, ror #43+ eor x22, x20, x15, ror #23+ eor v14.16b, v31.16b, v14.16b+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ bic x5, x13, x17, ror #63+ eor x5, x21, x5, ror #21+ bic v31.16b, v18.16b, v17.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ eor v16.16b, v31.16b, v21.16b+ bic x21, x21, x25, ror #50+ bic x20, x27, x4, ror #25+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ bic v31.16b, v20.16b, v19.16b+ eor x21, x17, x25, ror #30+ bic x19, x25, x19, ror #57+ eor v18.16b, v31.16b, v18.16b+ ldr x25, [sp, #0x18]+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ eor x17, x10, x9, ror #47+ ldr x9, [sp, #0x8]+ bic v31.16b, v22.16b, v1.16b+ eor x15, x20, x28, ror #27+ bic x20, x4, x28, ror #2+ eor v20.16b, v31.16b, v0.16b+ eor x10, x20, x1, ror #50+ bic x20, x11, x27, ror #60+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ eor x20, x20, x4, ror #21+ bic x4, x28, x1, ror #48+ bic v31.16b, v24.16b, v23.16b+ bic x1, x1, x11, ror #57+ ldr x28, [x9, x25, lsl #3]+ eor v22.16b, v31.16b, v22.16b+ ldr x9, [sp, #0xd0]+ add x25, x25, #0x1+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ str x25, [sp, #0x18]+ cmp x25, #0x17+ bic v31.16b, v1.16b, v0.16b+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ eor v24.16b, v31.16b, v24.16b+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ eor x11, x19, x13, ror #35+ bic v31.16b, v3.16b, v2.16b+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ eor v1.16b, v31.16b, v27.16b+ bic x27, x24, x9, ror #47+ bic x19, x23, x3, ror #9+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic v31.16b, v30.16b, v4.16b+ bic x29, x3, x29, ror #35+ eor x13, x13, x9, ror #57+ eor v3.16b, v31.16b, v3.16b+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ bic x14, x14, x8, ror #5+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop:+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor v30.16b, v30.16b, v15.16b+ eor x26, x8, x9, ror #57+ eor x27, x0, x14, ror #10+ eor v30.16b, v30.16b, v20.16b+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor x30, x23, x22, ror #50+ eor x0, x26, x10, ror #31+ eor v29.16b, v29.16b, v16.16b+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ eor v29.16b, v29.16b, v21.16b+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor x26, x30, x21, ror #26+ eor x26, x26, x25, ror #15+ eor v28.16b, v28.16b, v17.16b+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ eor v28.16b, v28.16b, v22.16b+ ror x26, x26, #0x3a+ eor x16, x30, x16+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor x28, x30, x28, ror #63+ str x28, [sp, #0xd0]+ eor v27.16b, v27.16b, v18.16b+ eor x29, x29, x17, ror #36+ eor x28, x1, x2, ror #61+ eor v27.16b, v27.16b, v23.16b+ eor x19, x30, x19, ror #37+ eor x29, x29, x20, ror #2+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor x28, x28, x4, ror #54+ eor x26, x26, x0, ror #55+ eor v26.16b, v26.16b, v19.16b+ eor x28, x28, x3, ror #39+ eor x28, x28, x5, ror #25+ eor v26.16b, v26.16b, v24.16b+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ rax1 v25.2d, v30.2d, v28.2d+ eor x27, x28, x27, ror #61+ eor x13, x0, x13, ror #46+ add v31.2d, v26.2d, v26.2d+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ sri v31.2d, v26.2d, #0x3f+ eor x20, x26, x3, ror #39+ eor x11, x0, x11, ror #50+ eor v28.16b, v31.16b, v28.16b+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ rax1 v26.2d, v26.2d, v29.2d+ eor x21, x26, x1+ add v31.2d, v27.2d, v27.2d+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ sri v31.2d, v27.2d, #0x3f+ eor x1, x30, x17, ror #36+ eor x14, x0, x14, ror #8+ eor v29.16b, v31.16b, v29.16b+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ rax1 v27.2d, v27.2d, v30.2d+ eor x17, x27, x7, ror #19+ eor x15, x0, x15, ror #62+ eor v30.16b, v0.16b, v26.16b+ bic x7, x20, x22, ror #47+ eor x4, x26, x4, ror #54+ eor v31.16b, v2.16b, v29.16b+ eor x0, x0, x12, ror #3+ eor x28, x28, x23, ror #58+ shl v0.2d, v31.2d, #0x3e+ eor x23, x26, x2, ror #61+ eor x26, x26, x5, ror #25+ sri v0.2d, v31.2d, #0x2+ eor x2, x7, x16, ror #39+ bic x7, x9, x20, ror #42+ xar v2.2d, v12.2d, v29.2d, #0x15+ bic x30, x15, x9, ror #16+ eor x7, x7, x22, ror #25+ eor v31.16b, v13.16b, v28.16b+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ shl v12.2d, v31.2d, #0x19+ eor x30, x27, x6, ror #43+ eor x22, x20, x15, ror #23+ sri v12.2d, v31.2d, #0x27+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ xar v13.2d, v19.2d, v27.2d, #0x38+ bic x5, x13, x17, ror #63+ eor x5, x21, x5, ror #21+ eor v31.16b, v23.16b, v28.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ shl v19.2d, v31.2d, #0x38+ bic x21, x21, x25, ror #50+ bic x20, x27, x4, ror #25+ sri v19.2d, v31.2d, #0x8+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ xar v23.2d, v15.2d, v26.2d, #0x17+ eor x21, x17, x25, ror #30+ bic x19, x25, x19, ror #57+ eor v31.16b, v1.16b, v25.16b+ ldr x25, [sp, #0x18]+ eor x17, x10, x9, ror #47+ shl v15.2d, v31.2d, #0x1+ ldr x9, [sp, #0x8]+ sri v15.2d, v31.2d, #0x3f+ eor x15, x20, x28, ror #27+ bic x20, x4, x28, ror #2+ xar v1.2d, v8.2d, v28.2d, #0x9+ eor x10, x20, x1, ror #50+ bic x20, x11, x27, ror #60+ eor v31.16b, v16.16b, v25.16b+ eor x20, x20, x4, ror #21+ bic x4, x28, x1, ror #48+ shl v8.2d, v31.2d, #0x2d+ bic x1, x1, x11, ror #57+ ldr x28, [x9, x25, lsl #3]+ sri v8.2d, v31.2d, #0x13+ ldr x9, [sp, #0xd0]+ add x25, x25, #0x1+ xar v16.2d, v7.2d, v29.2d, #0x3a+ str x25, [sp, #0x18]+ cmp x25, #0x17+ eor v31.16b, v10.16b, v26.16b+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ shl v7.2d, v31.2d, #0x3+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ sri v7.2d, v31.2d, #0x3d+ eor x11, x19, x13, ror #35+ bic x13, x26, x24, ror #10+ xar v10.2d, v3.2d, v28.2d, #0x24+ eor x28, x27, x24, ror #57+ bic x27, x24, x9, ror #47+ eor v31.16b, v18.16b, v28.16b+ bic x19, x23, x3, ror #9+ bic x4, x29, x14, ror #41+ shl v3.2d, v31.2d, #0x15+ eor x24, x19, x29, ror #44+ bic x29, x3, x29, ror #35+ sri v3.2d, v31.2d, #0x2b+ eor x13, x13, x9, ror #57+ eor x19, x29, x14, ror #12+ xar v18.2d, v17.2d, v29.2d, #0x31+ bic x29, x9, x0, ror #19+ bic x14, x14, x8, ror #5+ eor v31.16b, v11.16b, v25.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ shl v17.2d, v31.2d, #0xa+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ sri v17.2d, v31.2d, #0x36+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ xar v11.2d, v9.2d, v27.2d, #0x2c+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ eor v31.16b, v22.16b, v29.16b+ eor x0, x15, x11, ror #52+ shl v9.2d, v31.2d, #0x3d+ eor x0, x0, x13, ror #48+ eor x26, x8, x9, ror #57+ sri v9.2d, v31.2d, #0x3+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ xar v22.2d, v14.2d, v27.2d, #0x19+ eor x26, x26, x6, ror #51+ eor x30, x23, x22, ror #50+ eor v31.16b, v20.16b, v26.16b+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ shl v14.2d, v31.2d, #0x12+ eor x27, x27, x12, ror #5+ eor x30, x30, x24, ror #34+ sri v14.2d, v31.2d, #0x2e+ eor x0, x0, x7, ror #27+ eor x26, x30, x21, ror #26+ xar v20.2d, v4.2d, v27.2d, #0x25+ eor x26, x26, x25, ror #15+ ror x30, x27, #0x3e+ eor v31.16b, v24.16b, v27.16b+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ shl v4.2d, v31.2d, #0xe+ eor x16, x30, x16+ eor x28, x30, x28, ror #63+ sri v4.2d, v31.2d, #0x32+ str x28, [sp, #0xd0]+ eor x29, x29, x17, ror #36+ xar v24.2d, v21.2d, v25.2d, #0x3e+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ eor v31.16b, v5.16b, v26.16b+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ shl v21.2d, v31.2d, #0x24+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ sri v21.2d, v31.2d, #0x1c+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ xar v27.2d, v6.2d, v25.2d, #0x14+ eor x0, x0, x29, ror #63+ eor x27, x28, x27, ror #61+ bic v31.16b, v7.16b, v11.16b+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ eor v5.16b, v31.16b, v10.16b+ eor x29, x30, x20, ror #2+ eor x20, x26, x3, ror #39+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ eor x11, x0, x11, ror #50+ eor x25, x28, x25, ror #9+ bic v31.16b, v9.16b, v8.16b+ eor x3, x28, x21, ror #20+ eor v7.16b, v31.16b, v7.16b+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ eor x24, x28, x24, ror #28+ eor x1, x30, x17, ror #36+ bic v31.16b, v11.16b, v10.16b+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ eor v9.16b, v31.16b, v9.16b+ eor x8, x27, x8, ror #56+ eor x17, x27, x7, ror #19+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ bic v31.16b, v13.16b, v12.16b+ eor x4, x26, x4, ror #54+ eor x0, x0, x12, ror #3+ eor v11.16b, v31.16b, v16.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ eor x26, x26, x5, ror #25+ eor x2, x7, x16, ror #39+ bic v31.16b, v15.16b, v14.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ eor v13.16b, v31.16b, v13.16b+ eor x7, x7, x22, ror #25+ eor x12, x30, x20, ror #58+ bic v31.16b, v16.16b, v15.16b+ bic x20, x22, x16, ror #56+ eor x30, x27, x6, ror #43+ eor v14.16b, v31.16b, v14.16b+ eor x22, x20, x15, ror #23+ bic x6, x19, x13, ror #42+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ eor x6, x6, x17, ror #41+ bic x5, x13, x17, ror #63+ bic v31.16b, v18.16b, v17.16b+ eor x5, x21, x5, ror #21+ bic x17, x17, x21, ror #44+ eor v16.16b, v31.16b, v21.16b+ eor x27, x27, x10, ror #23+ bic x21, x21, x25, ror #50+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ bic v31.16b, v20.16b, v19.16b+ eor x16, x21, x19, ror #43+ eor x21, x17, x25, ror #30+ eor v18.16b, v31.16b, v18.16b+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x18]+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ eor x17, x10, x9, ror #47+ bic v31.16b, v22.16b, v1.16b+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ eor v20.16b, v31.16b, v0.16b+ bic x20, x4, x28, ror #2+ eor x10, x20, x1, ror #50+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ bic v31.16b, v24.16b, v23.16b+ bic x4, x28, x1, ror #48+ bic x1, x1, x11, ror #57+ eor v22.16b, v31.16b, v22.16b+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0xd0]+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ add x25, x25, #0x1+ str x25, [sp, #0x18]+ bic v31.16b, v1.16b, v0.16b+ cmp x25, #0x17+ eor x25, x1, x27, ror #53+ eor v24.16b, v31.16b, v24.16b+ bic x27, x30, x26, ror #47+ eor x1, x5, x28+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ eor x5, x4, x11, ror #41+ eor x11, x19, x13, ror #35+ bic v31.16b, v3.16b, v2.16b+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ eor v1.16b, v31.16b, v27.16b+ bic x27, x24, x9, ror #47+ bic x19, x23, x3, ror #9+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic v31.16b, v30.16b, v4.16b+ bic x29, x3, x29, ror #35+ eor x13, x13, x9, ror #57+ eor v3.16b, v31.16b, v3.16b+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ bic x14, x14, x8, ror #5+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop_end:+ b.le Lmld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop+ ror x2, x2, #0x3d+ ror x3, x3, #0x27+ ror x4, x4, #0x36+ ror x5, x5, #0x19+ ror x6, x6, #0x2b+ ror x7, x7, #0x13+ ror x8, x8, #0x38+ ror x9, x9, #0x31+ ror x10, x10, #0x17+ ror x11, x11, #0x32+ ror x12, x12, #0x3+ ror x13, x13, #0x2e+ ror x14, x14, #0x8+ ror x15, x15, #0x3e+ ror x17, x17, #0x24+ ror x28, x28, #0x3f+ ror x19, x19, #0x25+ ror x20, x20, #0x2+ ror x21, x21, #0x14+ ror x22, x22, #0x2c+ ror x23, x23, #0x3a+ ror x24, x24, #0x1c+ ror x25, x25, #0x9+ ldr x30, [sp, #0x20]+ cmp x30, #0x1+ b.eq Lmld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_done+ mov x30, #0x1 // =1+ str x30, [sp, #0x20]+ ldr x0, [sp]+ add x0, x0, #0x190+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x190+ add x0, x0, #0x258+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x258+ b Lmld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_initial++Lmld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_done:+ ldr x0, [sp]+ add x0, x0, #0x258+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x258+ add x4, x0, #0xc8+ trn1 v25.2d, v0.2d, v1.2d+ trn1 v26.2d, v2.2d, v3.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v0.2d, v1.2d+ trn2 v28.2d, v2.2d, v3.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v4.2d, v5.2d+ trn1 v26.2d, v6.2d, v7.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v4.2d, v5.2d+ trn2 v28.2d, v6.2d, v7.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v8.2d, v9.2d+ trn1 v26.2d, v10.2d, v11.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v8.2d, v9.2d+ trn2 v28.2d, v10.2d, v11.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v12.2d, v13.2d+ trn1 v26.2d, v14.2d, v15.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v12.2d, v13.2d+ trn2 v28.2d, v14.2d, v15.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v16.2d, v17.2d+ trn1 v26.2d, v18.2d, v19.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v16.2d, v17.2d+ trn2 v28.2d, v18.2d, v19.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v20.2d, v21.2d+ trn1 v26.2d, v22.2d, v23.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v20.2d, v21.2d+ trn2 v28.2d, v22.2d, v23.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ str d24, [x0]+ trn2 v25.2d, v24.2d, v24.2d+ str d25, [x4]+ ldp d8, d9, [sp, #0x90]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0xa0]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0xb0]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0xc0]+ .cfi_restore d14+ .cfi_restore d15+ ldp x19, x20, [sp, #0x30]+ .cfi_restore x19+ .cfi_restore x20+ ldp x21, x22, [sp, #0x40]+ .cfi_restore x21+ .cfi_restore x22+ ldp x23, x24, [sp, #0x50]+ .cfi_restore x23+ .cfi_restore x24+ ldp x25, x26, [sp, #0x60]+ .cfi_restore x25+ .cfi_restore x26+ ldp x27, x28, [sp, #0x70]+ .cfi_restore x27+ .cfi_restore x28+ ldp x29, x30, [sp, #0x80]+ .cfi_restore x29+ .cfi_restore x30+ add sp, sp, #0xe0+ .cfi_adjust_cfa_offset -0xe0+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)++#endif /* __ARM_FEATURE_SHA3 */++#endif /* MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/fips202/native/aarch64/src/keccakf1600_round_constants.c view
@@ -0,0 +1,47 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../../common.h"++#if (defined(MLD_FIPS202_AARCH64_NEED_X1_SCALAR) || \+ defined(MLD_FIPS202_AARCH64_NEED_X1_V84A) || \+ defined(MLD_FIPS202_AARCH64_NEED_X2_V84A) || \+ defined(MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID) || \+ defined(MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID)) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include "fips202_native_aarch64.h"++MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint64_t+ mld_keccakf1600_round_constants[24] = {+ 0x0000000000000001, 0x0000000000008082, 0x800000000000808a,+ 0x8000000080008000, 0x000000000000808b, 0x0000000080000001,+ 0x8000000080008081, 0x8000000000008009, 0x000000000000008a,+ 0x0000000000000088, 0x0000000080008009, 0x000000008000000a,+ 0x000000008000808b, 0x800000000000008b, 0x8000000000008089,+ 0x8000000000008003, 0x8000000000008002, 0x8000000000000080,+ 0x000000000000800a, 0x800000008000000a, 0x8000000080008081,+ 0x8000000000008080, 0x0000000080000001, 0x8000000080008008,+};++#else /* (MLD_FIPS202_AARCH64_NEED_X1_SCALAR || \+ MLD_FIPS202_AARCH64_NEED_X1_V84A || MLD_FIPS202_AARCH64_NEED_X2_V84A \+ || MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID || \+ MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID) && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(fips202_aarch64_round_constants)++#endif /* !((MLD_FIPS202_AARCH64_NEED_X1_SCALAR || \+ MLD_FIPS202_AARCH64_NEED_X1_V84A || MLD_FIPS202_AARCH64_NEED_X2_V84A \+ || MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID || \+ MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID) && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mldsa/src/fips202/native/aarch64/x1_scalar.h view
@@ -0,0 +1,27 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_AARCH64_X1_SCALAR_H+#define MLD_FIPS202_NATIVE_AARCH64_X1_SCALAR_H++/* Part of backend API */+#define MLD_USE_NATIVE_FIPS202_X1+/* Guard for assembly file */+#define MLD_FIPS202_AARCH64_NEED_X1_SCALAR++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x1_native(uint64_t *state)+{+ mld_keccak_f1600_x1_scalar_aarch64_asm(state,+ mld_keccakf1600_round_constants);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLD_FIPS202_NATIVE_AARCH64_X1_SCALAR_H */
+ cbits/mldsa/src/fips202/native/aarch64/x1_v84a.h view
@@ -0,0 +1,36 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_AARCH64_X1_V84A_H+#define MLD_FIPS202_NATIVE_AARCH64_X1_V84A_H++#if !defined(__ARM_FEATURE_SHA3)+#error This backend can only be used if SHA3 extensions are available.+#endif++/* Part of backend API */+#define MLD_USE_NATIVE_FIPS202_X1+/* Guard for assembly file */+#define MLD_FIPS202_AARCH64_NEED_X1_V84A++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x1_native(uint64_t *state)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON) ||+ !mld_sys_check_capability(MLD_SYS_CAP_AARCH64_SHA3))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ mld_keccak_f1600_x1_v84a_aarch64_asm(state, mld_keccakf1600_round_constants);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLD_FIPS202_NATIVE_AARCH64_X1_V84A_H */
+ cbits/mldsa/src/fips202/native/aarch64/x2_v84a.h view
@@ -0,0 +1,40 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_AARCH64_X2_V84A_H+#define MLD_FIPS202_NATIVE_AARCH64_X2_V84A_H++#if !defined(__ARM_FEATURE_SHA3)+#error This backend can only be used if SHA3 extensions are available.+#endif++/* Part of backend API */+#define MLD_USE_NATIVE_FIPS202_X4+/* Guard for assembly file */+#define MLD_FIPS202_AARCH64_NEED_X2_V84A++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON) ||+ !mld_sys_check_capability(MLD_SYS_CAP_AARCH64_SHA3))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ mld_keccak_f1600_x2_v84a_aarch64_asm(state + 0 * 25,+ mld_keccakf1600_round_constants);+ mld_keccak_f1600_x2_v84a_aarch64_asm(state + 2 * 25,+ mld_keccakf1600_round_constants);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLD_FIPS202_NATIVE_AARCH64_X2_V84A_H */
+ cbits/mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h view
@@ -0,0 +1,32 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H+#define MLD_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H++/* Part of backend API */+#define MLD_USE_NATIVE_FIPS202_X4+/* Guard for assembly file */+#define MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ mld_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(+ state, mld_keccakf1600_round_constants);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLD_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H */
+ cbits/mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h view
@@ -0,0 +1,37 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H+#define MLD_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H++#if !defined(__ARM_FEATURE_SHA3)+#error This backend can only be used if SHA3 extensions are available.+#endif++/* Part of backend API */+#define MLD_USE_NATIVE_FIPS202_X4+/* Guard for assembly file */+#define MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON) ||+ !mld_sys_check_capability(MLD_SYS_CAP_AARCH64_SHA3))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ mld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm(+ state, mld_keccakf1600_round_constants);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLD_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H */
+ cbits/mldsa/src/fips202/native/api.h view
@@ -0,0 +1,129 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_API_H+#define MLD_FIPS202_NATIVE_API_H+/*+ * FIPS-202 native interface+ *+ * This header is primarily for documentation purposes.+ * It should not be included by backend implementations.+ */++#include "../../cbmc.h"+#include "../../common.h"++/* Backends must return MLD_NATIVE_FUNC_SUCCESS upon success. */+#define MLD_NATIVE_FUNC_SUCCESS (0)+/* Backends may return MLD_NATIVE_FUNC_FALLBACK to signal to the frontend that+ * the target/parameters are unsupported; typically, this would be because of+ * dependencies on CPU features not detected on the host CPU. In this case,+ * the frontend falls back to the default C implementation.+ *+ * IMPORTANT: Backend implementations must ensure that the decision of whether+ * to fallback (return MLD_NATIVE_FUNC_FALLBACK) or not must never depend on+ * the input data itself. Fallback decisions may only depend on system+ * capabilities (e.g., CPU features) and, where present, length information.+ * This requirement applies to all backend functions to maintain constant-time+ * properties.+ */+#define MLD_NATIVE_FUNC_FALLBACK (-1)++/*+ * This is the C<->native interface allowing for the drop-in+ * of custom Keccak-F1600 implementations.+ *+ * A _backend_ is a specific implementation of parts of this interface.+ *+ * You can replace 1-fold or 4-fold batched Keccak-F1600.+ * To enable, set MLD_USE_NATIVE_FIPS202_X1 or MLD_USE_NATIVE_FIPS202_X4+ * in your backend, and define the inline wrappers mld_keccak_f1600_x1_native()+ * and/or mld_keccak_f1600_x4_native(), respectively, to forward to your+ * implementation.+ */++#if defined(MLD_USE_NATIVE_FIPS202_X1)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x1_native(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 1))+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 1))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged_u64(state, 25 * 1))+);+#endif /* MLD_USE_NATIVE_FIPS202_X1 */+#if defined(MLD_USE_NATIVE_FIPS202_X4)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x4_native(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged_u64(state, 25 * 4))+);+#endif /* MLD_USE_NATIVE_FIPS202_X4 */++/*+ * Native x4 XOR bytes and extract bytes interface.+ *+ * These functions allow backends to provide optimized implementations for+ * XORing input data into the state and extracting output data from the state.+ * This is particularly useful for backends that use a different internal state+ * representation (e.g., bit-interleaved), as conversion can happen during+ * XOR/extract rather than before/after each permutation.+ *+ * NOTE: We assume that the custom representation of the zero state is the+ * all-zero state.+ *+ * MLD_USE_NATIVE_FIPS202_X4_XOR_BYTES: Backend provides native XOR bytes+ * MLD_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES: Backend provides native extract+ * bytes+ */++#if defined(MLD_USE_NATIVE_FIPS202_X4_XOR_BYTES)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccakf1600_xor_bytes_x4_native(+ uint64_t *state, const unsigned char *data0, const unsigned char *data1,+ const unsigned char *data2, const unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= 25 * sizeof(uint64_t) &&+ 0 <= length && length <= 25 * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(memory_no_alias(data0, length))+ requires((data0 == data1 &&+ data0 == data2 &&+ data0 == data3) ||+ (memory_no_alias(data1, length) &&+ memory_no_alias(data2, length) &&+ memory_no_alias(data3, length)))+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged_u64(state, 25 * 4)));+#endif /* MLD_USE_NATIVE_FIPS202_X4_XOR_BYTES */++#if defined(MLD_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccakf1600_extract_bytes_x4_native(+ uint64_t *state, unsigned char *data0, unsigned char *data1,+ unsigned char *data2, unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= 25 * sizeof(uint64_t) &&+ 0 <= length && length <= 25 * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(memory_no_alias(data0, length))+ requires(memory_no_alias(data1, length))+ requires(memory_no_alias(data2, length))+ requires(memory_no_alias(data3, length))+ assigns(memory_slice(data0, length))+ assigns(memory_slice(data1, length))+ assigns(memory_slice(data2, length))+ assigns(memory_slice(data3, length))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS));+#endif /* MLD_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES */++#endif /* !MLD_FIPS202_NATIVE_API_H */
+ cbits/mldsa/src/fips202/native/auto.h view
@@ -0,0 +1,35 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_AUTO_H+#define MLD_FIPS202_NATIVE_AUTO_H++/*+ * Default FIPS202 backend+ */+#include "../../sys.h"++#if defined(MLD_SYS_AARCH64)+#include "aarch64/auto.h"+#endif++/* The x86_64 backend requires toolchain support for the SysV ABI */+#if defined(MLD_SYS_X86_64_AVX2) && defined(MLD_SYSV_ABI_SUPPORTED) && \+ (!defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_REDUCE_RAM)) && \+ !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+#include "x86_64/keccak_f1600_x4_avx2.h"+#endif /* MLD_SYS_X86_64_AVX2 && MLD_SYSV_ABI_SUPPORTED && \+ (!MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_REDUCE_RAM) && !MLD_CONFIG_SERIAL_FIPS202_ONLY */++/* We do not yet include the FIPS202 backend for Armv8.1-M+MVE by default+ * as it is still experimental and undergoing review. */+/* #if defined(MLD_SYS_ARMV81M_MVE) */+/* #include "armv81m/mve.h" */+/* #endif */++#endif /* !MLD_FIPS202_NATIVE_AUTO_H */
+ cbits/mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h view
@@ -0,0 +1,34 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H+#define MLD_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H++#include "../../../common.h"++#define MLD_FIPS202_X86_64_NEED_X4_AVX2++/* Part of backend API */+#define MLD_USE_NATIVE_FIPS202_X4++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_x86_64.h"+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ mld_keccak_f1600_x4_avx2_asm(state, mld_keccakf1600_round_constants,+ mld_keccak_rho8, mld_keccak_rho56);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLD_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H */
+ cbits/mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h view
@@ -0,0 +1,45 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H+#define MLD_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H++#include "../../../../cbmc.h"+#include "../../../../common.h"++/* TODO: Reconsider whether this check is needed -- x86_64 is always+ * little-endian, so the backend selection already implies this. */+#ifndef MLD_SYS_LITTLE_ENDIAN+#error Expecting a little-endian platform+#endif++#define mld_keccakf1600_round_constants \+ MLD_NAMESPACE(keccakf1600_round_constants)+MLD_INTERNAL_DATA_DECLARATION const uint64_t+ mld_keccakf1600_round_constants[24];++#define mld_keccak_rho8 MLD_NAMESPACE(keccak_rho8)+MLD_INTERNAL_DATA_DECLARATION const uint64_t mld_keccak_rho8[4];++#define mld_keccak_rho56 MLD_NAMESPACE(keccak_rho56)+MLD_INTERNAL_DATA_DECLARATION const uint64_t mld_keccak_rho56[4];++#define mld_keccak_f1600_x4_avx2_asm MLD_NAMESPACE(keccak_f1600_x4_avx2_asm)+MLD_SYSV_ABI+void mld_keccak_f1600_x4_avx2_asm(uint64_t states[100], const uint64_t rc[24],+ const uint64_t rho8[4],+ const uint64_t rho56[4])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/keccak_f1600_x4_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(states, sizeof(uint64_t) * 25 * 4))+ requires(rc == mld_keccakf1600_round_constants)+ requires(rho8 == mld_keccak_rho8)+ requires(rho56 == mld_keccak_rho56)+ assigns(memory_slice(states, sizeof(uint64_t) * 25 * 4))+);++#endif /* !MLD_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H */
+ cbits/mldsa/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S view
@@ -0,0 +1,488 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+/*yaml+ Name: keccak_f1600_x4_avx2_asm+ Description: x86_64 AVX2 Keccak-f[1600] permutation for four sequential states+ Signature: void mld_keccak_f1600_x4_avx2_asm(uint64_t states[100], const uint64_t rc[24], const uint64_t rho8[4], const uint64_t rho56[4])+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 800+ permissions: read/write+ c_parameter: uint64_t states[100]+ description: Four sequential Keccak states (4 x 25 x uint64_t)+ rsi:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: const uint64_t rc[24]+ description: Round constants (24 x uint64_t)+ rdx:+ type: buffer+ size_bytes: 32+ permissions: read-only+ c_parameter: const uint64_t rho8[4]+ description: Rotation constant rho8 (4 x uint64_t)+ rcx:+ type: buffer+ size_bytes: 32+ permissions: read-only+ c_parameter: const uint64_t rho56[4]+ description: Rotation constant rho56 (4 x uint64_t)+*/++#include "../../../../common.h"++#if defined(MLD_FIPS202_X86_64_NEED_X4_AVX2) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/fips202/x86_64/src/keccak_f1600_x4_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(keccak_f1600_x4_avx2_asm)+MLD_ASM_FN_SYMBOL(keccak_f1600_x4_avx2_asm)++ .cfi_startproc+ movq %rsp, %r11+ .cfi_def_cfa_register %r11+ andq $-0x20, %rsp+ subq $0x300, %rsp # imm = 0x300+ vmovdqu (%rdi), %ymm0+ vmovdqu 0xc8(%rdi), %ymm3+ vmovdqu 0x190(%rdi), %ymm1+ vmovdqu 0x258(%rdi), %ymm4+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm7 # ymm7 = ymm2[0,1],ymm3[0,1]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vmovdqu 0x278(%rdi), %ymm4+ vmovdqu %ymm3, 0x40(%rsp)+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm3 # ymm3 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm7, (%rsp)+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu 0x20(%rdi), %ymm0+ vmovdqu 0x1b0(%rdi), %ymm1+ vmovdqu %ymm3, 0x60(%rsp)+ vmovdqu 0xe8(%rdi), %ymm3+ vmovdqu %ymm7, 0x20(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm7 # ymm7 = ymm2[0,1],ymm3[0,1]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vmovdqu 0x298(%rdi), %ymm4+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm14 # ymm14 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm7, 0x80(%rsp)+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu 0x40(%rdi), %ymm0+ vmovdqu 0x1d0(%rdi), %ymm1+ vmovdqu %ymm3, 0xc0(%rsp)+ vmovdqu 0x108(%rdi), %ymm3+ vmovdqu %ymm14, %ymm10+ vmovdqu %ymm7, 0xa0(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm11 # ymm11 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu %ymm3, 0x100(%rsp)+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm8 # ymm8 = ymm0[2,3],ymm1[2,3]+ vmovdqu 0x128(%rdi), %ymm3+ vmovdqu 0x60(%rdi), %ymm0+ vmovdqu 0x1f0(%rdi), %ymm1+ vmovdqu %ymm7, 0xe0(%rsp)+ vmovdqu %ymm11, %ymm14+ vmovdqu 0x2b8(%rdi), %ymm4+ vmovdqu 0x2f8(%rdi), %ymm5+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vmovdqu 0x2d8(%rdi), %ymm4+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm15 # ymm15 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm9 # ymm9 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm3, 0x140(%rsp)+ vmovdqu 0x80(%rdi), %ymm0+ vmovdqu 0x148(%rdi), %ymm3+ vmovdqu 0x210(%rdi), %ymm1+ vmovdqu %ymm7, 0x120(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm7 # ymm7 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm13 # ymm13 = ymm2[2,3],ymm3[2,3]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm3 # ymm3 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm7, 0x160(%rsp)+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu 0xa0(%rdi), %ymm0+ vmovdqu 0x230(%rdi), %ymm1+ vmovdqu %ymm3, 0x1a0(%rsp)+ vmovdqu 0x168(%rdi), %ymm3+ vpunpcklqdq %ymm5, %ymm1, %ymm4 # ymm4 = ymm1[0],ymm5[0],ymm1[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm5[1],ymm1[3],ymm5[3]+ vmovdqu %ymm7, 0x180(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm12 # ymm12 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm3 # ymm3 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm7 # ymm7 = ymm2[2,3],ymm4[2,3]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm4 # ymm4 = ymm0[2,3],ymm1[2,3]+ vmovq 0x250(%rdi), %xmm0+ vmovq 0xc0(%rdi), %xmm1+ vmovdqu %ymm12, 0x1c0(%rsp)+ vmovdqu %ymm4, 0x1e0(%rsp)+ vpinsrq $0x1, 0x318(%rdi), %xmm0, %xmm0+ vpinsrq $0x1, 0x188(%rdi), %xmm1, %xmm1+ vinserti128 $0x1, %xmm0, %ymm1, %ymm2+ movq $0x0, %r10++LLmld_keccak_f1600_x4_avx2_asm:+ vmovdqu 0xa0(%rsp), %ymm4+ vpxor 0x1c0(%rsp), %ymm9, %ymm0+ vmovdqu %ymm9, 0x200(%rsp)+ vmovdqu %ymm10, %ymm9+ vmovdqu 0xc0(%rsp), %ymm11+ vmovdqu 0x160(%rsp), %ymm12+ vmovdqu %ymm3, 0x240(%rsp)+ vpxor 0x100(%rsp), %ymm4, %ymm1+ vmovdqu 0x40(%rsp), %ymm10+ vmovdqu %ymm4, 0x220(%rsp)+ vpxor %ymm3, %ymm12, %ymm12+ vmovdqu 0x20(%rsp), %ymm6+ vmovdqu 0x140(%rsp), %ymm4+ vmovdqu %ymm14, 0x2a0(%rsp)+ vpxor %ymm1, %ymm0, %ymm0+ vpxor %ymm8, %ymm11, %ymm1+ vpxor 0x180(%rsp), %ymm7, %ymm11+ vmovdqu %ymm10, 0x280(%rsp)+ vpxor %ymm1, %ymm12, %ymm12+ vpxor %ymm15, %ymm9, %ymm1+ vmovdqu 0xe0(%rsp), %ymm3+ vmovdqu %ymm8, 0x260(%rsp)+ vpxor %ymm1, %ymm11, %ymm11+ vpxor 0x120(%rsp), %ymm14, %ymm1+ vpxor %ymm6, %ymm12, %ymm12+ vmovdqu 0x60(%rsp), %ymm8+ vpxor %ymm10, %ymm11, %ymm11+ vpxor 0x1e0(%rsp), %ymm13, %ymm10+ vpxor %ymm4, %ymm3, %ymm3+ vmovdqu %ymm4, 0x2c0(%rsp)+ vpsrlq $0x3f, %ymm12, %ymm4+ vpsrlq $0x3f, %ymm11, %ymm5+ vpxor (%rsp), %ymm0, %ymm0+ vpxor %ymm1, %ymm10, %ymm10+ vmovdqu 0x80(%rsp), %ymm1+ vpxor %ymm8, %ymm10, %ymm10+ vmovdqu %ymm1, %ymm14+ vpxor 0x1a0(%rsp), %ymm2, %ymm1+ vmovdqu %ymm14, 0x2e0(%rsp)+ vpxor %ymm3, %ymm1, %ymm1+ vpsllq $0x1, %ymm12, %ymm3+ vpor %ymm4, %ymm3, %ymm3+ vpsllq $0x1, %ymm11, %ymm4+ vpxor %ymm14, %ymm1, %ymm1+ vpor %ymm5, %ymm4, %ymm4+ vpsrlq $0x3f, %ymm10, %ymm14+ vpxor %ymm1, %ymm3, %ymm3+ vpsllq $0x1, %ymm10, %ymm5+ vpxor %ymm0, %ymm4, %ymm4+ vpor %ymm14, %ymm5, %ymm5+ vpxor %ymm6, %ymm4, %ymm6+ vpxor %ymm12, %ymm5, %ymm5+ vpsrlq $0x3f, %ymm1, %ymm12+ vpsllq $0x1, %ymm1, %ymm1+ vpxor %ymm7, %ymm5, %ymm7+ vpxor %ymm9, %ymm5, %ymm9+ vpor %ymm12, %ymm1, %ymm1+ vpxor (%rsp), %ymm3, %ymm12+ vpxor %ymm11, %ymm1, %ymm1+ vpsrlq $0x3f, %ymm0, %ymm11+ vpsllq $0x1, %ymm0, %ymm0+ vpxor %ymm13, %ymm1, %ymm13+ vpxor %ymm8, %ymm1, %ymm8+ vpor %ymm11, %ymm0, %ymm0+ vpxor %ymm10, %ymm0, %ymm0+ vpxor 0xc0(%rsp), %ymm4, %ymm10+ vpxor %ymm2, %ymm0, %ymm2+ vpsrlq $0x14, %ymm10, %ymm11+ vpsllq $0x2c, %ymm10, %ymm10+ vpor %ymm11, %ymm10, %ymm10+ vpxor %ymm15, %ymm5, %ymm11+ vpbroadcastq (%rsi), %ymm15+ vpsrlq $0x15, %ymm11, %ymm14+ vpsllq $0x2b, %ymm11, %ymm11+ vpor %ymm14, %ymm11, %ymm11+ vpandn %ymm11, %ymm10, %ymm14+ vpxor %ymm15, %ymm14, %ymm14+ vpxor %ymm12, %ymm14, %ymm15+ vpsrlq $0x2b, %ymm13, %ymm14+ vpsllq $0x15, %ymm13, %ymm13+ vmovdqu %ymm15, (%rsp)+ vpor %ymm14, %ymm13, %ymm13+ vpandn %ymm13, %ymm11, %ymm14+ vpxor %ymm10, %ymm14, %ymm15+ vpsrlq $0x32, %ymm2, %ymm14+ vpsllq $0xe, %ymm2, %ymm2+ vmovdqu %ymm15, 0x20(%rsp)+ vpor %ymm14, %ymm2, %ymm2+ vpandn %ymm2, %ymm13, %ymm14+ vpxor %ymm11, %ymm14, %ymm11+ vmovdqu %ymm11, 0x40(%rsp)+ vpandn %ymm12, %ymm2, %ymm11+ vpandn %ymm10, %ymm12, %ymm12+ vpxor %ymm13, %ymm11, %ymm11+ vmovdqu %ymm11, 0x60(%rsp)+ vpxor %ymm2, %ymm12, %ymm11+ vpsrlq $0x24, %ymm8, %ymm2+ vpsllq $0x1c, %ymm8, %ymm8+ vmovdqu %ymm11, 0x80(%rsp)+ vpor %ymm2, %ymm8, %ymm8+ vpxor 0xe0(%rsp), %ymm0, %ymm2+ vpsrlq $0x2c, %ymm2, %ymm10+ vpsllq $0x14, %ymm2, %ymm2+ vpor %ymm10, %ymm2, %ymm2+ vpxor 0x100(%rsp), %ymm3, %ymm10+ vpsrlq $0x3d, %ymm10, %ymm11+ vpsllq $0x3, %ymm10, %ymm10+ vpor %ymm11, %ymm10, %ymm10+ vpandn %ymm10, %ymm2, %ymm11+ vpxor %ymm8, %ymm11, %ymm11+ vmovdqu %ymm11, 0xa0(%rsp)+ vpxor 0x160(%rsp), %ymm4, %ymm11+ vpsrlq $0x13, %ymm11, %ymm12+ vpsllq $0x2d, %ymm11, %ymm11+ vpor %ymm12, %ymm11, %ymm11+ vpandn %ymm11, %ymm10, %ymm12+ vpxor %ymm2, %ymm12, %ymm12+ vmovdqu %ymm12, 0xc0(%rsp)+ vpsrlq $0x3, %ymm7, %ymm12+ vpsllq $0x3d, %ymm7, %ymm7+ vpor %ymm12, %ymm7, %ymm7+ vpandn %ymm7, %ymm11, %ymm12+ vpxor %ymm10, %ymm12, %ymm10+ vpandn %ymm8, %ymm7, %ymm12+ vpandn %ymm2, %ymm8, %ymm8+ vpsrlq $0x3f, %ymm6, %ymm2+ vpsllq $0x1, %ymm6, %ymm6+ vpxor %ymm11, %ymm12, %ymm14+ vpor %ymm2, %ymm6, %ymm6+ vpsrlq $0x3a, %ymm9, %ymm2+ vpxor %ymm7, %ymm8, %ymm12+ vpsllq $0x6, %ymm9, %ymm9+ vmovdqu %ymm12, 0xe0(%rsp)+ vpxor 0x1a0(%rsp), %ymm0, %ymm7+ vpor %ymm2, %ymm9, %ymm9+ vpxor 0x120(%rsp), %ymm1, %ymm2+ vpshufb (%rdx), %ymm7, %ymm7+ vpsrlq $0x27, %ymm2, %ymm11+ vpsllq $0x19, %ymm2, %ymm2+ vpor %ymm2, %ymm11, %ymm11+ vpandn %ymm11, %ymm9, %ymm2+ vpandn %ymm7, %ymm11, %ymm8+ vpxor %ymm6, %ymm2, %ymm12+ vpxor 0x1c0(%rsp), %ymm3, %ymm2+ vpxor %ymm9, %ymm8, %ymm8+ vmovdqu %ymm12, 0x100(%rsp)+ vpsrlq $0x2e, %ymm2, %ymm12+ vpsllq $0x12, %ymm2, %ymm2+ vpor %ymm2, %ymm12, %ymm2+ vpandn %ymm2, %ymm7, %ymm12+ vpxor %ymm11, %ymm12, %ymm15+ vpandn %ymm6, %ymm2, %ymm11+ vpandn %ymm9, %ymm6, %ymm6+ vpxor %ymm7, %ymm11, %ymm12+ vmovdqu %ymm12, 0x120(%rsp)+ vpxor %ymm2, %ymm6, %ymm12+ vpxor 0x2e0(%rsp), %ymm0, %ymm6+ vpxor 0x2c0(%rsp), %ymm0, %ymm0+ vmovdqu %ymm12, 0x140(%rsp)+ vpsrlq $0x25, %ymm6, %ymm2+ vpsllq $0x1b, %ymm6, %ymm6+ vpor %ymm6, %ymm2, %ymm2+ vpxor 0x220(%rsp), %ymm3, %ymm6+ vpxor 0x200(%rsp), %ymm3, %ymm3+ vpsrlq $0x1c, %ymm6, %ymm7+ vpsllq $0x24, %ymm6, %ymm6+ vpor %ymm6, %ymm7, %ymm7+ vpxor 0x260(%rsp), %ymm4, %ymm6+ vpxor 0x240(%rsp), %ymm4, %ymm4+ vpsrlq $0x36, %ymm6, %ymm12+ vpsllq $0xa, %ymm6, %ymm6+ vpor %ymm6, %ymm12, %ymm12+ vpxor 0x180(%rsp), %ymm5, %ymm6+ vpxor 0x280(%rsp), %ymm5, %ymm5+ vpandn %ymm12, %ymm7, %ymm9+ vpsrlq $0x31, %ymm6, %ymm11+ vpsllq $0xf, %ymm6, %ymm6+ vpxor %ymm2, %ymm9, %ymm9+ vpor %ymm6, %ymm11, %ymm11+ vpandn %ymm11, %ymm12, %ymm6+ vpxor %ymm7, %ymm6, %ymm6+ vmovdqu %ymm6, 0x160(%rsp)+ vpxor 0x1e0(%rsp), %ymm1, %ymm6+ vpxor 0x2a0(%rsp), %ymm1, %ymm1+ vpshufb (%rcx), %ymm6, %ymm6+ vpandn %ymm6, %ymm11, %ymm13+ vpxor %ymm12, %ymm13, %ymm13+ vmovdqu %ymm13, 0x180(%rsp)+ vpandn %ymm2, %ymm6, %ymm13+ vpandn %ymm7, %ymm2, %ymm2+ vpxor %ymm6, %ymm2, %ymm2+ vpsrlq $0x3e, %ymm4, %ymm6+ vpxor %ymm11, %ymm13, %ymm13+ vmovdqu %ymm2, 0x1a0(%rsp)+ vpsrlq $0x2, %ymm5, %ymm2+ vpsllq $0x3e, %ymm5, %ymm5+ vpor %ymm5, %ymm2, %ymm2+ vpsrlq $0x9, %ymm1, %ymm5+ vpsllq $0x37, %ymm1, %ymm1+ vpsllq $0x2, %ymm4, %ymm4+ vpor %ymm1, %ymm5, %ymm1+ vpsrlq $0x19, %ymm0, %ymm5+ vpor %ymm4, %ymm6, %ymm4+ vpsllq $0x27, %ymm0, %ymm0+ vpor %ymm0, %ymm5, %ymm5+ vpandn %ymm5, %ymm1, %ymm0+ vpxor %ymm2, %ymm0, %ymm0+ vmovdqu %ymm0, 0x1c0(%rsp)+ vpsrlq $0x17, %ymm3, %ymm0+ vpsllq $0x29, %ymm3, %ymm3+ vpor %ymm3, %ymm0, %ymm0+ vpandn %ymm4, %ymm0, %ymm7+ vpandn %ymm0, %ymm5, %ymm3+ vpxor %ymm5, %ymm7, %ymm7+ vpandn %ymm2, %ymm4, %ymm5+ vpandn %ymm1, %ymm2, %ymm2+ vpxor %ymm0, %ymm5, %ymm5+ vpxor %ymm1, %ymm3, %ymm3+ vpxor %ymm4, %ymm2, %ymm2+ vmovdqu %ymm5, 0x1e0(%rsp)+ addq $0x8, %rsi+ addq $0x1, %r10+ cmpq $0x18, %r10+ jne LLmld_keccak_f1600_x4_avx2_asm+ vmovdqu (%rsp), %ymm4+ vmovdqu 0x40(%rsp), %ymm5+ vmovdqu 0x20(%rsp), %ymm0+ vmovdqu 0x60(%rsp), %ymm1+ vmovdqu 0x1c0(%rsp), %ymm12+ vmovdqu %ymm2, 0x1c0(%rsp)+ vpunpcklqdq %ymm0, %ymm4, %ymm2 # ymm2 = ymm4[0],ymm0[0],ymm4[2],ymm0[2]+ vpunpckhqdq %ymm0, %ymm4, %ymm0 # ymm0 = ymm4[1],ymm0[1],ymm4[3],ymm0[3]+ vpunpcklqdq %ymm1, %ymm5, %ymm4 # ymm4 = ymm5[0],ymm1[0],ymm5[2],ymm1[2]+ vpunpckhqdq %ymm1, %ymm5, %ymm1 # ymm1 = ymm5[1],ymm1[1],ymm5[3],ymm1[3]+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vmovdqu 0x80(%rsp), %ymm4+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm5 # ymm5 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm0 # ymm0 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm6, (%rdi)+ vmovdqu %ymm5, 0xc8(%rdi)+ vmovdqu %ymm2, 0x190(%rdi)+ vmovdqu %ymm0, 0x258(%rdi)+ vmovdqu 0xa0(%rsp), %ymm0+ vpunpcklqdq %ymm0, %ymm4, %ymm2 # ymm2 = ymm4[0],ymm0[0],ymm4[2],ymm0[2]+ vpunpckhqdq %ymm0, %ymm4, %ymm1 # ymm1 = ymm4[1],ymm0[1],ymm4[3],ymm0[3]+ vmovdqu 0xc0(%rsp), %ymm0+ vpunpcklqdq %ymm10, %ymm0, %ymm4 # ymm4 = ymm0[0],ymm10[0],ymm0[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm10[1],ymm0[3],ymm10[3]+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x20, %ymm0, %ymm1, %ymm5 # ymm5 = ymm1[0,1],ymm0[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vmovdqu 0xe0(%rsp), %ymm4+ vperm2i128 $0x31, %ymm0, %ymm1, %ymm1 # ymm1 = ymm1[2,3],ymm0[2,3]+ vmovdqu 0x100(%rsp), %ymm0+ vmovdqu %ymm2, 0x1b0(%rdi)+ vmovdqu %ymm1, 0x278(%rdi)+ vpunpcklqdq %ymm4, %ymm14, %ymm2 # ymm2 = ymm14[0],ymm4[0],ymm14[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm14, %ymm1 # ymm1 = ymm14[1],ymm4[1],ymm14[3],ymm4[3]+ vpunpcklqdq %ymm8, %ymm0, %ymm4 # ymm4 = ymm0[0],ymm8[0],ymm0[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm8[1],ymm0[3],ymm8[3]+ vmovdqu %ymm6, 0x20(%rdi)+ vmovdqu %ymm5, 0xe8(%rdi)+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x20, %ymm0, %ymm1, %ymm5 # ymm5 = ymm1[0,1],ymm0[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vperm2i128 $0x31, %ymm0, %ymm1, %ymm1 # ymm1 = ymm1[2,3],ymm0[2,3]+ vmovdqu 0x120(%rsp), %ymm4+ vmovdqu 0x140(%rsp), %ymm0+ vmovdqu %ymm2, 0x1d0(%rdi)+ vmovdqu %ymm1, 0x298(%rdi)+ vpunpcklqdq %ymm4, %ymm15, %ymm2 # ymm2 = ymm15[0],ymm4[0],ymm15[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm15, %ymm1 # ymm1 = ymm15[1],ymm4[1],ymm15[3],ymm4[3]+ vpunpcklqdq %ymm9, %ymm0, %ymm4 # ymm4 = ymm0[0],ymm9[0],ymm0[2],ymm9[2]+ vmovdqu %ymm5, 0x108(%rdi)+ vpunpckhqdq %ymm9, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm9[1],ymm0[3],ymm9[3]+ vmovdqu %ymm6, 0x40(%rdi)+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vperm2i128 $0x20, %ymm0, %ymm1, %ymm5 # ymm5 = ymm1[0,1],ymm0[0,1]+ vmovdqu 0x160(%rsp), %ymm4+ vperm2i128 $0x31, %ymm0, %ymm1, %ymm1 # ymm1 = ymm1[2,3],ymm0[2,3]+ vmovdqu 0x180(%rsp), %ymm0+ vmovdqu %ymm5, 0x128(%rdi)+ vmovdqu 0x1a0(%rsp), %ymm5+ vmovdqu %ymm2, 0x1f0(%rdi)+ vpunpcklqdq %ymm0, %ymm4, %ymm2 # ymm2 = ymm4[0],ymm0[0],ymm4[2],ymm0[2]+ vpunpckhqdq %ymm0, %ymm4, %ymm0 # ymm0 = ymm4[1],ymm0[1],ymm4[3],ymm0[3]+ vpunpcklqdq %ymm5, %ymm13, %ymm4 # ymm4 = ymm13[0],ymm5[0],ymm13[2],ymm5[2]+ vmovdqu %ymm6, 0x60(%rdi)+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vmovdqu %ymm1, 0x2b8(%rdi)+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vpunpckhqdq %ymm5, %ymm13, %ymm1 # ymm1 = ymm13[1],ymm5[1],ymm13[3],ymm5[3]+ vmovdqu %ymm6, 0x80(%rdi)+ vmovdqu 0x1e0(%rsp), %ymm4+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm5 # ymm5 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm0 # ymm0 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm2, 0x210(%rdi)+ vpunpcklqdq %ymm3, %ymm12, %ymm2 # ymm2 = ymm12[0],ymm3[0],ymm12[2],ymm3[2]+ vmovdqu %ymm0, 0x2d8(%rdi)+ vpunpckhqdq %ymm3, %ymm12, %ymm0 # ymm0 = ymm12[1],ymm3[1],ymm12[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm7, %ymm3 # ymm3 = ymm7[0],ymm4[0],ymm7[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm7, %ymm1 # ymm1 = ymm7[1],ymm4[1],ymm7[3],ymm4[3]+ vmovdqu %ymm5, 0x148(%rdi)+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm5 # ymm5 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm3[2,3]+ vmovdqu 0x1c0(%rsp), %ymm3+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm4 # ymm4 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm0 # ymm0 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm5, 0xa0(%rdi)+ vextracti128 $0x1, %ymm3, %xmm15+ vmovdqu %ymm4, 0x168(%rdi)+ vmovdqu %ymm2, 0x230(%rdi)+ vmovdqu %ymm0, 0x2f8(%rdi)+ vmovq %xmm3, 0xc0(%rdi)+ vmovhpd %xmm3, 0x188(%rdi)+ vmovq %xmm15, 0x250(%rdi)+ vmovhpd %xmm15, 0x318(%rdi)+ movq %r11, %rsp+ .cfi_def_cfa_register %rsp+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(keccak_f1600_x4_avx2_asm)++#endif /* MLD_FIPS202_X86_64_NEED_X4_AVX2 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/fips202/native/x86_64/src/keccakf1600_constants.c view
@@ -0,0 +1,52 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../../common.h"+#if defined(MLD_FIPS202_X86_64_NEED_X4_AVX2) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include <stdint.h>++#include "fips202_native_x86_64.h"++MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint64_t+ mld_keccakf1600_round_constants[24] = {+ 0x0000000000000001, 0x0000000000008082, 0x800000000000808a,+ 0x8000000080008000, 0x000000000000808b, 0x0000000080000001,+ 0x8000000080008081, 0x8000000000008009, 0x000000000000008a,+ 0x0000000000000088, 0x0000000080008009, 0x000000008000000a,+ 0x000000008000808b, 0x800000000000008b, 0x8000000000008089,+ 0x8000000000008003, 0x8000000000008002, 0x8000000000000080,+ 0x000000000000800a, 0x800000008000000a, 0x8000000080008081,+ 0x8000000000008080, 0x0000000080000001, 0x8000000080008008,+};++MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint64_t mld_keccak_rho8[4] = {+ 0x0605040302010007,+ 0x0e0d0c0b0a09080f,+ 0x1615141312111017,+ 0x1e1d1c1b1a19181f,+};++MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint64_t mld_keccak_rho56[4] = {+ 0x0007060504030201,+ 0x080f0e0d0c0b0a09,+ 0x1017161514131211,+ 0x181f1e1d1c1b1a19,+};++#else /* MLD_FIPS202_X86_64_NEED_X4_AVX2 && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++MLD_EMPTY_CU(fips202_x86_64_constants)++#endif /* !(MLD_FIPS202_X86_64_NEED_X4_AVX2 && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mldsa/src/native/aarch64/meta.h view
@@ -0,0 +1,314 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_NATIVE_AARCH64_META_H+#define MLD_NATIVE_AARCH64_META_H++/* Set of primitives that this backend replaces */+#define MLD_USE_NATIVE_NTT+#define MLD_USE_NATIVE_INTT+#define MLD_USE_NATIVE_REJ_UNIFORM+#define MLD_USE_NATIVE_REJ_UNIFORM_ETA2+#define MLD_USE_NATIVE_REJ_UNIFORM_ETA4+#define MLD_USE_NATIVE_POLY_DECOMPOSE_32+#define MLD_USE_NATIVE_POLY_DECOMPOSE_88+#define MLD_USE_NATIVE_POLY_CADDQ+#define MLD_USE_NATIVE_POLY_USE_HINT_32+#define MLD_USE_NATIVE_POLY_USE_HINT_88+#define MLD_USE_NATIVE_POLY_CHKNORM+#define MLD_USE_NATIVE_POLYZ_UNPACK_17+#define MLD_USE_NATIVE_POLYZ_UNPACK_19+#define MLD_USE_NATIVE_POINTWISE_MONTGOMERY+#define MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4+#define MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5+#define MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7++/* Identifier for this backend so that source and assembly files+ * in the build can be appropriately guarded. */+#define MLD_ARITH_BACKEND_AARCH64+++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/arith_native_aarch64.h"++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_ntt_native(int32_t data[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_ntt_aarch64_asm(data, mld_aarch64_ntt_zetas_layer123456,+ mld_aarch64_ntt_zetas_layer78);+ return MLD_NATIVE_FUNC_SUCCESS;+}++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_intt_native(int32_t data[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_intt_aarch64_asm(data, mld_aarch64_intt_zetas_layer78,+ mld_aarch64_intt_zetas_layer123456);+ return MLD_NATIVE_FUNC_SUCCESS;+}++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON) || len != MLDSA_N ||+ buflen % 24 != 0)+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ /* Safety: outlen is at most MLDSA_N, hence, this cast is safe. */+ return (int)mld_rej_uniform_aarch64_asm(r, buf, buflen,+ mld_rej_uniform_table);+}++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 2+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_eta2_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ uint64_t outlen;+ /* AArch64 implementation assumes specific buffer lengths */+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON) || len != MLDSA_N ||+ buflen != MLD_AARCH64_REJ_UNIFORM_ETA2_BUFLEN)+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ /* Constant time: Inputs and outputs to this function are secret.+ * It is safe to leak which coefficients are accepted/rejected.+ * The assembly implementation must not leak any other information about the+ * accepted coefficients. Constant-time testing cannot cover this, and we+ * hence have to manually verify the assembly.+ * We declassify prior the input data and mark the outputs as secret.+ */+ MLD_CT_TESTING_DECLASSIFY(buf, buflen);+ outlen = mld_rej_uniform_eta2_aarch64_asm(r, buf, buflen,+ mld_rej_uniform_eta_table);+ MLD_CT_TESTING_SECRET(r, sizeof(int32_t) * outlen);+ /* Safety: outlen is at most MLDSA_N and, hence, this cast is safe. */+ return (int)outlen;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 4+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_eta4_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ uint64_t outlen;+ /* AArch64 implementation assumes specific buffer lengths */+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON) || len != MLDSA_N ||+ buflen != MLD_AARCH64_REJ_UNIFORM_ETA4_BUFLEN)+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ /* Constant time: Inputs and outputs to this function are secret.+ * It is safe to leak which coefficients are accepted/rejected.+ * The assembly implementation must not leak any other information about the+ * accepted coefficients. Constant-time testing cannot cover this, and we+ * hence have to manually verify the assembly.+ * We declassify prior the input data and mark the outputs as secret.+ */+ MLD_CT_TESTING_DECLASSIFY(buf, buflen);+ outlen = mld_rej_uniform_eta4_aarch64_asm(r, buf, buflen,+ mld_rej_uniform_eta_table);+ MLD_CT_TESTING_SECRET(r, sizeof(int32_t) * outlen);+ /* Safety: outlen is at most MLDSA_N and, hence, this cast is safe. */+ return (int)outlen;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 4 */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_decompose_32_native(int32_t *a1, int32_t *a0)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_decompose_32_aarch64_asm(a1, a0);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_decompose_88_native(int32_t *a1, int32_t *a0)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_decompose_88_aarch64_asm(a1, a0);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */+#endif /* !MLD_CONFIG_NO_SIGN_API */++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_caddq_native(int32_t a[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_caddq_aarch64_asm(a);+ return MLD_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_use_hint_32_native(int32_t *a, const int32_t *h)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_use_hint_32_aarch64_asm(a, h);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_use_hint_88_native(int32_t *a, const int32_t *h)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_use_hint_88_aarch64_asm(a, h);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */+#endif /* !MLD_CONFIG_NO_VERIFY_API */++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_chknorm_native(const int32_t *a, int32_t B)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ return mld_poly_chknorm_aarch64_asm(a, B);+}++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyz_unpack_17_native(int32_t *r, const uint8_t *buf)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_polyz_unpack_17_aarch64_asm(r, buf, mld_polyz_unpack_17_indices);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyz_unpack_19_native(int32_t *r, const uint8_t *buf)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_polyz_unpack_19_aarch64_asm(r, buf, mld_polyz_unpack_19_indices);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_pointwise_montgomery_native(+ int32_t a[MLDSA_N], const int32_t b[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_pointwise_montgomery_aarch64_asm(a, b);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 4+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l4_native(+ int32_t w[MLDSA_N], const int32_t u[4][MLDSA_N],+ const int32_t v[4][MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm(w, u, v);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 4 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 5+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l5_native(+ int32_t w[MLDSA_N], const int32_t u[5][MLDSA_N],+ const int32_t v[5][MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm(w, u, v);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 5 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 7+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l7_native(+ int32_t w[MLDSA_N], const int32_t u[7][MLDSA_N],+ const int32_t v[7][MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm(w, u, v);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 7 */++#endif /* !__ASSEMBLER__ */+#endif /* !MLD_NATIVE_AARCH64_META_H */
+ cbits/mldsa/src/native/aarch64/src/aarch64_zetas.c view
@@ -0,0 +1,248 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include "arith_native_aarch64.h"++/*+ * Table of zeta values used in the AArch64 forward NTT+ * See autogen for details.+ */+MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const int32_t+ mld_aarch64_ntt_zetas_layer123456[144] = {+ -3572223, -915382907, 3765607, 964937599, 3761513, 963888510,+ -3201494, -820383522, -2883726, -738955404, -3145678, -806080660,+ -3201430, -820367122, 0, 0, -601683, -154181397,+ -3370349, -863652652, -4063053, -1041158200, 3602218, 923069133,+ 3182878, 815613168, 2740543, 702264730, -3586446, -919027554,+ 0, 0, 3542485, 907762539, 2663378, 682491182,+ -1674615, -429120452, -3110818, -797147778, 2101410, 538486762,+ 3704823, 949361686, 1159875, 297218217, 0, 0,+ 2682288, 687336873, -3524442, -903139016, -434125, -111244624,+ 394148, 101000509, 928749, 237992130, 1095468, 280713909,+ -3506380, -898510625, 0, 0, 2129892, 545785280,+ 676590, 173376332, -1335936, -342333886, 2071829, 530906624,+ -4018989, -1029866791, 3241972, 830756018, 2156050, 552488273,+ 0, 0, 3764867, 964747974, -3227876, -827143915,+ 1714295, 439288460, 3415069, 875112161, 1759347, 450833045,+ -817536, -209493775, -3574466, -915957677, 0, 0,+ -1005239, -257592709, 2453983, 628833668, 1460718, 374309300,+ 3756790, 962678241, -1935799, -496048908, -1716988, -439978542,+ -3950053, -1012201926, 0, 0, 557458, 142848732,+ -642628, -164673562, -3585098, -918682129, -2897314, -742437332,+ 3192354, 818041395, 556856, 142694469, 3870317, 991769559,+ 0, 0, -1221177, -312926867, 2815639, 721508096,+ 2283733, 585207070, 2917338, 747568486, 1853806, 475038184,+ 3345963, 857403734, 1858416, 476219497, 0, 0,+};++MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const int32_t+ mld_aarch64_ntt_zetas_layer78[384] = {+ 3073009, 1277625, -2635473, 3852015, 787459213,+ 327391679, -675340520, 987079667, 1753, -2659525,+ 2660408, -59148, 449207, -681503850, 681730119,+ -15156688, -1935420, -1455890, -1780227, 2772600,+ -495951789, -373072124, -456183549, 710479343, 4183372,+ -3222807, -3121440, -274060, 1071989969, -825844983,+ -799869667, -70227934, 1182243, 636927, -3956745,+ -3284915, 302950022, 163212680, -1013916752, -841760171,+ 87208, -3965306, -2296397, -3716946, 22347069,+ -1016110510, -588452222, -952468207, 2508980, 2028118,+ 1937570, -3815725, 642926661, 519705671, 496502727,+ -977780347, -27812, 1009365, -1979497, -3956944,+ -7126831, 258649997, -507246529, -1013967746, 822541,+ -2454145, 1596822, -3759465, 210776307, -628875181,+ 409185979, -963363710, 2811291, -2983781, -1109516,+ 4158088, 720393920, -764594519, -284313712, 1065510939,+ -1685153, 2678278, -3551006, -250446, -431820817,+ 686309310, -909946047, -64176841, -3410568, -3768948,+ 635956, -2455377, -873958779, -965793731, 162963861,+ -629190881, 1528066, 482649, 1148858, -2962264,+ 391567239, 123678909, 294395108, -759080783, -4146264,+ 2192938, 2387513, -268456, -1062481036, 561940831,+ 611800717, -68791907, -1772588, -1727088, -3611750,+ -3180456, -454226054, -442566669, -925511710, -814992530,+ -565603, 169688, 2462444, -3334383, -144935890,+ 43482586, 631001801, -854436357, 3747250, 1239911,+ 3195676, 1254190, 960233614, 317727459, 818892658,+ 321386456, 2296099, -3838479, 2642980, -12417,+ 588375860, -983611064, 677264190, -3181859, -4166425,+ -3488383, 1987814, -3197248, -1067647297, -893898890,+ 509377762, -819295484, 2998219, -89301, -1354892,+ -1310261, 768294260, -22883400, -347191365, -335754661,+ 141835, 2513018, 613238, -2218467, 36345249,+ 643961400, 157142369, -568482643, 1736313, 235407,+ -3250154, 3258457, 444930577, 60323094, -832852657,+ 834980303, -458740, 4040196, 2039144, -818761,+ -117552223, 1035301089, 522531086, -209807681, -1921994,+ -3472069, -1879878, -2178965, -492511373, -889718424,+ -481719139, -558360247, -2579253, 1787943, -2391089,+ -2254727, -660934133, 458160776, -612717067, -577774276,+ -1623354, -2374402, 586241, 527981, -415984810,+ -608441020, 150224382, 135295244, 2105286, -2033807,+ -1179613, -2743411, 539479988, -521163479, -302276083,+ -702999655, 3482206, -4182915, -1300016, -2362063,+ 892316032, -1071872863, -333129378, -605279149, -1476985,+ 2491325, 507927, -724804, -378477722, 638402564,+ 130156402, -185731180, 1994046, -1393159, -1187885,+ -1834526, 510974714, -356997292, -304395785, -470097680,+ -1317678, 2461387, 3035980, 621164, -337655269,+ 630730945, 777970524, 159173408, -3033742, 2647994,+ -2612853, 749577, -777397036, 678549029, -669544140,+ 192079267, -338420, 3009748, 4148469, -4022750,+ -86720197, 771248568, 1063046068, -1030830548, 3901472,+ -1226661, 2925816, 3374250, 999753034, -314332144,+ 749740976, 864652284, 3980599, -1615530, 1665318,+ 1163598, 1020029345, -413979908, 426738094, 298172236,+ 2569011, 1723229, 2028038, -3369273, 658309618,+ 441577800, 519685171, -863376927, 1356448, -2775755,+ 2683270, -2778788, 347590090, -711287812, 687588511,+ -712065019, 3994671, -1370517, 3363542, 545376,+ 1023635298, -351195274, 861908357, 139752717, -11879,+ 3020393, 214880, -770441, -3043996, 773976352,+ 55063046, -197425671, -3467665, 2312838, -653275,+ -459163, -888589898, 592665232, -167401858, -117660617,+ 3105558, 508145, 860144, 140244, 795799901,+ 130212265, 220412084, 35937555, -1103344, -553718,+ 3430436, -1514152, -282732136, -141890356, 879049958,+ -388001774, 348812, -327848, 1011223, -2354215,+ 89383150, -84011120, 259126110, -603268097, -2185084,+ 2358373, -3014420, 2926054, -559928242, 604333585,+ -772445769, 749801963, 3123762, -2193087, -1716814,+ -392707, 800464680, -561979013, -439933955, -100631253,+ -3818627, -1922253, -2236726, 1744507, -978523985,+ -492577742, -573161516, 447030292, -303005, -3974485,+ 1900052, 1054478, -77645096, -1018462631, 486888731,+ 270210213, 3531229, -3773731, -781875, -731434,+ 904878186, -967019376, -200355636, -187430119,+};++MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const int32_t+ mld_aarch64_intt_zetas_layer78[384] = {+ -1744507, 2236726, 1922253, 3818627, -447030292,+ 573161516, 492577742, 978523985, 731434, 781875,+ 3773731, -3531229, 187430119, 200355636, 967019376,+ -904878186, -1054478, -1900052, 3974485, 303005,+ -270210213, -486888731, 1018462631, 77645096, 2354215,+ -1011223, 327848, -348812, 603268097, -259126110,+ 84011120, -89383150, 392707, 1716814, 2193087,+ -3123762, 100631253, 439933955, 561979013, -800464680,+ -2926054, 3014420, -2358373, 2185084, -749801963,+ 772445769, -604333585, 559928242, 459163, 653275,+ -2312838, 3467665, 117660617, 167401858, -592665232,+ 888589898, 1514152, -3430436, 553718, 1103344,+ 388001774, -879049958, 141890356, 282732136, -140244,+ -860144, -508145, -3105558, -35937555, -220412084,+ -130212265, -795799901, 2778788, -2683270, 2775755,+ -1356448, 712065019, -687588511, 711287812, -347590090,+ 770441, -214880, -3020393, 11879, 197425671,+ -55063046, -773976352, 3043996, -545376, -3363542,+ 1370517, -3994671, -139752717, -861908357, 351195274,+ -1023635298, -3374250, -2925816, 1226661, -3901472,+ -864652284, -749740976, 314332144, -999753034, 3369273,+ -2028038, -1723229, -2569011, 863376927, -519685171,+ -441577800, -658309618, -1163598, -1665318, 1615530,+ -3980599, -298172236, -426738094, 413979908, -1020029345,+ -621164, -3035980, -2461387, 1317678, -159173408,+ -777970524, -630730945, 337655269, 4022750, -4148469,+ -3009748, 338420, 1030830548, -1063046068, -771248568,+ 86720197, -749577, 2612853, -2647994, 3033742,+ -192079267, 669544140, -678549029, 777397036, 2362063,+ 1300016, 4182915, -3482206, 605279149, 333129378,+ 1071872863, -892316032, 1834526, 1187885, 1393159,+ -1994046, 470097680, 304395785, 356997292, -510974714,+ 724804, -507927, -2491325, 1476985, 185731180,+ -130156402, -638402564, 378477722, 2254727, 2391089,+ -1787943, 2579253, 577774276, 612717067, -458160776,+ 660934133, 2743411, 1179613, 2033807, -2105286,+ 702999655, 302276083, 521163479, -539479988, -527981,+ -586241, 2374402, 1623354, -135295244, -150224382,+ 608441020, 415984810, -3258457, 3250154, -235407,+ -1736313, -834980303, 832852657, -60323094, -444930577,+ 2178965, 1879878, 3472069, 1921994, 558360247,+ 481719139, 889718424, 492511373, 818761, -2039144,+ -4040196, 458740, 209807681, -522531086, -1035301089,+ 117552223, 3197248, -1987814, 3488383, 4166425,+ 819295484, -509377762, 893898890, 1067647297, 2218467,+ -613238, -2513018, -141835, 568482643, -157142369,+ -643961400, -36345249, 1310261, 1354892, 89301,+ -2998219, 335754661, 347191365, 22883400, -768294260,+ 3334383, -2462444, -169688, 565603, 854436357,+ -631001801, -43482586, 144935890, 12417, -2642980,+ 3838479, -2296099, 3181859, -677264190, 983611064,+ -588375860, -1254190, -3195676, -1239911, -3747250,+ -321386456, -818892658, -317727459, -960233614, 2962264,+ -1148858, -482649, -1528066, 759080783, -294395108,+ -123678909, -391567239, 3180456, 3611750, 1727088,+ 1772588, 814992530, 925511710, 442566669, 454226054,+ 268456, -2387513, -2192938, 4146264, 68791907,+ -611800717, -561940831, 1062481036, -4158088, 1109516,+ 2983781, -2811291, -1065510939, 284313712, 764594519,+ -720393920, 2455377, -635956, 3768948, 3410568,+ 629190881, -162963861, 965793731, 873958779, 250446,+ 3551006, -2678278, 1685153, 64176841, 909946047,+ -686309310, 431820817, 3815725, -1937570, -2028118,+ -2508980, 977780347, -496502727, -519705671, -642926661,+ 3759465, -1596822, 2454145, -822541, 963363710,+ -409185979, 628875181, -210776307, 3956944, 1979497,+ -1009365, 27812, 1013967746, 507246529, -258649997,+ 7126831, 274060, 3121440, 3222807, -4183372,+ 70227934, 799869667, 825844983, -1071989969, 3716946,+ 2296397, 3965306, -87208, 952468207, 588452222,+ 1016110510, -22347069, 3284915, 3956745, -636927,+ -1182243, 841760171, 1013916752, -163212680, -302950022,+ -3852015, 2635473, -1277625, -3073009, -987079667,+ 675340520, -327391679, -787459213, -2772600, 1780227,+ 1455890, 1935420, -710479343, 456183549, 373072124,+ 495951789, 59148, -2660408, 2659525, -1753,+ 15156688, -681730119, 681503850, -449207,+};++MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const int32_t+ mld_aarch64_intt_zetas_layer123456[160] = {+ -2283733, -585207070, 0, 0, -1858416, -476219497,+ -3345963, -857403734, -2815639, -721508096, 0, 0,+ -1853806, -475038184, -2917338, -747568486, 3585098, 918682129,+ 0, 0, -3870317, -991769559, -556856, -142694469,+ 642628, 164673562, 0, 0, -3192354, -818041395,+ 2897314, 742437332, -1460718, -374309300, 0, 0,+ 3950053, 1012201926, 1716988, 439978542, -2453983, -628833668,+ 0, 0, 1935799, 496048908, -3756790, -962678241,+ -1714295, -439288460, 0, 0, 3574466, 915957677,+ 817536, 209493775, 3227876, 827143915, 0, 0,+ -1759347, -450833045, -3415069, -875112161, 1335936, 342333886,+ 0, 0, -2156050, -552488273, -3241972, -830756018,+ -676590, -173376332, 0, 0, 4018989, 1029866791,+ -2071829, -530906624, 434125, 111244624, 0, 0,+ 3506380, 898510625, -1095468, -280713909, 3524442, 903139016,+ 0, 0, -928749, -237992130, -394148, -101000509,+ 1674615, 429120452, 0, 0, -1159875, -297218217,+ -3704823, -949361686, -2663378, -682491182, 0, 0,+ -2101410, -538486762, 3110818, 797147778, 4063053, 1041158200,+ 0, 0, 3586446, 919027554, -2740543, -702264730,+ 3370349, 863652652, 0, 0, -3182878, -815613168,+ -3602218, -923069133, -294725, -75523344, -3761513, -963888510,+ -3765607, -964937599, 3201430, 820367122, 3145678, 806080660,+ 2883726, 738955404, 3201494, 820383522, 1221177, 312926867,+ -557458, -142848732, 1005239, 257592709, -3764867, -964747974,+ -2129892, -545785280, -2682288, -687336873, -3542485, -907762539,+ 601683, 154181397, 0, 0,+};++#else /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(aarch64_zetas)++#endif /* !(MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mldsa/src/native/aarch64/src/arith_native_aarch64.h view
@@ -0,0 +1,367 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H+#define MLD_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H++#include "../../../cbmc.h"+#include "../../../common.h"++#define mld_aarch64_ntt_zetas_layer123456 \+ MLD_NAMESPACE(aarch64_ntt_zetas_layer123456)+#define mld_aarch64_ntt_zetas_layer78 MLD_NAMESPACE(aarch64_ntt_zetas_layer78)++#define mld_aarch64_intt_zetas_layer78 MLD_NAMESPACE(aarch64_intt_zetas_layer78)+#define mld_aarch64_intt_zetas_layer123456 \+ MLD_NAMESPACE(aarch64_intt_zetas_layer123456)++MLD_INTERNAL_DATA_DECLARATION const int32_t+ mld_aarch64_ntt_zetas_layer123456[144];+MLD_INTERNAL_DATA_DECLARATION const int32_t mld_aarch64_ntt_zetas_layer78[384];++MLD_INTERNAL_DATA_DECLARATION const int32_t mld_aarch64_intt_zetas_layer78[384];+MLD_INTERNAL_DATA_DECLARATION const int32_t+ mld_aarch64_intt_zetas_layer123456[160];++#define mld_rej_uniform_table MLD_NAMESPACE(rej_uniform_table)+MLD_INTERNAL_DATA_DECLARATION const uint8_t mld_rej_uniform_table[256];+#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_rej_uniform_eta_table MLD_NAMESPACE(rej_uniform_eta_table)+MLD_INTERNAL_DATA_DECLARATION const uint8_t mld_rej_uniform_eta_table[4096];+#endif++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+#define mld_polyz_unpack_17_indices MLD_NAMESPACE(polyz_unpack_17_indices)+MLD_INTERNAL_DATA_DECLARATION const uint8_t mld_polyz_unpack_17_indices[64];+#endif+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+#define mld_polyz_unpack_19_indices MLD_NAMESPACE(polyz_unpack_19_indices)+MLD_INTERNAL_DATA_DECLARATION const uint8_t mld_polyz_unpack_19_indices[64];+#endif+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */+++/*+ * Sampling 256 coefficients mod 15 using rejection sampling from 4 bits.+ * Expected number of required bytes: (256 * (16/15))/2 = 136.5 bytes.+ * We sample 1 block (=136 bytes) of SHAKE256_RATE output initially.+ * Sampling 2 blocks initially results in slightly worse performance.+ */+#define MLD_AARCH64_REJ_UNIFORM_ETA2_BUFLEN (1 * 136)+/*+ * Sampling 256 coefficients mod 9 using rejection sampling from 4 bits.+ * Expected number of required bytes: (256 * (16/9))/2 = 227.5 bytes.+ * We sample 2 blocks (=272 bytes) of SHAKE256_RATE output initially.+ */+#define MLD_AARCH64_REJ_UNIFORM_ETA4_BUFLEN (2 * 136)++#define mld_ntt_aarch64_asm MLD_NAMESPACE(ntt_aarch64_asm)+void mld_ntt_aarch64_asm(int32_t r[MLDSA_N], const int32_t zetas_l123456[144],+ const int32_t zetas_l78[384])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_ntt_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 8380417 == MLDSA_Q */+ requires(array_abs_bound(r, 0, MLDSA_N, 8380417))+ requires(zetas_l123456 == mld_aarch64_ntt_zetas_layer123456)+ requires(zetas_l78 == mld_aarch64_ntt_zetas_layer78)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ /* check-magic: off */+ ensures(array_abs_bound(r, 0, MLDSA_N, 75423753))+ /* check-magic: on */+);++#define mld_intt_aarch64_asm MLD_NAMESPACE(intt_aarch64_asm)+void mld_intt_aarch64_asm(int32_t r[MLDSA_N], const int32_t zetas_l78[384],+ const int32_t zetas_l123456[160])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_intt_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(r, 0, MLDSA_N, 8380417))+ requires(zetas_l78 == mld_aarch64_intt_zetas_layer78)+ requires(zetas_l123456 == mld_aarch64_intt_zetas_layer123456)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ /* check-magic: off */+ ensures(array_abs_bound(r, 0, MLDSA_N, 8380417))+ /* check-magic: on */+);++#define mld_rej_uniform_aarch64_asm MLD_NAMESPACE(rej_uniform_aarch64_asm)+MLD_MUST_CHECK_RETURN_VALUE+uint64_t mld_rej_uniform_aarch64_asm(int32_t r[MLDSA_N], const uint8_t *buf,+ unsigned buflen, const uint8_t table[256])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_rej_uniform_aarch64_asm.ml. */+__contract__(+ requires(buflen % 24 == 0)+ requires(memory_no_alias(buf, buflen))+ requires(table == mld_rej_uniform_table)+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value <= MLDSA_N)+ ensures(array_bound(r, 0, (unsigned) return_value, 0, MLDSA_Q))+);++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_rej_uniform_eta2_aarch64_asm \+ MLD_NAMESPACE(rej_uniform_eta2_aarch64_asm)+MLD_MUST_CHECK_RETURN_VALUE+uint64_t mld_rej_uniform_eta2_aarch64_asm(int32_t r[MLDSA_N],+ const uint8_t *buf, unsigned buflen,+ const uint8_t table[4096])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_rej_uniform_eta2_aarch64_asm.ml */+__contract__(+ requires(buflen % 8 == 0)+ requires(buflen >= 8)+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(buf, buflen))+ requires(table == mld_rej_uniform_eta_table)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value <= MLDSA_N)+ /* check-magic: 3 == 2 + 1 (asm is eta=2-specific) */+ ensures(array_abs_bound(r, 0, return_value, 3))+);++#define mld_rej_uniform_eta4_aarch64_asm \+ MLD_NAMESPACE(rej_uniform_eta4_aarch64_asm)+MLD_MUST_CHECK_RETURN_VALUE+uint64_t mld_rej_uniform_eta4_aarch64_asm(int32_t r[MLDSA_N],+ const uint8_t *buf, unsigned buflen,+ const uint8_t table[4096])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_rej_uniform_eta4_aarch64_asm.ml */+__contract__(+ requires(buflen % 8 == 0)+ requires(buflen >= 8)+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(buf, buflen))+ requires(table == mld_rej_uniform_eta_table)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value <= MLDSA_N)+ /* check-magic: 5 == 4 + 1 (asm is eta=4-specific) */+ ensures(array_abs_bound(r, 0, return_value, 5))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_poly_decompose_32_aarch64_asm \+ MLD_NAMESPACE(poly_decompose_32_aarch64_asm)+void mld_poly_decompose_32_aarch64_asm(int32_t a1[MLDSA_N], int32_t a0[MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_poly_decompose_32_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(a1, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a0, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(int32_t) * MLDSA_N))+ assigns(memory_slice(a0, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 16 == (MLDSA_Q - 1) / (2 * ((MLDSA_Q - 1) / 32)) */+ ensures(array_bound(a1, 0, MLDSA_N, 0, 16))+ /* check-magic: 261889 == (MLDSA_Q - 1) / 32 + 1 */+ ensures(array_abs_bound(a0, 0, MLDSA_N, 261889))+);++#define mld_poly_decompose_88_aarch64_asm \+ MLD_NAMESPACE(poly_decompose_88_aarch64_asm)+void mld_poly_decompose_88_aarch64_asm(int32_t a1[MLDSA_N], int32_t a0[MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_poly_decompose_88_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(a1, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a0, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(int32_t) * MLDSA_N))+ assigns(memory_slice(a0, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 44 == (MLDSA_Q - 1) / (2 * ((MLDSA_Q - 1) / 88)) */+ ensures(array_bound(a1, 0, MLDSA_N, 0, 44))+ /* check-magic: 95233 == (MLDSA_Q - 1) / 88 + 1 */+ ensures(array_abs_bound(a0, 0, MLDSA_N, 95233))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#define mld_poly_caddq_aarch64_asm MLD_NAMESPACE(poly_caddq_aarch64_asm)+void mld_poly_caddq_aarch64_asm(int32_t a[MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_poly_caddq_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(a, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+);++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_poly_use_hint_32_aarch64_asm \+ MLD_NAMESPACE(poly_use_hint_32_aarch64_asm)+void mld_poly_use_hint_32_aarch64_asm(int32_t a[MLDSA_N],+ const int32_t h[MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_poly_use_hint_32_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(h, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(a, 0, MLDSA_N, 0, 16))+);++#define mld_poly_use_hint_88_aarch64_asm \+ MLD_NAMESPACE(poly_use_hint_88_aarch64_asm)+void mld_poly_use_hint_88_aarch64_asm(int32_t a[MLDSA_N],+ const int32_t h[MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_poly_use_hint_88_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(h, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(a, 0, MLDSA_N, 0, 44))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#define mld_poly_chknorm_aarch64_asm MLD_NAMESPACE(poly_chknorm_aarch64_asm)+MLD_MUST_CHECK_RETURN_VALUE+int mld_poly_chknorm_aarch64_asm(const int32_t a[MLDSA_N], int32_t B)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_poly_chknorm_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ /* HOL Light precondition: abs(ival(x i)) < 2^31, i.e., a[i] != INT32_MIN */+ requires(forall(k0, 0, MLDSA_N, a[k0] > INT32_MIN))+ ensures(return_value == 0 || return_value == 1)+ ensures((return_value == 0) == array_abs_bound(a, 0, MLDSA_N, B))+);++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+#define mld_polyz_unpack_17_aarch64_asm \+ MLD_NAMESPACE(polyz_unpack_17_aarch64_asm)+void mld_polyz_unpack_17_aarch64_asm(int32_t r[MLDSA_N], const uint8_t buf[576],+ const uint8_t indices[64])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_polyz_unpack_17_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(buf, 576))+ requires(indices == mld_polyz_unpack_17_indices)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(r, 0, MLDSA_N, -((1 << 17) - 1), (1 << 17) + 1))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+#define mld_polyz_unpack_19_aarch64_asm \+ MLD_NAMESPACE(polyz_unpack_19_aarch64_asm)+void mld_polyz_unpack_19_aarch64_asm(int32_t r[MLDSA_N], const uint8_t buf[640],+ const uint8_t indices[64])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_polyz_unpack_19_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(buf, 640))+ requires(indices == mld_polyz_unpack_19_indices)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(r, 0, MLDSA_N, -((1 << 19) - 1), (1 << 19) + 1))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+#define mld_poly_pointwise_montgomery_aarch64_asm \+ MLD_NAMESPACE(poly_pointwise_montgomery_aarch64_asm)+void mld_poly_pointwise_montgomery_aarch64_asm(int32_t a[MLDSA_N],+ const int32_t b[MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_pointwise_montgomery_aarch64_asm.ml+ */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * MLDSA_N))+ /* Input bound MLD_NTT_BOUND = 9 * MLD_FQMUL_BOUND, the guaranteed bound of+ * any forward NTT implementation. Hardcoded here to keep this header free+ * of poly.h. */+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(array_abs_bound(a, 0, MLDSA_N, 94279698))+ requires(array_abs_bound(b, 0, MLDSA_N, 94279698))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(a, 0, MLDSA_N, 8380417))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#define mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm \+ MLD_NAMESPACE(polyvecl_pointwise_acc_montgomery_l4_aarch64_asm)+void mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm(+ int32_t r[MLDSA_N], const int32_t a[4][MLDSA_N],+ const int32_t b[4][MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in+ * proofs/hol_light/aarch64/proofs/mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.ml+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, sizeof(int32_t) * 4 * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * 4 * MLDSA_N))+ requires(forall(l0, 0, 4, array_abs_bound(a[l0], 0, MLDSA_N, 8380417)))+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(forall(l1, 0, 4, array_abs_bound(b[l1], 0, MLDSA_N, 94279698)))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(r, 0, MLDSA_N, 8380417))+);++#define mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm \+ MLD_NAMESPACE(polyvecl_pointwise_acc_montgomery_l5_aarch64_asm)+void mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm(+ int32_t r[MLDSA_N], const int32_t a[5][MLDSA_N],+ const int32_t b[5][MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in+ * proofs/hol_light/aarch64/proofs/mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.ml+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, sizeof(int32_t) * 5 * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * 5 * MLDSA_N))+ requires(forall(l0, 0, 5, array_abs_bound(a[l0], 0, MLDSA_N, 8380417)))+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(forall(l1, 0, 5, array_abs_bound(b[l1], 0, MLDSA_N, 94279698)))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(r, 0, MLDSA_N, 8380417))+);++#define mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm \+ MLD_NAMESPACE(polyvecl_pointwise_acc_montgomery_l7_aarch64_asm)+void mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm(+ int32_t r[MLDSA_N], const int32_t a[7][MLDSA_N],+ const int32_t b[7][MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in+ * proofs/hol_light/aarch64/proofs/mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.ml+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, sizeof(int32_t) * 7 * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * 7 * MLDSA_N))+ requires(forall(l0, 0, 7, array_abs_bound(a[l0], 0, MLDSA_N, 8380417)))+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(forall(l1, 0, 7, array_abs_bound(b[l1], 0, MLDSA_N, 94279698)))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(r, 0, MLDSA_N, 8380417))+);++#endif /* !MLD_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H */
+ cbits/mldsa/src/native/aarch64/src/mldsa_intt_aarch64_asm.S view
@@ -0,0 +1,786 @@+/* Copyright (c) 2022 Arm Limited+ * Copyright (c) 2022 Hanno Becker+ * Copyright (c) 2023 Amin Abdulrahman, Matthias Kannwischer+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ *+ * - [NeonNTT_Autoformalised]+ * Neon NTT - (Auto)formalised+ * Hanno Becker+ * https://eprint.iacr.org/2026/1223+ *+ * - [SLOTHY_Paper]+ * Fast and Clean: Auditable high-performance assembly via constraint solving+ * Abdulrahman, Becker, Kannwischer, Klein+ * https://eprint.iacr.org/2022/1303+ */++/* AArch64 ML-DSA inverse NTT following @[NeonNTT], @[SLOTHY_Paper], and @[NeonNTT_Autoformalised] */++/*yaml+ Name: intt_aarch64_asm+ Description: AArch64 ML-DSA inverse NTT+ Signature: void mld_intt_aarch64_asm(int32_t r[256], const int32_t zetas_l78[384], const int32_t zetas_l123456[160])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t r[256]+ description: Input/output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 1536+ permissions: read-only+ c_parameter: const int32_t zetas_l78[384]+ description: Twiddle factors for layers 7-8 (384 x int32_t)+ x2:+ type: buffer+ size_bytes: 640+ permissions: read-only+ c_parameter: const int32_t zetas_l123456[160]+ description: Twiddle factors for layers 1-6 (160 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_intt_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(intt_aarch64_asm)+MLD_ASM_FN_SYMBOL(intt_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ mov w5, #0xe001 // =57345+ movk w5, #0x7f, lsl #16+ dup v31.4s, w5+ mov x3, x0+ mov x4, #0x10 // =16+ ldr q27, [x3, #0x10]+ ldr q18, [x3]+ ldr q3, [x3, #0x20]+ ldr q13, [x3, #0x30]+ ldr q2, [x1, #0x30]+ ldr q8, [x3, #0x70]+ ldr q21, [x3, #0x60]+ trn1 v10.4s, v18.4s, v27.4s+ trn2 v23.4s, v18.4s, v27.4s+ trn1 v6.4s, v3.4s, v13.4s+ trn2 v18.4s, v3.4s, v13.4s+ ldr q12, [x1, #0x50]+ ldr q17, [x1, #0x40]+ trn2 v29.2d, v10.2d, v6.2d+ trn2 v14.2d, v23.2d, v18.2d+ trn1 v10.2d, v10.2d, v6.2d+ trn1 v26.2d, v23.2d, v18.2d+ sub v3.4s, v29.4s, v14.4s+ trn1 v13.4s, v21.4s, v8.4s+ ldr q6, [x1, #0x10]+ add v24.4s, v10.4s, v26.4s+ sub v30.4s, v10.4s, v26.4s+ sqrdmulh v12.4s, v3.4s, v12.4s+ ldr q9, [x1, #0x20]+ sqrdmulh v5.4s, v30.4s, v2.4s+ mul v4.4s, v3.4s, v17.4s+ ldr q3, [x3, #0x50]+ add v10.4s, v29.4s, v14.4s+ ldr q26, [x3, #0x40]+ mul v15.4s, v30.4s, v9.4s+ trn2 v30.4s, v21.4s, v8.4s+ sub v21.4s, v24.4s, v10.4s+ add v29.4s, v24.4s, v10.4s+ mls v15.4s, v5.4s, v31.s[0]+ ldr q16, [x1], #0x60+ trn2 v10.4s, v26.4s, v3.4s+ mls v4.4s, v12.4s, v31.s[0]+ trn1 v3.4s, v26.4s, v3.4s+ ldr q0, [x1, #0x50]+ trn2 v25.2d, v10.2d, v30.2d+ sqrdmulh v12.4s, v21.4s, v6.4s+ trn2 v1.2d, v3.2d, v13.2d+ mul v21.4s, v21.4s, v16.4s+ sub v23.4s, v1.4s, v25.4s+ sub v2.4s, v15.4s, v4.4s+ add v20.4s, v15.4s, v4.4s+ sqrdmulh v4.4s, v23.4s, v0.4s+ trn1 v7.2d, v3.2d, v13.2d+ sqrdmulh v3.4s, v2.4s, v6.4s+ trn1 v15.2d, v10.2d, v30.2d+ mls v21.4s, v12.4s, v31.s[0]+ ldr q12, [x1, #0x30]+ sub v13.4s, v7.4s, v15.4s+ mul v10.4s, v2.4s, v16.4s+ ldr q16, [x1, #0x40]+ ldr q5, [x1, #0x20]+ mls v10.4s, v3.4s, v31.s[0]+ trn2 v11.4s, v29.4s, v20.4s+ sqrdmulh v2.4s, v13.4s, v12.4s+ ldr d9, [x2], #0x20+ mul v17.4s, v13.4s, v5.4s+ trn1 v24.4s, v29.4s, v20.4s+ trn1 v12.4s, v21.4s, v10.4s+ trn2 v10.4s, v21.4s, v10.4s+ mul v22.4s, v23.4s, v16.4s+ ldur q26, [x2, #-0x10]+ trn1 v13.2d, v24.2d, v12.2d+ trn1 v3.2d, v11.2d, v10.2d+ mls v17.4s, v2.4s, v31.s[0]+ mls v22.4s, v4.4s, v31.s[0]+ sub v23.4s, v13.4s, v3.4s+ trn2 v10.2d, v11.2d, v10.2d+ sqrdmulh v0.4s, v23.4s, v26.s[1]+ trn2 v21.2d, v24.2d, v12.2d+ mul v29.4s, v23.4s, v26.s[0]+ sub v23.4s, v21.4s, v10.4s+ sqrdmulh v8.4s, v23.4s, v26.s[3]+ add v6.4s, v1.4s, v25.4s+ add v11.4s, v21.4s, v10.4s+ mls v29.4s, v0.4s, v31.s[0]+ add v30.4s, v13.4s, v3.4s+ ldr q2, [x1, #0x10]+ mul v3.4s, v23.4s, v26.s[2]+ add v12.4s, v7.4s, v15.4s+ mls v3.4s, v8.4s, v31.s[0]+ sub v13.4s, v30.4s, v11.4s+ sub v21.4s, v12.4s, v6.4s+ add v26.4s, v17.4s, v22.4s+ sqrdmulh v28.4s, v13.4s, v9.s[1]+ mul v18.4s, v13.4s, v9.s[0]+ sub v8.4s, v29.4s, v3.4s+ sqrdmulh v24.4s, v21.4s, v2.4s+ add v15.4s, v30.4s, v11.4s+ add v10.4s, v29.4s, v3.4s+ add v14.4s, v12.4s, v6.4s+ sqrdmulh v16.4s, v8.4s, v9.s[1]+ sub v3.4s, v17.4s, v22.4s+ mul v8.4s, v8.4s, v9.s[0]+ ldr q17, [x1], #0x60+ sub x4, x4, #0x2++Lmld_intt_layer5678_start:+ ldr d4, [x2], #0x20+ mul v0.4s, v21.4s, v17.4s+ ldr q7, [x3, #0xa0]+ ldr q27, [x3, #0xb0]+ mls v8.4s, v16.4s, v31.s[0]+ ldr q29, [x3, #0x80]+ trn1 v9.4s, v14.4s, v26.4s+ ldr q25, [x3, #0x90]+ mls v0.4s, v24.4s, v31.s[0]+ str q15, [x3], #0x40+ trn1 v1.4s, v7.4s, v27.4s+ ldr q15, [x1, #0x50]+ trn2 v12.4s, v7.4s, v27.4s+ sqrdmulh v16.4s, v3.4s, v2.4s+ trn1 v21.4s, v29.4s, v25.4s+ stur q8, [x3, #-0x10]+ trn2 v19.4s, v29.4s, v25.4s+ mls v18.4s, v28.4s, v31.s[0]+ ldr q22, [x1, #0x30]+ trn2 v13.2d, v21.2d, v1.2d+ trn2 v5.2d, v19.2d, v12.2d+ mul v23.4s, v3.4s, v17.4s+ trn1 v17.2d, v21.2d, v1.2d+ ldr q29, [x1, #0x40]+ mls v23.4s, v16.4s, v31.s[0]+ sub v30.4s, v13.4s, v5.4s+ trn1 v24.2d, v19.2d, v12.2d+ stur q10, [x3, #-0x30]+ sqrdmulh v16.4s, v30.4s, v15.4s+ ldr q10, [x1, #0x20]+ trn2 v12.4s, v14.4s, v26.4s+ stur q18, [x3, #-0x20]+ mul v11.4s, v30.4s, v29.4s+ sub v8.4s, v17.4s, v24.4s+ trn1 v20.4s, v0.4s, v23.4s+ ldr q2, [x1, #0x10]+ trn2 v26.4s, v0.4s, v23.4s+ sqrdmulh v19.4s, v8.4s, v22.4s+ ldur q14, [x2, #-0x10]+ trn1 v21.2d, v9.2d, v20.2d+ mul v8.4s, v8.4s, v10.4s+ trn1 v6.2d, v12.2d, v26.2d+ trn2 v0.2d, v9.2d, v20.2d+ mls v11.4s, v16.4s, v31.s[0]+ sub v7.4s, v21.4s, v6.4s+ trn2 v28.2d, v12.2d, v26.2d+ add v10.4s, v21.4s, v6.4s+ mul v18.4s, v7.4s, v14.s[0]+ add v20.4s, v13.4s, v5.4s+ sqrdmulh v30.4s, v7.4s, v14.s[1]+ sub v21.4s, v0.4s, v28.4s+ add v13.4s, v0.4s, v28.4s+ add v9.4s, v17.4s, v24.4s+ sqrdmulh v1.4s, v21.4s, v14.s[3]+ ldr q17, [x1], #0x60+ mul v14.4s, v21.4s, v14.s[2]+ sub v21.4s, v9.4s, v20.4s+ mls v18.4s, v30.4s, v31.s[0]+ mls v14.4s, v1.4s, v31.s[0]+ mls v8.4s, v19.4s, v31.s[0]+ sub v3.4s, v10.4s, v13.4s+ add v15.4s, v10.4s, v13.4s+ sqrdmulh v28.4s, v3.4s, v4.s[1]+ add v10.4s, v18.4s, v14.4s+ sub v19.4s, v18.4s, v14.4s+ mul v18.4s, v3.4s, v4.s[0]+ sub v3.4s, v8.4s, v11.4s+ add v26.4s, v8.4s, v11.4s+ sqrdmulh v16.4s, v19.4s, v4.s[1]+ add v14.4s, v9.4s, v20.4s+ mul v8.4s, v19.4s, v4.s[0]+ sqrdmulh v24.4s, v21.4s, v2.4s+ subs x4, x4, #0x1+ cbnz x4, Lmld_intt_layer5678_start+ sqrdmulh v19.4s, v3.4s, v2.4s+ trn2 v0.4s, v14.4s, v26.4s+ str q15, [x3], #0x40+ trn1 v30.4s, v14.4s, v26.4s+ mul v26.4s, v21.4s, v17.4s+ stur q10, [x3, #-0x30]+ ldr d29, [x2], #0x20+ mls v26.4s, v24.4s, v31.s[0]+ mul v10.4s, v3.4s, v17.4s+ ldur q14, [x2, #-0x10]+ mls v10.4s, v19.4s, v31.s[0]+ mls v8.4s, v16.4s, v31.s[0]+ trn2 v22.4s, v26.4s, v10.4s+ trn1 v25.4s, v26.4s, v10.4s+ trn1 v1.2d, v30.2d, v25.2d+ trn1 v2.2d, v0.2d, v22.2d+ trn2 v13.2d, v30.2d, v25.2d+ trn2 v7.2d, v0.2d, v22.2d+ mls v18.4s, v28.4s, v31.s[0]+ sub v22.4s, v1.4s, v2.4s+ add v3.4s, v13.4s, v7.4s+ sqrdmulh v27.4s, v22.4s, v14.s[1]+ sub v4.4s, v13.4s, v7.4s+ stur q8, [x3, #-0x10]+ sqrdmulh v0.4s, v4.4s, v14.s[3]+ add v16.4s, v1.4s, v2.4s+ stur q18, [x3, #-0x20]+ mul v23.4s, v4.4s, v14.s[2]+ add v4.4s, v16.4s, v3.4s+ sub v17.4s, v16.4s, v3.4s+ mul v14.4s, v22.4s, v14.s[0]+ str q4, [x3], #0x40+ mls v23.4s, v0.4s, v31.s[0]+ mls v14.4s, v27.4s, v31.s[0]+ mul v11.4s, v17.4s, v29.s[0]+ sqrdmulh v10.4s, v17.4s, v29.s[1]+ sub v0.4s, v14.4s, v23.4s+ add v6.4s, v14.4s, v23.4s+ sqrdmulh v12.4s, v0.4s, v29.s[1]+ mul v0.4s, v0.4s, v29.s[0]+ stur q6, [x3, #-0x30]+ mls v11.4s, v10.4s, v31.s[0]+ mls v0.4s, v12.4s, v31.s[0]+ stur q11, [x3, #-0x20]+ stur q0, [x3, #-0x10]+ mov w5, #0x3ffe // =16382+ dup v29.4s, w5+ mov w5, #0xe03 // =3587+ movk w5, #0x40, lsl #16+ dup v30.4s, w5+ mov x4, #0x4 // =4+ ldr q0, [x2], #0x80+ ldur q1, [x2, #-0x70]+ ldur q2, [x2, #-0x60]+ ldur q3, [x2, #-0x50]+ ldur q4, [x2, #-0x40]+ ldur q5, [x2, #-0x30]+ ldur q6, [x2, #-0x20]+ ldur q7, [x2, #-0x10]+ ldr q8, [x0, #0xc0]+ ldr q27, [x0, #0x80]+ ldr q20, [x0, #0x1c0]+ ldr q23, [x0, #0x180]+ ldr q24, [x0, #0x3c0]+ ldr q28, [x0, #0x40]+ ldr q25, [x0, #0x340]+ ldr q10, [x0, #0x380]+ sub v15.4s, v27.4s, v8.4s+ ldr q26, [x0]+ ldr q18, [x0, #0x300]+ add v9.4s, v23.4s, v20.4s+ mul v11.4s, v15.4s, v4.s[0]+ sub v19.4s, v23.4s, v20.4s+ sub v22.4s, v10.4s, v24.4s+ ldr q13, [x0, #0x240]+ sqrdmulh v12.4s, v15.4s, v4.s[1]+ sub v14.4s, v26.4s, v28.4s+ ldr q17, [x0, #0x200]+ add v23.4s, v18.4s, v25.4s+ sqrdmulh v15.4s, v14.4s, v3.s[3]+ sub v18.4s, v18.4s, v25.4s+ mul v25.4s, v14.4s, v3.s[2]+ sub v16.4s, v17.4s, v13.4s+ mls v11.4s, v12.4s, v31.s[0]+ add v21.4s, v10.4s, v24.4s+ mls v25.4s, v15.4s, v31.s[0]+ mul v12.4s, v16.4s, v5.s[2]+ sub v10.4s, v23.4s, v21.4s+ sqrdmulh v14.4s, v10.4s, v3.s[1]+ add v23.4s, v23.4s, v21.4s+ add v27.4s, v27.4s, v8.4s+ mul v15.4s, v10.4s, v3.s[0]+ add v24.4s, v25.4s, v11.4s+ sub v10.4s, v25.4s, v11.4s+ ldr q25, [x0, #0x100]+ sqrdmulh v8.4s, v18.4s, v6.s[3]+ add v21.4s, v26.4s, v28.4s+ ldr q26, [x0, #0x140]+ sqrdmulh v20.4s, v16.4s, v5.s[3]+ sqrdmulh v28.4s, v22.4s, v7.s[1]+ add v11.4s, v25.4s, v26.4s+ mls v15.4s, v14.4s, v31.s[0]+ add v17.4s, v17.4s, v13.4s+ mul v13.4s, v22.4s, v7.s[0]+ ldr q14, [x0, #0x280]+ ldr q22, [x0, #0x2c0]+ mls v13.4s, v28.4s, v31.s[0]+ sub v16.4s, v25.4s, v26.4s+ sub v28.4s, v21.4s, v27.4s+ mls v12.4s, v20.4s, v31.s[0]+ add v20.4s, v11.4s, v9.4s+ add v26.4s, v14.4s, v22.4s+ sub v14.4s, v14.4s, v22.4s+ sqrdmulh v22.4s, v28.4s, v1.s[3]+ sub v9.4s, v11.4s, v9.4s+ mul v25.4s, v14.4s, v6.s[0]+ sub v11.4s, v17.4s, v26.4s+ add v17.4s, v17.4s, v26.4s+ add v26.4s, v21.4s, v27.4s+ sqrdmulh v27.4s, v11.4s, v2.s[3]+ mul v21.4s, v11.4s, v2.s[2]+ mul v11.4s, v18.4s, v6.s[2]+ mls v11.4s, v8.4s, v31.s[0]+ sqrdmulh v8.4s, v14.4s, v6.s[1]+ add v18.4s, v26.4s, v20.4s+ mls v21.4s, v27.4s, v31.s[0]+ sub v27.4s, v17.4s, v23.4s+ mul v14.4s, v28.4s, v1.s[2]+ sub v28.4s, v26.4s, v20.4s+ add v23.4s, v17.4s, v23.4s+ mls v25.4s, v8.4s, v31.s[0]+ add v26.4s, v21.4s, v15.4s+ sub v15.4s, v21.4s, v15.4s+ mul v8.4s, v19.4s, v5.s[0]+ mls v14.4s, v22.4s, v31.s[0]+ sub v22.4s, v11.4s, v13.4s+ sub v20.4s, v12.4s, v25.4s+ add v21.4s, v12.4s, v25.4s+ sqrdmulh v12.4s, v22.4s, v3.s[1]+ mul v17.4s, v16.4s, v4.s[2]+ sqrdmulh v25.4s, v16.4s, v4.s[3]+ add v16.4s, v11.4s, v13.4s+ mul v11.4s, v22.4s, v3.s[0]+ sqrdmulh v19.4s, v19.4s, v5.s[1]+ sqrdmulh v22.4s, v15.4s, v1.s[1]+ sqrdmulh v13.4s, v20.4s, v2.s[3]+ mul v20.4s, v20.4s, v2.s[2]+ mls v11.4s, v12.4s, v31.s[0]+ mls v20.4s, v13.4s, v31.s[0]+ sub v13.4s, v18.4s, v23.4s+ add v23.4s, v18.4s, v23.4s+ mls v17.4s, v25.4s, v31.s[0]+ mls v8.4s, v19.4s, v31.s[0]+ sub v18.4s, v20.4s, v11.4s+ add v20.4s, v20.4s, v11.4s+ sqrdmulh v12.4s, v9.4s, v2.s[1]+ mul v15.4s, v15.4s, v1.s[0]+ sub v11.4s, v17.4s, v8.4s+ mls v15.4s, v22.4s, v31.s[0]+ add v19.4s, v17.4s, v8.4s+ sqrdmulh v25.4s, v11.4s, v2.s[1]+ add v17.4s, v24.4s, v19.4s+ sub v24.4s, v24.4s, v19.4s+ mul v19.4s, v11.4s, v2.s[0]+ sqrdmulh v8.4s, v24.4s, v0.s[3]+ mls v19.4s, v25.4s, v31.s[0]+ mul v25.4s, v9.4s, v2.s[0]+ sub x4, x4, #0x1++Lmld_intt_layer1234_start:+ sub v22.4s, v21.4s, v16.4s+ mls v25.4s, v12.4s, v31.s[0]+ add v12.4s, v21.4s, v16.4s+ mul v9.4s, v24.4s, v0.s[2]+ mls v9.4s, v8.4s, v31.s[0]+ sub v16.4s, v14.4s, v25.4s+ add v14.4s, v14.4s, v25.4s+ sqrdmulh v24.4s, v22.4s, v1.s[1]+ sub v11.4s, v14.4s, v26.4s+ sqrdmulh v21.4s, v16.4s, v0.s[3]+ add v25.4s, v14.4s, v26.4s+ mul v14.4s, v22.4s, v1.s[0]+ mls v14.4s, v24.4s, v31.s[0]+ mul v16.4s, v16.4s, v0.s[2]+ mls v16.4s, v21.4s, v31.s[0]+ add v24.4s, v9.4s, v14.4s+ sqrdmulh v21.4s, v25.4s, v30.4s+ sqrdmulh v26.4s, v11.4s, v0.s[1]+ mul v8.4s, v25.4s, v29.4s+ mls v8.4s, v21.4s, v31.s[0]+ mul v25.4s, v24.4s, v29.4s+ mul v22.4s, v11.4s, v0.s[0]+ sub v11.4s, v17.4s, v12.4s+ str q8, [x0, #0x80]+ sub v8.4s, v9.4s, v14.4s+ add v14.4s, v16.4s, v15.4s+ sqrdmulh v9.4s, v11.4s, v0.s[1]+ sub v21.4s, v16.4s, v15.4s+ mls v22.4s, v26.4s, v31.s[0]+ mul v15.4s, v11.4s, v0.s[0]+ mls v15.4s, v9.4s, v31.s[0]+ add v11.4s, v17.4s, v12.4s+ mul v16.4s, v10.4s, v1.s[2]+ str q22, [x0, #0x280]+ sqrdmulh v26.4s, v10.4s, v1.s[3]+ str q15, [x0, #0x240]+ sqrdmulh v9.4s, v8.4s, v0.s[1]+ mul v12.4s, v8.4s, v0.s[0]+ mls v16.4s, v26.4s, v31.s[0]+ sqrdmulh v10.4s, v27.4s, v1.s[1]+ sqrdmulh v15.4s, v14.4s, v30.4s+ add v22.4s, v16.4s, v19.4s+ sub v19.4s, v16.4s, v19.4s+ mls v12.4s, v9.4s, v31.s[0]+ add v8.4s, v22.4s, v20.4s+ sub v26.4s, v22.4s, v20.4s+ mul v27.4s, v27.4s, v1.s[0]+ sqrdmulh v17.4s, v11.4s, v30.4s+ mls v27.4s, v10.4s, v31.s[0]+ mul v20.4s, v11.4s, v29.4s+ sqrdmulh v22.4s, v24.4s, v30.4s+ mls v20.4s, v17.4s, v31.s[0]+ mul v14.4s, v14.4s, v29.4s+ sqrdmulh v16.4s, v23.4s, v30.4s+ str q20, [x0, #0x40]+ sqrdmulh v11.4s, v21.4s, v0.s[1]+ sqrdmulh v20.4s, v28.4s, v0.s[3]+ mul v17.4s, v23.4s, v29.4s+ mul v23.4s, v28.4s, v0.s[2]+ mls v17.4s, v16.4s, v31.s[0]+ mul v10.4s, v21.4s, v0.s[0]+ str q12, [x0, #0x340]+ mls v25.4s, v22.4s, v31.s[0]+ str q17, [x0], #0x10+ ldr q17, [x0, #0x2c0]+ mls v14.4s, v15.4s, v31.s[0]+ ldr q12, [x0, #0x200]+ sqrdmulh v16.4s, v13.4s, v0.s[1]+ str q25, [x0, #0x130]+ mul v28.4s, v13.4s, v0.s[0]+ ldr q13, [x0, #0x240]+ str q14, [x0, #0x170]+ mul v14.4s, v18.4s, v1.s[0]+ mls v28.4s, v16.4s, v31.s[0]+ ldr q21, [x0, #0xc0]+ mls v10.4s, v11.4s, v31.s[0]+ mls v23.4s, v20.4s, v31.s[0]+ str q28, [x0, #0x1f0]+ ldr q11, [x0, #0x300]+ sqrdmulh v28.4s, v18.4s, v1.s[1]+ str q10, [x0, #0x370]+ mul v16.4s, v19.4s, v0.s[2]+ sqrdmulh v19.4s, v19.4s, v0.s[3]+ sqrdmulh v10.4s, v26.4s, v0.s[1]+ add v18.4s, v12.4s, v13.4s+ sub v12.4s, v12.4s, v13.4s+ mls v16.4s, v19.4s, v31.s[0]+ ldr q9, [x0, #0x280]+ ldr q20, [x0, #0x380]+ add v24.4s, v23.4s, v27.4s+ sub v25.4s, v23.4s, v27.4s+ mul v19.4s, v26.4s, v0.s[0]+ mls v14.4s, v28.4s, v31.s[0]+ add v13.4s, v9.4s, v17.4s+ sub v28.4s, v18.4s, v13.4s+ mls v19.4s, v10.4s, v31.s[0]+ sqrdmulh v23.4s, v25.4s, v0.s[1]+ mul v15.4s, v25.4s, v0.s[0]+ str q19, [x0, #0x2b0]+ ldr q19, [x0, #0x80]+ sqrdmulh v22.4s, v8.4s, v30.4s+ mls v15.4s, v23.4s, v31.s[0]+ add v26.4s, v16.4s, v14.4s+ mul v10.4s, v8.4s, v29.4s+ add v8.4s, v18.4s, v13.4s+ sqrdmulh v13.4s, v26.4s, v30.4s+ str q15, [x0, #0x2f0]+ sub v18.4s, v9.4s, v17.4s+ mul v27.4s, v26.4s, v29.4s+ sqrdmulh v23.4s, v12.4s, v5.s[3]+ mls v27.4s, v13.4s, v31.s[0]+ mul v25.4s, v18.4s, v6.s[0]+ sqrdmulh v13.4s, v18.4s, v6.s[1]+ str q27, [x0, #0x1b0]+ sub v27.4s, v19.4s, v21.4s+ add v21.4s, v19.4s, v21.4s+ mul v18.4s, v12.4s, v5.s[2]+ sub v9.4s, v16.4s, v14.4s+ mul v26.4s, v24.4s, v29.4s+ ldr q12, [x0, #0x3c0]+ mls v25.4s, v13.4s, v31.s[0]+ sub v16.4s, v20.4s, v12.4s+ sqrdmulh v17.4s, v24.4s, v30.4s+ mul v13.4s, v16.4s, v7.s[0]+ mls v18.4s, v23.4s, v31.s[0]+ mls v26.4s, v17.4s, v31.s[0]+ sqrdmulh v17.4s, v9.4s, v0.s[1]+ sqrdmulh v24.4s, v16.4s, v7.s[1]+ ldr q16, [x0, #0x340]+ str q26, [x0, #0xf0]+ sub v15.4s, v18.4s, v25.4s+ mul v23.4s, v9.4s, v0.s[0]+ ldr q19, [x0, #0x180]+ ldr q14, [x0, #0x1c0]+ sqrdmulh v9.4s, v15.4s, v2.s[3]+ add v20.4s, v20.4s, v12.4s+ add v12.4s, v11.4s, v16.4s+ mul v26.4s, v15.4s, v2.s[2]+ sub v16.4s, v11.4s, v16.4s+ sub v11.4s, v12.4s, v20.4s+ mls v23.4s, v17.4s, v31.s[0]+ add v20.4s, v12.4s, v20.4s+ sub v12.4s, v19.4s, v14.4s+ mul v17.4s, v27.4s, v4.s[0]+ mls v10.4s, v22.4s, v31.s[0]+ add v22.4s, v19.4s, v14.4s+ ldr q14, [x0, #0x40]+ str q23, [x0, #0x3b0]+ sqrdmulh v23.4s, v27.4s, v4.s[1]+ ldr q19, [x0, #0x100]+ ldr q15, [x0, #0x140]+ mls v26.4s, v9.4s, v31.s[0]+ str q10, [x0, #0xb0]+ sqrdmulh v10.4s, v11.4s, v3.s[1]+ sub v27.4s, v8.4s, v20.4s+ mls v13.4s, v24.4s, v31.s[0]+ add v8.4s, v8.4s, v20.4s+ sub v20.4s, v19.4s, v15.4s+ sqrdmulh v9.4s, v16.4s, v6.s[3]+ add v19.4s, v19.4s, v15.4s+ ldr q15, [x0]+ mul v24.4s, v20.4s, v4.s[2]+ mls v17.4s, v23.4s, v31.s[0]+ add v23.4s, v15.4s, v14.4s+ sub v14.4s, v15.4s, v14.4s+ mul v15.4s, v11.4s, v3.s[0]+ sub v11.4s, v23.4s, v21.4s+ mul v16.4s, v16.4s, v6.s[2]+ add v23.4s, v23.4s, v21.4s+ add v21.4s, v18.4s, v25.4s+ sqrdmulh v25.4s, v20.4s, v4.s[3]+ sqrdmulh v18.4s, v12.4s, v5.s[1]+ sqrdmulh v20.4s, v14.4s, v3.s[3]+ mul v12.4s, v12.4s, v5.s[0]+ mls v16.4s, v9.4s, v31.s[0]+ mls v12.4s, v18.4s, v31.s[0]+ mls v24.4s, v25.4s, v31.s[0]+ sub v25.4s, v16.4s, v13.4s+ add v16.4s, v16.4s, v13.4s+ mul v13.4s, v28.4s, v2.s[2]+ sqrdmulh v9.4s, v25.4s, v3.s[1]+ mul v18.4s, v25.4s, v3.s[0]+ sqrdmulh v25.4s, v28.4s, v2.s[3]+ mls v18.4s, v9.4s, v31.s[0]+ mls v15.4s, v10.4s, v31.s[0]+ sub v9.4s, v24.4s, v12.4s+ mul v28.4s, v14.4s, v3.s[2]+ add v12.4s, v24.4s, v12.4s+ mls v28.4s, v20.4s, v31.s[0]+ add v20.4s, v26.4s, v18.4s+ add v24.4s, v19.4s, v22.4s+ mls v13.4s, v25.4s, v31.s[0]+ sub v18.4s, v26.4s, v18.4s+ sqrdmulh v26.4s, v11.4s, v1.s[3]+ add v25.4s, v23.4s, v24.4s+ mul v14.4s, v11.4s, v1.s[2]+ sub v10.4s, v28.4s, v17.4s+ add v17.4s, v28.4s, v17.4s+ sub v28.4s, v23.4s, v24.4s+ sqrdmulh v11.4s, v9.4s, v2.s[1]+ sub v22.4s, v19.4s, v22.4s+ sub v24.4s, v17.4s, v12.4s+ mls v14.4s, v26.4s, v31.s[0]+ add v17.4s, v17.4s, v12.4s+ sqrdmulh v12.4s, v22.4s, v2.s[1]+ sub v23.4s, v13.4s, v15.4s+ add v26.4s, v13.4s, v15.4s+ mul v19.4s, v9.4s, v2.s[0]+ sqrdmulh v9.4s, v23.4s, v1.s[1]+ mul v15.4s, v23.4s, v1.s[0]+ mls v19.4s, v11.4s, v31.s[0]+ sub v13.4s, v25.4s, v8.4s+ mls v15.4s, v9.4s, v31.s[0]+ add v23.4s, v25.4s, v8.4s+ sqrdmulh v8.4s, v24.4s, v0.s[3]+ mul v25.4s, v22.4s, v2.s[0]+ subs x4, x4, #0x1+ cbnz x4, Lmld_intt_layer1234_start+ mul v22.4s, v24.4s, v0.s[2]+ sqrdmulh v9.4s, v23.4s, v30.4s+ mls v25.4s, v12.4s, v31.s[0]+ mul v23.4s, v23.4s, v29.4s+ mls v23.4s, v9.4s, v31.s[0]+ mls v22.4s, v8.4s, v31.s[0]+ sqrdmulh v11.4s, v10.4s, v1.s[3]+ sub v9.4s, v14.4s, v25.4s+ str q23, [x0], #0x10+ mul v23.4s, v10.4s, v1.s[2]+ sqrdmulh v24.4s, v9.4s, v0.s[3]+ mls v23.4s, v11.4s, v31.s[0]+ mul v9.4s, v9.4s, v0.s[2]+ mls v9.4s, v24.4s, v31.s[0]+ sub v24.4s, v23.4s, v19.4s+ add v19.4s, v23.4s, v19.4s+ mul v8.4s, v28.4s, v0.s[2]+ add v23.4s, v14.4s, v25.4s+ mul v11.4s, v13.4s, v0.s[0]+ sub v25.4s, v21.4s, v16.4s+ sub v10.4s, v9.4s, v15.4s+ add v15.4s, v9.4s, v15.4s+ add v16.4s, v21.4s, v16.4s+ sqrdmulh v28.4s, v28.4s, v0.s[3]+ add v9.4s, v23.4s, v26.4s+ add v14.4s, v17.4s, v16.4s+ sqrdmulh v21.4s, v13.4s, v0.s[1]+ sub v13.4s, v17.4s, v16.4s+ mul v16.4s, v14.4s, v29.4s+ sub v12.4s, v23.4s, v26.4s+ mul v17.4s, v13.4s, v0.s[0]+ mls v8.4s, v28.4s, v31.s[0]+ sqrdmulh v28.4s, v14.4s, v30.4s+ sub v14.4s, v19.4s, v20.4s+ add v20.4s, v19.4s, v20.4s+ sqrdmulh v19.4s, v13.4s, v0.s[1]+ sqrdmulh v26.4s, v27.4s, v1.s[1]+ mls v16.4s, v28.4s, v31.s[0]+ sqrdmulh v28.4s, v12.4s, v0.s[1]+ mul v27.4s, v27.4s, v1.s[0]+ str q16, [x0, #0x30]+ mul v12.4s, v12.4s, v0.s[0]+ mls v12.4s, v28.4s, v31.s[0]+ mls v27.4s, v26.4s, v31.s[0]+ sqrdmulh v26.4s, v25.4s, v1.s[1]+ str q12, [x0, #0x270]+ sqrdmulh v12.4s, v14.4s, v0.s[1]+ sub v13.4s, v8.4s, v27.4s+ add v27.4s, v8.4s, v27.4s+ mul v16.4s, v14.4s, v0.s[0]+ mul v8.4s, v20.4s, v29.4s+ mls v16.4s, v12.4s, v31.s[0]+ mul v12.4s, v18.4s, v1.s[0]+ sqrdmulh v14.4s, v20.4s, v30.4s+ str q16, [x0, #0x2b0]+ sqrdmulh v20.4s, v18.4s, v1.s[1]+ mul v25.4s, v25.4s, v1.s[0]+ mls v25.4s, v26.4s, v31.s[0]+ mls v17.4s, v19.4s, v31.s[0]+ mul v18.4s, v9.4s, v29.4s+ add v28.4s, v22.4s, v25.4s+ sqrdmulh v9.4s, v9.4s, v30.4s+ sub v23.4s, v22.4s, v25.4s+ str q17, [x0, #0x230]+ mls v11.4s, v21.4s, v31.s[0]+ mls v8.4s, v14.4s, v31.s[0]+ sqrdmulh v21.4s, v28.4s, v30.4s+ str q11, [x0, #0x1f0]+ mul v17.4s, v28.4s, v29.4s+ str q8, [x0, #0xb0]+ mul v28.4s, v10.4s, v0.s[0]+ mls v17.4s, v21.4s, v31.s[0]+ sqrdmulh v21.4s, v10.4s, v0.s[1]+ mul v8.4s, v15.4s, v29.4s+ str q17, [x0, #0x130]+ sqrdmulh v15.4s, v15.4s, v30.4s+ mul v10.4s, v24.4s, v0.s[2]+ sqrdmulh v24.4s, v24.4s, v0.s[3]+ mls v12.4s, v20.4s, v31.s[0]+ mls v8.4s, v15.4s, v31.s[0]+ mls v10.4s, v24.4s, v31.s[0]+ sqrdmulh v24.4s, v27.4s, v30.4s+ str q8, [x0, #0x170]+ mul v14.4s, v13.4s, v0.s[0]+ sub v16.4s, v10.4s, v12.4s+ sqrdmulh v17.4s, v13.4s, v0.s[1]+ add v10.4s, v10.4s, v12.4s+ mul v13.4s, v27.4s, v29.4s+ sqrdmulh v20.4s, v10.4s, v30.4s+ mls v14.4s, v17.4s, v31.s[0]+ mls v13.4s, v24.4s, v31.s[0]+ sqrdmulh v24.4s, v23.4s, v0.s[1]+ str q14, [x0, #0x2f0]+ sqrdmulh v8.4s, v16.4s, v0.s[1]+ str q13, [x0, #0xf0]+ mls v18.4s, v9.4s, v31.s[0]+ mul v9.4s, v23.4s, v0.s[0]+ mul v14.4s, v16.4s, v0.s[0]+ str q18, [x0, #0x70]+ mul v27.4s, v10.4s, v29.4s+ mls v28.4s, v21.4s, v31.s[0]+ mls v14.4s, v8.4s, v31.s[0]+ mls v27.4s, v20.4s, v31.s[0]+ str q28, [x0, #0x370]+ mls v9.4s, v24.4s, v31.s[0]+ str q14, [x0, #0x3b0]+ str q27, [x0, #0x1b0]+ str q9, [x0, #0x330]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(intt_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_ntt_aarch64_asm.S view
@@ -0,0 +1,686 @@+/* Copyright (c) 2022 Arm Limited+ * Copyright (c) 2022 Hanno Becker+ * Copyright (c) 2023 Amin Abdulrahman, Matthias Kannwischer+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ *+ * - [NeonNTT_Autoformalised]+ * Neon NTT - (Auto)formalised+ * Hanno Becker+ * https://eprint.iacr.org/2026/1223+ *+ * - [SLOTHY_Paper]+ * Fast and Clean: Auditable high-performance assembly via constraint solving+ * Abdulrahman, Becker, Kannwischer, Klein+ * https://eprint.iacr.org/2022/1303+ */++/* AArch64 ML-DSA forward NTT following @[NeonNTT], @[SLOTHY_Paper], and @[NeonNTT_Autoformalised] */++/*yaml+ Name: ntt_aarch64_asm+ Description: AArch64 ML-DSA forward NTT+ Signature: void mld_ntt_aarch64_asm(int32_t r[256], const int32_t zetas_l123456[144], const int32_t zetas_l78[384])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t r[256]+ description: Input/output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 576+ permissions: read-only+ c_parameter: const int32_t zetas_l123456[144]+ description: Twiddle factors for layers 1-6 (144 x int32_t)+ x2:+ type: buffer+ size_bytes: 1536+ permissions: read-only+ c_parameter: const int32_t zetas_l78[384]+ description: Twiddle factors for layers 7-8 (384 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_ntt_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(ntt_aarch64_asm)+MLD_ASM_FN_SYMBOL(ntt_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ mov w5, #0xe001 // =57345+ movk w5, #0x7f, lsl #16+ dup v7.4s, w5+ mov x3, x0+ mov x4, #0x8 // =8+ ldr q0, [x1], #0x40+ ldur q1, [x1, #-0x30]+ ldur q2, [x1, #-0x20]+ ldur q3, [x1, #-0x10]+ ldr q23, [x0, #0x390]+ ldr q13, [x0, #0x380]+ ldr q22, [x0, #0x80]+ ldr q26, [x0, #0x190]+ ldr q8, [x0, #0x280]+ ldr q6, [x0, #0x210]+ mul v10.4s, v13.4s, v0.s[0]+ sqrdmulh v13.4s, v13.4s, v0.s[1]+ mul v12.4s, v8.4s, v0.s[0]+ sqrdmulh v27.4s, v8.4s, v0.s[1]+ mul v4.4s, v6.4s, v0.s[0]+ mls v10.4s, v13.4s, v7.s[0]+ ldr q13, [x0, #0x180]+ sqrdmulh v14.4s, v23.4s, v0.s[1]+ mls v12.4s, v27.4s, v7.s[0]+ add v31.4s, v13.4s, v10.4s+ sub v13.4s, v13.4s, v10.4s+ mul v10.4s, v23.4s, v0.s[0]+ sqrdmulh v8.4s, v13.4s, v1.s[1]+ sub v18.4s, v22.4s, v12.4s+ mls v10.4s, v14.4s, v7.s[0]+ mul v13.4s, v13.4s, v1.s[0]+ mls v13.4s, v8.4s, v7.s[0]+ sub v29.4s, v26.4s, v10.4s+ add v25.4s, v26.4s, v10.4s+ mul v10.4s, v31.4s, v0.s[2]+ mul v14.4s, v25.4s, v0.s[2]+ add v17.4s, v18.4s, v13.4s+ sub v15.4s, v18.4s, v13.4s+ sqrdmulh v13.4s, v31.4s, v0.s[3]+ sqrdmulh v20.4s, v15.4s, v3.s[1]+ sqrdmulh v5.4s, v17.4s, v2.s[3]+ mls v10.4s, v13.4s, v7.s[0]+ ldr q13, [x0, #0x300]+ mul v18.4s, v17.4s, v2.s[2]+ add v31.4s, v22.4s, v12.4s+ mul v23.4s, v15.4s, v3.s[0]+ ldr q17, [x0, #0x90]+ add v19.4s, v31.4s, v10.4s+ sub v16.4s, v31.4s, v10.4s+ mul v10.4s, v13.4s, v0.s[0]+ sqrdmulh v13.4s, v13.4s, v0.s[1]+ sqrdmulh v27.4s, v16.4s, v2.s[1]+ mul v11.4s, v16.4s, v2.s[0]+ mls v10.4s, v13.4s, v7.s[0]+ ldr q13, [x0, #0x290]+ ldr q22, [x0, #0x100]+ mls v11.4s, v27.4s, v7.s[0]+ sqrdmulh v15.4s, v13.4s, v0.s[1]+ sub v12.4s, v22.4s, v10.4s+ add v30.4s, v22.4s, v10.4s+ mul v10.4s, v13.4s, v0.s[0]+ ldr q28, [x0]+ sqrdmulh v13.4s, v25.4s, v0.s[3]+ sqrdmulh v27.4s, v30.4s, v0.s[3]+ mls v10.4s, v15.4s, v7.s[0]+ mls v14.4s, v13.4s, v7.s[0]+ ldr q13, [x0, #0x200]+ sqrdmulh v25.4s, v12.4s, v1.s[1]+ add v24.4s, v17.4s, v10.4s+ sub v21.4s, v17.4s, v10.4s+ sqrdmulh v8.4s, v13.4s, v0.s[1]+ sub v9.4s, v24.4s, v14.4s+ mul v26.4s, v12.4s, v1.s[0]+ mul v13.4s, v13.4s, v0.s[0]+ mls v13.4s, v8.4s, v7.s[0]+ mul v8.4s, v30.4s, v0.s[2]+ mls v8.4s, v27.4s, v7.s[0]+ add v16.4s, v28.4s, v13.4s+ sub v10.4s, v28.4s, v13.4s+ mls v26.4s, v25.4s, v7.s[0]+ sqrdmulh v12.4s, v19.4s, v1.s[3]+ sub v25.4s, v16.4s, v8.4s+ mls v23.4s, v20.4s, v7.s[0]+ sub v22.4s, v25.4s, v11.4s+ sqrdmulh v20.4s, v9.4s, v2.s[1]+ sub v15.4s, v10.4s, v26.4s+ sub x4, x4, #0x2++Lmld_ntt_layer123_start:+ add v31.4s, v10.4s, v26.4s+ mul v17.4s, v19.4s, v1.s[2]+ add v26.4s, v15.4s, v23.4s+ ldr q30, [x0, #0x2a0]+ sub v13.4s, v15.4s, v23.4s+ mul v23.4s, v29.4s, v1.s[0]+ add v25.4s, v25.4s, v11.4s+ str q22, [x0, #0x180]+ mul v11.4s, v9.4s, v2.s[0]+ str q13, [x0, #0x380]+ ldr q28, [x0, #0x10]+ add v10.4s, v16.4s, v8.4s+ mls v17.4s, v12.4s, v7.s[0]+ ldr q13, [x0, #0x3a0]+ str q26, [x0, #0x300]+ sqrdmulh v27.4s, v30.4s, v0.s[1]+ mls v18.4s, v5.4s, v7.s[0]+ ldr q9, [x0, #0x1a0]+ sub v16.4s, v10.4s, v17.4s+ add v15.4s, v10.4s, v17.4s+ sqrdmulh v10.4s, v6.4s, v0.s[1]+ str q16, [x0, #0x80]+ str q15, [x0], #0x10+ sqrdmulh v19.4s, v13.4s, v0.s[1]+ sub v15.4s, v31.4s, v18.4s+ mul v8.4s, v13.4s, v0.s[0]+ add v26.4s, v31.4s, v18.4s+ str q15, [x0, #0x270]+ sqrdmulh v13.4s, v29.4s, v1.s[1]+ str q26, [x0, #0x1f0]+ mls v8.4s, v19.4s, v7.s[0]+ mls v11.4s, v20.4s, v7.s[0]+ mls v23.4s, v13.4s, v7.s[0]+ add v22.4s, v9.4s, v8.4s+ ldr q6, [x0, #0x210]+ sub v29.4s, v9.4s, v8.4s+ mul v17.4s, v30.4s, v0.s[0]+ ldr q9, [x0, #0x300]+ sqrdmulh v13.4s, v22.4s, v0.s[3]+ add v18.4s, v21.4s, v23.4s+ mls v4.4s, v10.4s, v7.s[0]+ sub v31.4s, v21.4s, v23.4s+ sqrdmulh v16.4s, v31.4s, v3.s[1]+ add v19.4s, v24.4s, v14.4s+ mul v14.4s, v22.4s, v0.s[2]+ sub v10.4s, v28.4s, v4.4s+ mls v14.4s, v13.4s, v7.s[0]+ ldr q13, [x0, #0x100]+ sqrdmulh v22.4s, v9.4s, v0.s[1]+ mul v8.4s, v9.4s, v0.s[0]+ mul v23.4s, v31.4s, v3.s[0]+ mls v8.4s, v22.4s, v7.s[0]+ mls v23.4s, v16.4s, v7.s[0]+ add v16.4s, v28.4s, v4.4s+ ldr q22, [x0, #0x90]+ mul v4.4s, v6.4s, v0.s[0]+ mls v17.4s, v27.4s, v7.s[0]+ add v21.4s, v13.4s, v8.4s+ sub v27.4s, v13.4s, v8.4s+ sqrdmulh v31.4s, v21.4s, v0.s[3]+ str q25, [x0, #0xf0]+ mul v8.4s, v21.4s, v0.s[2]+ add v24.4s, v22.4s, v17.4s+ sub v21.4s, v22.4s, v17.4s+ sqrdmulh v5.4s, v18.4s, v2.s[3]+ mls v8.4s, v31.4s, v7.s[0]+ sub v9.4s, v24.4s, v14.4s+ sqrdmulh v20.4s, v27.4s, v1.s[1]+ mul v26.4s, v27.4s, v1.s[0]+ sub v25.4s, v16.4s, v8.4s+ mul v18.4s, v18.4s, v2.s[2]+ sub v22.4s, v25.4s, v11.4s+ mls v26.4s, v20.4s, v7.s[0]+ sqrdmulh v20.4s, v9.4s, v2.s[1]+ sqrdmulh v12.4s, v19.4s, v1.s[3]+ sub v15.4s, v10.4s, v26.4s+ subs x4, x4, #0x1+ cbnz x4, Lmld_ntt_layer123_start+ add v13.4s, v10.4s, v26.4s+ mls v18.4s, v5.4s, v7.s[0]+ str q22, [x0, #0x180]+ add v27.4s, v16.4s, v8.4s+ mul v22.4s, v19.4s, v1.s[2]+ add v26.4s, v24.4s, v14.4s+ ldr q31, [x0, #0x110]+ sub v14.4s, v15.4s, v23.4s+ add v17.4s, v15.4s, v23.4s+ mls v22.4s, v12.4s, v7.s[0]+ add v28.4s, v13.4s, v18.4s+ str q14, [x0, #0x380]+ sqrdmulh v24.4s, v6.4s, v0.s[1]+ add v5.4s, v25.4s, v11.4s+ sub v19.4s, v13.4s, v18.4s+ str q17, [x0, #0x300]+ str q5, [x0, #0x100]+ mul v16.4s, v9.4s, v2.s[0]+ ldr q18, [x0, #0x310]+ str q19, [x0, #0x280]+ mls v16.4s, v20.4s, v7.s[0]+ str q28, [x0, #0x200]+ add v13.4s, v27.4s, v22.4s+ ldr q15, [x0, #0x10]+ sub v10.4s, v27.4s, v22.4s+ mls v4.4s, v24.4s, v7.s[0]+ str q13, [x0], #0x10+ str q10, [x0, #0x70]+ sqrdmulh v12.4s, v29.4s, v1.s[1]+ mul v23.4s, v29.4s, v1.s[0]+ mul v8.4s, v26.4s, v1.s[2]+ add v20.4s, v15.4s, v4.4s+ sub v6.4s, v15.4s, v4.4s+ mls v23.4s, v12.4s, v7.s[0]+ sqrdmulh v22.4s, v18.4s, v0.s[1]+ mul v5.4s, v18.4s, v0.s[0]+ sub v28.4s, v21.4s, v23.4s+ sqrdmulh v10.4s, v26.4s, v1.s[3]+ mls v5.4s, v22.4s, v7.s[0]+ sqrdmulh v30.4s, v28.4s, v3.s[1]+ add v4.4s, v21.4s, v23.4s+ mls v8.4s, v10.4s, v7.s[0]+ add v12.4s, v31.4s, v5.4s+ sub v9.4s, v31.4s, v5.4s+ sqrdmulh v25.4s, v4.4s, v2.s[3]+ sqrdmulh v15.4s, v9.4s, v1.s[1]+ sqrdmulh v31.4s, v12.4s, v0.s[3]+ mul v18.4s, v12.4s, v0.s[2]+ mul v11.4s, v9.4s, v1.s[0]+ mls v18.4s, v31.4s, v7.s[0]+ mul v29.4s, v4.4s, v2.s[2]+ mls v29.4s, v25.4s, v7.s[0]+ add v23.4s, v20.4s, v18.4s+ mls v11.4s, v15.4s, v7.s[0]+ sub v31.4s, v20.4s, v18.4s+ add v17.4s, v23.4s, v8.4s+ add v5.4s, v31.4s, v16.4s+ mul v24.4s, v28.4s, v3.s[0]+ str q17, [x0], #0x10+ sub v19.4s, v31.4s, v16.4s+ mls v24.4s, v30.4s, v7.s[0]+ str q5, [x0, #0xf0]+ add v31.4s, v6.4s, v11.4s+ sub v26.4s, v23.4s, v8.4s+ str q19, [x0, #0x170]+ add v4.4s, v31.4s, v29.4s+ sub v13.4s, v6.4s, v11.4s+ str q26, [x0, #0x70]+ sub v11.4s, v31.4s, v29.4s+ sub v22.4s, v13.4s, v24.4s+ add v23.4s, v13.4s, v24.4s+ str q4, [x0, #0x1f0]+ str q11, [x0, #0x270]+ str q23, [x0, #0x2f0]+ str q22, [x0, #0x370]+ mov x0, x3+ mov x4, #0x8 // =8+ ldr q9, [x0, #0x40]+ ldr q23, [x1], #0x40+ ldr q21, [x2, #0x60]+ ldr q1, [x0, #0x20]+ ldur q14, [x1, #-0x30]+ ldr q13, [x0]+ ldr q11, [x2, #0x50]+ sqrdmulh v16.4s, v9.4s, v23.s[1]+ ldr q17, [x0, #0x50]+ mul v15.4s, v9.4s, v23.s[0]+ ldr q30, [x0, #0x70]+ ldr q27, [x0, #0x60]+ ldr q8, [x2, #0x30]+ sqrdmulh v12.4s, v17.4s, v23.s[1]+ ldr q6, [x0, #0x30]+ mls v15.4s, v16.4s, v7.s[0]+ sqrdmulh v18.4s, v27.4s, v23.s[1]+ sqrdmulh v19.4s, v30.4s, v23.s[1]+ add v5.4s, v13.4s, v15.4s+ mul v25.4s, v27.4s, v23.s[0]+ sub v26.4s, v13.4s, v15.4s+ mls v25.4s, v18.4s, v7.s[0]+ mul v10.4s, v17.4s, v23.s[0]+ mls v10.4s, v12.4s, v7.s[0]+ mul v4.4s, v30.4s, v23.s[0]+ sub v22.4s, v1.4s, v25.4s+ mls v4.4s, v19.4s, v7.s[0]+ add v28.4s, v1.4s, v25.4s+ sqrdmulh v19.4s, v28.4s, v23.s[3]+ sqrdmulh v9.4s, v22.4s, v14.s[1]+ add v2.4s, v6.4s, v4.4s+ mul v0.4s, v28.4s, v23.s[2]+ sqrdmulh v27.4s, v2.4s, v23.s[3]+ sub v17.4s, v6.4s, v4.4s+ mul v3.4s, v2.4s, v23.s[2]+ sqrdmulh v20.4s, v17.4s, v14.s[1]+ ldr q1, [x0, #0x10]+ mls v3.4s, v27.4s, v7.s[0]+ mls v0.4s, v19.4s, v7.s[0]+ ldur q16, [x1, #-0x20]+ add v31.4s, v1.4s, v10.4s+ mul v30.4s, v17.4s, v14.s[0]+ mls v30.4s, v20.4s, v7.s[0]+ add v27.4s, v31.4s, v3.4s+ sub v23.4s, v1.4s, v10.4s+ sub v24.4s, v31.4s, v3.4s+ sqrdmulh v4.4s, v27.4s, v14.s[3]+ sqrdmulh v10.4s, v24.4s, v16.s[1]+ mul v18.4s, v24.4s, v16.s[0]+ add v15.4s, v23.4s, v30.4s+ sub v23.4s, v23.4s, v30.4s+ mul v29.4s, v27.4s, v14.s[2]+ sub v2.4s, v5.4s, v0.4s+ add v12.4s, v5.4s, v0.4s+ mls v18.4s, v10.4s, v7.s[0]+ ldur q3, [x1, #-0x10]+ mls v29.4s, v4.4s, v7.s[0]+ mul v4.4s, v22.4s, v14.s[0]+ add v1.4s, v2.4s, v18.4s+ sub v24.4s, v2.4s, v18.4s+ mls v4.4s, v9.4s, v7.s[0]+ ldr q20, [x2, #0x10]+ add v25.4s, v12.4s, v29.4s+ mul v9.4s, v23.4s, v3.s[0]+ sub v5.4s, v12.4s, v29.4s+ sqrdmulh v31.4s, v23.4s, v3.s[1]+ trn2 v6.4s, v1.4s, v24.4s+ trn2 v10.4s, v25.4s, v5.4s+ sqrdmulh v13.4s, v15.4s, v16.s[3]+ trn2 v30.2d, v10.2d, v6.2d+ ldr q3, [x2], #0xc0+ mul v12.4s, v15.4s, v16.s[2]+ trn1 v27.2d, v10.2d, v6.2d+ mls v9.4s, v31.4s, v7.s[0]+ trn1 v22.4s, v25.4s, v5.4s+ sub v6.4s, v26.4s, v4.4s+ mls v12.4s, v13.4s, v7.s[0]+ trn1 v1.4s, v1.4s, v24.4s+ add v13.4s, v26.4s, v4.4s+ trn2 v10.2d, v22.2d, v1.2d+ mul v28.4s, v30.4s, v3.4s+ sub v31.4s, v6.4s, v9.4s+ sub x4, x4, #0x1++Lmld_ntt_layer45678_start:+ add v2.4s, v13.4s, v12.4s+ sqrdmulh v5.4s, v30.4s, v20.4s+ sub v25.4s, v13.4s, v12.4s+ add v17.4s, v6.4s, v9.4s+ mul v19.4s, v10.4s, v3.4s+ trn2 v4.4s, v2.4s, v25.4s+ ldur q24, [x2, #-0x50]+ trn2 v29.4s, v17.4s, v31.4s+ sqrdmulh v15.4s, v10.4s, v20.4s+ mls v28.4s, v5.4s, v7.s[0]+ trn2 v3.2d, v4.2d, v29.2d+ sqrdmulh v12.4s, v3.4s, v24.4s+ mul v16.4s, v3.4s, v21.4s+ mls v19.4s, v15.4s, v7.s[0]+ ldur q10, [x2, #-0xa0]+ add v13.4s, v27.4s, v28.4s+ mls v16.4s, v12.4s, v7.s[0]+ sqrdmulh v9.4s, v13.4s, v8.4s+ sub v30.4s, v27.4s, v28.4s+ ldr q18, [x1], #0x40+ mul v8.4s, v13.4s, v10.4s+ ldr q10, [x0, #0xd0]+ sqrdmulh v14.4s, v30.4s, v11.4s+ ldr q23, [x0, #0xe0]+ sqrdmulh v13.4s, v10.4s, v18.s[1]+ sqrdmulh v12.4s, v23.4s, v18.s[1]+ ldur q6, [x2, #-0x80]+ mul v3.4s, v10.4s, v18.s[0]+ mls v3.4s, v13.4s, v7.s[0]+ ldr q13, [x0, #0xf0]+ trn1 v27.4s, v2.4s, v25.4s+ mul v2.4s, v30.4s, v6.4s+ trn1 v20.4s, v17.4s, v31.4s+ trn1 v25.2d, v4.2d, v29.2d+ sqrdmulh v10.4s, v13.4s, v18.s[1]+ trn2 v5.2d, v27.2d, v20.2d+ ldur q6, [x2, #-0x10]+ mls v8.4s, v9.4s, v7.s[0]+ sub v15.4s, v25.4s, v16.4s+ sqrdmulh v31.4s, v5.4s, v24.4s+ sqrdmulh v30.4s, v15.4s, v6.4s+ ldur q9, [x2, #-0x30]+ mul v4.4s, v5.4s, v21.4s+ ldur q21, [x2, #-0x40]+ ldur q6, [x2, #-0x20]+ add v5.4s, v25.4s, v16.4s+ mls v4.4s, v31.4s, v7.s[0]+ mul v0.4s, v5.4s, v21.4s+ mul v17.4s, v13.4s, v18.s[0]+ mls v17.4s, v10.4s, v7.s[0]+ ldr q28, [x0, #0xb0]+ sqrdmulh v26.4s, v5.4s, v9.4s+ mul v9.4s, v15.4s, v6.4s+ trn1 v6.2d, v22.2d, v1.2d+ mls v9.4s, v30.4s, v7.s[0]+ add v25.4s, v28.4s, v17.4s+ mls v2.4s, v14.4s, v7.s[0]+ trn1 v5.2d, v27.2d, v20.2d+ ldr q20, [x2, #0x10]+ mul v29.4s, v25.4s, v18.s[2]+ add v15.4s, v6.4s, v19.4s+ ldr q30, [x0, #0xc0]+ sub v19.4s, v6.4s, v19.4s+ add v31.4s, v15.4s, v8.4s+ mls v0.4s, v26.4s, v7.s[0]+ ldur q14, [x1, #-0x30]+ add v21.4s, v19.4s, v2.4s+ sub v24.4s, v19.4s, v2.4s+ sqrdmulh v27.4s, v25.4s, v18.s[3]+ sub v26.4s, v15.4s, v8.4s+ ldr q2, [x0, #0x90]+ mul v16.4s, v30.4s, v18.s[0]+ sub v25.4s, v28.4s, v17.4s+ trn1 v11.4s, v31.4s, v26.4s+ ldr q1, [x0, #0xa0]+ trn1 v6.4s, v21.4s, v24.4s+ sqrdmulh v13.4s, v25.4s, v14.s[1]+ add v8.4s, v2.4s, v3.4s+ trn2 v28.2d, v11.2d, v6.2d+ sqrdmulh v19.4s, v30.4s, v18.s[1]+ sub v10.4s, v5.4s, v4.4s+ ldur q22, [x1, #-0x20]+ str q28, [x0, #0x20]+ mls v29.4s, v27.4s, v7.s[0]+ add v15.4s, v10.4s, v9.4s+ mul v25.4s, v25.4s, v14.s[0]+ ldur q27, [x1, #-0x10]+ trn2 v17.4s, v31.4s, v26.4s+ trn2 v21.4s, v21.4s, v24.4s+ mls v16.4s, v19.4s, v7.s[0]+ sub v24.4s, v8.4s, v29.4s+ sub v10.4s, v10.4s, v9.4s+ mls v25.4s, v13.4s, v7.s[0]+ trn1 v13.2d, v11.2d, v6.2d+ ldr q28, [x0, #0x80]+ sqrdmulh v30.4s, v24.4s, v22.s[1]+ trn2 v19.2d, v17.2d, v21.2d+ trn1 v6.2d, v17.2d, v21.2d+ mul v31.4s, v23.4s, v18.s[0]+ str q13, [x0], #0x80+ stur q6, [x0, #-0x70]+ stur q19, [x0, #-0x50]+ ldr q11, [x2, #0x50]+ mls v31.4s, v12.4s, v7.s[0]+ ldr q21, [x2, #0x60]+ trn1 v9.4s, v15.4s, v10.4s+ trn2 v6.4s, v15.4s, v10.4s+ mul v24.4s, v24.4s, v22.s[0]+ sub v10.4s, v2.4s, v3.4s+ ldr q3, [x2], #0xc0+ mls v24.4s, v30.4s, v7.s[0]+ add v26.4s, v8.4s, v29.4s+ ldur q8, [x2, #-0x90]+ add v17.4s, v5.4s, v4.4s+ sqrdmulh v2.4s, v26.4s, v14.s[3]+ sub v13.4s, v1.4s, v31.4s+ add v30.4s, v1.4s, v31.4s+ add v15.4s, v10.4s, v25.4s+ sqrdmulh v19.4s, v13.4s, v14.s[1]+ sub v25.4s, v10.4s, v25.4s+ mul v29.4s, v13.4s, v14.s[0]+ sub v5.4s, v28.4s, v16.4s+ sqrdmulh v4.4s, v30.4s, v18.s[3]+ sub v23.4s, v17.4s, v0.4s+ add v31.4s, v17.4s, v0.4s+ mul v18.4s, v30.4s, v18.s[2]+ add v1.4s, v28.4s, v16.4s+ trn2 v12.4s, v31.4s, v23.4s+ mls v29.4s, v19.4s, v7.s[0]+ trn1 v13.4s, v31.4s, v23.4s+ trn2 v30.2d, v12.2d, v6.2d+ mls v18.4s, v4.4s, v7.s[0]+ trn2 v10.2d, v13.2d, v9.2d+ trn1 v31.2d, v13.2d, v9.2d+ mul v19.4s, v26.4s, v14.s[2]+ trn1 v12.2d, v12.2d, v6.2d+ sub v6.4s, v5.4s, v29.4s+ mls v19.4s, v2.4s, v7.s[0]+ add v13.4s, v5.4s, v29.4s+ stur q10, [x0, #-0x20]+ sub v10.4s, v1.4s, v18.4s+ add v28.4s, v1.4s, v18.4s+ sqrdmulh v5.4s, v25.4s, v27.s[1]+ stur q31, [x0, #-0x40]+ add v26.4s, v10.4s, v24.4s+ sub v31.4s, v10.4s, v24.4s+ mul v9.4s, v25.4s, v27.s[0]+ stur q12, [x0, #-0x30]+ sub v24.4s, v28.4s, v19.4s+ sqrdmulh v10.4s, v15.4s, v22.s[3]+ trn1 v1.4s, v26.4s, v31.4s+ stur q30, [x0, #-0x10]+ add v30.4s, v28.4s, v19.4s+ mls v9.4s, v5.4s, v7.s[0]+ trn2 v25.4s, v26.4s, v31.4s+ trn2 v14.4s, v30.4s, v24.4s+ mul v12.4s, v15.4s, v22.s[2]+ trn1 v22.4s, v30.4s, v24.4s+ trn1 v27.2d, v14.2d, v25.2d+ mls v12.4s, v10.4s, v7.s[0]+ trn2 v30.2d, v14.2d, v25.2d+ sub v31.4s, v6.4s, v9.4s+ trn2 v10.2d, v22.2d, v1.2d+ mul v28.4s, v30.4s, v3.4s+ subs x4, x4, #0x1+ cbnz x4, Lmld_ntt_layer45678_start+ add v9.4s, v6.4s, v9.4s+ sqrdmulh v6.4s, v30.4s, v20.4s+ ldur q24, [x2, #-0xa0]+ add v25.4s, v13.4s, v12.4s+ sub v15.4s, v13.4s, v12.4s+ mul v19.4s, v10.4s, v3.4s+ trn2 v5.4s, v9.4s, v31.4s+ sqrdmulh v3.4s, v10.4s, v20.4s+ trn2 v10.4s, v25.4s, v15.4s+ mls v28.4s, v6.4s, v7.s[0]+ trn2 v13.2d, v10.2d, v5.2d+ ldur q30, [x2, #-0x50]+ mul v12.4s, v13.4s, v21.4s+ mls v19.4s, v3.4s, v7.s[0]+ add v20.4s, v27.4s, v28.4s+ sqrdmulh v13.4s, v13.4s, v30.4s+ sub v3.4s, v27.4s, v28.4s+ mul v24.4s, v20.4s, v24.4s+ sqrdmulh v6.4s, v3.4s, v11.4s+ ldur q27, [x2, #-0x80]+ mls v12.4s, v13.4s, v7.s[0]+ trn1 v25.4s, v25.4s, v15.4s+ mul v27.4s, v3.4s, v27.4s+ trn1 v31.4s, v9.4s, v31.4s+ trn1 v3.2d, v10.2d, v5.2d+ ldur q13, [x2, #-0x30]+ ldur q15, [x2, #-0x40]+ sqrdmulh v9.4s, v20.4s, v8.4s+ trn2 v20.2d, v25.2d, v31.2d+ ldur q10, [x2, #-0x10]+ mls v27.4s, v6.4s, v7.s[0]+ add v5.4s, v3.4s, v12.4s+ sub v6.4s, v3.4s, v12.4s+ sqrdmulh v3.4s, v20.4s, v30.4s+ trn1 v12.2d, v22.2d, v1.2d+ sqrdmulh v10.4s, v6.4s, v10.4s+ mls v24.4s, v9.4s, v7.s[0]+ sub v9.4s, v12.4s, v19.4s+ trn1 v25.2d, v25.2d, v31.2d+ sqrdmulh v31.4s, v5.4s, v13.4s+ add v30.4s, v9.4s, v27.4s+ add v13.4s, v12.4s, v19.4s+ mul v1.4s, v20.4s, v21.4s+ ldur q12, [x2, #-0x20]+ add v21.4s, v13.4s, v24.4s+ sub v13.4s, v13.4s, v24.4s+ mls v1.4s, v3.4s, v7.s[0]+ sub v3.4s, v9.4s, v27.4s+ mul v9.4s, v6.4s, v12.4s+ trn2 v12.4s, v21.4s, v13.4s+ trn1 v6.4s, v30.4s, v3.4s+ trn2 v30.4s, v30.4s, v3.4s+ mls v9.4s, v10.4s, v7.s[0]+ trn1 v13.4s, v21.4s, v13.4s+ mul v15.4s, v5.4s, v15.4s+ sub v3.4s, v25.4s, v1.4s+ add v5.4s, v25.4s, v1.4s+ mls v15.4s, v31.4s, v7.s[0]+ trn1 v21.2d, v13.2d, v6.2d+ trn2 v6.2d, v13.2d, v6.2d+ add v10.4s, v3.4s, v9.4s+ sub v13.4s, v3.4s, v9.4s+ str q21, [x0], #0x80+ trn1 v3.2d, v12.2d, v30.2d+ trn2 v31.2d, v12.2d, v30.2d+ trn1 v21.4s, v10.4s, v13.4s+ sub v30.4s, v5.4s, v15.4s+ add v12.4s, v5.4s, v15.4s+ stur q3, [x0, #-0x70]+ trn2 v13.4s, v10.4s, v13.4s+ trn1 v19.4s, v12.4s, v30.4s+ trn2 v12.4s, v12.4s, v30.4s+ stur q6, [x0, #-0x60]+ stur q31, [x0, #-0x50]+ trn1 v10.2d, v19.2d, v21.2d+ trn2 v3.2d, v19.2d, v21.2d+ trn1 v21.2d, v12.2d, v13.2d+ trn2 v13.2d, v12.2d, v13.2d+ stur q10, [x0, #-0x40]+ stur q3, [x0, #-0x20]+ stur q13, [x0, #-0x10]+ stur q21, [x0, #-0x30]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(ntt_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_pointwise_montgomery_aarch64_asm.S view
@@ -0,0 +1,106 @@+/* Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: poly_pointwise_montgomery_aarch64_asm+ Description: AArch64 pointwise Montgomery multiplication of two polynomials+ Signature: void mld_poly_pointwise_montgomery_aarch64_asm(int32_t a[256], const int32_t b[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t a[256]+ description: Input/output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t b[256]+ description: Input polynomial (256 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ (!defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_pointwise_montgomery_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_pointwise_montgomery_aarch64_asm)+MLD_ASM_FN_SYMBOL(poly_pointwise_montgomery_aarch64_asm)++ .cfi_startproc+ mov w3, #0xe001 // =57345+ movk w3, #0x7f, lsl #16+ dup v0.4s, w3+ mov w3, #0x2001 // =8193+ movk w3, #0x380, lsl #16+ dup v1.4s, w3+ mov x3, #0x40 // =64++Lmld_poly_pointwise_montgomery_loop_start:+ ldr q16, [x0]+ ldr q17, [x0, #0x10]+ ldr q18, [x0, #0x20]+ ldr q19, [x0, #0x30]+ ldr q21, [x1, #0x10]+ ldr q22, [x1, #0x20]+ ldr q23, [x1, #0x30]+ ldr q20, [x1], #0x40+ smull v24.2d, v16.2s, v20.2s+ smull2 v25.2d, v16.4s, v20.4s+ smull v26.2d, v17.2s, v21.2s+ smull2 v27.2d, v17.4s, v21.4s+ smull v28.2d, v18.2s, v22.2s+ smull2 v29.2d, v18.4s, v22.4s+ smull v30.2d, v19.2s, v23.2s+ smull2 v31.2d, v19.4s, v23.4s+ uzp1 v16.4s, v24.4s, v25.4s+ mul v16.4s, v16.4s, v1.4s+ smlsl v24.2d, v16.2s, v0.2s+ smlsl2 v25.2d, v16.4s, v0.4s+ uzp2 v16.4s, v24.4s, v25.4s+ uzp1 v17.4s, v26.4s, v27.4s+ mul v17.4s, v17.4s, v1.4s+ smlsl v26.2d, v17.2s, v0.2s+ smlsl2 v27.2d, v17.4s, v0.4s+ uzp2 v17.4s, v26.4s, v27.4s+ uzp1 v18.4s, v28.4s, v29.4s+ mul v18.4s, v18.4s, v1.4s+ smlsl v28.2d, v18.2s, v0.2s+ smlsl2 v29.2d, v18.4s, v0.4s+ uzp2 v18.4s, v28.4s, v29.4s+ uzp1 v19.4s, v30.4s, v31.4s+ mul v19.4s, v19.4s, v1.4s+ smlsl v30.2d, v19.2s, v0.2s+ smlsl2 v31.2d, v19.4s, v0.4s+ uzp2 v19.4s, v30.4s, v31.4s+ str q17, [x0, #0x10]+ str q18, [x0, #0x20]+ str q19, [x0, #0x30]+ str q16, [x0], #0x40+ subs x3, x3, #0x4+ cbnz x3, Lmld_poly_pointwise_montgomery_loop_start+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_pointwise_montgomery_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && (!MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_NO_VERIFY_API || MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) \+ && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_poly_caddq_aarch64_asm.S view
@@ -0,0 +1,69 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+/*yaml+ Name: poly_caddq_aarch64_asm+ Description: AArch64 conditional addition of q to each coefficient+ Signature: void mld_poly_caddq_aarch64_asm(int32_t a[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t a[256]+ description: Input/output polynomial (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_poly_caddq_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_caddq_aarch64_asm)+MLD_ASM_FN_SYMBOL(poly_caddq_aarch64_asm)++ .cfi_startproc+ mov w9, #0xe001 // =57345+ movk w9, #0x7f, lsl #16+ dup v4.4s, w9+ mov x1, #0x10 // =16++Lmld_poly_caddq_loop:+ ldr q0, [x0]+ ldr q1, [x0, #0x10]+ ldr q2, [x0, #0x20]+ ldr q3, [x0, #0x30]+ ushr v5.4s, v0.4s, #0x1f+ mla v0.4s, v5.4s, v4.4s+ ushr v5.4s, v1.4s, #0x1f+ mla v1.4s, v5.4s, v4.4s+ ushr v5.4s, v2.4s, #0x1f+ mla v2.4s, v5.4s, v4.4s+ ushr v5.4s, v3.4s, #0x1f+ mla v3.4s, v5.4s, v4.4s+ str q1, [x0, #0x10]+ str q2, [x0, #0x20]+ str q3, [x0, #0x30]+ str q0, [x0], #0x40+ subs x1, x1, #0x1+ b.ne Lmld_poly_caddq_loop+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_caddq_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_poly_chknorm_aarch64_asm.S view
@@ -0,0 +1,76 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+/*yaml+ Name: poly_chknorm_aarch64_asm+ Description: AArch64 infinity-norm bound check on polynomial coefficients+ Signature: int mld_poly_chknorm_aarch64_asm(const int32_t a[256], int32_t B)+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t a[256]+ description: Input polynomial (256 x int32_t)+ x1:+ type: scalar+ c_parameter: int32_t B+ description: Norm bound+ test_with: 131072 # representative non-negative bound (1 << 17)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_poly_chknorm_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_chknorm_aarch64_asm)+MLD_ASM_FN_SYMBOL(poly_chknorm_aarch64_asm)++ .cfi_startproc+ dup v20.4s, w1+ eor v21.16b, v21.16b, v21.16b+ mov x2, #0x10 // =16++Lmld_poly_chknorm_loop:+ ldr q1, [x0, #0x10]+ ldr q2, [x0, #0x20]+ ldr q3, [x0, #0x30]+ ldr q0, [x0], #0x40+ abs v1.4s, v1.4s+ cmge v1.4s, v1.4s, v20.4s+ orr v21.16b, v21.16b, v1.16b+ abs v2.4s, v2.4s+ cmge v2.4s, v2.4s, v20.4s+ orr v21.16b, v21.16b, v2.16b+ abs v3.4s, v3.4s+ cmge v3.4s, v3.4s, v20.4s+ orr v21.16b, v21.16b, v3.16b+ abs v0.4s, v0.4s+ cmge v0.4s, v0.4s, v20.4s+ orr v21.16b, v21.16b, v0.16b+ subs x2, x2, #0x1+ b.ne Lmld_poly_chknorm_loop+ umaxv s21, v21.4s+ fmov w0, s21+ and w0, w0, #0x1+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_chknorm_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_poly_decompose_32_aarch64_asm.S view
@@ -0,0 +1,108 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+/*yaml+ Name: poly_decompose_32_aarch64_asm+ Description: AArch64 coefficient decomposition (alpha = (Q-1)/32)+ Signature: void mld_poly_decompose_32_aarch64_asm(int32_t a1[256], int32_t a0[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t a1[256]+ description: Output high-part polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t a0[256]+ description: Input polynomial / output low-part (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_SIGN_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_poly_decompose_32_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_decompose_32_aarch64_asm)+MLD_ASM_FN_SYMBOL(poly_decompose_32_aarch64_asm)++ .cfi_startproc+ mov w4, #0xe001 // =57345+ movk w4, #0x7f, lsl #16+ dup v20.4s, w4+ mov w5, #0xe100 // =57600+ movk w5, #0x7b, lsl #16+ dup v21.4s, w5+ mov w7, #0xfe00 // =65024+ movk w7, #0x7, lsl #16+ dup v22.4s, w7+ mov w11, #0x401 // =1025+ movk w11, #0x4010, lsl #16+ dup v23.4s, w11+ mov x3, #0x10 // =16++Lmld_poly_decompose_32_loop:+ ldr q0, [x1]+ ldr q1, [x1, #0x10]+ ldr q2, [x1, #0x20]+ ldr q3, [x1, #0x30]+ sqdmulh v5.4s, v1.4s, v23.4s+ srshr v5.4s, v5.4s, #0x12+ cmgt v24.4s, v1.4s, v21.4s+ mls v1.4s, v5.4s, v22.4s+ bic v5.16b, v5.16b, v24.16b+ add v1.4s, v1.4s, v24.4s+ sqdmulh v6.4s, v2.4s, v23.4s+ srshr v6.4s, v6.4s, #0x12+ cmgt v24.4s, v2.4s, v21.4s+ mls v2.4s, v6.4s, v22.4s+ bic v6.16b, v6.16b, v24.16b+ add v2.4s, v2.4s, v24.4s+ sqdmulh v7.4s, v3.4s, v23.4s+ srshr v7.4s, v7.4s, #0x12+ cmgt v24.4s, v3.4s, v21.4s+ mls v3.4s, v7.4s, v22.4s+ bic v7.16b, v7.16b, v24.16b+ add v3.4s, v3.4s, v24.4s+ sqdmulh v4.4s, v0.4s, v23.4s+ srshr v4.4s, v4.4s, #0x12+ cmgt v24.4s, v0.4s, v21.4s+ mls v0.4s, v4.4s, v22.4s+ bic v4.16b, v4.16b, v24.16b+ add v0.4s, v0.4s, v24.4s+ str q5, [x0, #0x10]+ str q6, [x0, #0x20]+ str q7, [x0, #0x30]+ str q4, [x0], #0x40+ str q1, [x1, #0x10]+ str q2, [x1, #0x20]+ str q3, [x1, #0x30]+ str q0, [x1], #0x40+ subs x3, x3, #0x1+ b.ne Lmld_poly_decompose_32_loop+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_decompose_32_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_SIGN_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_poly_decompose_88_aarch64_asm.S view
@@ -0,0 +1,108 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+/*yaml+ Name: poly_decompose_88_aarch64_asm+ Description: AArch64 coefficient decomposition (alpha = (Q-1)/88)+ Signature: void mld_poly_decompose_88_aarch64_asm(int32_t a1[256], int32_t a0[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t a1[256]+ description: Output high-part polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t a0[256]+ description: Input polynomial / output low-part (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_SIGN_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_poly_decompose_88_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_decompose_88_aarch64_asm)+MLD_ASM_FN_SYMBOL(poly_decompose_88_aarch64_asm)++ .cfi_startproc+ mov w4, #0xe001 // =57345+ movk w4, #0x7f, lsl #16+ dup v20.4s, w4+ mov w5, #0x6c00 // =27648+ movk w5, #0x7e, lsl #16+ dup v21.4s, w5+ mov w7, #0xe800 // =59392+ movk w7, #0x2, lsl #16+ dup v22.4s, w7+ mov w11, #0x581 // =1409+ movk w11, #0x5816, lsl #16+ dup v23.4s, w11+ mov x3, #0x10 // =16++Lmld_poly_decompose_88_loop:+ ldr q0, [x1]+ ldr q1, [x1, #0x10]+ ldr q2, [x1, #0x20]+ ldr q3, [x1, #0x30]+ sqdmulh v5.4s, v1.4s, v23.4s+ srshr v5.4s, v5.4s, #0x11+ cmgt v24.4s, v1.4s, v21.4s+ mls v1.4s, v5.4s, v22.4s+ bic v5.16b, v5.16b, v24.16b+ add v1.4s, v1.4s, v24.4s+ sqdmulh v6.4s, v2.4s, v23.4s+ srshr v6.4s, v6.4s, #0x11+ cmgt v24.4s, v2.4s, v21.4s+ mls v2.4s, v6.4s, v22.4s+ bic v6.16b, v6.16b, v24.16b+ add v2.4s, v2.4s, v24.4s+ sqdmulh v7.4s, v3.4s, v23.4s+ srshr v7.4s, v7.4s, #0x11+ cmgt v24.4s, v3.4s, v21.4s+ mls v3.4s, v7.4s, v22.4s+ bic v7.16b, v7.16b, v24.16b+ add v3.4s, v3.4s, v24.4s+ sqdmulh v4.4s, v0.4s, v23.4s+ srshr v4.4s, v4.4s, #0x11+ cmgt v24.4s, v0.4s, v21.4s+ mls v0.4s, v4.4s, v22.4s+ bic v4.16b, v4.16b, v24.16b+ add v0.4s, v0.4s, v24.4s+ str q5, [x0, #0x10]+ str q6, [x0, #0x20]+ str q7, [x0, #0x30]+ str q4, [x0], #0x40+ str q1, [x1, #0x10]+ str q2, [x1, #0x20]+ str q3, [x1, #0x30]+ str q0, [x1], #0x40+ subs x3, x3, #0x1+ b.ne Lmld_poly_decompose_88_loop+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_decompose_88_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_SIGN_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_poly_use_hint_32_aarch64_asm.S view
@@ -0,0 +1,125 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+/*yaml+ Name: poly_use_hint_32_aarch64_asm+ Description: AArch64 hint application (alpha = (Q-1)/32)+ Signature: void mld_poly_use_hint_32_aarch64_asm(int32_t a[256], const int32_t h[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t a[256]+ description: Input/output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t h[256]+ description: Hint polynomial (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_VERIFY_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_poly_use_hint_32_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_use_hint_32_aarch64_asm)+MLD_ASM_FN_SYMBOL(poly_use_hint_32_aarch64_asm)++ .cfi_startproc+ mov w4, #0xe001 // =57345+ movk w4, #0x7f, lsl #16+ dup v20.4s, w4+ mov w5, #0xe100 // =57600+ movk w5, #0x7b, lsl #16+ dup v21.4s, w5+ mov w7, #0xfe00 // =65024+ movk w7, #0x7, lsl #16+ dup v22.4s, w7+ mov w11, #0x401 // =1025+ movk w11, #0x4010, lsl #16+ dup v23.4s, w11+ movi v24.4s, #0xf+ mov x3, #0x10 // =16++Lmld_poly_use_hint_32_loop:+ ldr q1, [x0, #0x10]+ ldr q2, [x0, #0x20]+ ldr q3, [x0, #0x30]+ ldr q0, [x0]+ ldr q5, [x1, #0x10]+ ldr q6, [x1, #0x20]+ ldr q7, [x1, #0x30]+ ldr q4, [x1], #0x40+ sqdmulh v17.4s, v1.4s, v23.4s+ srshr v17.4s, v17.4s, #0x12+ cmgt v25.4s, v1.4s, v21.4s+ mls v1.4s, v17.4s, v22.4s+ bic v17.16b, v17.16b, v25.16b+ add v1.4s, v1.4s, v25.4s+ cmle v1.4s, v1.4s, #0+ orr v1.4s, #0x1+ mla v17.4s, v1.4s, v5.4s+ and v17.16b, v17.16b, v24.16b+ sqdmulh v18.4s, v2.4s, v23.4s+ srshr v18.4s, v18.4s, #0x12+ cmgt v25.4s, v2.4s, v21.4s+ mls v2.4s, v18.4s, v22.4s+ bic v18.16b, v18.16b, v25.16b+ add v2.4s, v2.4s, v25.4s+ cmle v2.4s, v2.4s, #0+ orr v2.4s, #0x1+ mla v18.4s, v2.4s, v6.4s+ and v18.16b, v18.16b, v24.16b+ sqdmulh v19.4s, v3.4s, v23.4s+ srshr v19.4s, v19.4s, #0x12+ cmgt v25.4s, v3.4s, v21.4s+ mls v3.4s, v19.4s, v22.4s+ bic v19.16b, v19.16b, v25.16b+ add v3.4s, v3.4s, v25.4s+ cmle v3.4s, v3.4s, #0+ orr v3.4s, #0x1+ mla v19.4s, v3.4s, v7.4s+ and v19.16b, v19.16b, v24.16b+ sqdmulh v16.4s, v0.4s, v23.4s+ srshr v16.4s, v16.4s, #0x12+ cmgt v25.4s, v0.4s, v21.4s+ mls v0.4s, v16.4s, v22.4s+ bic v16.16b, v16.16b, v25.16b+ add v0.4s, v0.4s, v25.4s+ cmle v0.4s, v0.4s, #0+ orr v0.4s, #0x1+ mla v16.4s, v0.4s, v4.4s+ and v16.16b, v16.16b, v24.16b+ str q17, [x0, #0x10]+ str q18, [x0, #0x20]+ str q19, [x0, #0x30]+ str q16, [x0], #0x40+ subs x3, x3, #0x1+ b.ne Lmld_poly_use_hint_32_loop+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_use_hint_32_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_VERIFY_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_poly_use_hint_88_aarch64_asm.S view
@@ -0,0 +1,133 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+/*yaml+ Name: poly_use_hint_88_aarch64_asm+ Description: AArch64 hint application (alpha = (Q-1)/88)+ Signature: void mld_poly_use_hint_88_aarch64_asm(int32_t a[256], const int32_t h[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t a[256]+ description: Input/output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t h[256]+ description: Hint polynomial (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_VERIFY_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_poly_use_hint_88_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_use_hint_88_aarch64_asm)+MLD_ASM_FN_SYMBOL(poly_use_hint_88_aarch64_asm)++ .cfi_startproc+ mov w4, #0xe001 // =57345+ movk w4, #0x7f, lsl #16+ dup v20.4s, w4+ mov w5, #0x6c00 // =27648+ movk w5, #0x7e, lsl #16+ dup v21.4s, w5+ mov w7, #0xe800 // =59392+ movk w7, #0x2, lsl #16+ dup v22.4s, w7+ mov w11, #0x581 // =1409+ movk w11, #0x5816, lsl #16+ dup v23.4s, w11+ movi v24.4s, #0x2b+ mov x3, #0x10 // =16++Lmld_poly_use_hint_88_loop:+ ldr q1, [x0, #0x10]+ ldr q2, [x0, #0x20]+ ldr q3, [x0, #0x30]+ ldr q0, [x0]+ ldr q5, [x1, #0x10]+ ldr q6, [x1, #0x20]+ ldr q7, [x1, #0x30]+ ldr q4, [x1], #0x40+ sqdmulh v17.4s, v1.4s, v23.4s+ srshr v17.4s, v17.4s, #0x11+ cmgt v25.4s, v1.4s, v21.4s+ mls v1.4s, v17.4s, v22.4s+ bic v17.16b, v17.16b, v25.16b+ add v1.4s, v1.4s, v25.4s+ cmle v1.4s, v1.4s, #0+ orr v1.4s, #0x1+ mla v17.4s, v1.4s, v5.4s+ cmgt v25.4s, v17.4s, v24.4s+ bic v17.16b, v17.16b, v25.16b+ umin v17.4s, v17.4s, v24.4s+ sqdmulh v18.4s, v2.4s, v23.4s+ srshr v18.4s, v18.4s, #0x11+ cmgt v25.4s, v2.4s, v21.4s+ mls v2.4s, v18.4s, v22.4s+ bic v18.16b, v18.16b, v25.16b+ add v2.4s, v2.4s, v25.4s+ cmle v2.4s, v2.4s, #0+ orr v2.4s, #0x1+ mla v18.4s, v2.4s, v6.4s+ cmgt v25.4s, v18.4s, v24.4s+ bic v18.16b, v18.16b, v25.16b+ umin v18.4s, v18.4s, v24.4s+ sqdmulh v19.4s, v3.4s, v23.4s+ srshr v19.4s, v19.4s, #0x11+ cmgt v25.4s, v3.4s, v21.4s+ mls v3.4s, v19.4s, v22.4s+ bic v19.16b, v19.16b, v25.16b+ add v3.4s, v3.4s, v25.4s+ cmle v3.4s, v3.4s, #0+ orr v3.4s, #0x1+ mla v19.4s, v3.4s, v7.4s+ cmgt v25.4s, v19.4s, v24.4s+ bic v19.16b, v19.16b, v25.16b+ umin v19.4s, v19.4s, v24.4s+ sqdmulh v16.4s, v0.4s, v23.4s+ srshr v16.4s, v16.4s, #0x11+ cmgt v25.4s, v0.4s, v21.4s+ mls v0.4s, v16.4s, v22.4s+ bic v16.16b, v16.16b, v25.16b+ add v0.4s, v0.4s, v25.4s+ cmle v0.4s, v0.4s, #0+ orr v0.4s, #0x1+ mla v16.4s, v0.4s, v4.4s+ cmgt v25.4s, v16.4s, v24.4s+ bic v16.16b, v16.16b, v25.16b+ umin v16.4s, v16.4s, v24.4s+ str q17, [x0, #0x10]+ str q18, [x0, #0x20]+ str q19, [x0, #0x30]+ str q16, [x0], #0x40+ subs x3, x3, #0x1+ b.ne Lmld_poly_use_hint_88_loop+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_use_hint_88_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_VERIFY_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S view
@@ -0,0 +1,157 @@+/* Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: polyvecl_pointwise_acc_montgomery_l4_aarch64_asm+ Description: AArch64 pointwise multiply-accumulate of length-4 polynomial vectors+ Signature: void mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm(int32_t r[256], const int32_t a[4][256], const int32_t b[4][256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 4096+ permissions: read-only+ c_parameter: const int32_t a[4][256]+ description: Input polynomial vector a (4 x 256 x int32_t)+ x2:+ type: buffer+ size_bytes: 4096+ permissions: read-only+ c_parameter: const int32_t b[4][256]+ description: Input polynomial vector b (4 x 256 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 4)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(polyvecl_pointwise_acc_montgomery_l4_aarch64_asm)+MLD_ASM_FN_SYMBOL(polyvecl_pointwise_acc_montgomery_l4_aarch64_asm)++ .cfi_startproc+ mov w3, #0xe001 // =57345+ movk w3, #0x7f, lsl #16+ dup v0.4s, w3+ mov w3, #0x2001 // =8193+ movk w3, #0x380, lsl #16+ dup v1.4s, w3+ mov x3, #0x40 // =64++Lmld_polyvecl_pointwise_acc_montgomery_l4_loop_start:+ ldr q17, [x1, #0x10]+ ldr q18, [x1, #0x20]+ ldr q19, [x1, #0x30]+ ldr q16, [x1], #0x40+ ldr q21, [x2, #0x10]+ ldr q22, [x2, #0x20]+ ldr q23, [x2, #0x30]+ ldr q20, [x2], #0x40+ smull v24.2d, v16.2s, v20.2s+ smull2 v25.2d, v16.4s, v20.4s+ smull v26.2d, v17.2s, v21.2s+ smull2 v27.2d, v17.4s, v21.4s+ smull v28.2d, v18.2s, v22.2s+ smull2 v29.2d, v18.4s, v22.4s+ smull v30.2d, v19.2s, v23.2s+ smull2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x3c0]+ ldr q17, [x1, #0x3d0]+ ldr q18, [x1, #0x3e0]+ ldr q19, [x1, #0x3f0]+ ldr q20, [x2, #0x3c0]+ ldr q21, [x2, #0x3d0]+ ldr q22, [x2, #0x3e0]+ ldr q23, [x2, #0x3f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x7c0]+ ldr q17, [x1, #0x7d0]+ ldr q18, [x1, #0x7e0]+ ldr q19, [x1, #0x7f0]+ ldr q20, [x2, #0x7c0]+ ldr q21, [x2, #0x7d0]+ ldr q22, [x2, #0x7e0]+ ldr q23, [x2, #0x7f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0xbc0]+ ldr q17, [x1, #0xbd0]+ ldr q18, [x1, #0xbe0]+ ldr q19, [x1, #0xbf0]+ ldr q20, [x2, #0xbc0]+ ldr q21, [x2, #0xbd0]+ ldr q22, [x2, #0xbe0]+ ldr q23, [x2, #0xbf0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ uzp1 v16.4s, v24.4s, v25.4s+ mul v16.4s, v16.4s, v1.4s+ smlsl v24.2d, v16.2s, v0.2s+ smlsl2 v25.2d, v16.4s, v0.4s+ uzp2 v16.4s, v24.4s, v25.4s+ uzp1 v17.4s, v26.4s, v27.4s+ mul v17.4s, v17.4s, v1.4s+ smlsl v26.2d, v17.2s, v0.2s+ smlsl2 v27.2d, v17.4s, v0.4s+ uzp2 v17.4s, v26.4s, v27.4s+ uzp1 v18.4s, v28.4s, v29.4s+ mul v18.4s, v18.4s, v1.4s+ smlsl v28.2d, v18.2s, v0.2s+ smlsl2 v29.2d, v18.4s, v0.4s+ uzp2 v18.4s, v28.4s, v29.4s+ uzp1 v19.4s, v30.4s, v31.4s+ mul v19.4s, v19.4s, v1.4s+ smlsl v30.2d, v19.2s, v0.2s+ smlsl2 v31.2d, v19.4s, v0.4s+ uzp2 v19.4s, v30.4s, v31.4s+ str q17, [x0, #0x10]+ str q18, [x0, #0x20]+ str q19, [x0, #0x30]+ str q16, [x0], #0x40+ subs x3, x3, #0x4+ cbnz x3, Lmld_polyvecl_pointwise_acc_montgomery_l4_loop_start+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(polyvecl_pointwise_acc_montgomery_l4_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 4) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S view
@@ -0,0 +1,173 @@+/* Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: polyvecl_pointwise_acc_montgomery_l5_aarch64_asm+ Description: AArch64 pointwise multiply-accumulate of length-5 polynomial vectors+ Signature: void mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm(int32_t r[256], const int32_t a[5][256], const int32_t b[5][256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 5120+ permissions: read-only+ c_parameter: const int32_t a[5][256]+ description: Input polynomial vector a (5 x 256 x int32_t)+ x2:+ type: buffer+ size_bytes: 5120+ permissions: read-only+ c_parameter: const int32_t b[5][256]+ description: Input polynomial vector b (5 x 256 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 5)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(polyvecl_pointwise_acc_montgomery_l5_aarch64_asm)+MLD_ASM_FN_SYMBOL(polyvecl_pointwise_acc_montgomery_l5_aarch64_asm)++ .cfi_startproc+ mov w3, #0xe001 // =57345+ movk w3, #0x7f, lsl #16+ dup v0.4s, w3+ mov w3, #0x2001 // =8193+ movk w3, #0x380, lsl #16+ dup v1.4s, w3+ mov x3, #0x40 // =64++Lmld_polyvecl_pointwise_acc_montgomery_l5_loop_start:+ ldr q17, [x1, #0x10]+ ldr q18, [x1, #0x20]+ ldr q19, [x1, #0x30]+ ldr q16, [x1], #0x40+ ldr q21, [x2, #0x10]+ ldr q22, [x2, #0x20]+ ldr q23, [x2, #0x30]+ ldr q20, [x2], #0x40+ smull v24.2d, v16.2s, v20.2s+ smull2 v25.2d, v16.4s, v20.4s+ smull v26.2d, v17.2s, v21.2s+ smull2 v27.2d, v17.4s, v21.4s+ smull v28.2d, v18.2s, v22.2s+ smull2 v29.2d, v18.4s, v22.4s+ smull v30.2d, v19.2s, v23.2s+ smull2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x3c0]+ ldr q17, [x1, #0x3d0]+ ldr q18, [x1, #0x3e0]+ ldr q19, [x1, #0x3f0]+ ldr q20, [x2, #0x3c0]+ ldr q21, [x2, #0x3d0]+ ldr q22, [x2, #0x3e0]+ ldr q23, [x2, #0x3f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x7c0]+ ldr q17, [x1, #0x7d0]+ ldr q18, [x1, #0x7e0]+ ldr q19, [x1, #0x7f0]+ ldr q20, [x2, #0x7c0]+ ldr q21, [x2, #0x7d0]+ ldr q22, [x2, #0x7e0]+ ldr q23, [x2, #0x7f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0xbc0]+ ldr q17, [x1, #0xbd0]+ ldr q18, [x1, #0xbe0]+ ldr q19, [x1, #0xbf0]+ ldr q20, [x2, #0xbc0]+ ldr q21, [x2, #0xbd0]+ ldr q22, [x2, #0xbe0]+ ldr q23, [x2, #0xbf0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0xfc0]+ ldr q17, [x1, #0xfd0]+ ldr q18, [x1, #0xfe0]+ ldr q19, [x1, #0xff0]+ ldr q20, [x2, #0xfc0]+ ldr q21, [x2, #0xfd0]+ ldr q22, [x2, #0xfe0]+ ldr q23, [x2, #0xff0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ uzp1 v16.4s, v24.4s, v25.4s+ mul v16.4s, v16.4s, v1.4s+ smlsl v24.2d, v16.2s, v0.2s+ smlsl2 v25.2d, v16.4s, v0.4s+ uzp2 v16.4s, v24.4s, v25.4s+ uzp1 v17.4s, v26.4s, v27.4s+ mul v17.4s, v17.4s, v1.4s+ smlsl v26.2d, v17.2s, v0.2s+ smlsl2 v27.2d, v17.4s, v0.4s+ uzp2 v17.4s, v26.4s, v27.4s+ uzp1 v18.4s, v28.4s, v29.4s+ mul v18.4s, v18.4s, v1.4s+ smlsl v28.2d, v18.2s, v0.2s+ smlsl2 v29.2d, v18.4s, v0.4s+ uzp2 v18.4s, v28.4s, v29.4s+ uzp1 v19.4s, v30.4s, v31.4s+ mul v19.4s, v19.4s, v1.4s+ smlsl v30.2d, v19.2s, v0.2s+ smlsl2 v31.2d, v19.4s, v0.4s+ uzp2 v19.4s, v30.4s, v31.4s+ str q17, [x0, #0x10]+ str q18, [x0, #0x20]+ str q19, [x0, #0x30]+ str q16, [x0], #0x40+ subs x3, x3, #0x4+ cbnz x3, Lmld_polyvecl_pointwise_acc_montgomery_l5_loop_start+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(polyvecl_pointwise_acc_montgomery_l5_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 5) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S view
@@ -0,0 +1,205 @@+/* Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: polyvecl_pointwise_acc_montgomery_l7_aarch64_asm+ Description: AArch64 pointwise multiply-accumulate of length-7 polynomial vectors+ Signature: void mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm(int32_t r[256], const int32_t a[7][256], const int32_t b[7][256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 7168+ permissions: read-only+ c_parameter: const int32_t a[7][256]+ description: Input polynomial vector a (7 x 256 x int32_t)+ x2:+ type: buffer+ size_bytes: 7168+ permissions: read-only+ c_parameter: const int32_t b[7][256]+ description: Input polynomial vector b (7 x 256 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 7)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(polyvecl_pointwise_acc_montgomery_l7_aarch64_asm)+MLD_ASM_FN_SYMBOL(polyvecl_pointwise_acc_montgomery_l7_aarch64_asm)++ .cfi_startproc+ mov w3, #0xe001 // =57345+ movk w3, #0x7f, lsl #16+ dup v0.4s, w3+ mov w3, #0x2001 // =8193+ movk w3, #0x380, lsl #16+ dup v1.4s, w3+ mov x3, #0x40 // =64++Lmld_polyvecl_pointwise_acc_montgomery_l7_loop_start:+ ldr q17, [x1, #0x10]+ ldr q18, [x1, #0x20]+ ldr q19, [x1, #0x30]+ ldr q16, [x1], #0x40+ ldr q21, [x2, #0x10]+ ldr q22, [x2, #0x20]+ ldr q23, [x2, #0x30]+ ldr q20, [x2], #0x40+ smull v24.2d, v16.2s, v20.2s+ smull2 v25.2d, v16.4s, v20.4s+ smull v26.2d, v17.2s, v21.2s+ smull2 v27.2d, v17.4s, v21.4s+ smull v28.2d, v18.2s, v22.2s+ smull2 v29.2d, v18.4s, v22.4s+ smull v30.2d, v19.2s, v23.2s+ smull2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x3c0]+ ldr q17, [x1, #0x3d0]+ ldr q18, [x1, #0x3e0]+ ldr q19, [x1, #0x3f0]+ ldr q20, [x2, #0x3c0]+ ldr q21, [x2, #0x3d0]+ ldr q22, [x2, #0x3e0]+ ldr q23, [x2, #0x3f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x7c0]+ ldr q17, [x1, #0x7d0]+ ldr q18, [x1, #0x7e0]+ ldr q19, [x1, #0x7f0]+ ldr q20, [x2, #0x7c0]+ ldr q21, [x2, #0x7d0]+ ldr q22, [x2, #0x7e0]+ ldr q23, [x2, #0x7f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0xbc0]+ ldr q17, [x1, #0xbd0]+ ldr q18, [x1, #0xbe0]+ ldr q19, [x1, #0xbf0]+ ldr q20, [x2, #0xbc0]+ ldr q21, [x2, #0xbd0]+ ldr q22, [x2, #0xbe0]+ ldr q23, [x2, #0xbf0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0xfc0]+ ldr q17, [x1, #0xfd0]+ ldr q18, [x1, #0xfe0]+ ldr q19, [x1, #0xff0]+ ldr q20, [x2, #0xfc0]+ ldr q21, [x2, #0xfd0]+ ldr q22, [x2, #0xfe0]+ ldr q23, [x2, #0xff0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x13c0]+ ldr q17, [x1, #0x13d0]+ ldr q18, [x1, #0x13e0]+ ldr q19, [x1, #0x13f0]+ ldr q20, [x2, #0x13c0]+ ldr q21, [x2, #0x13d0]+ ldr q22, [x2, #0x13e0]+ ldr q23, [x2, #0x13f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x17c0]+ ldr q17, [x1, #0x17d0]+ ldr q18, [x1, #0x17e0]+ ldr q19, [x1, #0x17f0]+ ldr q20, [x2, #0x17c0]+ ldr q21, [x2, #0x17d0]+ ldr q22, [x2, #0x17e0]+ ldr q23, [x2, #0x17f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ uzp1 v16.4s, v24.4s, v25.4s+ mul v16.4s, v16.4s, v1.4s+ smlsl v24.2d, v16.2s, v0.2s+ smlsl2 v25.2d, v16.4s, v0.4s+ uzp2 v16.4s, v24.4s, v25.4s+ uzp1 v17.4s, v26.4s, v27.4s+ mul v17.4s, v17.4s, v1.4s+ smlsl v26.2d, v17.2s, v0.2s+ smlsl2 v27.2d, v17.4s, v0.4s+ uzp2 v17.4s, v26.4s, v27.4s+ uzp1 v18.4s, v28.4s, v29.4s+ mul v18.4s, v18.4s, v1.4s+ smlsl v28.2d, v18.2s, v0.2s+ smlsl2 v29.2d, v18.4s, v0.4s+ uzp2 v18.4s, v28.4s, v29.4s+ uzp1 v19.4s, v30.4s, v31.4s+ mul v19.4s, v19.4s, v1.4s+ smlsl v30.2d, v19.2s, v0.2s+ smlsl2 v31.2d, v19.4s, v0.4s+ uzp2 v19.4s, v30.4s, v31.4s+ str q17, [x0, #0x10]+ str q18, [x0, #0x20]+ str q19, [x0, #0x30]+ str q16, [x0], #0x40+ subs x3, x3, #0x4+ cbnz x3, Lmld_polyvecl_pointwise_acc_montgomery_l7_loop_start+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(polyvecl_pointwise_acc_montgomery_l7_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 7) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_polyz_unpack_17_aarch64_asm.S view
@@ -0,0 +1,103 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: polyz_unpack_17_aarch64_asm+ Description: AArch64 unpacking of 17-bit packed coefficients+ Signature: void mld_polyz_unpack_17_aarch64_asm(int32_t r[256], const uint8_t buf[576], const uint8_t indices[64])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 576+ permissions: read-only+ c_parameter: const uint8_t buf[576]+ description: Packed input bytes+ x2:+ type: buffer+ size_bytes: 64+ permissions: read-only+ c_parameter: const uint8_t indices[64]+ description: Permutation index table (64 x uint8_t)+*/++ #include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ (!defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_polyz_unpack_17_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(polyz_unpack_17_aarch64_asm)+MLD_ASM_FN_SYMBOL(polyz_unpack_17_aarch64_asm)++ .cfi_startproc+ ldr q24, [x2]+ ldr q25, [x2, #0x10]+ ldr q26, [x2, #0x20]+ ldr q27, [x2, #0x30]+ mov x3, #0xfe00000000 // =1090921693184+ mov v28.d[0], x3+ mov x3, #0xfc // =252+ movk x3, #0xfa, lsl #32+ mov v28.d[1], x3+ movi v29.4s, #0x3, msl #16+ movi v30.4s, #0x2, lsl #16+ mov x9, #0x10 // =16++Lmld_polyz_unpack_17_loop:+ ld1 { v0.16b, v1.16b }, [x1]+ add x1, x1, #0x14+ ld1 { v2.16b }, [x1], #16+ tbl v4.16b, { v0.16b }, v24.16b+ tbl v5.16b, { v0.16b, v1.16b }, v25.16b+ tbl v6.16b, { v1.16b }, v26.16b+ tbl v7.16b, { v1.16b, v2.16b }, v27.16b+ ushl v4.4s, v4.4s, v28.4s+ and v4.16b, v4.16b, v29.16b+ sub v4.4s, v30.4s, v4.4s+ ushl v5.4s, v5.4s, v28.4s+ and v5.16b, v5.16b, v29.16b+ sub v5.4s, v30.4s, v5.4s+ ushl v6.4s, v6.4s, v28.4s+ and v6.16b, v6.16b, v29.16b+ sub v6.4s, v30.4s, v6.4s+ ushl v7.4s, v7.4s, v28.4s+ and v7.16b, v7.16b, v29.16b+ sub v7.4s, v30.4s, v7.4s+ str q5, [x0, #0x10]+ str q6, [x0, #0x20]+ str q7, [x0, #0x30]+ str q4, [x0], #0x40+ subs x9, x9, #0x1+ b.ne Lmld_polyz_unpack_17_loop+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(polyz_unpack_17_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && (!MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_polyz_unpack_19_aarch64_asm.S view
@@ -0,0 +1,100 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: polyz_unpack_19_aarch64_asm+ Description: AArch64 unpacking of 19-bit packed coefficients+ Signature: void mld_polyz_unpack_19_aarch64_asm(int32_t r[256], const uint8_t buf[640], const uint8_t indices[64])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 640+ permissions: read-only+ c_parameter: const uint8_t buf[640]+ description: Packed input bytes+ x2:+ type: buffer+ size_bytes: 64+ permissions: read-only+ c_parameter: const uint8_t indices[64]+ description: Permutation index table (64 x uint8_t)+*/++ #include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ (!defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_polyz_unpack_19_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(polyz_unpack_19_aarch64_asm)+MLD_ASM_FN_SYMBOL(polyz_unpack_19_aarch64_asm)++ .cfi_startproc+ ldr q24, [x2]+ ldr q25, [x2, #0x10]+ ldr q26, [x2, #0x20]+ ldr q27, [x2, #0x30]+ mov x3, #0xfc00000000 // =1082331758592+ dup v28.2d, x3+ movi v29.4s, #0xf, msl #16+ movi v30.4s, #0x8, lsl #16+ mov x9, #0x10 // =16++Lmld_polyz_unpack_19_loop:+ ld1 { v0.16b, v1.16b }, [x1]+ add x1, x1, #0x18+ ld1 { v2.16b }, [x1], #16+ tbl v4.16b, { v0.16b }, v24.16b+ tbl v5.16b, { v0.16b, v1.16b }, v25.16b+ tbl v6.16b, { v1.16b }, v26.16b+ tbl v7.16b, { v1.16b, v2.16b }, v27.16b+ ushl v4.4s, v4.4s, v28.4s+ and v4.16b, v4.16b, v29.16b+ sub v4.4s, v30.4s, v4.4s+ ushl v5.4s, v5.4s, v28.4s+ and v5.16b, v5.16b, v29.16b+ sub v5.4s, v30.4s, v5.4s+ ushl v6.4s, v6.4s, v28.4s+ and v6.16b, v6.16b, v29.16b+ sub v6.4s, v30.4s, v6.4s+ ushl v7.4s, v7.4s, v28.4s+ and v7.16b, v7.16b, v29.16b+ sub v7.4s, v30.4s, v7.4s+ str q5, [x0, #0x10]+ str q6, [x0, #0x20]+ str q7, [x0, #0x30]+ str q4, [x0], #0x40+ subs x9, x9, #0x1+ b.ne Lmld_polyz_unpack_19_loop+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(polyz_unpack_19_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && (!MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_rej_uniform_aarch64_asm.S view
@@ -0,0 +1,222 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: rej_uniform_aarch64_asm+ Description: AArch64 rejection sampling of uniform coefficients mod q+ Signature: uint64_t mld_rej_uniform_aarch64_asm(int32_t r[256], const uint8_t *buf, unsigned buflen, const uint8_t table[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output buffer (256 x int32_t)+ x1:+ type: buffer+ size_bytes: x2+ permissions: read-only+ c_parameter: const uint8_t *buf+ description: Input buffer+ x2:+ type: scalar+ c_parameter: unsigned buflen+ description: Length of input buffer (must be a multiple of 24)+ test_with: 840 # MLD_POLY_UNIFORM_NBLOCKS * SHAKE128_RATE = 5 * 168+ x3:+ type: buffer+ size_bytes: 256+ permissions: read-only+ c_parameter: const uint8_t table[256]+ description: Lookup table (256 x uint8_t)+*/++ #include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_rej_uniform_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(rej_uniform_aarch64_asm)+MLD_ASM_FN_SYMBOL(rej_uniform_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x440+ .cfi_adjust_cfa_offset 0x440+ mov x7, #0x1 // =1+ movk x7, #0x2, lsl #32+ mov v31.d[0], x7+ mov x7, #0x4 // =4+ movk x7, #0x8, lsl #32+ mov v31.d[1], x7+ mov w7, #0xe001 // =57345+ movk w7, #0x7f, lsl #16+ dup v30.4s, w7+ mov x8, sp+ mov x7, x8+ mov x11, #0x0 // =0+ eor v16.16b, v16.16b, v16.16b++Lmld_rej_uniform_initial_zero:+ str q16, [x7], #0x40+ stur q16, [x7, #-0x30]+ stur q16, [x7, #-0x20]+ stur q16, [x7, #-0x10]+ add x11, x11, #0x10+ cmp x11, #0x100+ b.lt Lmld_rej_uniform_initial_zero+ mov x7, x8+ mov x9, #0x0 // =0+ mov x4, #0x100 // =256+ cmp x2, #0x30+ b.lo Lmld_rej_uniform_loop48_end++Lmld_rej_uniform_loop48:+ cmp x9, x4+ b.hs Lmld_rej_uniform_memory_copy+ sub x2, x2, #0x30+ ld3 { v0.16b, v1.16b, v2.16b }, [x1], #48+ movi v4.16b, #0x80+ bic v2.16b, v2.16b, v4.16b+ zip1 v4.16b, v0.16b, v1.16b+ zip2 v5.16b, v0.16b, v1.16b+ ushll v6.8h, v2.8b, #0x0+ ushll2 v7.8h, v2.16b, #0x0+ zip1 v16.8h, v4.8h, v6.8h+ zip2 v17.8h, v4.8h, v6.8h+ zip1 v18.8h, v5.8h, v7.8h+ zip2 v19.8h, v5.8h, v7.8h+ cmhi v4.4s, v30.4s, v16.4s+ cmhi v5.4s, v30.4s, v17.4s+ cmhi v6.4s, v30.4s, v18.4s+ cmhi v7.4s, v30.4s, v19.4s+ and v4.16b, v4.16b, v31.16b+ and v5.16b, v5.16b, v31.16b+ and v6.16b, v6.16b, v31.16b+ and v7.16b, v7.16b, v31.16b+ uaddlv d20, v4.4s+ uaddlv d21, v5.4s+ uaddlv d22, v6.4s+ uaddlv d23, v7.4s+ fmov x12, d20+ fmov x13, d21+ fmov x14, d22+ fmov x15, d23+ ldr q24, [x3, x12, lsl #4]+ ldr q25, [x3, x13, lsl #4]+ ldr q26, [x3, x14, lsl #4]+ ldr q27, [x3, x15, lsl #4]+ cnt v4.16b, v4.16b+ cnt v5.16b, v5.16b+ cnt v6.16b, v6.16b+ cnt v7.16b, v7.16b+ uaddlv d20, v4.4s+ uaddlv d21, v5.4s+ uaddlv d22, v6.4s+ uaddlv d23, v7.4s+ fmov x12, d20+ fmov x13, d21+ fmov x14, d22+ fmov x15, d23+ tbl v16.16b, { v16.16b }, v24.16b+ tbl v17.16b, { v17.16b }, v25.16b+ tbl v18.16b, { v18.16b }, v26.16b+ tbl v19.16b, { v19.16b }, v27.16b+ st1 { v16.4s }, [x7]+ add x7, x7, x12, lsl #2+ st1 { v17.4s }, [x7]+ add x7, x7, x13, lsl #2+ st1 { v18.4s }, [x7]+ add x7, x7, x14, lsl #2+ st1 { v19.4s }, [x7]+ add x7, x7, x15, lsl #2+ add x12, x12, x13+ add x14, x14, x15+ add x9, x9, x12+ add x9, x9, x14+ cmp x2, #0x30+ b.hs Lmld_rej_uniform_loop48++Lmld_rej_uniform_loop48_end:+ cmp x9, x4+ b.hs Lmld_rej_uniform_memory_copy+ cmp x2, #0x18+ b.lo Lmld_rej_uniform_memory_copy+ sub x2, x2, #0x18+ ld3 { v0.8b, v1.8b, v2.8b }, [x1], #24+ movi v4.16b, #0x80+ bic v2.16b, v2.16b, v4.16b+ zip1 v4.16b, v0.16b, v1.16b+ ushll v6.8h, v2.8b, #0x0+ zip1 v16.8h, v4.8h, v6.8h+ zip2 v17.8h, v4.8h, v6.8h+ cmhi v4.4s, v30.4s, v16.4s+ cmhi v5.4s, v30.4s, v17.4s+ and v4.16b, v4.16b, v31.16b+ and v5.16b, v5.16b, v31.16b+ uaddlv d20, v4.4s+ uaddlv d21, v5.4s+ fmov x12, d20+ fmov x13, d21+ ldr q24, [x3, x12, lsl #4]+ ldr q25, [x3, x13, lsl #4]+ cnt v4.16b, v4.16b+ cnt v5.16b, v5.16b+ uaddlv d20, v4.4s+ uaddlv d21, v5.4s+ fmov x12, d20+ fmov x13, d21+ tbl v16.16b, { v16.16b }, v24.16b+ tbl v17.16b, { v17.16b }, v25.16b+ st1 { v16.4s }, [x7]+ add x7, x7, x12, lsl #2+ st1 { v17.4s }, [x7]+ add x7, x7, x13, lsl #2+ add x9, x9, x12+ add x9, x9, x13++Lmld_rej_uniform_memory_copy:+ cmp x9, x4+ csel x9, x9, x4, lo+ mov x11, #0x0 // =0+ mov x7, x8++Lmld_rej_uniform_final_copy:+ ldr q16, [x7], #0x40+ ldur q17, [x7, #-0x30]+ ldur q18, [x7, #-0x20]+ ldur q19, [x7, #-0x10]+ str q16, [x0], #0x40+ stur q17, [x0, #-0x30]+ stur q18, [x0, #-0x20]+ stur q19, [x0, #-0x10]+ add x11, x11, #0x10+ cmp x11, #0x100+ b.lt Lmld_rej_uniform_final_copy+ mov x0, x9+ b Lmld_rej_uniform_return++Lmld_rej_uniform_return:+ add sp, sp, #0x440+ .cfi_adjust_cfa_offset -0x440+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(rej_uniform_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_rej_uniform_eta2_aarch64_asm.S view
@@ -0,0 +1,170 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: rej_uniform_eta2_aarch64_asm+ Description: AArch64 rejection sampling of eta=2 secret coefficients+ Signature: uint64_t mld_rej_uniform_eta2_aarch64_asm(int32_t r[256], const uint8_t *buf, unsigned buflen, const uint8_t table[4096])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output buffer (256 x int32_t)+ x1:+ type: buffer+ size_bytes: x2+ permissions: read-only+ c_parameter: const uint8_t *buf+ description: Input buffer+ x2:+ type: scalar+ c_parameter: unsigned buflen+ description: Length of input buffer (must be a multiple of 8)+ test_with: 136 # MLD_AARCH64_REJ_UNIFORM_ETA2_BUFLEN+ x3:+ type: buffer+ size_bytes: 4096+ permissions: read-only+ c_parameter: const uint8_t table[4096]+ description: Lookup table (4096 x uint8_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ !defined(MLD_CONFIG_NO_KEYPAIR_API) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 2)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_rej_uniform_eta2_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(rej_uniform_eta2_aarch64_asm)+MLD_ASM_FN_SYMBOL(rej_uniform_eta2_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x240+ .cfi_adjust_cfa_offset 0x240+ mov x7, #0x1 // =1+ movk x7, #0x2, lsl #16+ movk x7, #0x4, lsl #32+ movk x7, #0x8, lsl #48+ mov v31.d[0], x7+ mov x7, #0x10 // =16+ movk x7, #0x20, lsl #16+ movk x7, #0x40, lsl #32+ movk x7, #0x80, lsl #48+ mov v31.d[1], x7+ movi v30.8h, #0xf+ mov x8, sp+ mov x7, x8+ mov x11, #0x0 // =0+ eor v16.16b, v16.16b, v16.16b++Lmld_rej_uniform_eta2_initial_zero:+ str q16, [x7], #0x40+ stur q16, [x7, #-0x30]+ stur q16, [x7, #-0x20]+ stur q16, [x7, #-0x10]+ add x11, x11, #0x20+ cmp x11, #0x100+ b.lt Lmld_rej_uniform_eta2_initial_zero+ mov x7, x8+ mov x9, #0x0 // =0+ mov x4, #0x100 // =256++Lmld_rej_uniform_eta2_loop8:+ cmp x9, x4+ b.hs Lmld_rej_uniform_eta2_memory_copy+ sub x2, x2, #0x8+ ld1 { v0.8b }, [x1], #8+ movi v26.8b, #0xf+ and v27.8b, v0.8b, v26.8b+ ushr v28.8b, v0.8b, #0x4+ zip1 v26.8b, v27.8b, v28.8b+ zip2 v29.8b, v27.8b, v28.8b+ ushll v16.8h, v26.8b, #0x0+ ushll v17.8h, v29.8b, #0x0+ cmhi v4.8h, v30.8h, v16.8h+ cmhi v5.8h, v30.8h, v17.8h+ and v4.16b, v4.16b, v31.16b+ and v5.16b, v5.16b, v31.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ fmov w12, s20+ fmov w13, s21+ ldr q24, [x3, x12, lsl #4]+ ldr q25, [x3, x13, lsl #4]+ cnt v4.16b, v4.16b+ cnt v5.16b, v5.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ fmov w12, s20+ fmov w13, s21+ tbl v16.16b, { v16.16b }, v24.16b+ tbl v17.16b, { v17.16b }, v25.16b+ st1 { v16.8h }, [x7]+ add x7, x7, x12, lsl #1+ st1 { v17.8h }, [x7]+ add x7, x7, x13, lsl #1+ add x12, x12, x13+ add x9, x9, x12+ cmp x2, #0x8+ b.hs Lmld_rej_uniform_eta2_loop8++Lmld_rej_uniform_eta2_memory_copy:+ cmp x9, x4+ csel x9, x9, x4, lo+ mov w7, #0x199a // =6554+ dup v26.8h, w7+ movi v27.8h, #0x5+ movi v7.8h, #0x2+ mov x11, #0x0 // =0+ mov x7, x8++Lmld_rej_uniform_eta2_final_copy:+ ldr q16, [x7], #0x20+ ldur q18, [x7, #-0x10]+ sqdmulh v28.8h, v16.8h, v26.8h+ mls v16.8h, v28.8h, v27.8h+ sqdmulh v28.8h, v18.8h, v26.8h+ mls v18.8h, v28.8h, v27.8h+ sub v16.8h, v7.8h, v16.8h+ sub v18.8h, v7.8h, v18.8h+ sshll2 v17.4s, v16.8h, #0x0+ sshll v16.4s, v16.4h, #0x0+ sshll2 v19.4s, v18.8h, #0x0+ sshll v18.4s, v18.4h, #0x0+ str q16, [x0], #0x40+ stur q17, [x0, #-0x30]+ stur q18, [x0, #-0x20]+ stur q19, [x0, #-0x10]+ add x11, x11, #0x10+ cmp x11, #0x100+ b.lt Lmld_rej_uniform_eta2_final_copy+ mov x0, x9+ add sp, sp, #0x240+ .cfi_adjust_cfa_offset -0x240+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(rej_uniform_eta2_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_KEYPAIR_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/mldsa_rej_uniform_eta4_aarch64_asm.S view
@@ -0,0 +1,163 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: rej_uniform_eta4_aarch64_asm+ Description: AArch64 rejection sampling of eta=4 secret coefficients+ Signature: uint64_t mld_rej_uniform_eta4_aarch64_asm(int32_t r[256], const uint8_t *buf, unsigned buflen, const uint8_t table[4096])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output buffer (256 x int32_t)+ x1:+ type: buffer+ size_bytes: x2+ permissions: read-only+ c_parameter: const uint8_t *buf+ description: Input buffer+ x2:+ type: scalar+ c_parameter: unsigned buflen+ description: Length of input buffer (must be a multiple of 8)+ test_with: 272 # MLD_AARCH64_REJ_UNIFORM_ETA4_BUFLEN+ x3:+ type: buffer+ size_bytes: 4096+ permissions: read-only+ c_parameter: const uint8_t table[4096]+ description: Lookup table (4096 x uint8_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ !defined(MLD_CONFIG_NO_KEYPAIR_API) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 4)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_rej_uniform_eta4_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(rej_uniform_eta4_aarch64_asm)+MLD_ASM_FN_SYMBOL(rej_uniform_eta4_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x240+ .cfi_adjust_cfa_offset 0x240+ mov x7, #0x1 // =1+ movk x7, #0x2, lsl #16+ movk x7, #0x4, lsl #32+ movk x7, #0x8, lsl #48+ mov v31.d[0], x7+ mov x7, #0x10 // =16+ movk x7, #0x20, lsl #16+ movk x7, #0x40, lsl #32+ movk x7, #0x80, lsl #48+ mov v31.d[1], x7+ movi v30.8h, #0x9+ movi v7.8h, #0x4+ mov x8, sp+ mov x7, x8+ mov x11, #0x0 // =0+ eor v16.16b, v16.16b, v16.16b++Lmld_rej_uniform_eta4_initial_zero:+ str q16, [x7], #0x40+ stur q16, [x7, #-0x30]+ stur q16, [x7, #-0x20]+ stur q16, [x7, #-0x10]+ add x11, x11, #0x20+ cmp x11, #0x100+ b.lt Lmld_rej_uniform_eta4_initial_zero+ mov x7, x8+ mov x9, #0x0 // =0+ mov x4, #0x100 // =256++Lmld_rej_uniform_eta4_loop8:+ cmp x9, x4+ b.hs Lmld_rej_uniform_eta4_memory_copy+ sub x2, x2, #0x8+ ld1 { v0.8b }, [x1], #8+ movi v26.8b, #0xf+ and v27.8b, v0.8b, v26.8b+ ushr v28.8b, v0.8b, #0x4+ zip1 v26.8b, v27.8b, v28.8b+ zip2 v29.8b, v27.8b, v28.8b+ ushll v16.8h, v26.8b, #0x0+ ushll v17.8h, v29.8b, #0x0+ cmhi v4.8h, v30.8h, v16.8h+ cmhi v5.8h, v30.8h, v17.8h+ and v4.16b, v4.16b, v31.16b+ and v5.16b, v5.16b, v31.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ fmov w12, s20+ fmov w13, s21+ ldr q24, [x3, x12, lsl #4]+ ldr q25, [x3, x13, lsl #4]+ cnt v4.16b, v4.16b+ cnt v5.16b, v5.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ fmov w12, s20+ fmov w13, s21+ tbl v16.16b, { v16.16b }, v24.16b+ tbl v17.16b, { v17.16b }, v25.16b+ st1 { v16.8h }, [x7]+ add x7, x7, x12, lsl #1+ st1 { v17.8h }, [x7]+ add x7, x7, x13, lsl #1+ add x12, x12, x13+ add x9, x9, x12+ cmp x2, #0x8+ b.hs Lmld_rej_uniform_eta4_loop8++Lmld_rej_uniform_eta4_memory_copy:+ cmp x9, x4+ csel x9, x9, x4, lo+ mov x11, #0x0 // =0+ mov x7, x8++Lmld_rej_uniform_eta4_final_copy:+ ldr q16, [x7], #0x20+ ldur q18, [x7, #-0x10]+ sub v16.8h, v7.8h, v16.8h+ sub v18.8h, v7.8h, v18.8h+ sshll2 v17.4s, v16.8h, #0x0+ sshll v16.4s, v16.4h, #0x0+ sshll2 v19.4s, v18.8h, #0x0+ sshll v18.4s, v18.4h, #0x0+ str q16, [x0], #0x40+ stur q17, [x0, #-0x30]+ stur q18, [x0, #-0x20]+ stur q19, [x0, #-0x10]+ add x11, x11, #0x10+ cmp x11, #0x100+ b.lt Lmld_rej_uniform_eta4_final_copy+ mov x0, x9+ add sp, sp, #0x240+ .cfi_adjust_cfa_offset -0x240+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(rej_uniform_eta4_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_KEYPAIR_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 4) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/aarch64/src/polyz_unpack_table.c view
@@ -0,0 +1,52 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include "arith_native_aarch64.h"++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+/* Table of indices used for tbl instructions in polyz_unpack_{17,19}.+ * See autogen for details. */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint8_t+ mld_polyz_unpack_17_indices[64] = {+ 0, 1, 2, 255, 2, 3, 4, 255, 4, 5, 6, 255, 6, 7, 8, 255,+ 9, 10, 11, 255, 11, 12, 13, 255, 13, 14, 15, 255, 15, 16, 17, 255,+ 2, 3, 4, 255, 4, 5, 6, 255, 6, 7, 8, 255, 8, 9, 10, 255,+ 11, 12, 13, 255, 13, 14, 15, 255, 15, 28, 29, 255, 29, 30, 31, 255,+};+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint8_t+ mld_polyz_unpack_19_indices[64] = {+ 0, 1, 2, 255, 2, 3, 4, 255, 5, 6, 7, 255, 7, 8, 9, 255,+ 10, 11, 12, 255, 12, 13, 14, 255, 15, 16, 17, 255, 17, 18, 19, 255,+ 4, 5, 6, 255, 6, 7, 8, 255, 9, 10, 11, 255, 11, 12, 13, 255,+ 14, 15, 24, 255, 24, 25, 26, 255, 27, 28, 29, 255, 29, 30, 31, 255,+};+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */++#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#else /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(aarch64_polyz_unpack_table)++#endif /* !(MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mldsa/src/native/aarch64/src/rej_uniform_eta_table.c view
@@ -0,0 +1,547 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ !defined(MLD_CONFIG_NO_KEYPAIR_API) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include "arith_native_aarch64.h"++/*+ * Lookup table used by 16-bit rejection sampling (rej_eta).+ * Adapted from ML-KEM for ML-DSA eta rejection sampling.+ * See autogen for details.+ */+MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint8_t+ mld_rej_uniform_eta_table[4096] = {+ 255, 255, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 0 */,+ 0, 1, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 1 */,+ 2, 3, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 2 */,+ 0, 1, 2, 3, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 3 */,+ 4, 5, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 4 */,+ 0, 1, 4, 5, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 5 */,+ 2, 3, 4, 5, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 6 */,+ 0, 1, 2, 3, 4, 5, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 7 */,+ 6, 7, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 8 */,+ 0, 1, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 9 */,+ 2, 3, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 10 */,+ 0, 1, 2, 3, 6, 7, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 11 */,+ 4, 5, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 12 */,+ 0, 1, 4, 5, 6, 7, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 13 */,+ 2, 3, 4, 5, 6, 7, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 14 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 15 */,+ 8, 9, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 16 */,+ 0, 1, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 17 */,+ 2, 3, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 18 */,+ 0, 1, 2, 3, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 19 */,+ 4, 5, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 20 */,+ 0, 1, 4, 5, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 21 */,+ 2, 3, 4, 5, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 22 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 23 */,+ 6, 7, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 24 */,+ 0, 1, 6, 7, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 25 */,+ 2, 3, 6, 7, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 26 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 27 */,+ 4, 5, 6, 7, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 28 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 29 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 30 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 255, 255, 255, 255, 255, 255 /* 31 */,+ 10, 11, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 32 */,+ 0, 1, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 33 */,+ 2, 3, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 34 */,+ 0, 1, 2, 3, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 35 */,+ 4, 5, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 36 */,+ 0, 1, 4, 5, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 37 */,+ 2, 3, 4, 5, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 38 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 39 */,+ 6, 7, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 40 */,+ 0, 1, 6, 7, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 41 */,+ 2, 3, 6, 7, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 42 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 43 */,+ 4, 5, 6, 7, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 44 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 45 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 46 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 47 */,+ 8, 9, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 48 */,+ 0, 1, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 49 */,+ 2, 3, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 50 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 51 */,+ 4, 5, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 52 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 53 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 54 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 55 */,+ 6, 7, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 56 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 57 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 58 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 59 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 60 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 61 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 62 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 255, 255, 255, 255 /* 63 */,+ 12, 13, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 64 */,+ 0, 1, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 65 */,+ 2, 3, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 66 */,+ 0, 1, 2, 3, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 67 */,+ 4, 5, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 68 */,+ 0, 1, 4, 5, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 69 */,+ 2, 3, 4, 5, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 70 */,+ 0, 1, 2, 3, 4, 5, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 71 */,+ 6, 7, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 72 */,+ 0, 1, 6, 7, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 73 */,+ 2, 3, 6, 7, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 74 */,+ 0, 1, 2, 3, 6, 7, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 75 */,+ 4, 5, 6, 7, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 76 */,+ 0, 1, 4, 5, 6, 7, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 77 */,+ 2, 3, 4, 5, 6, 7, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 78 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 79 */,+ 8, 9, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 80 */,+ 0, 1, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 81 */,+ 2, 3, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 82 */,+ 0, 1, 2, 3, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 83 */,+ 4, 5, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 84 */,+ 0, 1, 4, 5, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 85 */,+ 2, 3, 4, 5, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 86 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 87 */,+ 6, 7, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 88 */,+ 0, 1, 6, 7, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 89 */,+ 2, 3, 6, 7, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 90 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 91 */,+ 4, 5, 6, 7, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 92 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 93 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 94 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 12, 13, 255, 255, 255, 255 /* 95 */,+ 10, 11, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 96 */,+ 0, 1, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 97 */,+ 2, 3, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 98 */,+ 0, 1, 2, 3, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 99 */,+ 4, 5, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 100 */,+ 0, 1, 4, 5, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 101 */,+ 2, 3, 4, 5, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 102 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 103 */,+ 6, 7, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 104 */,+ 0, 1, 6, 7, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 105 */,+ 2, 3, 6, 7, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 106 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 107 */,+ 4, 5, 6, 7, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 108 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 109 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 110 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 111 */,+ 8, 9, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 112 */,+ 0, 1, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 113 */,+ 2, 3, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 114 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 115 */,+ 4, 5, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 116 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 117 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 118 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 119 */,+ 6, 7, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 120 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 121 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 122 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 123 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 124 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 125 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 126 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 12, 13, 255, 255 /* 127 */,+ 14, 15, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 128 */,+ 0, 1, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 129 */,+ 2, 3, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 130 */,+ 0, 1, 2, 3, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 131 */,+ 4, 5, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 132 */,+ 0, 1, 4, 5, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 133 */,+ 2, 3, 4, 5, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 134 */,+ 0, 1, 2, 3, 4, 5, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 135 */,+ 6, 7, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 136 */,+ 0, 1, 6, 7, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 137 */,+ 2, 3, 6, 7, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 138 */,+ 0, 1, 2, 3, 6, 7, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 139 */,+ 4, 5, 6, 7, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 140 */,+ 0, 1, 4, 5, 6, 7, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 141 */,+ 2, 3, 4, 5, 6, 7, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 142 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 143 */,+ 8, 9, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 144 */,+ 0, 1, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 145 */,+ 2, 3, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 146 */,+ 0, 1, 2, 3, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 147 */,+ 4, 5, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 148 */,+ 0, 1, 4, 5, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 149 */,+ 2, 3, 4, 5, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 150 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 151 */,+ 6, 7, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 152 */,+ 0, 1, 6, 7, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 153 */,+ 2, 3, 6, 7, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 154 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 155 */,+ 4, 5, 6, 7, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 156 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 157 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 158 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 14, 15, 255, 255, 255, 255 /* 159 */,+ 10, 11, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 160 */,+ 0, 1, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 161 */,+ 2, 3, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 162 */,+ 0, 1, 2, 3, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 163 */,+ 4, 5, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 164 */,+ 0, 1, 4, 5, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 165 */,+ 2, 3, 4, 5, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 166 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 167 */,+ 6, 7, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 168 */,+ 0, 1, 6, 7, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 169 */,+ 2, 3, 6, 7, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 170 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 171 */,+ 4, 5, 6, 7, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 172 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 173 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 174 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 175 */,+ 8, 9, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 176 */,+ 0, 1, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 177 */,+ 2, 3, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 178 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 179 */,+ 4, 5, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 180 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 181 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 182 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 183 */,+ 6, 7, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 184 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 185 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 186 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 187 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 188 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 189 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 190 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 14, 15, 255, 255 /* 191 */,+ 12, 13, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 192 */,+ 0, 1, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 193 */,+ 2, 3, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 194 */,+ 0, 1, 2, 3, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 195 */,+ 4, 5, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 196 */,+ 0, 1, 4, 5, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 197 */,+ 2, 3, 4, 5, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 198 */,+ 0, 1, 2, 3, 4, 5, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 199 */,+ 6, 7, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 200 */,+ 0, 1, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 201 */,+ 2, 3, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 202 */,+ 0, 1, 2, 3, 6, 7, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 203 */,+ 4, 5, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 204 */,+ 0, 1, 4, 5, 6, 7, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 205 */,+ 2, 3, 4, 5, 6, 7, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 206 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 207 */,+ 8, 9, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 208 */,+ 0, 1, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 209 */,+ 2, 3, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 210 */,+ 0, 1, 2, 3, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 211 */,+ 4, 5, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 212 */,+ 0, 1, 4, 5, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 213 */,+ 2, 3, 4, 5, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 214 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 215 */,+ 6, 7, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 216 */,+ 0, 1, 6, 7, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 217 */,+ 2, 3, 6, 7, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 218 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 219 */,+ 4, 5, 6, 7, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 220 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 221 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 222 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 12, 13, 14, 15, 255, 255 /* 223 */,+ 10, 11, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 224 */,+ 0, 1, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 225 */,+ 2, 3, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 226 */,+ 0, 1, 2, 3, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 227 */,+ 4, 5, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 228 */,+ 0, 1, 4, 5, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 229 */,+ 2, 3, 4, 5, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 230 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 231 */,+ 6, 7, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 232 */,+ 0, 1, 6, 7, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 233 */,+ 2, 3, 6, 7, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 234 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 235 */,+ 4, 5, 6, 7, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 236 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 237 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 238 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 239 */,+ 8, 9, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 240 */,+ 0, 1, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 241 */,+ 2, 3, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 242 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 243 */,+ 4, 5, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 244 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 245 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 246 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 247 */,+ 6, 7, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 248 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 249 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 250 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 251 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 252 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 253 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 254 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 12, 13, 14, 15 /* 255 */,+};++#else /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_KEYPAIR_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(aarch64_rej_uniform_eta_table)++#endif /* !(MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_KEYPAIR_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mldsa/src/native/aarch64/src/rej_uniform_table.c view
@@ -0,0 +1,63 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include "arith_native_aarch64.h"++/*+ * Lookup table used by rejection sampling of the public matrix.+ * See autogen for details.+ */+MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint8_t+ mld_rej_uniform_table[256] = {+ 255, 255, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 0 */,+ 0, 1, 2, 3, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 1 */,+ 4, 5, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 2 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 3 */,+ 8, 9, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 4 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 5 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 6 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 255, 255, 255, 255 /* 7 */,+ 12, 13, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 8 */,+ 0, 1, 2, 3, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 9 */,+ 4, 5, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 10 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 11 */,+ 8, 9, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 12 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 13 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 14 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 12, 13, 14, 15 /* 15 */,+};++#else /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(aarch64_rej_uniform_table)++#endif /* !(MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mldsa/src/native/api.h view
@@ -0,0 +1,617 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_NATIVE_API_H+#define MLD_NATIVE_API_H+/*+ * Native arithmetic interface+ *+ * This header is primarily for documentation purposes.+ * It should not be included by backend implementations.+ *+ * To ensure consistency with backends, the header will be+ * included automatically after inclusion of the active+ * backend, to ensure consistency of function signatures,+ * and run sanity checks.+ */++#include "../cbmc.h"+#include "../common.h"++/* Backends must return MLD_NATIVE_FUNC_SUCCESS upon success. */+#define MLD_NATIVE_FUNC_SUCCESS (0)+/* Backends may return MLD_NATIVE_FUNC_FALLBACK to signal to the frontend that+ * the target/parameters are unsupported; typically, this would be because of+ * dependencies on CPU features not detected on the host CPU. In this case,+ * the frontend falls back to the default C implementation.+ *+ * IMPORTANT: Backend implementations must ensure that the decision of whether+ * to fallback (return MLD_NATIVE_FUNC_FALLBACK) or not must never depend on+ * the input data itself. Fallback decisions may only depend on system+ * capabilities (e.g., CPU features) and, where present, length information.+ * This requirement applies to all backend functions to maintain constant-time+ * properties.+ */+#define MLD_NATIVE_FUNC_FALLBACK (-1)++/* Absolute exclusive upper bound for the output of fqmul.+ *+ * NOTE: This is the same bound as in poly.h and has to be kept+ * in sync. */+#define MLD_FQMUL_BOUND ((5 * MLDSA_Q + 3) / 4)++/* Bound on absolute value of coefficients after NTT.+ *+ * NOTE: This is the same bound as in poly.h and has to be kept+ * in sync. */+#define MLD_NTT_BOUND (9 * MLD_FQMUL_BOUND)++/* Absolute exclusive upper bound for the output of the inverse NTT+ *+ * NOTE: This is the same bound as in poly.h and has to be kept+ * in sync. */+#define MLD_INTT_BOUND MLDSA_Q++/* Absolute bound for range of mld_reduce32()+ *+ * NOTE: This is the same bound as in reduce.h and has to be kept+ * in sync. */+/* check-magic: 6283009 == (MLD_REDUCE32_DOMAIN_MAX - 255 * MLDSA_Q + 1) */+#define MLD_REDUCE32_RANGE_MAX 6283009+/*+ * This is the C<->native interface allowing for the drop-in of+ * native code for performance-critical arithmetic components of ML-DSA.+ *+ * A _backend_ is a specific implementation of (part of) this interface.+ *+ * To add a function to a backend, define MLD_USE_NATIVE_XXX and+ * implement `static inline xxx(...)` in the profile header.+ */++/*+ * Those functions are meant to be trivial wrappers around the chosen native+ * implementation. The are static inline to avoid unnecessary calls.+ * The macro before each declaration controls whether a native+ * implementation is present.+ */++#if defined(MLD_USE_NATIVE_NTT)+/**+ * Computes negacyclic number-theoretic transform (NTT) of a polynomial+ * in place.+ *+ * The input polynomial is assumed to be in normal order. The output+ * polynomial is in bitreversed order.+ *+ * @param[in,out] p Pointer to in/output polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_ntt_native(int32_t p[MLDSA_N])+__contract__(+ requires(memory_no_alias(p, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(p, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(p, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(p, 0, MLDSA_N, MLD_NTT_BOUND))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_abs_bound(p, 0, MLDSA_N, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(p, MLDSA_N))+);+#endif /* MLD_USE_NATIVE_NTT */+++#if defined(MLD_USE_NATIVE_NTT_CUSTOM_ORDER)+/*+ * This must only be set if NTT and INTT have native implementations+ * that are adapted to the custom order.+ */+#if !defined(MLD_USE_NATIVE_NTT) || !defined(MLD_USE_NATIVE_INTT)+#error \+ "Invalid native profile: MLD_USE_NATIVE_NTT_CUSTOM_ORDER can only be \+set if there are native implementations for NTT and INTT."+#endif++/**+ * When MLD_USE_NATIVE_NTT_CUSTOM_ORDER is defined, convert a polynomial in+ * NTT domain from bitreversed order to the custom order output by the native+ * NTT.+ *+ * This must only be defined if there is native code for both the NTT and+ * INTT.+ *+ * @param[in,out] p Pointer to in/output polynomial.+ */+static MLD_INLINE void mld_poly_permute_bitrev_to_custom(int32_t p[MLDSA_N])+__contract__(+ /* We don't specify that this should be a permutation, but only+ * that it does not change the bound established at the end of+ * mld_polyvec_matrix_expand.+ */+ requires(memory_no_alias(p, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(p, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(p, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(p, 0, MLDSA_N, 0, MLDSA_Q)));+#endif /* MLD_USE_NATIVE_NTT_CUSTOM_ORDER */+++#if defined(MLD_USE_NATIVE_INTT)+/**+ * Computes inverse of negacyclic number-theoretic transform (NTT) of a+ * polynomial in place.+ *+ * The input polynomial is in bitreversed order. The output polynomial is+ * assumed to be in normal order.+ *+ * @param[in,out] p Pointer to in/output polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_intt_native(int32_t p[MLDSA_N])+__contract__(+ requires(memory_no_alias(p, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(p, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(p, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(p, 0, MLDSA_N, MLD_INTT_BOUND))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_abs_bound(p, 0, MLDSA_N, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(p, MLDSA_N))+);+#endif /* MLD_USE_NATIVE_INTT */++#if defined(MLD_USE_NATIVE_REJ_UNIFORM)+/**+ * Run rejection sampling on uniform random bytes to generate uniform random+ * integers in [0, MLDSA_Q-1].+ *+ * @param[out] r Pointer to output buffer.+ * @param len Requested number of 32-bit integers (uniform mod+ * MLDSA_Q).+ * @param[in] buf Pointer to input buffer (assumed to be uniform random+ * bytes).+ * @param buflen Length of input buffer in bytes.+ *+ * @return - MLD_NATIVE_FUNC_FALLBACK if the native implementation does not+ * support the input lengths.+ * - Otherwise, the non-negative number of sampled 32-bit integers+ * (at most len).+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+__contract__(+ requires(len <= MLDSA_N)+ requires(buflen <= ( 5 * 168) && buflen % 3 == 0)+ requires(memory_no_alias(r, sizeof(int32_t) * len))+ requires(memory_no_alias(buf, buflen))+ assigns(memory_slice(r, sizeof(int32_t) * len))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || (0 <= return_value && return_value <= len))+ ensures((return_value != MLD_NATIVE_FUNC_FALLBACK) ==> array_bound(r, 0, (unsigned) return_value, 0, MLDSA_Q))+);+#endif /* MLD_USE_NATIVE_REJ_UNIFORM */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#if defined(MLD_USE_NATIVE_REJ_UNIFORM_ETA2)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 2+/**+ * Run rejection sampling on uniform random bytes to generate uniform random+ * integers in [-2, +2].+ *+ * @param[out] r Pointer to output buffer.+ * @param len Requested number of 32-bit integers (uniform in+ * [-2, +2]).+ * @param[in] buf Pointer to input buffer (assumed to be uniform random+ * bytes).+ * @param buflen Length of input buffer in bytes.+ *+ * @return - MLD_NATIVE_FUNC_FALLBACK if the native implementation does not+ * support the input lengths.+ * - Otherwise, the non-negative number of sampled 32-bit integers+ * (at most len).+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_eta2_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+__contract__(+ requires(len <= MLDSA_N)+ requires(buflen <= (2 * 136))+ requires(memory_no_alias(r, sizeof(int32_t) * len))+ requires(memory_no_alias(buf, buflen))+ assigns(memory_slice(r, sizeof(int32_t) * len))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || (0 <= return_value && return_value <= len))+ /* check-magic: 3 == 2 + 1 (decl gated on MLDSA_ETA == 2) */+ ensures((return_value != MLD_NATIVE_FUNC_FALLBACK) ==> (array_abs_bound(r, 0, return_value, 3)))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2 */+#endif /* MLD_USE_NATIVE_REJ_UNIFORM_ETA2 */++#if defined(MLD_USE_NATIVE_REJ_UNIFORM_ETA4)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 4+/**+ * Run rejection sampling on uniform random bytes to generate uniform random+ * integers in [-4, +4].+ *+ * @param[out] r Pointer to output buffer.+ * @param len Requested number of 32-bit integers (uniform in+ * [-4, +4]).+ * @param[in] buf Pointer to input buffer (assumed to be uniform random+ * bytes).+ * @param buflen Length of input buffer in bytes.+ *+ * @return - MLD_NATIVE_FUNC_FALLBACK if the native implementation does not+ * support the input lengths.+ * - Otherwise, the non-negative number of sampled 32-bit integers+ * (at most len).+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_eta4_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+__contract__(+ requires(len <= MLDSA_N)+ requires(buflen <= (2 * 136))+ requires(memory_no_alias(r, sizeof(int32_t) * len))+ requires(memory_no_alias(buf, buflen))+ assigns(memory_slice(r, sizeof(int32_t) * len))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || (0 <= return_value && return_value <= len))+ /* check-magic: 5 == 4 + 1 (decl gated on MLDSA_ETA == 4) */+ ensures((return_value != MLD_NATIVE_FUNC_FALLBACK) ==> (array_abs_bound(r, 0, return_value, 5)))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 4 */+#endif /* MLD_USE_NATIVE_REJ_UNIFORM_ETA4 */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#if defined(MLD_USE_NATIVE_POLY_DECOMPOSE_32)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+/**+ * Native implementation of poly_decompose for GAMMA2 = (MLDSA_Q-1)/32.+ *+ * For all coefficients c of the input polynomial, compute high and low bits+ * c0, c1 such c mod MLDSA_Q = c1*(2*GAMMA2) + c0 with+ * -(2*GAMMA2)/2 < c0 <= (2*GAMMA2)/2 except c1 = (MLDSA_Q-1)/(2*GAMMA2) where+ * we set c1 = 0 and -(2*GAMMA2)/2 <= c0 = c mod MLDSA_Q - MLDSA_Q < 0.+ * Assumes coefficients to be standard representatives.+ *+ * @param[out] a1 Output polynomial with coefficients c1.+ * @param[in,out] a0 Input/output polynomial. Output has coefficients c0.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_decompose_32_native(int32_t *a1, int32_t *a0)+__contract__(+ requires(memory_no_alias(a1, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a0, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(int32_t) * MLDSA_N))+ assigns(memory_slice(a0, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_bound(a1, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(a0, 0, MLDSA_N, MLDSA_GAMMA2+1))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(a0, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* MLD_USE_NATIVE_POLY_DECOMPOSE_32 */++#if defined(MLD_USE_NATIVE_POLY_DECOMPOSE_88)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+/**+ * Native implementation of poly_decompose for GAMMA2 = (MLDSA_Q-1)/88.+ *+ * For all coefficients c of the input polynomial, compute high and low bits+ * c0, c1 such c mod MLDSA_Q = c1*(2*GAMMA2) + c0 with+ * -(2*GAMMA2)/2 < c0 <= (2*GAMMA2)/2 except c1 = (MLDSA_Q-1)/(2*GAMMA2) where+ * we set c1 = 0 and -(2*GAMMA2)/2 <= c0 = c mod MLDSA_Q - MLDSA_Q < 0.+ * Assumes coefficients to be standard representatives.+ *+ * @param[out] a1 Output polynomial with coefficients c1.+ * @param[in,out] a0 Input/output polynomial. Output has coefficients c0.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_decompose_88_native(int32_t *a1, int32_t *a0)+__contract__(+ requires(memory_no_alias(a1, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a0, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(int32_t) * MLDSA_N))+ assigns(memory_slice(a0, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_bound(a1, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(a0, 0, MLDSA_N, MLDSA_GAMMA2+1))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(a0, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */+#endif /* MLD_USE_NATIVE_POLY_DECOMPOSE_88 */+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if defined(MLD_USE_NATIVE_POLY_CADDQ)+/**+ * For all coefficients of in/out polynomial add Q if coefficient is negative.+ *+ * @param[in,out] a Pointer to input/output polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_caddq_native(int32_t a[MLDSA_N])+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(a, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_abs_bound(a, 0, MLDSA_N, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(a, MLDSA_N))+);+#endif /* MLD_USE_NATIVE_POLY_CADDQ */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_USE_NATIVE_POLY_USE_HINT_32)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+/**+ * Native implementation of poly_use_hint for GAMMA2 = (MLDSA_Q-1)/32.+ *+ * Use hint h to correct the high bits of a in-place.+ *+ * @param[in,out] a Input/output polynomial.+ * @param[in] h Hint polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_use_hint_32_native(int32_t *a, const int32_t *h)+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(h, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_bound(a, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(a, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* MLD_USE_NATIVE_POLY_USE_HINT_32 */++#if defined(MLD_USE_NATIVE_POLY_USE_HINT_88)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+/**+ * Native implementation of poly_use_hint for GAMMA2 = (MLDSA_Q-1)/88.+ *+ * Use hint h to correct the high bits of a in-place.+ *+ * @param[in,out] a Input/output polynomial.+ * @param[in] h Hint polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_use_hint_88_native(int32_t *a, const int32_t *h)+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(h, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_bound(a, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(a, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */+#endif /* MLD_USE_NATIVE_POLY_USE_HINT_88 */+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if defined(MLD_USE_NATIVE_POLY_CHKNORM)+/**+ * Check infinity norm of polynomial against given bound. Assumes input+ * coefficients were reduced by mld_reduce32().+ *+ * @param[in] a Pointer to polynomial.+ * @param B Norm bound, which must be in the range+ * 0 .. MLDSA_Q - MLD_REDUCE32_RANGE_MAX inclusive.+ *+ * @return - MLD_NATIVE_FUNC_FALLBACK if the target CPU cannot support a+ * native implementation of this function.+ * - MLD_NATIVE_FUNC_SUCCESS if the infinity norm is strictly smaller+ * than B.+ * - 1 otherwise.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_chknorm_native(const int32_t *a, int32_t B)+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(0 <= B && B <= MLDSA_Q - MLD_REDUCE32_RANGE_MAX)+ requires(array_bound(a, 0, MLDSA_N, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == 0 ||+ return_value == 1)+ ensures((return_value != MLD_NATIVE_FUNC_FALLBACK) ==>+ ((return_value == 0) == array_abs_bound(a, 0, MLDSA_N, B)))+);+#endif /* MLD_USE_NATIVE_POLY_CHKNORM */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_USE_NATIVE_POLYZ_UNPACK_17)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+/**+ * Native implementation of polyz_unpack for GAMMA1 = 2^17.+ *+ * Unpack polynomial z with coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1].+ *+ * @param[out] r Pointer to output polynomial.+ * @param[in] a Byte array with bit-packed polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyz_unpack_17_native(int32_t *r, const uint8_t *a)+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, MLDSA_POLYZ_PACKEDBYTES))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_bound(r, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(r, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */+#endif /* MLD_USE_NATIVE_POLYZ_UNPACK_17 */++#if defined(MLD_USE_NATIVE_POLYZ_UNPACK_19)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+/**+ * Native implementation of polyz_unpack for GAMMA1 = 2^19.+ *+ * Unpack polynomial z with coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1].+ *+ * @param[out] r Pointer to output polynomial.+ * @param[in] a Byte array with bit-packed polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyz_unpack_19_native(int32_t *r, const uint8_t *a)+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, MLDSA_POLYZ_PACKEDBYTES))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_bound(r, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(r, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* MLD_USE_NATIVE_POLYZ_UNPACK_19 */+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+#if defined(MLD_USE_NATIVE_POINTWISE_MONTGOMERY)+/**+ * Pointwise multiplication of polynomials in NTT domain with Montgomery+ * reduction. Destructive in the first argument.+ *+ * Computes a[i] = a[i] * b[i] * R^(-1) mod MLDSA_Q for all i, where R = 2^32.+ *+ * @param[in,out] a First input/output polynomial.+ * @param[in] b Second input polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_pointwise_montgomery_native(+ int32_t a[MLDSA_N], const int32_t b[MLDSA_N])+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(a, 0, MLDSA_N, MLD_NTT_BOUND))+ requires(array_abs_bound(b, 0, MLDSA_N, MLD_NTT_BOUND))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(a, 0, MLDSA_N, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_abs_bound(a, 0, MLDSA_N, MLD_NTT_BOUND))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_abs_bound(b, 0, MLDSA_N, MLD_NTT_BOUND))+);+#endif /* MLD_USE_NATIVE_POINTWISE_MONTGOMERY */+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if defined(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 4+/**+ * Native implementation of polyvecl_pointwise_acc_montgomery for MLDSA_L = 4.+ *+ * Pointwise multiply vectors of polynomials of length MLDSA_L, multiply+ * resulting vector by 2^{-32} and add (accumulate) polynomials in it.+ * Input/output vectors are in NTT domain representation.+ *+ * @param[out] w Output polynomial.+ * @param[in] u First input vector.+ * @param[in] v Second input vector.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l4_native(+ int32_t w[MLDSA_N], const int32_t u[4][MLDSA_N],+ const int32_t v[4][MLDSA_N])+__contract__(+ requires(memory_no_alias(w, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(u, sizeof(int32_t) * 4 * MLDSA_N))+ requires(memory_no_alias(v, sizeof(int32_t) * 4 * MLDSA_N))+ requires(forall(l0, 0, 4,+ array_bound(u[l0], 0, MLDSA_N, 0, MLDSA_Q)))+ requires(forall(l1, 0, 4,+ array_abs_bound(v[l1], 0, MLDSA_N, MLD_NTT_BOUND)))+ assigns(memory_slice(w, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(w, 0, MLDSA_N, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(w, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 4 */+#endif /* MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4 */++#if defined(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 5+/**+ * Native implementation of polyvecl_pointwise_acc_montgomery for MLDSA_L = 5.+ *+ * Pointwise multiply vectors of polynomials of length MLDSA_L, multiply+ * resulting vector by 2^{-32} and add (accumulate) polynomials in it.+ * Input/output vectors are in NTT domain representation.+ *+ * @param[out] w Output polynomial.+ * @param[in] u First input vector.+ * @param[in] v Second input vector.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l5_native(+ int32_t w[MLDSA_N], const int32_t u[5][MLDSA_N],+ const int32_t v[5][MLDSA_N])+__contract__(+ requires(memory_no_alias(w, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(u, sizeof(int32_t) * 5 * MLDSA_N))+ requires(memory_no_alias(v, sizeof(int32_t) * 5 * MLDSA_N))+ requires(forall(l0, 0, 5,+ array_bound(u[l0], 0, MLDSA_N, 0, MLDSA_Q)))+ requires(forall(l1, 0, 5,+ array_abs_bound(v[l1], 0, MLDSA_N, MLD_NTT_BOUND)))+ assigns(memory_slice(w, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(w, 0, MLDSA_N, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(w, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 5 */+#endif /* MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5 */++#if defined(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 7+/**+ * Native implementation of polyvecl_pointwise_acc_montgomery for MLDSA_L = 7.+ *+ * Pointwise multiply vectors of polynomials of length MLDSA_L, multiply+ * resulting vector by 2^{-32} and add (accumulate) polynomials in it.+ * Input/output vectors are in NTT domain representation.+ *+ * @param[out] w Output polynomial.+ * @param[in] u First input vector.+ * @param[in] v Second input vector.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l7_native(+ int32_t w[MLDSA_N], const int32_t u[7][MLDSA_N],+ const int32_t v[7][MLDSA_N])+__contract__(+ requires(memory_no_alias(w, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(u, sizeof(int32_t) * 7 * MLDSA_N))+ requires(memory_no_alias(v, sizeof(int32_t) * 7 * MLDSA_N))+ requires(forall(l0, 0, 7,+ array_bound(u[l0], 0, MLDSA_N, 0, MLDSA_Q)))+ requires(forall(l1, 0, 7,+ array_abs_bound(v[l1], 0, MLDSA_N, MLD_NTT_BOUND)))+ assigns(memory_slice(w, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(w, 0, MLDSA_N, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(w, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 7 */+#endif /* MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7 */++#endif /* !MLD_NATIVE_API_H */
+ cbits/mldsa/src/native/meta.h view
@@ -0,0 +1,24 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_NATIVE_META_H+#define MLD_NATIVE_META_H++/*+ * Default arithmetic backend+ */+#include "../sys.h"++#ifdef MLD_SYS_AARCH64_NEON+#include "aarch64/meta.h"+#endif++/* The x86_64 backend requires toolchain support for the SysV ABI */+#if defined(MLD_SYS_X86_64_AVX2) && defined(MLD_SYSV_ABI_SUPPORTED)+#include "x86_64/meta.h"+#endif++#endif /* !MLD_NATIVE_META_H */
+ cbits/mldsa/src/native/x86_64/meta.h view
@@ -0,0 +1,323 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_NATIVE_X86_64_META_H+#define MLD_NATIVE_X86_64_META_H++/* Identifier for this backend so that source and assembly files+ * in the build can be appropriately guarded. */+#define MLD_ARITH_BACKEND_X86_64_DEFAULT++#define MLD_USE_NATIVE_NTT_CUSTOM_ORDER+#define MLD_USE_NATIVE_NTT+#define MLD_USE_NATIVE_INTT+#define MLD_USE_NATIVE_REJ_UNIFORM+#define MLD_USE_NATIVE_REJ_UNIFORM_ETA2+#define MLD_USE_NATIVE_REJ_UNIFORM_ETA4+#define MLD_USE_NATIVE_POLY_DECOMPOSE_32+#define MLD_USE_NATIVE_POLY_DECOMPOSE_88+#define MLD_USE_NATIVE_POLY_CADDQ+#define MLD_USE_NATIVE_POLY_USE_HINT_32+#define MLD_USE_NATIVE_POLY_USE_HINT_88+#define MLD_USE_NATIVE_POLY_CHKNORM+#define MLD_USE_NATIVE_POLYZ_UNPACK_17+#define MLD_USE_NATIVE_POLYZ_UNPACK_19+#define MLD_USE_NATIVE_POINTWISE_MONTGOMERY+#define MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4+#define MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5+#define MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7++#if !defined(__ASSEMBLER__)+#include <string.h>+#include "../../common.h"+#include "../api.h"+#include "src/arith_native_x86_64.h"++static MLD_INLINE void mld_poly_permute_bitrev_to_custom(int32_t data[MLDSA_N])+{+ if (mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ mld_nttunpack_avx2_asm(data);+ }+}++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_ntt_native(int32_t data[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ mld_ntt_avx2_asm(data, mld_qdata);+ return MLD_NATIVE_FUNC_SUCCESS;+}++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_intt_native(int32_t data[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_invntt_avx2_asm(data, mld_qdata);+ return MLD_NATIVE_FUNC_SUCCESS;+}++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ /* AVX2 implementation assumes specific buffer lengths */+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2) || len != MLDSA_N ||+ buflen != MLD_AVX2_REJ_UNIFORM_BUFLEN)+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ /* Safety: outlen is at most MLDSA_N and, hence, this cast is safe. */+ return (int)mld_rej_uniform_avx2_asm(r, buf, mld_rej_uniform_table);+}++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 2+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_eta2_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ unsigned int outlen;+ /* AVX2 implementation assumes specific buffer lengths */+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2) || len != MLDSA_N ||+ buflen != MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN)+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ /* Constant time: Inputs and outputs to this function are secret.+ * It is safe to leak which coefficients are accepted/rejected.+ * The assembly implementation must not leak any other information about the+ * accepted coefficients. Constant-time testing cannot cover this, and we+ * hence have to manually verify the assembly.+ * We declassify prior the input data and mark the outputs as secret.+ */+ MLD_CT_TESTING_DECLASSIFY(buf, buflen);+ outlen = mld_rej_uniform_eta2_avx2_asm(r, buf, mld_rej_uniform_table);+ MLD_CT_TESTING_SECRET(r, sizeof(int32_t) * outlen);+ /* Safety: outlen is at most MLDSA_N and, hence, this cast is safe. */+ return (int)outlen;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 4+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_eta4_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ unsigned int outlen;+ /* AVX2 implementation assumes specific buffer lengths */+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2) || len != MLDSA_N ||+ buflen != MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN)+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ /* Constant time: Inputs and outputs to this function are secret.+ * It is safe to leak which coefficients are accepted/rejected.+ * The assembly implementation must not leak any other information about the+ * accepted coefficients. Constant-time testing cannot cover this, and we+ * hence have to manually verify the assembly.+ * We declassify prior the input data and mark the outputs as secret.+ */+ MLD_CT_TESTING_DECLASSIFY(buf, buflen);+ outlen = mld_rej_uniform_eta4_avx2_asm(r, buf, mld_rej_uniform_table);+ MLD_CT_TESTING_SECRET(r, sizeof(int32_t) * outlen);+ /* Safety: outlen is at most MLDSA_N and, hence, this cast is safe. */+ return (int)outlen;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 4 */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_decompose_32_native(int32_t *a1, int32_t *a0)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_decompose_32_avx2_asm(a1, a0);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_decompose_88_native(int32_t *a1, int32_t *a0)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_decompose_88_avx2_asm(a1, a0);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */+#endif /* !MLD_CONFIG_NO_SIGN_API */++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_caddq_native(int32_t a[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_caddq_avx2_asm(a);+ return MLD_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_use_hint_32_native(int32_t *a, const int32_t *h)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_use_hint_32_avx2_asm(a, h);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_use_hint_88_native(int32_t *a, const int32_t *h)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_use_hint_88_avx2_asm(a, h);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */+#endif /* !MLD_CONFIG_NO_VERIFY_API */++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_chknorm_native(const int32_t *a, int32_t B)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ return mld_poly_chknorm_avx2_asm(a, B);+}++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyz_unpack_17_native(int32_t *r, const uint8_t *a)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_polyz_unpack_17_avx2_asm(r, a);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyz_unpack_19_native(int32_t *r, const uint8_t *a)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_polyz_unpack_19_avx2_asm(r, a);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_pointwise_montgomery_native(+ int32_t a[MLDSA_N], const int32_t b[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_pointwise_avx2_asm(a, b, mld_qdata);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 4+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l4_native(+ int32_t w[MLDSA_N], const int32_t u[4][MLDSA_N],+ const int32_t v[4][MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_pointwise_acc_l4_avx2_asm(w, u, v, mld_qdata);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 4 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 5+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l5_native(+ int32_t w[MLDSA_N], const int32_t u[5][MLDSA_N],+ const int32_t v[5][MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_pointwise_acc_l5_avx2_asm(w, u, v, mld_qdata);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 5 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 7+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l7_native(+ int32_t w[MLDSA_N], const int32_t u[7][MLDSA_N],+ const int32_t v[7][MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_pointwise_acc_l7_avx2_asm(w, u, v, mld_qdata);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 7 */++#endif /* !__ASSEMBLER__ */++#endif /* !MLD_NATIVE_X86_64_META_H */
+ cbits/mldsa/src/native/x86_64/src/arith_native_x86_64.h view
@@ -0,0 +1,330 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H+#define MLD_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H+#include "../../../common.h"++#include "consts.h"++#define MLD_AVX2_REJ_UNIFORM_BUFLEN \+ (5 * 168) /* REJ_UNIFORM_NBLOCKS * SHAKE128_RATE */+++/*+ * Sampling 256 coefficients mod 15 using rejection sampling from 4 bits.+ * Expected number of required bytes: (256 * (16/15))/2 = 136.5 bytes.+ * We sample 1 block (=136 bytes) of SHAKE256_RATE output initially.+ * Sampling 2 blocks initially results in slightly worse performance.+ */+#define MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN (1 * 136)++/*+ * Sampling 256 coefficients mod 9 using rejection sampling from 4 bits.+ * Expected number of required bytes: (256 * (16/9))/2 = 227.5 bytes.+ * We sample 2 blocks (=272 bytes) of SHAKE256_RATE output initially.+ */+#define MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN (2 * 136)++#define mld_rej_uniform_table MLD_NAMESPACE(mld_rej_uniform_table)+MLD_INTERNAL_DATA_DECLARATION const uint8_t mld_rej_uniform_table[256][8];++#define mld_ntt_avx2_asm MLD_NAMESPACE(ntt_avx2_asm)+MLD_SYSV_ABI+void mld_ntt_avx2_asm(int32_t *r, const int32_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_ntt_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 8380417 == MLDSA_Q */+ requires(array_abs_bound(r, 0, MLDSA_N, 8380417))+ requires(qdata == mld_qdata)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ /* check-magic: off */+ ensures(array_abs_bound(r, 0, MLDSA_N, 42035262))+ /* check-magic: on */+);++#define mld_invntt_avx2_asm MLD_NAMESPACE(invntt_avx2_asm)+MLD_SYSV_ABI+void mld_invntt_avx2_asm(int32_t *r, const int32_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_intt_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(r, 0, MLDSA_N, 8380417))+ requires(qdata == mld_qdata)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ /* check-magic: off */+ ensures(array_abs_bound(r, 0, MLDSA_N, 6285313))+ /* check-magic: on */+);++#define mld_nttunpack_avx2_asm MLD_NAMESPACE(nttunpack_avx2_asm)+MLD_SYSV_ABI+void mld_nttunpack_avx2_asm(int32_t *r)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_nttunpack_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(r, 0, MLDSA_N, 8380417))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ /* Output is a permutation of input: every output coefficient+ * is some input coefficient */+ ensures(forall(i, 0, MLDSA_N, exists(j, 0, MLDSA_N,+ r[i] == old(*(int32_t (*)[MLDSA_N])r)[j])))+);++#define mld_rej_uniform_avx2_asm MLD_NAMESPACE(rej_uniform_avx2_asm)+/* This contract must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_rej_uniform_avx2_asm.ml */+MLD_MUST_CHECK_RETURN_VALUE MLD_SYSV_ABI+unsigned mld_rej_uniform_avx2_asm(+ int32_t *r, const uint8_t buf[MLD_AVX2_REJ_UNIFORM_BUFLEN],+ const uint8_t table[256][8])+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(buf, 840))+ requires(table == mld_rej_uniform_table)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value <= MLDSA_N)+ ensures(array_bound(r, 0, return_value, 0, MLDSA_Q))+);++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_rej_uniform_eta2_avx2_asm MLD_NAMESPACE(rej_uniform_eta2_avx2_asm)+/* This contract must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_rej_uniform_eta2_avx2_asm.ml */+MLD_MUST_CHECK_RETURN_VALUE MLD_SYSV_ABI+unsigned mld_rej_uniform_eta2_avx2_asm(+ int32_t *r, const uint8_t buf[MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN],+ const uint8_t table[256][8])+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(buf, MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN))+ requires(table == mld_rej_uniform_table)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value <= MLDSA_N)+ ensures(array_abs_bound(r, 0, return_value, 3))+);++#define mld_rej_uniform_eta4_avx2_asm MLD_NAMESPACE(rej_uniform_eta4_avx2_asm)+/* This contract must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_rej_uniform_eta4_avx2_asm.ml */+MLD_MUST_CHECK_RETURN_VALUE MLD_SYSV_ABI+unsigned mld_rej_uniform_eta4_avx2_asm(+ int32_t *r, const uint8_t buf[MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN],+ const uint8_t table[256][8])+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(buf, MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN))+ requires(table == mld_rej_uniform_table)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value <= MLDSA_N)+ ensures(array_abs_bound(r, 0, return_value, 5))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_poly_decompose_32_avx2_asm MLD_NAMESPACE(poly_decompose_32_avx2_asm)+MLD_SYSV_ABI+void mld_poly_decompose_32_avx2_asm(int32_t *a1, int32_t *a0)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_poly_decompose_32_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(a1, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a0, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(int32_t) * MLDSA_N))+ assigns(memory_slice(a0, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 16 == (MLDSA_Q - 1) / (2 * ((MLDSA_Q - 1) / 32)) */+ ensures(array_bound(a1, 0, MLDSA_N, 0, 16))+ /* check-magic: 261889 == (MLDSA_Q - 1) / 32 + 1 */+ ensures(array_abs_bound(a0, 0, MLDSA_N, 261889))+);++#define mld_poly_decompose_88_avx2_asm MLD_NAMESPACE(poly_decompose_88_avx2_asm)+MLD_SYSV_ABI+void mld_poly_decompose_88_avx2_asm(int32_t *a1, int32_t *a0)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_poly_decompose_88_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(a1, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a0, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(int32_t) * MLDSA_N))+ assigns(memory_slice(a0, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 44 == (MLDSA_Q - 1) / (2 * ((MLDSA_Q - 1) / 88)) */+ ensures(array_bound(a1, 0, MLDSA_N, 0, 44))+ /* check-magic: 95233 == (MLDSA_Q - 1) / 88 + 1 */+ ensures(array_abs_bound(a0, 0, MLDSA_N, 95233))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#define mld_poly_caddq_avx2_asm MLD_NAMESPACE(poly_caddq_avx2_asm)+MLD_SYSV_ABI+void mld_poly_caddq_avx2_asm(int32_t *r)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_poly_caddq_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(r, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(r, 0, MLDSA_N, 0, MLDSA_Q))+);++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_poly_use_hint_32_avx2_asm MLD_NAMESPACE(poly_use_hint_32_avx2_asm)+MLD_SYSV_ABI+void mld_poly_use_hint_32_avx2_asm(int32_t *a, const int32_t *h)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_poly_use_hint_32_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(h, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 16 == (MLDSA_Q - 1) / (2 * ((MLDSA_Q - 1) / 32)) */+ ensures(array_bound(a, 0, MLDSA_N, 0, 16))+);++#define mld_poly_use_hint_88_avx2_asm MLD_NAMESPACE(poly_use_hint_88_avx2_asm)+MLD_SYSV_ABI+void mld_poly_use_hint_88_avx2_asm(int32_t *a, const int32_t *h)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_poly_use_hint_88_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(h, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 44 == (MLDSA_Q - 1) / (2 * ((MLDSA_Q - 1) / 88)) */+ ensures(array_bound(a, 0, MLDSA_N, 0, 44))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#define mld_poly_chknorm_avx2_asm MLD_NAMESPACE(poly_chknorm_avx2_asm)+MLD_MUST_CHECK_RETURN_VALUE MLD_SYSV_ABI+int mld_poly_chknorm_avx2_asm(const int32_t *a, int32_t B)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_poly_chknorm_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ /* HOL Light precondition: abs(ival(x i)) < 2^31, i.e., a[i] != INT32_MIN */+ requires(forall(k0, 0, MLDSA_N, a[k0] > INT32_MIN))+ /* HOL Light precondition: 0 <= ival bound (asm computes B-1 internally) */+ requires(B >= 0)+ ensures(return_value == 0 || return_value == 1)+ ensures((return_value == 0) == array_abs_bound(a, 0, MLDSA_N, B))+);++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_polyz_unpack_17_avx2_asm MLD_NAMESPACE(polyz_unpack_17_avx2_asm)+MLD_SYSV_ABI+void mld_polyz_unpack_17_avx2_asm(int32_t *r, const uint8_t *a)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_polyz_unpack_17_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, 576))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(r, 0, MLDSA_N, -((1 << 17) - 1), (1 << 17) + 1))+);++#define mld_polyz_unpack_19_avx2_asm MLD_NAMESPACE(polyz_unpack_19_avx2_asm)+MLD_SYSV_ABI+void mld_polyz_unpack_19_avx2_asm(int32_t *r, const uint8_t *a)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_polyz_unpack_19_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, 640))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(r, 0, MLDSA_N, -((1 << 19) - 1), (1 << 19) + 1))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#define mld_pointwise_avx2_asm MLD_NAMESPACE(pointwise_avx2_asm)+MLD_SYSV_ABI+void mld_pointwise_avx2_asm(int32_t *a, const int32_t *b, const int32_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_pointwise_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * MLDSA_N))+ /* Input bound MLD_NTT_BOUND = 9 * MLD_FQMUL_BOUND, the guaranteed bound of+ * any forward NTT implementation. Hardcoded here to keep this header free+ * of poly.h. */+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(array_abs_bound(a, 0, MLDSA_N, 94279698))+ requires(array_abs_bound(b, 0, MLDSA_N, 94279698))+ requires(qdata == mld_qdata)+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(a, 0, MLDSA_N, 8380417))+);++#define mld_pointwise_acc_l4_avx2_asm MLD_NAMESPACE(pointwise_acc_l4_avx2_asm)+MLD_SYSV_ABI+void mld_pointwise_acc_l4_avx2_asm(int32_t c[MLDSA_N],+ const int32_t a[4][MLDSA_N],+ const int32_t b[4][MLDSA_N],+ const int32_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_pointwise_acc_l4_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(c, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, sizeof(int32_t) * 4 * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * 4 * MLDSA_N))+ requires(forall(l0, 0, 4, array_abs_bound(a[l0], 0, MLDSA_N, 8380417)))+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(forall(l1, 0, 4, array_abs_bound(b[l1], 0, MLDSA_N, 94279698)))+ requires(qdata == mld_qdata)+ assigns(memory_slice(c, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(c, 0, MLDSA_N, 8380417))+);++#define mld_pointwise_acc_l5_avx2_asm MLD_NAMESPACE(pointwise_acc_l5_avx2_asm)+MLD_SYSV_ABI+void mld_pointwise_acc_l5_avx2_asm(int32_t c[MLDSA_N],+ const int32_t a[5][MLDSA_N],+ const int32_t b[5][MLDSA_N],+ const int32_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_pointwise_acc_l5_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(c, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, sizeof(int32_t) * 5 * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * 5 * MLDSA_N))+ requires(forall(l0, 0, 5, array_abs_bound(a[l0], 0, MLDSA_N, 8380417)))+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(forall(l1, 0, 5, array_abs_bound(b[l1], 0, MLDSA_N, 94279698)))+ requires(qdata == mld_qdata)+ assigns(memory_slice(c, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(c, 0, MLDSA_N, 8380417))+);++#define mld_pointwise_acc_l7_avx2_asm MLD_NAMESPACE(pointwise_acc_l7_avx2_asm)+MLD_SYSV_ABI+void mld_pointwise_acc_l7_avx2_asm(int32_t c[MLDSA_N],+ const int32_t a[7][MLDSA_N],+ const int32_t b[7][MLDSA_N],+ const int32_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_pointwise_acc_l7_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(c, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, sizeof(int32_t) * 7 * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * 7 * MLDSA_N))+ requires(forall(l0, 0, 7, array_abs_bound(a[l0], 0, MLDSA_N, 8380417)))+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(forall(l1, 0, 7, array_abs_bound(b[l1], 0, MLDSA_N, 94279698)))+ requires(qdata == mld_qdata)+ assigns(memory_slice(c, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(c, 0, MLDSA_N, 8380417))+);++#endif /* !MLD_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H */
+ cbits/mldsa/src/native/x86_64/src/consts.c view
@@ -0,0 +1,157 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include "consts.h"++/*+ * Table of zeta values used in the AVX2 forward and inverse NTT+ * See autogen for details.+ */+MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const int32_t mld_qdata[624] = {+ 8380417, 8380417, 8380417, 8380417, 8380417,+ 8380417, 8380417, 8380417, 58728449, 58728449,+ 58728449, 58728449, 58728449, 58728449, 58728449,+ 58728449, -8395782, -8395782, -8395782, -8395782,+ -8395782, -8395782, -8395782, -8395782, 41978,+ 41978, 41978, 41978, 41978, 41978,+ 41978, 41978, -151046689, 1830765815, -1929875198,+ -1927777021, 1640767044, 1477910808, 1612161320, 1640734244,+ 308362795, 308362795, 308362795, 308362795, -1815525077,+ -1815525077, -1815525077, -1815525077, -1374673747, -1374673747,+ -1374673747, -1374673747, -1091570561, -1091570561, -1091570561,+ -1091570561, -1929495947, -1929495947, -1929495947, -1929495947,+ 515185417, 515185417, 515185417, 515185417, -285697463,+ -285697463, -285697463, -285697463, 625853735, 625853735,+ 625853735, 625853735, 1727305304, 1727305304, 2082316400,+ 2082316400, -1364982364, -1364982364, 858240904, 858240904,+ 1806278032, 1806278032, 222489248, 222489248, -346752664,+ -346752664, 684667771, 684667771, 1654287830, 1654287830,+ -878576921, -878576921, -1257667337, -1257667337, -748618600,+ -748618600, 329347125, 329347125, 1837364258, 1837364258,+ -1443016191, -1443016191, -1170414139, -1170414139, -1846138265,+ -1631226336, -1404529459, 1838055109, 1594295555, -1076973524,+ -1898723372, -594436433, -202001019, -475984260, -561427818,+ 1797021249, -1061813248, 2059733581, -1661512036, -1104976547,+ -1750224323, -901666090, 418987550, 1831915353, -1925356481,+ 992097815, 879957084, 2024403852, 1484874664, -1636082790,+ -285388938, -1983539117, -1495136972, -950076368, -1714807468,+ -952438995, -1574918427, 1350681039, -2143979939, 1599739335,+ -1285853323, -993005454, -1440787840, 568627424, -783134478,+ -588790216, 289871779, -1262003603, 2135294594, -1018755525,+ -889861155, 1665705315, 1321868265, 1225434135, -1784632064,+ 666258756, 675310538, -1555941048, -1999506068, -1499481951,+ -695180180, -1375177022, 1777179795, 334803717, -178766299,+ -518252220, 1957047970, 1146323031, -654783359, -1974159335,+ 1651689966, 140455867, -1039411342, 1955560694, 1529189038,+ -2131021878, -247357819, 1518161567, -86965173, 1708872713,+ 1787797779, 1638590967, -120646188, -1669960606, -916321552,+ 1155548552, 2143745726, 1210558298, -1261461890, -318346816,+ 628664287, -1729304568, 1422575624, 1424130038, -1185330464,+ 235321234, 168022240, 1206536194, 985155484, -894060583,+ -898413, -1363460238, -605900043, 2027833504, 14253662,+ 1014493059, 863641633, 1819892093, 2124962073, -1223601433,+ -1920467227, -1637785316, -1536588520, 694382729, 235104446,+ -1045062172, 831969619, -300448763, 756955444, -260312805,+ 1554794072, 1339088280, -2040058690, -853476187, -2047270596,+ -1723816713, -1591599803, -440824168, 1119856484, 1544891539,+ 155290192, -973777462, 991903578, 912367099, -44694137,+ 1176904444, -421552614, -818371958, 1747917558, -325927722,+ 908452108, 1851023419, -1176751719, -1354528380, -72690498,+ -314284737, 985022747, 963438279, -1078959975, 604552167,+ -1021949428, 608791570, 173440395, -2126092136, -1316619236,+ -1039370342, 6087993, -110126092, 565464272, -1758099917,+ -1600929361, 879867909, -1809756372, 400711272, 1363007700,+ 30313375, -326425360, 1683520342, -517299994, 2027935492,+ -1372618620, 128353682, -1123881663, 137583815, -635454918,+ -642772911, 45766801, 671509323, -2070602178, 419615363,+ 1216882040, -270590488, -1276805128, 371462360, -1357098057,+ -384158533, 827959816, -596344473, 702390549, -279505433,+ -260424530, -71875110, -1208667171, -1499603926, 2036925262,+ -540420426, 746144248, -1420958686, 2032221021, 1904936414,+ 1257750362, 1926727420, 1931587462, 1258381762, 885133339,+ 1629985060, 1967222129, 6363718, -1287922800, 1136965286,+ 1779436847, 1116720494, 1042326957, 1405999311, 713994583,+ 940195359, -1542497137, 2061661095, -883155599, 1726753853,+ -1547952704, 394851342, 283780712, 776003547, 1123958025,+ 201262505, 1934038751, 374860238, -3975713, 25847,+ -2608894, -518909, 237124, -777960, -876248,+ 466468, 1826347, 1826347, 1826347, 1826347,+ 2353451, 2353451, 2353451, 2353451, -359251,+ -359251, -359251, -359251, -2091905, -2091905,+ -2091905, -2091905, 3119733, 3119733, 3119733,+ 3119733, -2884855, -2884855, -2884855, -2884855,+ 3111497, 3111497, 3111497, 3111497, 2680103,+ 2680103, 2680103, 2680103, 2725464, 2725464,+ 1024112, 1024112, -1079900, -1079900, 3585928,+ 3585928, -549488, -549488, -1119584, -1119584,+ 2619752, 2619752, -2108549, -2108549, -2118186,+ -2118186, -3859737, -3859737, -1399561, -1399561,+ -3277672, -3277672, 1757237, 1757237, -19422,+ -19422, 4010497, 4010497, 280005, 280005,+ 2706023, 95776, 3077325, 3530437, -1661693,+ -3592148, -2537516, 3915439, -3861115, -3043716,+ 3574422, -2867647, 3539968, -300467, 2348700,+ -539299, -1699267, -1643818, 3505694, -3821735,+ 3507263, -2140649, -1600420, 3699596, 811944,+ 531354, 954230, 3881043, 3900724, -2556880,+ 2071892, -2797779, -3930395, -3677745, -1452451,+ 2176455, -1257611, -4083598, -3190144, -3632928,+ 3412210, 2147896, -2967645, -411027, -671102,+ -22981, -381987, 1852771, -3343383, 508951,+ 44288, 904516, -3724342, 1653064, 2389356,+ 759969, 189548, 3159746, -2409325, 1315589,+ 1285669, -812732, -3019102, -3628969, -1528703,+ -3041255, 3475950, -1585221, 1939314, -1000202,+ -3157330, 126922, -983419, 2715295, -3693493,+ -2477047, -1228525, -1308169, 1349076, -1430430,+ 264944, 3097992, -1100098, 3958618, -8578,+ -3249728, -210977, -1316856, -3553272, -1851402,+ -177440, 1341330, -1584928, -1439742, -3881060,+ 3839961, 2091667, -3342478, 266997, -3520352,+ 900702, 495491, -655327, -3556995, 342297,+ 3437287, 2842341, 4055324, -3767016, -2994039,+ -1333058, -451100, -1279661, 1500165, -542412,+ -2584293, -2013608, 1957272, -3183426, 810149,+ -3038916, 2213111, -426683, -1667432, -2939036,+ 183443, -554416, 3937738, 3407706, 2244091,+ 2434439, -3759364, 1859098, -1613174, -3122442,+ -525098, 286988, -3342277, 2691481, 1247620,+ 1250494, 1869119, 1237275, 1312455, 1917081,+ 777191, -2831860, -3724270, 2432395, 3369112,+ 162844, 1652634, 3523897, -975884, 1723600,+ -1104333, -2235985, -976891, 3919660, 1400424,+ 2316500, -2446433, -1235728, -1197226, 909542,+ -43260, 2031748, -768622, -2437823, 1735879,+ -2590150, 2486353, 2635921, 1903435, -3318210,+ 3306115, -2546312, 2235880, -1671176, 594136,+ 2454455, 185531, 1616392, -3694233, 3866901,+ 1717735, -1803090, -260646, -420899, 1612842,+ -48306, -846154, 3817976, -3562462, 3513181,+ -3193378, 819034, -522500, 3207046, -3595838,+ 4108315, 203044, 1265009, 1595974, -3548272,+ -1050970, -1430225, -1962642, -1374803, 3406031,+ -1846953, -3776993, -164721, -1207385, 3014001,+ -1799107, 269760, 472078, 1910376, -3833893,+ -2286327, -3545687, -1362209, 1976782,+};++#else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++MLD_EMPTY_CU(avx2_consts)++#endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mldsa/src/native/x86_64/src/consts.h view
@@ -0,0 +1,27 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#ifndef MLD_NATIVE_X86_64_SRC_CONSTS_H+#define MLD_NATIVE_X86_64_SRC_CONSTS_H+#include "../../../common.h"+#define MLD_AVX2_BACKEND_DATA_OFFSET_8XQ 0+#define MLD_AVX2_BACKEND_DATA_OFFSET_8XQINV 8+#define MLD_AVX2_BACKEND_DATA_OFFSET_8XDIV_QINV 16+#define MLD_AVX2_BACKEND_DATA_OFFSET_8XDIV 24+#define MLD_AVX2_BACKEND_DATA_OFFSET_ZETAS_QINV 32+#define MLD_AVX2_BACKEND_DATA_OFFSET_ZETAS 328++#ifndef __ASSEMBLER__+#define mld_qdata MLD_NAMESPACE(qdata)+MLD_INTERNAL_DATA_DECLARATION const int32_t mld_qdata[624];+#endif++#endif /* !MLD_NATIVE_X86_64_SRC_CONSTS_H */
+ cbits/mldsa/src/native/x86_64/src/mldsa_intt_avx2_asm.S view
@@ -0,0 +1,2333 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++ /*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: invntt_avx2_asm+ Description: x86_64 AVX2 inverse NTT+ Signature: void mld_invntt_avx2_asm(int32_t *r, const int32_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *r+ description: Input/output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 2496+ permissions: read-only+ c_parameter: const int32_t *qdata+ description: Precomputed constants (624 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_intt_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(invntt_avx2_asm)+MLD_ASM_FN_SYMBOL(invntt_avx2_asm)++ .cfi_startproc+ vmovdqa (%rsi), %ymm0+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x20(%rdi), %ymm5+ vmovdqa 0x40(%rdi), %ymm6+ vmovdqa 0x60(%rdi), %ymm7+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm9+ vmovdqa 0xc0(%rdi), %ymm10+ vmovdqa 0xe0(%rdi), %ymm11+ vpermq $0x1b, 0x500(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x9a0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm5, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpermq $0x1b, 0x480(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x920(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x400(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x8a0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm9, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpermq $0x1b, 0x380(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x820(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm10, %ymm11, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x300(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x7a0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x280(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x720(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x200(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x6a0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovsldup %ymm5, %ymm3 # ymm3 = ymm5[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7]+ vpsrlq $0x20, %ymm4, %ymm4+ vpblendd $0xaa, %ymm5, %ymm4, %ymm5 # ymm5 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovsldup %ymm7, %ymm4 # ymm4 = ymm7[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7]+ vmovsldup %ymm9, %ymm6 # ymm6 = ymm9[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7]+ vpsrlq $0x20, %ymm8, %ymm8+ vpblendd $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7]+ vmovsldup %ymm11, %ymm8 # ymm8 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm8, %ymm10, %ymm8 # ymm8 = ymm10[0],ymm8[1],ymm10[2],ymm8[3],ymm10[4],ymm8[5],ymm10[6],ymm8[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vpermq $0x1b, 0x180(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x620(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm4, %ymm7, %ymm12+ vpaddd %ymm7, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm6, %ymm9, %ymm12+ vpaddd %ymm6, %ymm9, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm8, %ymm11, %ymm12+ vpaddd %ymm11, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpunpcklqdq %ymm4, %ymm3, %ymm10 # ymm10 = ymm3[0],ymm4[0],ymm3[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[1],ymm4[1],ymm3[3],ymm4[3]+ vpunpcklqdq %ymm8, %ymm6, %ymm3 # ymm3 = ymm6[0],ymm8[0],ymm6[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm6, %ymm8 # ymm8 = ymm6[1],ymm8[1],ymm6[3],ymm8[3]+ vpunpcklqdq %ymm7, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm7[0],ymm5[2],ymm7[2]+ vpunpckhqdq %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[1],ymm7[1],ymm5[3],ymm7[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vpermq $0x1b, 0x100(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x5a0(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm10, %ymm4, %ymm12+ vpaddd %ymm4, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm4 # ymm4 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm4, %ymm4+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm4, %ymm12, %ymm4 # ymm4 = ymm12[0],ymm4[1],ymm12[2],ymm4[3],ymm12[4],ymm4[5],ymm12[6],ymm4[7]+ vpsubd %ymm3, %ymm8, %ymm12+ vpaddd %ymm3, %ymm8, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm5, %ymm11, %ymm12+ vpaddd %ymm5, %ymm11, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vperm2i128 $0x20, %ymm3, %ymm10, %ymm9 # ymm9 = ymm10[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm5, %ymm6, %ymm10 # ymm10 = ymm6[0,1],ymm5[0,1]+ vperm2i128 $0x31, %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[2,3],ymm5[2,3]+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm6 # ymm6 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm4 # ymm4 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpbroadcastd 0x9c(%rsi), %ymm1+ vpbroadcastd 0x53c(%rsi), %ymm2+ vpsubd %ymm9, %ymm3, %ymm12+ vpaddd %ymm3, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm3 # ymm3 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm3, %ymm3+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm12, %ymm3 # ymm3 = ymm12[0],ymm3[1],ymm12[2],ymm3[3],ymm12[4],ymm3[5],ymm12[6],ymm3[7]+ vpsubd %ymm10, %ymm5, %ymm12+ vpaddd %ymm5, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm4, %ymm11, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm9, (%rdi)+ vmovdqa %ymm10, 0x20(%rdi)+ vmovdqa %ymm6, 0x40(%rdi)+ vmovdqa %ymm4, 0x60(%rdi)+ vmovdqa %ymm3, 0x80(%rdi)+ vmovdqa %ymm5, 0xa0(%rdi)+ vmovdqa %ymm8, 0xc0(%rdi)+ vmovdqa %ymm11, 0xe0(%rdi)+ vmovdqa 0x100(%rdi), %ymm4+ vmovdqa 0x120(%rdi), %ymm5+ vmovdqa 0x140(%rdi), %ymm6+ vmovdqa 0x160(%rdi), %ymm7+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1a0(%rdi), %ymm9+ vmovdqa 0x1c0(%rdi), %ymm10+ vmovdqa 0x1e0(%rdi), %ymm11+ vpermq $0x1b, 0x4e0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x980(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm5, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpermq $0x1b, 0x460(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x900(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x3e0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x880(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm9, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpermq $0x1b, 0x360(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x800(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm10, %ymm11, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x2e0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x780(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x260(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x700(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x1e0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x680(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovsldup %ymm5, %ymm3 # ymm3 = ymm5[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7]+ vpsrlq $0x20, %ymm4, %ymm4+ vpblendd $0xaa, %ymm5, %ymm4, %ymm5 # ymm5 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovsldup %ymm7, %ymm4 # ymm4 = ymm7[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7]+ vmovsldup %ymm9, %ymm6 # ymm6 = ymm9[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7]+ vpsrlq $0x20, %ymm8, %ymm8+ vpblendd $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7]+ vmovsldup %ymm11, %ymm8 # ymm8 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm8, %ymm10, %ymm8 # ymm8 = ymm10[0],ymm8[1],ymm10[2],ymm8[3],ymm10[4],ymm8[5],ymm10[6],ymm8[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vpermq $0x1b, 0x160(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x600(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm4, %ymm7, %ymm12+ vpaddd %ymm7, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm6, %ymm9, %ymm12+ vpaddd %ymm6, %ymm9, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm8, %ymm11, %ymm12+ vpaddd %ymm11, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpunpcklqdq %ymm4, %ymm3, %ymm10 # ymm10 = ymm3[0],ymm4[0],ymm3[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[1],ymm4[1],ymm3[3],ymm4[3]+ vpunpcklqdq %ymm8, %ymm6, %ymm3 # ymm3 = ymm6[0],ymm8[0],ymm6[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm6, %ymm8 # ymm8 = ymm6[1],ymm8[1],ymm6[3],ymm8[3]+ vpunpcklqdq %ymm7, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm7[0],ymm5[2],ymm7[2]+ vpunpckhqdq %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[1],ymm7[1],ymm5[3],ymm7[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vpermq $0x1b, 0xe0(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x580(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm10, %ymm4, %ymm12+ vpaddd %ymm4, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm4 # ymm4 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm4, %ymm4+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm4, %ymm12, %ymm4 # ymm4 = ymm12[0],ymm4[1],ymm12[2],ymm4[3],ymm12[4],ymm4[5],ymm12[6],ymm4[7]+ vpsubd %ymm3, %ymm8, %ymm12+ vpaddd %ymm3, %ymm8, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm5, %ymm11, %ymm12+ vpaddd %ymm5, %ymm11, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vperm2i128 $0x20, %ymm3, %ymm10, %ymm9 # ymm9 = ymm10[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm5, %ymm6, %ymm10 # ymm10 = ymm6[0,1],ymm5[0,1]+ vperm2i128 $0x31, %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[2,3],ymm5[2,3]+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm6 # ymm6 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm4 # ymm4 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpbroadcastd 0x98(%rsi), %ymm1+ vpbroadcastd 0x538(%rsi), %ymm2+ vpsubd %ymm9, %ymm3, %ymm12+ vpaddd %ymm3, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm3 # ymm3 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm3, %ymm3+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm12, %ymm3 # ymm3 = ymm12[0],ymm3[1],ymm12[2],ymm3[3],ymm12[4],ymm3[5],ymm12[6],ymm3[7]+ vpsubd %ymm10, %ymm5, %ymm12+ vpaddd %ymm5, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm4, %ymm11, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm9, 0x100(%rdi)+ vmovdqa %ymm10, 0x120(%rdi)+ vmovdqa %ymm6, 0x140(%rdi)+ vmovdqa %ymm4, 0x160(%rdi)+ vmovdqa %ymm3, 0x180(%rdi)+ vmovdqa %ymm5, 0x1a0(%rdi)+ vmovdqa %ymm8, 0x1c0(%rdi)+ vmovdqa %ymm11, 0x1e0(%rdi)+ vmovdqa 0x200(%rdi), %ymm4+ vmovdqa 0x220(%rdi), %ymm5+ vmovdqa 0x240(%rdi), %ymm6+ vmovdqa 0x260(%rdi), %ymm7+ vmovdqa 0x280(%rdi), %ymm8+ vmovdqa 0x2a0(%rdi), %ymm9+ vmovdqa 0x2c0(%rdi), %ymm10+ vmovdqa 0x2e0(%rdi), %ymm11+ vpermq $0x1b, 0x4c0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x960(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm5, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpermq $0x1b, 0x440(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x8e0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x3c0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x860(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm9, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpermq $0x1b, 0x340(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x7e0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm10, %ymm11, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x2c0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x760(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x240(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x6e0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x1c0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x660(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovsldup %ymm5, %ymm3 # ymm3 = ymm5[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7]+ vpsrlq $0x20, %ymm4, %ymm4+ vpblendd $0xaa, %ymm5, %ymm4, %ymm5 # ymm5 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovsldup %ymm7, %ymm4 # ymm4 = ymm7[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7]+ vmovsldup %ymm9, %ymm6 # ymm6 = ymm9[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7]+ vpsrlq $0x20, %ymm8, %ymm8+ vpblendd $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7]+ vmovsldup %ymm11, %ymm8 # ymm8 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm8, %ymm10, %ymm8 # ymm8 = ymm10[0],ymm8[1],ymm10[2],ymm8[3],ymm10[4],ymm8[5],ymm10[6],ymm8[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vpermq $0x1b, 0x140(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x5e0(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm4, %ymm7, %ymm12+ vpaddd %ymm7, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm6, %ymm9, %ymm12+ vpaddd %ymm6, %ymm9, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm8, %ymm11, %ymm12+ vpaddd %ymm11, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpunpcklqdq %ymm4, %ymm3, %ymm10 # ymm10 = ymm3[0],ymm4[0],ymm3[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[1],ymm4[1],ymm3[3],ymm4[3]+ vpunpcklqdq %ymm8, %ymm6, %ymm3 # ymm3 = ymm6[0],ymm8[0],ymm6[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm6, %ymm8 # ymm8 = ymm6[1],ymm8[1],ymm6[3],ymm8[3]+ vpunpcklqdq %ymm7, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm7[0],ymm5[2],ymm7[2]+ vpunpckhqdq %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[1],ymm7[1],ymm5[3],ymm7[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vpermq $0x1b, 0xc0(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x560(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm10, %ymm4, %ymm12+ vpaddd %ymm4, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm4 # ymm4 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm4, %ymm4+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm4, %ymm12, %ymm4 # ymm4 = ymm12[0],ymm4[1],ymm12[2],ymm4[3],ymm12[4],ymm4[5],ymm12[6],ymm4[7]+ vpsubd %ymm3, %ymm8, %ymm12+ vpaddd %ymm3, %ymm8, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm5, %ymm11, %ymm12+ vpaddd %ymm5, %ymm11, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vperm2i128 $0x20, %ymm3, %ymm10, %ymm9 # ymm9 = ymm10[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm5, %ymm6, %ymm10 # ymm10 = ymm6[0,1],ymm5[0,1]+ vperm2i128 $0x31, %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[2,3],ymm5[2,3]+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm6 # ymm6 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm4 # ymm4 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpbroadcastd 0x94(%rsi), %ymm1+ vpbroadcastd 0x534(%rsi), %ymm2+ vpsubd %ymm9, %ymm3, %ymm12+ vpaddd %ymm3, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm3 # ymm3 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm3, %ymm3+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm12, %ymm3 # ymm3 = ymm12[0],ymm3[1],ymm12[2],ymm3[3],ymm12[4],ymm3[5],ymm12[6],ymm3[7]+ vpsubd %ymm10, %ymm5, %ymm12+ vpaddd %ymm5, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm4, %ymm11, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm9, 0x200(%rdi)+ vmovdqa %ymm10, 0x220(%rdi)+ vmovdqa %ymm6, 0x240(%rdi)+ vmovdqa %ymm4, 0x260(%rdi)+ vmovdqa %ymm3, 0x280(%rdi)+ vmovdqa %ymm5, 0x2a0(%rdi)+ vmovdqa %ymm8, 0x2c0(%rdi)+ vmovdqa %ymm11, 0x2e0(%rdi)+ vmovdqa 0x300(%rdi), %ymm4+ vmovdqa 0x320(%rdi), %ymm5+ vmovdqa 0x340(%rdi), %ymm6+ vmovdqa 0x360(%rdi), %ymm7+ vmovdqa 0x380(%rdi), %ymm8+ vmovdqa 0x3a0(%rdi), %ymm9+ vmovdqa 0x3c0(%rdi), %ymm10+ vmovdqa 0x3e0(%rdi), %ymm11+ vpermq $0x1b, 0x4a0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x940(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm5, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpermq $0x1b, 0x420(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x8c0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x3a0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x840(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm9, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpermq $0x1b, 0x320(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x7c0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm10, %ymm11, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x2a0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x740(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x220(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x6c0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x1a0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x640(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovsldup %ymm5, %ymm3 # ymm3 = ymm5[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7]+ vpsrlq $0x20, %ymm4, %ymm4+ vpblendd $0xaa, %ymm5, %ymm4, %ymm5 # ymm5 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovsldup %ymm7, %ymm4 # ymm4 = ymm7[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7]+ vmovsldup %ymm9, %ymm6 # ymm6 = ymm9[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7]+ vpsrlq $0x20, %ymm8, %ymm8+ vpblendd $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7]+ vmovsldup %ymm11, %ymm8 # ymm8 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm8, %ymm10, %ymm8 # ymm8 = ymm10[0],ymm8[1],ymm10[2],ymm8[3],ymm10[4],ymm8[5],ymm10[6],ymm8[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vpermq $0x1b, 0x120(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x5c0(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm4, %ymm7, %ymm12+ vpaddd %ymm7, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm6, %ymm9, %ymm12+ vpaddd %ymm6, %ymm9, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm8, %ymm11, %ymm12+ vpaddd %ymm11, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpunpcklqdq %ymm4, %ymm3, %ymm10 # ymm10 = ymm3[0],ymm4[0],ymm3[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[1],ymm4[1],ymm3[3],ymm4[3]+ vpunpcklqdq %ymm8, %ymm6, %ymm3 # ymm3 = ymm6[0],ymm8[0],ymm6[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm6, %ymm8 # ymm8 = ymm6[1],ymm8[1],ymm6[3],ymm8[3]+ vpunpcklqdq %ymm7, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm7[0],ymm5[2],ymm7[2]+ vpunpckhqdq %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[1],ymm7[1],ymm5[3],ymm7[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vpermq $0x1b, 0xa0(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x540(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm10, %ymm4, %ymm12+ vpaddd %ymm4, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm4 # ymm4 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm4, %ymm4+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm4, %ymm12, %ymm4 # ymm4 = ymm12[0],ymm4[1],ymm12[2],ymm4[3],ymm12[4],ymm4[5],ymm12[6],ymm4[7]+ vpsubd %ymm3, %ymm8, %ymm12+ vpaddd %ymm3, %ymm8, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm5, %ymm11, %ymm12+ vpaddd %ymm5, %ymm11, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vperm2i128 $0x20, %ymm3, %ymm10, %ymm9 # ymm9 = ymm10[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm5, %ymm6, %ymm10 # ymm10 = ymm6[0,1],ymm5[0,1]+ vperm2i128 $0x31, %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[2,3],ymm5[2,3]+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm6 # ymm6 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm4 # ymm4 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpbroadcastd 0x90(%rsi), %ymm1+ vpbroadcastd 0x530(%rsi), %ymm2+ vpsubd %ymm9, %ymm3, %ymm12+ vpaddd %ymm3, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm3 # ymm3 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm3, %ymm3+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm12, %ymm3 # ymm3 = ymm12[0],ymm3[1],ymm12[2],ymm3[3],ymm12[4],ymm3[5],ymm12[6],ymm3[7]+ vpsubd %ymm10, %ymm5, %ymm12+ vpaddd %ymm5, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm4, %ymm11, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm9, 0x300(%rdi)+ vmovdqa %ymm10, 0x320(%rdi)+ vmovdqa %ymm6, 0x340(%rdi)+ vmovdqa %ymm4, 0x360(%rdi)+ vmovdqa %ymm3, 0x380(%rdi)+ vmovdqa %ymm5, 0x3a0(%rdi)+ vmovdqa %ymm8, 0x3c0(%rdi)+ vmovdqa %ymm11, 0x3e0(%rdi)+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x80(%rdi), %ymm5+ vmovdqa 0x100(%rdi), %ymm6+ vmovdqa 0x180(%rdi), %ymm7+ vmovdqa 0x200(%rdi), %ymm8+ vmovdqa 0x280(%rdi), %ymm9+ vmovdqa 0x300(%rdi), %ymm10+ vmovdqa 0x380(%rdi), %ymm11+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpbroadcastd 0x80(%rsi), %ymm1+ vpbroadcastd 0x520(%rsi), %ymm2+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm8, 0x200(%rdi)+ vmovdqa %ymm9, 0x280(%rdi)+ vmovdqa %ymm10, 0x300(%rdi)+ vmovdqa %ymm11, 0x380(%rdi)+ vmovdqa 0x40(%rsi), %ymm1+ vmovdqa 0x60(%rsi), %ymm2+ vpmuldq %ymm1, %ymm4, %ymm12+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm4, %ymm8 # ymm8 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm9 # ymm9 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm4, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm8[1],ymm4[2],ymm8[3],ymm4[4],ymm8[5],ymm4[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm9[1],ymm5[2],ymm9[3],ymm5[4],ymm9[5],ymm5[6],ymm9[7]+ vpmuldq %ymm1, %ymm6, %ymm12+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm6, %ymm8 # ymm8 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm9 # ymm9 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm6, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm8[1],ymm6[2],ymm8[3],ymm6[4],ymm8[5],ymm6[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vmovdqa %ymm4, (%rdi)+ vmovdqa %ymm5, 0x80(%rdi)+ vmovdqa %ymm6, 0x100(%rdi)+ vmovdqa %ymm7, 0x180(%rdi)+ vmovdqa 0x20(%rdi), %ymm4+ vmovdqa 0xa0(%rdi), %ymm5+ vmovdqa 0x120(%rdi), %ymm6+ vmovdqa 0x1a0(%rdi), %ymm7+ vmovdqa 0x220(%rdi), %ymm8+ vmovdqa 0x2a0(%rdi), %ymm9+ vmovdqa 0x320(%rdi), %ymm10+ vmovdqa 0x3a0(%rdi), %ymm11+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpbroadcastd 0x80(%rsi), %ymm1+ vpbroadcastd 0x520(%rsi), %ymm2+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm8, 0x220(%rdi)+ vmovdqa %ymm9, 0x2a0(%rdi)+ vmovdqa %ymm10, 0x320(%rdi)+ vmovdqa %ymm11, 0x3a0(%rdi)+ vmovdqa 0x40(%rsi), %ymm1+ vmovdqa 0x60(%rsi), %ymm2+ vpmuldq %ymm1, %ymm4, %ymm12+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm4, %ymm8 # ymm8 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm9 # ymm9 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm4, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm8[1],ymm4[2],ymm8[3],ymm4[4],ymm8[5],ymm4[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm9[1],ymm5[2],ymm9[3],ymm5[4],ymm9[5],ymm5[6],ymm9[7]+ vpmuldq %ymm1, %ymm6, %ymm12+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm6, %ymm8 # ymm8 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm9 # ymm9 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm6, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm8[1],ymm6[2],ymm8[3],ymm6[4],ymm8[5],ymm6[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vmovdqa %ymm4, 0x20(%rdi)+ vmovdqa %ymm5, 0xa0(%rdi)+ vmovdqa %ymm6, 0x120(%rdi)+ vmovdqa %ymm7, 0x1a0(%rdi)+ vmovdqa 0x40(%rdi), %ymm4+ vmovdqa 0xc0(%rdi), %ymm5+ vmovdqa 0x140(%rdi), %ymm6+ vmovdqa 0x1c0(%rdi), %ymm7+ vmovdqa 0x240(%rdi), %ymm8+ vmovdqa 0x2c0(%rdi), %ymm9+ vmovdqa 0x340(%rdi), %ymm10+ vmovdqa 0x3c0(%rdi), %ymm11+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpbroadcastd 0x80(%rsi), %ymm1+ vpbroadcastd 0x520(%rsi), %ymm2+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm8, 0x240(%rdi)+ vmovdqa %ymm9, 0x2c0(%rdi)+ vmovdqa %ymm10, 0x340(%rdi)+ vmovdqa %ymm11, 0x3c0(%rdi)+ vmovdqa 0x40(%rsi), %ymm1+ vmovdqa 0x60(%rsi), %ymm2+ vpmuldq %ymm1, %ymm4, %ymm12+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm4, %ymm8 # ymm8 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm9 # ymm9 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm4, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm8[1],ymm4[2],ymm8[3],ymm4[4],ymm8[5],ymm4[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm9[1],ymm5[2],ymm9[3],ymm5[4],ymm9[5],ymm5[6],ymm9[7]+ vpmuldq %ymm1, %ymm6, %ymm12+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm6, %ymm8 # ymm8 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm9 # ymm9 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm6, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm8[1],ymm6[2],ymm8[3],ymm6[4],ymm8[5],ymm6[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vmovdqa %ymm4, 0x40(%rdi)+ vmovdqa %ymm5, 0xc0(%rdi)+ vmovdqa %ymm6, 0x140(%rdi)+ vmovdqa %ymm7, 0x1c0(%rdi)+ vmovdqa 0x60(%rdi), %ymm4+ vmovdqa 0xe0(%rdi), %ymm5+ vmovdqa 0x160(%rdi), %ymm6+ vmovdqa 0x1e0(%rdi), %ymm7+ vmovdqa 0x260(%rdi), %ymm8+ vmovdqa 0x2e0(%rdi), %ymm9+ vmovdqa 0x360(%rdi), %ymm10+ vmovdqa 0x3e0(%rdi), %ymm11+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpbroadcastd 0x80(%rsi), %ymm1+ vpbroadcastd 0x520(%rsi), %ymm2+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm8, 0x260(%rdi)+ vmovdqa %ymm9, 0x2e0(%rdi)+ vmovdqa %ymm10, 0x360(%rdi)+ vmovdqa %ymm11, 0x3e0(%rdi)+ vmovdqa 0x40(%rsi), %ymm1+ vmovdqa 0x60(%rsi), %ymm2+ vpmuldq %ymm1, %ymm4, %ymm12+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm4, %ymm8 # ymm8 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm9 # ymm9 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm4, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm8[1],ymm4[2],ymm8[3],ymm4[4],ymm8[5],ymm4[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm9[1],ymm5[2],ymm9[3],ymm5[4],ymm9[5],ymm5[6],ymm9[7]+ vpmuldq %ymm1, %ymm6, %ymm12+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm6, %ymm8 # ymm8 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm9 # ymm9 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm6, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm8[1],ymm6[2],ymm8[3],ymm6[4],ymm8[5],ymm6[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vmovdqa %ymm4, 0x60(%rdi)+ vmovdqa %ymm5, 0xe0(%rdi)+ vmovdqa %ymm6, 0x160(%rdi)+ vmovdqa %ymm7, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(invntt_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_ntt_avx2_asm.S view
@@ -0,0 +1,2405 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++ /*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: ntt_avx2_asm+ Description: x86_64 AVX2 forward NTT+ Signature: void mld_ntt_avx2_asm(int32_t *r, const int32_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *r+ description: Input/output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 2496+ permissions: read-only+ c_parameter: const int32_t *qdata+ description: Precomputed constants (624 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_ntt_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(ntt_avx2_asm)+MLD_ASM_FN_SYMBOL(ntt_avx2_asm)++ .cfi_startproc+ vmovdqa (%rsi), %ymm0+ vpbroadcastd 0x84(%rsi), %ymm1+ vpbroadcastd 0x524(%rsi), %ymm2+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x80(%rdi), %ymm5+ vmovdqa 0x100(%rdi), %ymm6+ vmovdqa 0x180(%rdi), %ymm7+ vmovdqa 0x200(%rdi), %ymm8+ vmovdqa 0x280(%rdi), %ymm9+ vmovdqa 0x300(%rdi), %ymm10+ vmovdqa 0x380(%rdi), %ymm11+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm5, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm5, %ymm5+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa %ymm4, (%rdi)+ vmovdqa %ymm5, 0x80(%rdi)+ vmovdqa %ymm6, 0x100(%rdi)+ vmovdqa %ymm7, 0x180(%rdi)+ vmovdqa %ymm8, 0x200(%rdi)+ vmovdqa %ymm9, 0x280(%rdi)+ vmovdqa %ymm10, 0x300(%rdi)+ vmovdqa %ymm11, 0x380(%rdi)+ vpbroadcastd 0x84(%rsi), %ymm1+ vpbroadcastd 0x524(%rsi), %ymm2+ vmovdqa 0x20(%rdi), %ymm4+ vmovdqa 0xa0(%rdi), %ymm5+ vmovdqa 0x120(%rdi), %ymm6+ vmovdqa 0x1a0(%rdi), %ymm7+ vmovdqa 0x220(%rdi), %ymm8+ vmovdqa 0x2a0(%rdi), %ymm9+ vmovdqa 0x320(%rdi), %ymm10+ vmovdqa 0x3a0(%rdi), %ymm11+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm5, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm5, %ymm5+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa %ymm4, 0x20(%rdi)+ vmovdqa %ymm5, 0xa0(%rdi)+ vmovdqa %ymm6, 0x120(%rdi)+ vmovdqa %ymm7, 0x1a0(%rdi)+ vmovdqa %ymm8, 0x220(%rdi)+ vmovdqa %ymm9, 0x2a0(%rdi)+ vmovdqa %ymm10, 0x320(%rdi)+ vmovdqa %ymm11, 0x3a0(%rdi)+ vpbroadcastd 0x84(%rsi), %ymm1+ vpbroadcastd 0x524(%rsi), %ymm2+ vmovdqa 0x40(%rdi), %ymm4+ vmovdqa 0xc0(%rdi), %ymm5+ vmovdqa 0x140(%rdi), %ymm6+ vmovdqa 0x1c0(%rdi), %ymm7+ vmovdqa 0x240(%rdi), %ymm8+ vmovdqa 0x2c0(%rdi), %ymm9+ vmovdqa 0x340(%rdi), %ymm10+ vmovdqa 0x3c0(%rdi), %ymm11+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm5, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm5, %ymm5+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa %ymm4, 0x40(%rdi)+ vmovdqa %ymm5, 0xc0(%rdi)+ vmovdqa %ymm6, 0x140(%rdi)+ vmovdqa %ymm7, 0x1c0(%rdi)+ vmovdqa %ymm8, 0x240(%rdi)+ vmovdqa %ymm9, 0x2c0(%rdi)+ vmovdqa %ymm10, 0x340(%rdi)+ vmovdqa %ymm11, 0x3c0(%rdi)+ vpbroadcastd 0x84(%rsi), %ymm1+ vpbroadcastd 0x524(%rsi), %ymm2+ vmovdqa 0x60(%rdi), %ymm4+ vmovdqa 0xe0(%rdi), %ymm5+ vmovdqa 0x160(%rdi), %ymm6+ vmovdqa 0x1e0(%rdi), %ymm7+ vmovdqa 0x260(%rdi), %ymm8+ vmovdqa 0x2e0(%rdi), %ymm9+ vmovdqa 0x360(%rdi), %ymm10+ vmovdqa 0x3e0(%rdi), %ymm11+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm5, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm5, %ymm5+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa %ymm4, 0x60(%rdi)+ vmovdqa %ymm5, 0xe0(%rdi)+ vmovdqa %ymm6, 0x160(%rdi)+ vmovdqa %ymm7, 0x1e0(%rdi)+ vmovdqa %ymm8, 0x260(%rdi)+ vmovdqa %ymm9, 0x2e0(%rdi)+ vmovdqa %ymm10, 0x360(%rdi)+ vmovdqa %ymm11, 0x3e0(%rdi)+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x20(%rdi), %ymm5+ vmovdqa 0x40(%rdi), %ymm6+ vmovdqa 0x60(%rdi), %ymm7+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm9+ vmovdqa 0xc0(%rdi), %ymm10+ vmovdqa 0xe0(%rdi), %ymm11+ vpbroadcastd 0x90(%rsi), %ymm1+ vpbroadcastd 0x530(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vmovdqa 0xa0(%rsi), %ymm1+ vmovdqa 0x540(%rsi), %ymm2+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm3, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovdqa 0x120(%rsi), %ymm1+ vmovdqa 0x5c0(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm7, %ymm12+ vpaddd %ymm7, %ymm8, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm5, %ymm12+ vpaddd %ymm6, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm3, %ymm12+ vpaddd %ymm4, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm10, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm10, %ymm10+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa 0x1a0(%rsi), %ymm1+ vmovdqa 0x640(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm7, %ymm12+ vpaddd %ymm3, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm6, %ymm12+ vpaddd %ymm6, %ymm11, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm6, %ymm6+ vmovdqa 0x220(%rsi), %ymm1+ vmovdqa 0x6c0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm9, %ymm12+ vpaddd %ymm7, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm8, %ymm8+ vmovdqa 0x2a0(%rsi), %ymm1+ vmovdqa 0x740(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm3, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm4, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm4, %ymm4+ vmovdqa 0x320(%rsi), %ymm1+ vmovdqa 0x7c0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm8, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa 0x3a0(%rsi), %ymm1+ vmovdqa 0x840(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm6, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vmovdqa 0x420(%rsi), %ymm1+ vmovdqa 0x8c0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm4, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vmovdqa 0x4a0(%rsi), %ymm1+ vmovdqa 0x940(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm3, %ymm12+ vpaddd %ymm3, %ymm11, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm3, %ymm3+ vmovdqa %ymm9, (%rdi)+ vmovdqa %ymm8, 0x20(%rdi)+ vmovdqa %ymm7, 0x40(%rdi)+ vmovdqa %ymm6, 0x60(%rdi)+ vmovdqa %ymm5, 0x80(%rdi)+ vmovdqa %ymm4, 0xa0(%rdi)+ vmovdqa %ymm3, 0xc0(%rdi)+ vmovdqa %ymm11, 0xe0(%rdi)+ vmovdqa 0x100(%rdi), %ymm4+ vmovdqa 0x120(%rdi), %ymm5+ vmovdqa 0x140(%rdi), %ymm6+ vmovdqa 0x160(%rdi), %ymm7+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1a0(%rdi), %ymm9+ vmovdqa 0x1c0(%rdi), %ymm10+ vmovdqa 0x1e0(%rdi), %ymm11+ vpbroadcastd 0x94(%rsi), %ymm1+ vpbroadcastd 0x534(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vmovdqa 0xc0(%rsi), %ymm1+ vmovdqa 0x560(%rsi), %ymm2+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm3, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovdqa 0x140(%rsi), %ymm1+ vmovdqa 0x5e0(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm7, %ymm12+ vpaddd %ymm7, %ymm8, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm5, %ymm12+ vpaddd %ymm6, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm3, %ymm12+ vpaddd %ymm4, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm10, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm10, %ymm10+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa 0x1c0(%rsi), %ymm1+ vmovdqa 0x660(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm7, %ymm12+ vpaddd %ymm3, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm6, %ymm12+ vpaddd %ymm6, %ymm11, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm6, %ymm6+ vmovdqa 0x240(%rsi), %ymm1+ vmovdqa 0x6e0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm9, %ymm12+ vpaddd %ymm7, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm8, %ymm8+ vmovdqa 0x2c0(%rsi), %ymm1+ vmovdqa 0x760(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm3, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm4, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm4, %ymm4+ vmovdqa 0x340(%rsi), %ymm1+ vmovdqa 0x7e0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm8, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa 0x3c0(%rsi), %ymm1+ vmovdqa 0x860(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm6, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vmovdqa 0x440(%rsi), %ymm1+ vmovdqa 0x8e0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm4, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vmovdqa 0x4c0(%rsi), %ymm1+ vmovdqa 0x960(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm3, %ymm12+ vpaddd %ymm3, %ymm11, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm3, %ymm3+ vmovdqa %ymm9, 0x100(%rdi)+ vmovdqa %ymm8, 0x120(%rdi)+ vmovdqa %ymm7, 0x140(%rdi)+ vmovdqa %ymm6, 0x160(%rdi)+ vmovdqa %ymm5, 0x180(%rdi)+ vmovdqa %ymm4, 0x1a0(%rdi)+ vmovdqa %ymm3, 0x1c0(%rdi)+ vmovdqa %ymm11, 0x1e0(%rdi)+ vmovdqa 0x200(%rdi), %ymm4+ vmovdqa 0x220(%rdi), %ymm5+ vmovdqa 0x240(%rdi), %ymm6+ vmovdqa 0x260(%rdi), %ymm7+ vmovdqa 0x280(%rdi), %ymm8+ vmovdqa 0x2a0(%rdi), %ymm9+ vmovdqa 0x2c0(%rdi), %ymm10+ vmovdqa 0x2e0(%rdi), %ymm11+ vpbroadcastd 0x98(%rsi), %ymm1+ vpbroadcastd 0x538(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vmovdqa 0xe0(%rsi), %ymm1+ vmovdqa 0x580(%rsi), %ymm2+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm3, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovdqa 0x160(%rsi), %ymm1+ vmovdqa 0x600(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm7, %ymm12+ vpaddd %ymm7, %ymm8, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm5, %ymm12+ vpaddd %ymm6, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm3, %ymm12+ vpaddd %ymm4, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm10, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm10, %ymm10+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa 0x1e0(%rsi), %ymm1+ vmovdqa 0x680(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm7, %ymm12+ vpaddd %ymm3, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm6, %ymm12+ vpaddd %ymm6, %ymm11, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm6, %ymm6+ vmovdqa 0x260(%rsi), %ymm1+ vmovdqa 0x700(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm9, %ymm12+ vpaddd %ymm7, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm8, %ymm8+ vmovdqa 0x2e0(%rsi), %ymm1+ vmovdqa 0x780(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm3, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm4, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm4, %ymm4+ vmovdqa 0x360(%rsi), %ymm1+ vmovdqa 0x800(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm8, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa 0x3e0(%rsi), %ymm1+ vmovdqa 0x880(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm6, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vmovdqa 0x460(%rsi), %ymm1+ vmovdqa 0x900(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm4, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vmovdqa 0x4e0(%rsi), %ymm1+ vmovdqa 0x980(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm3, %ymm12+ vpaddd %ymm3, %ymm11, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm3, %ymm3+ vmovdqa %ymm9, 0x200(%rdi)+ vmovdqa %ymm8, 0x220(%rdi)+ vmovdqa %ymm7, 0x240(%rdi)+ vmovdqa %ymm6, 0x260(%rdi)+ vmovdqa %ymm5, 0x280(%rdi)+ vmovdqa %ymm4, 0x2a0(%rdi)+ vmovdqa %ymm3, 0x2c0(%rdi)+ vmovdqa %ymm11, 0x2e0(%rdi)+ vmovdqa 0x300(%rdi), %ymm4+ vmovdqa 0x320(%rdi), %ymm5+ vmovdqa 0x340(%rdi), %ymm6+ vmovdqa 0x360(%rdi), %ymm7+ vmovdqa 0x380(%rdi), %ymm8+ vmovdqa 0x3a0(%rdi), %ymm9+ vmovdqa 0x3c0(%rdi), %ymm10+ vmovdqa 0x3e0(%rdi), %ymm11+ vpbroadcastd 0x9c(%rsi), %ymm1+ vpbroadcastd 0x53c(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vmovdqa 0x100(%rsi), %ymm1+ vmovdqa 0x5a0(%rsi), %ymm2+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm3, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovdqa 0x180(%rsi), %ymm1+ vmovdqa 0x620(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm7, %ymm12+ vpaddd %ymm7, %ymm8, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm5, %ymm12+ vpaddd %ymm6, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm3, %ymm12+ vpaddd %ymm4, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm10, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm10, %ymm10+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa 0x200(%rsi), %ymm1+ vmovdqa 0x6a0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm7, %ymm12+ vpaddd %ymm3, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm6, %ymm12+ vpaddd %ymm6, %ymm11, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm6, %ymm6+ vmovdqa 0x280(%rsi), %ymm1+ vmovdqa 0x720(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm9, %ymm12+ vpaddd %ymm7, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm8, %ymm8+ vmovdqa 0x300(%rsi), %ymm1+ vmovdqa 0x7a0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm3, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm4, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm4, %ymm4+ vmovdqa 0x380(%rsi), %ymm1+ vmovdqa 0x820(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm8, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa 0x400(%rsi), %ymm1+ vmovdqa 0x8a0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm6, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vmovdqa 0x480(%rsi), %ymm1+ vmovdqa 0x920(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm4, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vmovdqa 0x500(%rsi), %ymm1+ vmovdqa 0x9a0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm3, %ymm12+ vpaddd %ymm3, %ymm11, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm3, %ymm3+ vmovdqa %ymm9, 0x300(%rdi)+ vmovdqa %ymm8, 0x320(%rdi)+ vmovdqa %ymm7, 0x340(%rdi)+ vmovdqa %ymm6, 0x360(%rdi)+ vmovdqa %ymm5, 0x380(%rdi)+ vmovdqa %ymm4, 0x3a0(%rdi)+ vmovdqa %ymm3, 0x3c0(%rdi)+ vmovdqa %ymm11, 0x3e0(%rdi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(ntt_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_nttunpack_avx2_asm.S view
@@ -0,0 +1,254 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: nttunpack_avx2_asm+ Description: x86_64 AVX2 NTT coefficient unpacking/permutation+ Signature: void mld_nttunpack_avx2_asm(int32_t *r)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *r+ description: Input/output polynomial (256 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_nttunpack_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(nttunpack_avx2_asm)+MLD_ASM_FN_SYMBOL(nttunpack_avx2_asm)++ .cfi_startproc+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x20(%rdi), %ymm5+ vmovdqa 0x40(%rdi), %ymm6+ vmovdqa 0x60(%rdi), %ymm7+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm9+ vmovdqa 0xc0(%rdi), %ymm10+ vmovdqa 0xe0(%rdi), %ymm11+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa %ymm9, (%rdi)+ vmovdqa %ymm8, 0x20(%rdi)+ vmovdqa %ymm7, 0x40(%rdi)+ vmovdqa %ymm6, 0x60(%rdi)+ vmovdqa %ymm5, 0x80(%rdi)+ vmovdqa %ymm4, 0xa0(%rdi)+ vmovdqa %ymm3, 0xc0(%rdi)+ vmovdqa %ymm11, 0xe0(%rdi)+ vmovdqa 0x100(%rdi), %ymm4+ vmovdqa 0x120(%rdi), %ymm5+ vmovdqa 0x140(%rdi), %ymm6+ vmovdqa 0x160(%rdi), %ymm7+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1a0(%rdi), %ymm9+ vmovdqa 0x1c0(%rdi), %ymm10+ vmovdqa 0x1e0(%rdi), %ymm11+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa %ymm9, 0x100(%rdi)+ vmovdqa %ymm8, 0x120(%rdi)+ vmovdqa %ymm7, 0x140(%rdi)+ vmovdqa %ymm6, 0x160(%rdi)+ vmovdqa %ymm5, 0x180(%rdi)+ vmovdqa %ymm4, 0x1a0(%rdi)+ vmovdqa %ymm3, 0x1c0(%rdi)+ vmovdqa %ymm11, 0x1e0(%rdi)+ vmovdqa 0x200(%rdi), %ymm4+ vmovdqa 0x220(%rdi), %ymm5+ vmovdqa 0x240(%rdi), %ymm6+ vmovdqa 0x260(%rdi), %ymm7+ vmovdqa 0x280(%rdi), %ymm8+ vmovdqa 0x2a0(%rdi), %ymm9+ vmovdqa 0x2c0(%rdi), %ymm10+ vmovdqa 0x2e0(%rdi), %ymm11+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa %ymm9, 0x200(%rdi)+ vmovdqa %ymm8, 0x220(%rdi)+ vmovdqa %ymm7, 0x240(%rdi)+ vmovdqa %ymm6, 0x260(%rdi)+ vmovdqa %ymm5, 0x280(%rdi)+ vmovdqa %ymm4, 0x2a0(%rdi)+ vmovdqa %ymm3, 0x2c0(%rdi)+ vmovdqa %ymm11, 0x2e0(%rdi)+ vmovdqa 0x300(%rdi), %ymm4+ vmovdqa 0x320(%rdi), %ymm5+ vmovdqa 0x340(%rdi), %ymm6+ vmovdqa 0x360(%rdi), %ymm7+ vmovdqa 0x380(%rdi), %ymm8+ vmovdqa 0x3a0(%rdi), %ymm9+ vmovdqa 0x3c0(%rdi), %ymm10+ vmovdqa 0x3e0(%rdi), %ymm11+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa %ymm9, 0x300(%rdi)+ vmovdqa %ymm8, 0x320(%rdi)+ vmovdqa %ymm7, 0x340(%rdi)+ vmovdqa %ymm6, 0x360(%rdi)+ vmovdqa %ymm5, 0x380(%rdi)+ vmovdqa %ymm4, 0x3a0(%rdi)+ vmovdqa %ymm3, 0x3c0(%rdi)+ vmovdqa %ymm11, 0x3e0(%rdi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(nttunpack_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_pointwise_acc_l4_avx2_asm.S view
@@ -0,0 +1,173 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: pointwise_acc_l4_avx2_asm+ Description: x86_64 AVX2 pointwise multiply-accumulate of length-4 polynomial vectors+ Signature: void mld_pointwise_acc_l4_avx2_asm(int32_t *c, const int32_t a[4][256], const int32_t b[4][256], const int32_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *c+ description: Output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 4096+ permissions: read-only+ c_parameter: const int32_t a[4][256]+ description: Input polynomial vector a (4 x 256 x int32_t)+ rdx:+ type: buffer+ size_bytes: 4096+ permissions: read-only+ c_parameter: const int32_t b[4][256]+ description: Input polynomial vector b (4 x 256 x int32_t)+ rcx:+ type: buffer+ size_bytes: 2496+ permissions: read-only+ c_parameter: const int32_t *qdata+ description: Precomputed constants (624 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 4)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_pointwise_acc_l4_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(pointwise_acc_l4_avx2_asm)+MLD_ASM_FN_SYMBOL(pointwise_acc_l4_avx2_asm)++ .cfi_startproc+ vmovdqa 0x20(%rcx), %ymm0+ vmovdqa (%rcx), %ymm1+ xorl %eax, %eax++Lmld_pointwise_acc_l4_avx2_looptop2:+ vmovdqa (%rsi), %ymm6+ vmovdqa 0x20(%rsi), %ymm8+ vmovdqa (%rdx), %ymm10+ vmovdqa 0x20(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vmovdqa %ymm6, %ymm2+ vmovdqa %ymm7, %ymm3+ vmovdqa %ymm8, %ymm4+ vmovdqa %ymm9, %ymm5+ vmovdqa 0x400(%rsi), %ymm6+ vmovdqa 0x420(%rsi), %ymm8+ vmovdqa 0x400(%rdx), %ymm10+ vmovdqa 0x420(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0x800(%rsi), %ymm6+ vmovdqa 0x820(%rsi), %ymm8+ vmovdqa 0x800(%rdx), %ymm10+ vmovdqa 0x820(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0xc00(%rsi), %ymm6+ vmovdqa 0xc20(%rsi), %ymm8+ vmovdqa 0xc00(%rdx), %ymm10+ vmovdqa 0xc20(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vpmuldq %ymm2, %ymm0, %ymm6+ vpmuldq %ymm3, %ymm0, %ymm7+ vpmuldq %ymm4, %ymm0, %ymm8+ vpmuldq %ymm5, %ymm0, %ymm9+ vpmuldq %ymm6, %ymm1, %ymm6+ vpmuldq %ymm7, %ymm1, %ymm7+ vpmuldq %ymm8, %ymm1, %ymm8+ vpmuldq %ymm9, %ymm1, %ymm9+ vpsubq %ymm6, %ymm2, %ymm2+ vpsubq %ymm7, %ymm3, %ymm3+ vpsubq %ymm8, %ymm4, %ymm4+ vpsubq %ymm9, %ymm5, %ymm5+ vpsrlq $0x20, %ymm2, %ymm2+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm2, %ymm2 # ymm2 = ymm2[0],ymm3[1],ymm2[2],ymm3[3],ymm2[4],ymm3[5],ymm2[6],ymm3[7]+ vpblendd $0xaa, %ymm5, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovdqa %ymm2, (%rdi)+ vmovdqa %ymm4, 0x20(%rdi)+ addq $0x40, %rsi+ addq $0x40, %rdx+ addq $0x40, %rdi+ addl $0x1, %eax+ cmpl $0x10, %eax+ jb Lmld_pointwise_acc_l4_avx2_looptop2+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(pointwise_acc_l4_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ && (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 4) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_pointwise_acc_l5_avx2_asm.S view
@@ -0,0 +1,189 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: pointwise_acc_l5_avx2_asm+ Description: x86_64 AVX2 pointwise multiply-accumulate of length-5 polynomial vectors+ Signature: void mld_pointwise_acc_l5_avx2_asm(int32_t *c, const int32_t a[5][256], const int32_t b[5][256], const int32_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *c+ description: Output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 5120+ permissions: read-only+ c_parameter: const int32_t a[5][256]+ description: Input polynomial vector a (5 x 256 x int32_t)+ rdx:+ type: buffer+ size_bytes: 5120+ permissions: read-only+ c_parameter: const int32_t b[5][256]+ description: Input polynomial vector b (5 x 256 x int32_t)+ rcx:+ type: buffer+ size_bytes: 2496+ permissions: read-only+ c_parameter: const int32_t *qdata+ description: Precomputed constants (624 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 5)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_pointwise_acc_l5_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(pointwise_acc_l5_avx2_asm)+MLD_ASM_FN_SYMBOL(pointwise_acc_l5_avx2_asm)++ .cfi_startproc+ vmovdqa 0x20(%rcx), %ymm0+ vmovdqa (%rcx), %ymm1+ xorl %eax, %eax++Lmld_pointwise_acc_l5_avx2_looptop2:+ vmovdqa (%rsi), %ymm6+ vmovdqa 0x20(%rsi), %ymm8+ vmovdqa (%rdx), %ymm10+ vmovdqa 0x20(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vmovdqa %ymm6, %ymm2+ vmovdqa %ymm7, %ymm3+ vmovdqa %ymm8, %ymm4+ vmovdqa %ymm9, %ymm5+ vmovdqa 0x400(%rsi), %ymm6+ vmovdqa 0x420(%rsi), %ymm8+ vmovdqa 0x400(%rdx), %ymm10+ vmovdqa 0x420(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0x800(%rsi), %ymm6+ vmovdqa 0x820(%rsi), %ymm8+ vmovdqa 0x800(%rdx), %ymm10+ vmovdqa 0x820(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0xc00(%rsi), %ymm6+ vmovdqa 0xc20(%rsi), %ymm8+ vmovdqa 0xc00(%rdx), %ymm10+ vmovdqa 0xc20(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0x1000(%rsi), %ymm6+ vmovdqa 0x1020(%rsi), %ymm8+ vmovdqa 0x1000(%rdx), %ymm10+ vmovdqa 0x1020(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vpmuldq %ymm2, %ymm0, %ymm6+ vpmuldq %ymm3, %ymm0, %ymm7+ vpmuldq %ymm4, %ymm0, %ymm8+ vpmuldq %ymm5, %ymm0, %ymm9+ vpmuldq %ymm6, %ymm1, %ymm6+ vpmuldq %ymm7, %ymm1, %ymm7+ vpmuldq %ymm8, %ymm1, %ymm8+ vpmuldq %ymm9, %ymm1, %ymm9+ vpsubq %ymm6, %ymm2, %ymm2+ vpsubq %ymm7, %ymm3, %ymm3+ vpsubq %ymm8, %ymm4, %ymm4+ vpsubq %ymm9, %ymm5, %ymm5+ vpsrlq $0x20, %ymm2, %ymm2+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm2, %ymm2 # ymm2 = ymm2[0],ymm3[1],ymm2[2],ymm3[3],ymm2[4],ymm3[5],ymm2[6],ymm3[7]+ vpblendd $0xaa, %ymm5, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovdqa %ymm2, (%rdi)+ vmovdqa %ymm4, 0x20(%rdi)+ addq $0x40, %rsi+ addq $0x40, %rdx+ addq $0x40, %rdi+ addl $0x1, %eax+ cmpl $0x10, %eax+ jb Lmld_pointwise_acc_l5_avx2_looptop2+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(pointwise_acc_l5_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ && (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 5) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_pointwise_acc_l7_avx2_asm.S view
@@ -0,0 +1,221 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: pointwise_acc_l7_avx2_asm+ Description: x86_64 AVX2 pointwise multiply-accumulate of length-7 polynomial vectors+ Signature: void mld_pointwise_acc_l7_avx2_asm(int32_t *c, const int32_t a[7][256], const int32_t b[7][256], const int32_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *c+ description: Output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 7168+ permissions: read-only+ c_parameter: const int32_t a[7][256]+ description: Input polynomial vector a (7 x 256 x int32_t)+ rdx:+ type: buffer+ size_bytes: 7168+ permissions: read-only+ c_parameter: const int32_t b[7][256]+ description: Input polynomial vector b (7 x 256 x int32_t)+ rcx:+ type: buffer+ size_bytes: 2496+ permissions: read-only+ c_parameter: const int32_t *qdata+ description: Precomputed constants (624 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 7)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_pointwise_acc_l7_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(pointwise_acc_l7_avx2_asm)+MLD_ASM_FN_SYMBOL(pointwise_acc_l7_avx2_asm)++ .cfi_startproc+ vmovdqa 0x20(%rcx), %ymm0+ vmovdqa (%rcx), %ymm1+ xorl %eax, %eax++Lmld_pointwise_acc_l7_avx2_looptop2:+ vmovdqa (%rsi), %ymm6+ vmovdqa 0x20(%rsi), %ymm8+ vmovdqa (%rdx), %ymm10+ vmovdqa 0x20(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vmovdqa %ymm6, %ymm2+ vmovdqa %ymm7, %ymm3+ vmovdqa %ymm8, %ymm4+ vmovdqa %ymm9, %ymm5+ vmovdqa 0x400(%rsi), %ymm6+ vmovdqa 0x420(%rsi), %ymm8+ vmovdqa 0x400(%rdx), %ymm10+ vmovdqa 0x420(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0x800(%rsi), %ymm6+ vmovdqa 0x820(%rsi), %ymm8+ vmovdqa 0x800(%rdx), %ymm10+ vmovdqa 0x820(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0xc00(%rsi), %ymm6+ vmovdqa 0xc20(%rsi), %ymm8+ vmovdqa 0xc00(%rdx), %ymm10+ vmovdqa 0xc20(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0x1000(%rsi), %ymm6+ vmovdqa 0x1020(%rsi), %ymm8+ vmovdqa 0x1000(%rdx), %ymm10+ vmovdqa 0x1020(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0x1400(%rsi), %ymm6+ vmovdqa 0x1420(%rsi), %ymm8+ vmovdqa 0x1400(%rdx), %ymm10+ vmovdqa 0x1420(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0x1800(%rsi), %ymm6+ vmovdqa 0x1820(%rsi), %ymm8+ vmovdqa 0x1800(%rdx), %ymm10+ vmovdqa 0x1820(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vpmuldq %ymm2, %ymm0, %ymm6+ vpmuldq %ymm3, %ymm0, %ymm7+ vpmuldq %ymm4, %ymm0, %ymm8+ vpmuldq %ymm5, %ymm0, %ymm9+ vpmuldq %ymm6, %ymm1, %ymm6+ vpmuldq %ymm7, %ymm1, %ymm7+ vpmuldq %ymm8, %ymm1, %ymm8+ vpmuldq %ymm9, %ymm1, %ymm9+ vpsubq %ymm6, %ymm2, %ymm2+ vpsubq %ymm7, %ymm3, %ymm3+ vpsubq %ymm8, %ymm4, %ymm4+ vpsubq %ymm9, %ymm5, %ymm5+ vpsrlq $0x20, %ymm2, %ymm2+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm2, %ymm2 # ymm2 = ymm2[0],ymm3[1],ymm2[2],ymm3[3],ymm2[4],ymm3[5],ymm2[6],ymm3[7]+ vpblendd $0xaa, %ymm5, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovdqa %ymm2, (%rdi)+ vmovdqa %ymm4, 0x20(%rdi)+ addq $0x40, %rsi+ addq $0x40, %rdx+ addq $0x40, %rdi+ addl $0x1, %eax+ cmpl $0x10, %eax+ jb Lmld_pointwise_acc_l7_avx2_looptop2+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(pointwise_acc_l7_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ && (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 7) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_pointwise_avx2_asm.S view
@@ -0,0 +1,158 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: pointwise_avx2_asm+ Description: x86_64 AVX2 pointwise Montgomery multiplication+ Signature: void mld_pointwise_avx2_asm(int32_t *a, const int32_t *b, const int32_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *a+ description: Input/output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t *b+ description: Input polynomial (256 x int32_t)+ rdx:+ type: buffer+ size_bytes: 2496+ permissions: read-only+ c_parameter: const int32_t *qdata+ description: Precomputed constants (624 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_pointwise_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(pointwise_avx2_asm)+MLD_ASM_FN_SYMBOL(pointwise_avx2_asm)++ .cfi_startproc+ vmovdqa 0x20(%rdx), %ymm0+ vmovdqa (%rdx), %ymm1+ xorl %eax, %eax++Lmld_pointwise_avx2_looptop1:+ vmovdqa (%rdi), %ymm2+ vmovdqa 0x20(%rdi), %ymm4+ vmovdqa 0x40(%rdi), %ymm6+ vmovdqa (%rsi), %ymm10+ vmovdqa 0x20(%rsi), %ymm12+ vmovdqa 0x40(%rsi), %ymm14+ vpsrlq $0x20, %ymm2, %ymm3+ vpsrlq $0x20, %ymm4, %ymm5+ vmovshdup %ymm6, %ymm7 # ymm7 = ymm6[1,1,3,3,5,5,7,7]+ vpsrlq $0x20, %ymm10, %ymm11+ vpsrlq $0x20, %ymm12, %ymm13+ vmovshdup %ymm14, %ymm15 # ymm15 = ymm14[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm2, %ymm2+ vpmuldq %ymm11, %ymm3, %ymm3+ vpmuldq %ymm12, %ymm4, %ymm4+ vpmuldq %ymm13, %ymm5, %ymm5+ vpmuldq %ymm14, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm0, %ymm10+ vpmuldq %ymm3, %ymm0, %ymm11+ vpmuldq %ymm4, %ymm0, %ymm12+ vpmuldq %ymm5, %ymm0, %ymm13+ vpmuldq %ymm6, %ymm0, %ymm14+ vpmuldq %ymm7, %ymm0, %ymm15+ vpmuldq %ymm10, %ymm1, %ymm10+ vpmuldq %ymm11, %ymm1, %ymm11+ vpmuldq %ymm12, %ymm1, %ymm12+ vpmuldq %ymm13, %ymm1, %ymm13+ vpmuldq %ymm14, %ymm1, %ymm14+ vpmuldq %ymm15, %ymm1, %ymm15+ vpsubq %ymm10, %ymm2, %ymm2+ vpsubq %ymm11, %ymm3, %ymm3+ vpsubq %ymm12, %ymm4, %ymm4+ vpsubq %ymm13, %ymm5, %ymm5+ vpsubq %ymm14, %ymm6, %ymm6+ vpsubq %ymm15, %ymm7, %ymm7+ vpsrlq $0x20, %ymm2, %ymm2+ vpsrlq $0x20, %ymm4, %ymm4+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm2, %ymm2 # ymm2 = ymm2[0],ymm3[1],ymm2[2],ymm3[3],ymm2[4],ymm3[5],ymm2[6],ymm3[7]+ vpblendd $0xaa, %ymm5, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vpblendd $0xaa, %ymm7, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7]+ vmovdqa %ymm2, (%rdi)+ vmovdqa %ymm4, 0x20(%rdi)+ vmovdqa %ymm6, 0x40(%rdi)+ addq $0x60, %rdi+ addq $0x60, %rsi+ addl $0x1, %eax+ cmpl $0xa, %eax+ jb Lmld_pointwise_avx2_looptop1+ vmovdqa (%rdi), %ymm2+ vmovdqa 0x20(%rdi), %ymm4+ vmovdqa (%rsi), %ymm10+ vmovdqa 0x20(%rsi), %ymm12+ vpsrlq $0x20, %ymm2, %ymm3+ vpsrlq $0x20, %ymm4, %ymm5+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm2, %ymm2+ vpmuldq %ymm11, %ymm3, %ymm3+ vpmuldq %ymm12, %ymm4, %ymm4+ vpmuldq %ymm13, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm0, %ymm10+ vpmuldq %ymm3, %ymm0, %ymm11+ vpmuldq %ymm4, %ymm0, %ymm12+ vpmuldq %ymm5, %ymm0, %ymm13+ vpmuldq %ymm10, %ymm1, %ymm10+ vpmuldq %ymm11, %ymm1, %ymm11+ vpmuldq %ymm12, %ymm1, %ymm12+ vpmuldq %ymm13, %ymm1, %ymm13+ vpsubq %ymm10, %ymm2, %ymm2+ vpsubq %ymm11, %ymm3, %ymm3+ vpsubq %ymm12, %ymm4, %ymm4+ vpsubq %ymm13, %ymm5, %ymm5+ vpsrlq $0x20, %ymm2, %ymm2+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0x55, %ymm2, %ymm3, %ymm2 # ymm2 = ymm2[0],ymm3[1],ymm2[2],ymm3[3],ymm2[4],ymm3[5],ymm2[6],ymm3[7]+ vpblendd $0x55, %ymm4, %ymm5, %ymm4 # ymm4 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovdqa %ymm2, (%rdi)+ vmovdqa %ymm4, 0x20(%rdi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(pointwise_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_poly_caddq_avx2_asm.S view
@@ -0,0 +1,199 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */+++/*yaml+ Name: poly_caddq_avx2_asm+ Description: x86_64 AVX2 conditional addition of q to each coefficient.+ For all coefficients of the in/out polynomial, add Q if the coefficient+ is negative.+ Signature: void mld_poly_caddq_avx2_asm(int32_t *r)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *r+ description: Input/output polynomial (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)+++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_poly_caddq_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_caddq_avx2_asm)+MLD_ASM_FN_SYMBOL(poly_caddq_avx2_asm)++ .cfi_startproc+ vpxor %xmm2, %xmm2, %xmm2+ movl $0x7fe001, %eax # imm = 0x7FE001+ vmovd %eax, %xmm1+ vpbroadcastd %xmm1, %ymm1+ vpcmpgtd (%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd (%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, (%rdi)+ vpcmpgtd 0x20(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0x20(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0x20(%rdi)+ vpcmpgtd 0x40(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0x40(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0x40(%rdi)+ vpcmpgtd 0x60(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0x60(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0x60(%rdi)+ vpcmpgtd 0x80(%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd 0x80(%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, 0x80(%rdi)+ vpcmpgtd 0xa0(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0xa0(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0xa0(%rdi)+ vpcmpgtd 0xc0(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0xc0(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0xc0(%rdi)+ vpcmpgtd 0xe0(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0xe0(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0xe0(%rdi)+ vpcmpgtd 0x100(%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd 0x100(%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, 0x100(%rdi)+ vpcmpgtd 0x120(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0x120(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0x120(%rdi)+ vpcmpgtd 0x140(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0x140(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0x140(%rdi)+ vpcmpgtd 0x160(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0x160(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0x160(%rdi)+ vpcmpgtd 0x180(%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd 0x180(%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, 0x180(%rdi)+ vpcmpgtd 0x1a0(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0x1a0(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0x1a0(%rdi)+ vpcmpgtd 0x1c0(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0x1c0(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0x1c0(%rdi)+ vpcmpgtd 0x1e0(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0x1e0(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0x1e0(%rdi)+ vpcmpgtd 0x200(%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd 0x200(%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, 0x200(%rdi)+ vpcmpgtd 0x220(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0x220(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0x220(%rdi)+ vpcmpgtd 0x240(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0x240(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0x240(%rdi)+ vpcmpgtd 0x260(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0x260(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0x260(%rdi)+ vpcmpgtd 0x280(%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd 0x280(%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, 0x280(%rdi)+ vpcmpgtd 0x2a0(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0x2a0(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0x2a0(%rdi)+ vpcmpgtd 0x2c0(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0x2c0(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0x2c0(%rdi)+ vpcmpgtd 0x2e0(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0x2e0(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0x2e0(%rdi)+ vpcmpgtd 0x300(%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd 0x300(%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, 0x300(%rdi)+ vpcmpgtd 0x320(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0x320(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0x320(%rdi)+ vpcmpgtd 0x340(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0x340(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0x340(%rdi)+ vpcmpgtd 0x360(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0x360(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0x360(%rdi)+ vpcmpgtd 0x380(%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd 0x380(%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, 0x380(%rdi)+ vpcmpgtd 0x3a0(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0x3a0(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0x3a0(%rdi)+ vpcmpgtd 0x3c0(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0x3c0(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0x3c0(%rdi)+ vpcmpgtd 0x3e0(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0x3e0(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0x3e0(%rdi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_caddq_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S view
@@ -0,0 +1,176 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */+++/*yaml+ Name: poly_chknorm_avx2_asm+ Description: x86_64 AVX2 infinity-norm bound check on polynomial coefficients.+ Check the infinity norm of the polynomial against the given bound B.+ Returns 0 if the norm is strictly smaller than B; otherwise returns 1+ (i.e. returns 1 if any |coefficient| >= B, 0 otherwise).+ Signature: int mld_poly_chknorm_avx2_asm(const int32_t *a, int32_t B)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t *a+ description: Input polynomial (256 x int32_t)+ rsi:+ type: scalar+ c_parameter: int32_t B+ description: Norm bound (must be non-negative)+ test_with: 131072 # representative non-negative bound (1 << 17)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)+++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_poly_chknorm_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_chknorm_avx2_asm)+MLD_ASM_FN_SYMBOL(poly_chknorm_avx2_asm)++ .cfi_startproc+ subl $0x1, %esi+ vpxor %xmm1, %xmm1, %xmm1+ vmovd %esi, %xmm2+ vpbroadcastd %xmm2, %ymm2+ vpabsd (%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0x20(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0x40(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0x60(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ vpabsd 0x80(%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0xa0(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0xc0(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0xe0(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ vpabsd 0x100(%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0x120(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0x140(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0x160(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ vpabsd 0x180(%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0x1a0(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0x1c0(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0x1e0(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ vpabsd 0x200(%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0x220(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0x240(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0x260(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ vpabsd 0x280(%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0x2a0(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0x2c0(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0x2e0(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ vpabsd 0x300(%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0x320(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0x340(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0x360(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ vpabsd 0x380(%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0x3a0(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0x3c0(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0x3e0(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ xorl %eax, %eax+ vptest %ymm1, %ymm1+ setne %al+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_chknorm_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S view
@@ -0,0 +1,490 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ *+ * The algorithm for Decompose(r) (more specifically the handling for the+ * wrap-around cases) is modified. See the AVX2 intrinsics version+ * (poly_decompose_32_avx2.c, predecessor of this file) for a more detailed+ * comparison.+ */+++/*yaml+ Name: poly_decompose_32_avx2_asm+ Description: x86_64 AVX2 coefficient decomposition (alpha = 2*(Q-1)/32).+ For all coefficients c of the input polynomial, compute high and low bits+ c0, c1 such that c = c1*ALPHA + c0 with -ALPHA/2 < c0 <= ALPHA/2, except if+ c1 = (Q-1)/ALPHA where we set c1 = 0 and -ALPHA/2 <= c0 = c - Q < 0.+ Assumes coefficients to be standard (unsigned canonical) representatives,+ i.e. 0 <= c < Q. For ML-DSA-65 / ML-DSA-87 (gamma2 = (Q-1)/32,+ alpha = 2*gamma2).+ Signature: void mld_poly_decompose_32_avx2_asm(int32_t *a1, int32_t *a0)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *a1+ description: Output high-part polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *a0+ description: Input polynomial / output low-part (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_NO_SIGN_API) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))+++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_decompose_32_avx2_asm)+MLD_ASM_FN_SYMBOL(poly_decompose_32_avx2_asm)++ .cfi_startproc+ movl $0x7f, %eax+ vmovd %eax, %xmm10+ vpbroadcastd %xmm10, %ymm10+ movl $0x401, %eax # imm = 0x401+ vmovd %eax, %xmm11+ vpbroadcastd %xmm11, %ymm11+ movl $0x200, %eax # imm = 0x200+ vmovd %eax, %xmm12+ vpbroadcastd %xmm12, %ymm12+ movl $0x7be100, %eax # imm = 0x7BE100+ vmovd %eax, %xmm13+ vpbroadcastd %xmm13, %ymm13+ movl $0x7fe00, %eax # imm = 0x7FE00+ vmovd %eax, %xmm14+ vpbroadcastd %xmm14, %ymm14+ vmovdqa (%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, (%rdi)+ vmovdqa %ymm2, (%rsi)+ vmovdqa 0x20(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x20(%rdi)+ vmovdqa %ymm2, 0x20(%rsi)+ vmovdqa 0x40(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x40(%rdi)+ vmovdqa %ymm2, 0x40(%rsi)+ vmovdqa 0x60(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x60(%rdi)+ vmovdqa %ymm2, 0x60(%rsi)+ vmovdqa 0x80(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x80(%rdi)+ vmovdqa %ymm2, 0x80(%rsi)+ vmovdqa 0xa0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0xa0(%rdi)+ vmovdqa %ymm2, 0xa0(%rsi)+ vmovdqa 0xc0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0xc0(%rdi)+ vmovdqa %ymm2, 0xc0(%rsi)+ vmovdqa 0xe0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0xe0(%rdi)+ vmovdqa %ymm2, 0xe0(%rsi)+ vmovdqa 0x100(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x100(%rdi)+ vmovdqa %ymm2, 0x100(%rsi)+ vmovdqa 0x120(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x120(%rdi)+ vmovdqa %ymm2, 0x120(%rsi)+ vmovdqa 0x140(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x140(%rdi)+ vmovdqa %ymm2, 0x140(%rsi)+ vmovdqa 0x160(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x160(%rdi)+ vmovdqa %ymm2, 0x160(%rsi)+ vmovdqa 0x180(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x180(%rdi)+ vmovdqa %ymm2, 0x180(%rsi)+ vmovdqa 0x1a0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x1a0(%rdi)+ vmovdqa %ymm2, 0x1a0(%rsi)+ vmovdqa 0x1c0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x1c0(%rdi)+ vmovdqa %ymm2, 0x1c0(%rsi)+ vmovdqa 0x1e0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x1e0(%rdi)+ vmovdqa %ymm2, 0x1e0(%rsi)+ vmovdqa 0x200(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x200(%rdi)+ vmovdqa %ymm2, 0x200(%rsi)+ vmovdqa 0x220(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x220(%rdi)+ vmovdqa %ymm2, 0x220(%rsi)+ vmovdqa 0x240(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x240(%rdi)+ vmovdqa %ymm2, 0x240(%rsi)+ vmovdqa 0x260(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x260(%rdi)+ vmovdqa %ymm2, 0x260(%rsi)+ vmovdqa 0x280(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x280(%rdi)+ vmovdqa %ymm2, 0x280(%rsi)+ vmovdqa 0x2a0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x2a0(%rdi)+ vmovdqa %ymm2, 0x2a0(%rsi)+ vmovdqa 0x2c0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x2c0(%rdi)+ vmovdqa %ymm2, 0x2c0(%rsi)+ vmovdqa 0x2e0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x2e0(%rdi)+ vmovdqa %ymm2, 0x2e0(%rsi)+ vmovdqa 0x300(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x300(%rdi)+ vmovdqa %ymm2, 0x300(%rsi)+ vmovdqa 0x320(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x320(%rdi)+ vmovdqa %ymm2, 0x320(%rsi)+ vmovdqa 0x340(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x340(%rdi)+ vmovdqa %ymm2, 0x340(%rsi)+ vmovdqa 0x360(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x360(%rdi)+ vmovdqa %ymm2, 0x360(%rsi)+ vmovdqa 0x380(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x380(%rdi)+ vmovdqa %ymm2, 0x380(%rsi)+ vmovdqa 0x3a0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x3a0(%rdi)+ vmovdqa %ymm2, 0x3a0(%rsi)+ vmovdqa 0x3c0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x3c0(%rdi)+ vmovdqa %ymm2, 0x3c0(%rsi)+ vmovdqa 0x3e0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x3e0(%rdi)+ vmovdqa %ymm2, 0x3e0(%rsi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_decompose_32_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_SIGN_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S view
@@ -0,0 +1,489 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ *+ * The algorithm for Decompose(r) (more specifically the handling for the+ * wrap-around cases) is modified. See the AVX2 intrinsics version+ * (poly_decompose_88_avx2.c, predecessor of this file) for a more detailed+ * comparison.+ */+++/*yaml+ Name: poly_decompose_88_avx2_asm+ Description: x86_64 AVX2 coefficient decomposition (alpha = 2*(Q-1)/88).+ For all coefficients c of the input polynomial, compute high and low bits+ c0, c1 such that c = c1*ALPHA + c0 with -ALPHA/2 < c0 <= ALPHA/2, except if+ c1 = (Q-1)/ALPHA where we set c1 = 0 and -ALPHA/2 <= c0 = c - Q < 0.+ Assumes coefficients to be standard (unsigned canonical) representatives,+ i.e. 0 <= c < Q. For ML-DSA-44 (gamma2 = (Q-1)/88, alpha = 2*gamma2).+ Signature: void mld_poly_decompose_88_avx2_asm(int32_t *a1, int32_t *a0)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *a1+ description: Output high-part polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *a0+ description: Input polynomial / output low-part (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_NO_SIGN_API) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ MLD_CONFIG_PARAMETER_SET == 44)+++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_decompose_88_avx2_asm)+MLD_ASM_FN_SYMBOL(poly_decompose_88_avx2_asm)++ .cfi_startproc+ movl $0x7f, %eax+ vmovd %eax, %xmm10+ vpbroadcastd %xmm10, %ymm10+ movl $0x2c0b, %eax # imm = 0x2C0B+ vmovd %eax, %xmm11+ vpbroadcastd %xmm11, %ymm11+ movl $0x80, %eax+ vmovd %eax, %xmm12+ vpbroadcastd %xmm12, %ymm12+ movl $0x7e6c00, %eax # imm = 0x7E6C00+ vmovd %eax, %xmm13+ vpbroadcastd %xmm13, %ymm13+ movl $0x2e800, %eax # imm = 0x2E800+ vmovd %eax, %xmm14+ vpbroadcastd %xmm14, %ymm14+ vmovdqa (%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, (%rdi)+ vmovdqa %ymm2, (%rsi)+ vmovdqa 0x20(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x20(%rdi)+ vmovdqa %ymm2, 0x20(%rsi)+ vmovdqa 0x40(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x40(%rdi)+ vmovdqa %ymm2, 0x40(%rsi)+ vmovdqa 0x60(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x60(%rdi)+ vmovdqa %ymm2, 0x60(%rsi)+ vmovdqa 0x80(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x80(%rdi)+ vmovdqa %ymm2, 0x80(%rsi)+ vmovdqa 0xa0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0xa0(%rdi)+ vmovdqa %ymm2, 0xa0(%rsi)+ vmovdqa 0xc0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0xc0(%rdi)+ vmovdqa %ymm2, 0xc0(%rsi)+ vmovdqa 0xe0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0xe0(%rdi)+ vmovdqa %ymm2, 0xe0(%rsi)+ vmovdqa 0x100(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x100(%rdi)+ vmovdqa %ymm2, 0x100(%rsi)+ vmovdqa 0x120(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x120(%rdi)+ vmovdqa %ymm2, 0x120(%rsi)+ vmovdqa 0x140(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x140(%rdi)+ vmovdqa %ymm2, 0x140(%rsi)+ vmovdqa 0x160(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x160(%rdi)+ vmovdqa %ymm2, 0x160(%rsi)+ vmovdqa 0x180(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x180(%rdi)+ vmovdqa %ymm2, 0x180(%rsi)+ vmovdqa 0x1a0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x1a0(%rdi)+ vmovdqa %ymm2, 0x1a0(%rsi)+ vmovdqa 0x1c0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x1c0(%rdi)+ vmovdqa %ymm2, 0x1c0(%rsi)+ vmovdqa 0x1e0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x1e0(%rdi)+ vmovdqa %ymm2, 0x1e0(%rsi)+ vmovdqa 0x200(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x200(%rdi)+ vmovdqa %ymm2, 0x200(%rsi)+ vmovdqa 0x220(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x220(%rdi)+ vmovdqa %ymm2, 0x220(%rsi)+ vmovdqa 0x240(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x240(%rdi)+ vmovdqa %ymm2, 0x240(%rsi)+ vmovdqa 0x260(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x260(%rdi)+ vmovdqa %ymm2, 0x260(%rsi)+ vmovdqa 0x280(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x280(%rdi)+ vmovdqa %ymm2, 0x280(%rsi)+ vmovdqa 0x2a0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x2a0(%rdi)+ vmovdqa %ymm2, 0x2a0(%rsi)+ vmovdqa 0x2c0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x2c0(%rdi)+ vmovdqa %ymm2, 0x2c0(%rsi)+ vmovdqa 0x2e0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x2e0(%rdi)+ vmovdqa %ymm2, 0x2e0(%rsi)+ vmovdqa 0x300(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x300(%rdi)+ vmovdqa %ymm2, 0x300(%rsi)+ vmovdqa 0x320(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x320(%rdi)+ vmovdqa %ymm2, 0x320(%rsi)+ vmovdqa 0x340(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x340(%rdi)+ vmovdqa %ymm2, 0x340(%rsi)+ vmovdqa 0x360(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x360(%rdi)+ vmovdqa %ymm2, 0x360(%rsi)+ vmovdqa 0x380(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x380(%rdi)+ vmovdqa %ymm2, 0x380(%rsi)+ vmovdqa 0x3a0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x3a0(%rdi)+ vmovdqa %ymm2, 0x3a0(%rsi)+ vmovdqa 0x3c0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x3c0(%rdi)+ vmovdqa %ymm2, 0x3c0(%rsi)+ vmovdqa 0x3e0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x3e0(%rdi)+ vmovdqa %ymm2, 0x3e0(%rsi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_decompose_88_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_SIGN_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S view
@@ -0,0 +1,123 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */+++/*yaml+ Name: poly_use_hint_32_avx2_asm+ Description: x86_64 AVX2 hint application (alpha = (Q-1)/32).+ Use the hint polynomial h to correct the high bits of the polynomial a,+ in place. Variant for parameter sets ML-DSA-65 and ML-DSA-87+ (GAMMA2 = (Q-1)/32).+ Signature: void mld_poly_use_hint_32_avx2_asm(int32_t *a, const int32_t *h)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *a+ description: Input/output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t *h+ description: Hint polynomial (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_NO_VERIFY_API) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))+++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_use_hint_32_avx2_asm)+MLD_ASM_FN_SYMBOL(poly_use_hint_32_avx2_asm)++ .cfi_startproc+ movl $0x7f, %ecx+ movl $0x401, %r8d # imm = 0x401+ vmovd %r8d, %xmm8+ vpbroadcastd %xmm8, %ymm8+ xorl %eax, %eax+ vpxor %xmm6, %xmm6, %xmm6+ vmovd %ecx, %xmm5+ movl $0x7be100, %ecx # imm = 0x7BE100+ movl $0x200, %r9d # imm = 0x200+ vmovd %r9d, %xmm7+ vpbroadcastd %xmm7, %ymm7+ vmovd %ecx, %xmm4+ movl $0xf, %ecx+ vpbroadcastd %xmm5, %ymm5+ vmovd %ecx, %xmm3+ vpbroadcastd %xmm4, %ymm4+ vpbroadcastd %xmm3, %ymm3++Lmld_poly_use_hint_32_avx2_asm_loop:+ vmovdqa (%rdi), %ymm0+ vmovdqa (%rsi), %ymm2+ vpaddd %ymm0, %ymm5, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm8, %ymm1, %ymm1+ vpmulhrsw %ymm7, %ymm1, %ymm1+ vpcmpgtd %ymm4, %ymm0, %ymm11+ vpandn %ymm1, %ymm11, %ymm9+ vpslld $0xa, %ymm1, %ymm10+ vpsubd %ymm1, %ymm10, %ymm1+ vpslld $0x9, %ymm1, %ymm1+ vpsubd %ymm1, %ymm0, %ymm0+ vpaddd %ymm11, %ymm0, %ymm0+ vpcmpgtd %ymm6, %ymm0, %ymm0+ vpandn %ymm2, %ymm0, %ymm0+ vpslld $0x1, %ymm0, %ymm0+ vpsubd %ymm0, %ymm2, %ymm2+ vpaddd %ymm9, %ymm2, %ymm2+ vpand %ymm3, %ymm2, %ymm2+ vmovdqa %ymm2, (%rdi)+ addq $0x20, %rdi+ addq $0x20, %rsi+ addq $0x20, %rax+ cmpq $0x400, %rax # imm = 0x400+ jne Lmld_poly_use_hint_32_avx2_asm_loop+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_use_hint_32_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_VERIFY_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S view
@@ -0,0 +1,125 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */+++/*yaml+ Name: poly_use_hint_88_avx2_asm+ Description: x86_64 AVX2 hint application (alpha = (Q-1)/88).+ Use the hint polynomial h to correct the high bits of the polynomial a,+ in place. Variant for parameter set ML-DSA-44 (GAMMA2 = (Q-1)/88).+ Signature: void mld_poly_use_hint_88_avx2_asm(int32_t *a, const int32_t *h)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *a+ description: Input/output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t *h+ description: Hint polynomial (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_NO_VERIFY_API) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ MLD_CONFIG_PARAMETER_SET == 44)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_use_hint_88_avx2_asm)+MLD_ASM_FN_SYMBOL(poly_use_hint_88_avx2_asm)++ .cfi_startproc+ movl $0x7f, %ecx+ xorl %eax, %eax+ vpxor %xmm5, %xmm5, %xmm5+ movl $0x2c0b, %r8d # imm = 0x2C0B+ vmovd %r8d, %xmm8+ vpbroadcastd %xmm8, %ymm8+ vmovd %ecx, %xmm4+ movl $0x7e6c00, %ecx # imm = 0x7E6C00+ movl $0x80, %r9d+ vmovd %r9d, %xmm7+ vpbroadcastd %xmm7, %ymm7+ movl $0x2b, %r10d+ vmovd %r10d, %xmm6+ vpbroadcastd %xmm6, %ymm6+ vmovd %ecx, %xmm3+ vpbroadcastd %xmm4, %ymm4+ vpbroadcastd %xmm3, %ymm3++Lmld_poly_use_hint_88_avx2_asm_loop:+ vmovdqa (%rdi), %ymm0+ vmovdqa (%rsi), %ymm1+ vpaddd %ymm0, %ymm4, %ymm10+ vpsrld $0x7, %ymm10, %ymm10+ vpmulhuw %ymm8, %ymm10, %ymm10+ vpmulhrsw %ymm7, %ymm10, %ymm10+ vpcmpgtd %ymm3, %ymm0, %ymm11+ vpandn %ymm10, %ymm11, %ymm9+ vpslld $0x1, %ymm10, %ymm12+ vpaddd %ymm10, %ymm12, %ymm12+ vpslld $0x5, %ymm12, %ymm10+ vpsubd %ymm12, %ymm10, %ymm10+ vpslld $0xb, %ymm10, %ymm10+ vpsubd %ymm10, %ymm0, %ymm0+ vpaddd %ymm11, %ymm0, %ymm0+ vpcmpgtd %ymm5, %ymm0, %ymm0+ vpandn %ymm1, %ymm0, %ymm0+ vpslld $0x1, %ymm0, %ymm0+ vpsubd %ymm0, %ymm1, %ymm0+ vpaddd %ymm9, %ymm0, %ymm0+ vpblendvb %ymm0, %ymm6, %ymm0, %ymm0+ vpcmpgtd %ymm6, %ymm0, %ymm1+ vpandn %ymm0, %ymm1, %ymm0+ vmovdqa %ymm0, (%rdi)+ addq $0x20, %rdi+ addq $0x20, %rsi+ addq $0x20, %rax+ cmpq $0x400, %rax # imm = 0x400+ jne Lmld_poly_use_hint_88_avx2_asm_loop+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_use_hint_88_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_VERIFY_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S view
@@ -0,0 +1,355 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */+++/*yaml+ Name: polyz_unpack_17_avx2_asm+ Description: x86_64 AVX2 unpacking of 17-bit packed coefficients.+ Unpack polynomial z with 18-bit packed coefficients (GAMMA1 = 2^17),+ mapping packed [0, 2^18-1] to signed [-(2^17-1), 2^17] via GAMMA1 - x.+ Signature: void mld_polyz_unpack_17_avx2_asm(int32_t *r, const uint8_t *a)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *r+ description: Output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 576+ permissions: read-only+ c_parameter: const uint8_t *a+ description: Packed input bytes+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ (!defined(MLD_CONFIG_NO_SIGN_API) || \+ !defined(MLD_CONFIG_NO_VERIFY_API)) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ MLD_CONFIG_PARAMETER_SET == 44)+++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(polyz_unpack_17_avx2_asm)+MLD_ASM_FN_SYMBOL(polyz_unpack_17_avx2_asm)++ .cfi_startproc+ movabsq $-0xfbfcfd00fdff00, %rax # imm = 0xFF040302FF020100+ vmovq %rax, %xmm1+ movabsq $-0xf7f8f900f9fafc, %rax # imm = 0xFF080706FF060504+ vpinsrq $0x1, %rax, %xmm1, %xmm1+ movabsq $-0xe4e5e600e6e7e9, %rax # imm = 0xFF1B1A19FF191817+ vmovq %rax, %xmm5+ movabsq $-0xe0e1e200e2e3e5, %rax # imm = 0xFF1F1E1DFF1D1C1B+ vpinsrq $0x1, %rax, %xmm5, %xmm5+ vinserti128 $0x1, %xmm5, %ymm1, %ymm1+ movabsq $0x200000000, %rax # imm = 0x200000000+ vmovq %rax, %xmm2+ movabsq $0x600000004, %rax # imm = 0x600000004+ vpinsrq $0x1, %rax, %xmm2, %xmm2+ vinserti128 $0x1, %xmm2, %ymm2, %ymm2+ movl $0x3ffff, %eax # imm = 0x3FFFF+ vmovd %eax, %xmm3+ vpbroadcastd %xmm3, %ymm3+ movl $0x20000, %eax # imm = 0x20000+ vmovd %eax, %xmm4+ vpbroadcastd %xmm4, %ymm4+ vmovdqu (%rsi), %xmm0+ vmovdqu 0x2(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, (%rdi)+ vmovdqu 0x12(%rsi), %xmm0+ vmovdqu 0x14(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x20(%rdi)+ vmovdqu 0x24(%rsi), %xmm0+ vmovdqu 0x26(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x40(%rdi)+ vmovdqu 0x36(%rsi), %xmm0+ vmovdqu 0x38(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x60(%rdi)+ vmovdqu 0x48(%rsi), %xmm0+ vmovdqu 0x4a(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x80(%rdi)+ vmovdqu 0x5a(%rsi), %xmm0+ vmovdqu 0x5c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0xa0(%rdi)+ vmovdqu 0x6c(%rsi), %xmm0+ vmovdqu 0x6e(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0xc0(%rdi)+ vmovdqu 0x7e(%rsi), %xmm0+ vmovdqu 0x80(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0xe0(%rdi)+ vmovdqu 0x90(%rsi), %xmm0+ vmovdqu 0x92(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x100(%rdi)+ vmovdqu 0xa2(%rsi), %xmm0+ vmovdqu 0xa4(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x120(%rdi)+ vmovdqu 0xb4(%rsi), %xmm0+ vmovdqu 0xb6(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x140(%rdi)+ vmovdqu 0xc6(%rsi), %xmm0+ vmovdqu 0xc8(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x160(%rdi)+ vmovdqu 0xd8(%rsi), %xmm0+ vmovdqu 0xda(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x180(%rdi)+ vmovdqu 0xea(%rsi), %xmm0+ vmovdqu 0xec(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x1a0(%rdi)+ vmovdqu 0xfc(%rsi), %xmm0+ vmovdqu 0xfe(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x1c0(%rdi)+ vmovdqu 0x10e(%rsi), %xmm0+ vmovdqu 0x110(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x1e0(%rdi)+ vmovdqu 0x120(%rsi), %xmm0+ vmovdqu 0x122(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x200(%rdi)+ vmovdqu 0x132(%rsi), %xmm0+ vmovdqu 0x134(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x220(%rdi)+ vmovdqu 0x144(%rsi), %xmm0+ vmovdqu 0x146(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x240(%rdi)+ vmovdqu 0x156(%rsi), %xmm0+ vmovdqu 0x158(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x260(%rdi)+ vmovdqu 0x168(%rsi), %xmm0+ vmovdqu 0x16a(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x280(%rdi)+ vmovdqu 0x17a(%rsi), %xmm0+ vmovdqu 0x17c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x2a0(%rdi)+ vmovdqu 0x18c(%rsi), %xmm0+ vmovdqu 0x18e(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x2c0(%rdi)+ vmovdqu 0x19e(%rsi), %xmm0+ vmovdqu 0x1a0(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x2e0(%rdi)+ vmovdqu 0x1b0(%rsi), %xmm0+ vmovdqu 0x1b2(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x300(%rdi)+ vmovdqu 0x1c2(%rsi), %xmm0+ vmovdqu 0x1c4(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x320(%rdi)+ vmovdqu 0x1d4(%rsi), %xmm0+ vmovdqu 0x1d6(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x340(%rdi)+ vmovdqu 0x1e6(%rsi), %xmm0+ vmovdqu 0x1e8(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x360(%rdi)+ vmovdqu 0x1f8(%rsi), %xmm0+ vmovdqu 0x1fa(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x380(%rdi)+ vmovdqu 0x20a(%rsi), %xmm0+ vmovdqu 0x20c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x3a0(%rdi)+ vmovdqu 0x21c(%rsi), %xmm0+ vmovdqu 0x21e(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x3c0(%rdi)+ vmovdqu 0x22e(%rsi), %xmm0+ vmovdqu 0x230(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x3e0(%rdi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(polyz_unpack_17_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S view
@@ -0,0 +1,355 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */+++/*yaml+ Name: polyz_unpack_19_avx2_asm+ Description: x86_64 AVX2 unpacking of 19-bit packed coefficients.+ Unpack polynomial z with 20-bit packed coefficients (GAMMA1 = 2^19),+ mapping packed [0, 2^20-1] to signed [-(2^19-1), 2^19] via GAMMA1 - x.+ Signature: void mld_polyz_unpack_19_avx2_asm(int32_t *r, const uint8_t *a)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *r+ description: Output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 640+ permissions: read-only+ c_parameter: const uint8_t *a+ description: Packed input bytes+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ (!defined(MLD_CONFIG_NO_SIGN_API) || \+ !defined(MLD_CONFIG_NO_VERIFY_API)) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))+++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(polyz_unpack_19_avx2_asm)+MLD_ASM_FN_SYMBOL(polyz_unpack_19_avx2_asm)++ .cfi_startproc+ movabsq $-0xfbfcfd00fdff00, %rax # imm = 0xFF040302FF020100+ vmovq %rax, %xmm1+ movabsq $-0xf6f7f800f8f9fb, %rax # imm = 0xFF090807FF070605+ vpinsrq $0x1, %rax, %xmm1, %xmm1+ movabsq $-0xe5e6e700e7e8ea, %rax # imm = 0xFF1A1918FF181716+ vmovq %rax, %xmm5+ movabsq $-0xe0e1e200e2e3e5, %rax # imm = 0xFF1F1E1DFF1D1C1B+ vpinsrq $0x1, %rax, %xmm5, %xmm5+ vinserti128 $0x1, %xmm5, %ymm1, %ymm1+ movabsq $0x400000000, %rax # imm = 0x400000000+ vmovq %rax, %xmm2+ movabsq $0x400000000, %rax # imm = 0x400000000+ vpinsrq $0x1, %rax, %xmm2, %xmm2+ vinserti128 $0x1, %xmm2, %ymm2, %ymm2+ movl $0xfffff, %eax # imm = 0xFFFFF+ vmovd %eax, %xmm3+ vpbroadcastd %xmm3, %ymm3+ movl $0x80000, %eax # imm = 0x80000+ vmovd %eax, %xmm4+ vpbroadcastd %xmm4, %ymm4+ vmovdqu (%rsi), %xmm0+ vmovdqu 0x4(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, (%rdi)+ vmovdqu 0x14(%rsi), %xmm0+ vmovdqu 0x18(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x20(%rdi)+ vmovdqu 0x28(%rsi), %xmm0+ vmovdqu 0x2c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x40(%rdi)+ vmovdqu 0x3c(%rsi), %xmm0+ vmovdqu 0x40(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x60(%rdi)+ vmovdqu 0x50(%rsi), %xmm0+ vmovdqu 0x54(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x80(%rdi)+ vmovdqu 0x64(%rsi), %xmm0+ vmovdqu 0x68(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0xa0(%rdi)+ vmovdqu 0x78(%rsi), %xmm0+ vmovdqu 0x7c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0xc0(%rdi)+ vmovdqu 0x8c(%rsi), %xmm0+ vmovdqu 0x90(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0xe0(%rdi)+ vmovdqu 0xa0(%rsi), %xmm0+ vmovdqu 0xa4(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x100(%rdi)+ vmovdqu 0xb4(%rsi), %xmm0+ vmovdqu 0xb8(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x120(%rdi)+ vmovdqu 0xc8(%rsi), %xmm0+ vmovdqu 0xcc(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x140(%rdi)+ vmovdqu 0xdc(%rsi), %xmm0+ vmovdqu 0xe0(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x160(%rdi)+ vmovdqu 0xf0(%rsi), %xmm0+ vmovdqu 0xf4(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x180(%rdi)+ vmovdqu 0x104(%rsi), %xmm0+ vmovdqu 0x108(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x1a0(%rdi)+ vmovdqu 0x118(%rsi), %xmm0+ vmovdqu 0x11c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x1c0(%rdi)+ vmovdqu 0x12c(%rsi), %xmm0+ vmovdqu 0x130(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x1e0(%rdi)+ vmovdqu 0x140(%rsi), %xmm0+ vmovdqu 0x144(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x200(%rdi)+ vmovdqu 0x154(%rsi), %xmm0+ vmovdqu 0x158(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x220(%rdi)+ vmovdqu 0x168(%rsi), %xmm0+ vmovdqu 0x16c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x240(%rdi)+ vmovdqu 0x17c(%rsi), %xmm0+ vmovdqu 0x180(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x260(%rdi)+ vmovdqu 0x190(%rsi), %xmm0+ vmovdqu 0x194(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x280(%rdi)+ vmovdqu 0x1a4(%rsi), %xmm0+ vmovdqu 0x1a8(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x2a0(%rdi)+ vmovdqu 0x1b8(%rsi), %xmm0+ vmovdqu 0x1bc(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x2c0(%rdi)+ vmovdqu 0x1cc(%rsi), %xmm0+ vmovdqu 0x1d0(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x2e0(%rdi)+ vmovdqu 0x1e0(%rsi), %xmm0+ vmovdqu 0x1e4(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x300(%rdi)+ vmovdqu 0x1f4(%rsi), %xmm0+ vmovdqu 0x1f8(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x320(%rdi)+ vmovdqu 0x208(%rsi), %xmm0+ vmovdqu 0x20c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x340(%rdi)+ vmovdqu 0x21c(%rsi), %xmm0+ vmovdqu 0x220(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x360(%rdi)+ vmovdqu 0x230(%rsi), %xmm0+ vmovdqu 0x234(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x380(%rdi)+ vmovdqu 0x244(%rsi), %xmm0+ vmovdqu 0x248(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x3a0(%rdi)+ vmovdqu 0x258(%rsi), %xmm0+ vmovdqu 0x25c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x3c0(%rdi)+ vmovdqu 0x26c(%rsi), %xmm0+ vmovdqu 0x270(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x3e0(%rdi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(polyz_unpack_19_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_rej_uniform_avx2_asm.S view
@@ -0,0 +1,132 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: rej_uniform_avx2_asm+ Description: x86_64 AVX2 rejection sampling of uniform coefficients mod q.+ Extract 23-bit values from the input byte buffer and accept those that are+ < MLDSA_Q, writing accepted coefficients to the output buffer. Returns the+ number of valid coefficients written (at most 256).+ Signature: unsigned mld_rej_uniform_avx2_asm(int32_t *r, const uint8_t buf[840], const uint8_t table[256][8])+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *r+ description: Output buffer (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 840+ permissions: read-only+ c_parameter: const uint8_t buf[840]+ description: Input buffer (MLD_POLY_UNIFORM_NBLOCKS * SHAKE128_RATE = 5 * 168)+ rdx:+ type: buffer+ size_bytes: 2048+ permissions: read-only+ c_parameter: const uint8_t table[256][8]+ description: Lookup table (256 x 8 x uint8_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_rej_uniform_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(rej_uniform_avx2_asm)+MLD_ASM_FN_SYMBOL(rej_uniform_avx2_asm)++ .cfi_startproc+ movabsq $-0xfafbfc00fdff00, %r10 # imm = 0xFF050403FF020100+ vmovq %r10, %xmm0+ movabsq $-0xf4f5f600f7f8fa, %r10 # imm = 0xFF0B0A09FF080706+ vpinsrq $0x1, %r10, %xmm0, %xmm0+ movabsq $-0xf6f7f800f9fafc, %r10 # imm = 0xFF090807FF060504+ vmovq %r10, %xmm3+ movabsq $-0xf0f1f200f3f4f6, %r10 # imm = 0xFF0F0E0DFF0C0B0A+ vpinsrq $0x1, %r10, %xmm3, %xmm3+ vinserti128 $0x1, %xmm3, %ymm0, %ymm0+ movl $0x7fffff, %r8d # imm = 0x7FFFFF+ vmovd %r8d, %xmm1+ vpbroadcastd %xmm1, %ymm1+ movl $0x7fe001, %r8d # imm = 0x7FE001+ vmovd %r8d, %xmm2+ vpbroadcastd %xmm2, %ymm2+ xorl %eax, %eax+ xorl %ecx, %ecx++Lmld_rej_uniform_avx2_asm_loop:+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_avx2_asm_scalar+ cmpl $0x328, %ecx # imm = 0x328+ ja Lmld_rej_uniform_avx2_asm_scalar+ vmovdqu (%rsi,%rcx), %ymm3+ addl $0x18, %ecx+ vpermq $0x94, %ymm3, %ymm3 # ymm3 = ymm3[0,1,1,2]+ vpshufb %ymm0, %ymm3, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpsubd %ymm2, %ymm3, %ymm4+ vmovmskps %ymm4, %r8d+ popcntl %r8d, %r9d+ vmovq (%rdx,%r8,8), %xmm4+ vpmovzxbd %xmm4, %ymm4 # ymm4 = xmm4[0],zero,zero,zero,xmm4[1],zero,zero,zero,xmm4[2],zero,zero,zero,xmm4[3],zero,zero,zero,xmm4[4],zero,zero,zero,xmm4[5],zero,zero,zero,xmm4[6],zero,zero,zero,xmm4[7],zero,zero,zero+ vpermd %ymm3, %ymm4, %ymm3+ vmovdqu %ymm3, (%rdi,%rax,4)+ addl %r9d, %eax+ jmp Lmld_rej_uniform_avx2_asm_loop++Lmld_rej_uniform_avx2_asm_scalar:+ cmpl $0x100, %eax # imm = 0x100+ jae Lmld_rej_uniform_avx2_asm_done+ cmpl $0x345, %ecx # imm = 0x345+ ja Lmld_rej_uniform_avx2_asm_done+ movzwl (%rsi,%rcx), %r8d+ movzbl 0x2(%rsi,%rcx), %r9d+ shll $0x10, %r9d+ orl %r9d, %r8d+ andl $0x7fffff, %r8d # imm = 0x7FFFFF+ addl $0x3, %ecx+ cmpl $0x7fe001, %r8d # imm = 0x7FE001+ jae Lmld_rej_uniform_avx2_asm_scalar+ movl %r8d, (%rdi,%rax,4)+ addl $0x1, %eax+ jmp Lmld_rej_uniform_avx2_asm_scalar++Lmld_rej_uniform_avx2_asm_done:+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(rej_uniform_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S view
@@ -0,0 +1,205 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: rej_uniform_eta2_avx2_asm+ Description: x86_64 AVX2 rejection sampling of eta=2 secret coefficients.+ Extracts 4-bit nibbles from the input byte buffer, accepts those < 15,+ and maps an accepted nibble n to the coefficient 2 - (n mod 5) via a+ centered modulo-5 reduction, producing values in [-2, 2].+ Signature: unsigned mld_rej_uniform_eta2_avx2_asm(int32_t *r, const uint8_t *buf, const uint8_t table[256][8])+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *r+ description: Output coefficient buffer (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 136+ permissions: read-only+ c_parameter: const uint8_t *buf+ description: Input byte buffer (MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN = 136)+ rdx:+ type: buffer+ size_bytes: 2048+ permissions: read-only+ c_parameter: const uint8_t *table+ description: Lookup table (256 x 8 uint8_t = mld_rej_uniform_table)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(rej_uniform_eta2_avx2_asm)+MLD_ASM_FN_SYMBOL(rej_uniform_eta2_avx2_asm)++ .cfi_startproc+ movl $0xf0f0f0f, %r8d # imm = 0xF0F0F0F+ vmovd %r8d, %xmm3+ vpbroadcastd %xmm3, %ymm3+ movl $0x2020202, %r8d # imm = 0x2020202+ vmovd %r8d, %xmm4+ vpbroadcastd %xmm4, %ymm4+ movl $0xf0f0f0f, %r8d # imm = 0xF0F0F0F+ vmovd %r8d, %xmm5+ vpbroadcastd %xmm5, %ymm5+ movl $0xffffe660, %r8d # imm = 0xFFFFE660+ vpinsrw $0x0, %r8d, %xmm6, %xmm6+ vpbroadcastw %xmm6, %ymm6+ movl $0x5, %r8d+ vpinsrw $0x0, %r8d, %xmm7, %xmm7+ vpbroadcastw %xmm7, %ymm7+ xorl %eax, %eax+ xorl %ecx, %ecx++Lmld_rej_uniform_eta2_avx2_asm_loop:+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta2_avx2_asm_scalar+ cmpl $0x78, %ecx+ ja Lmld_rej_uniform_eta2_avx2_asm_scalar+ vpmovzxbw (%rsi,%rcx), %ymm0+ vpsllw $0x4, %ymm0, %ymm1+ vpor %ymm1, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubb %ymm5, %ymm0, %ymm1+ vpsubb %ymm0, %ymm4, %ymm0+ vpmovmskb %ymm1, %r8d+ vextracti128 $0x0, %ymm0, %xmm8+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm9+ vpshufb %xmm9, %xmm8, %xmm9+ vpmovsxbd %xmm9, %ymm1+ vpmulhrsw %ymm6, %ymm1, %ymm2+ vpmullw %ymm7, %ymm2, %ymm2+ vpaddd %ymm2, %ymm1, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ shrl $0x8, %r8d+ addl $0x4, %ecx+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta2_avx2_asm_scalar+ vpsrldq $0x8, %xmm8, %xmm8 # xmm8 = xmm8[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm9+ vpshufb %xmm9, %xmm8, %xmm9+ vpmovsxbd %xmm9, %ymm1+ vpmulhrsw %ymm6, %ymm1, %ymm2+ vpmullw %ymm7, %ymm2, %ymm2+ vpaddd %ymm2, %ymm1, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ shrl $0x8, %r8d+ addl $0x4, %ecx+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta2_avx2_asm_scalar+ vextracti128 $0x1, %ymm0, %xmm8+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm9+ vpshufb %xmm9, %xmm8, %xmm9+ vpmovsxbd %xmm9, %ymm1+ vpmulhrsw %ymm6, %ymm1, %ymm2+ vpmullw %ymm7, %ymm2, %ymm2+ vpaddd %ymm2, %ymm1, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ shrl $0x8, %r8d+ addl $0x4, %ecx+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta2_avx2_asm_scalar+ vpsrldq $0x8, %xmm8, %xmm8 # xmm8 = xmm8[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm9+ vpshufb %xmm9, %xmm8, %xmm9+ vpmovsxbd %xmm9, %ymm1+ vpmulhrsw %ymm6, %ymm1, %ymm2+ vpmullw %ymm7, %ymm2, %ymm2+ vpaddd %ymm2, %ymm1, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ addl $0x4, %ecx+ jmp Lmld_rej_uniform_eta2_avx2_asm_loop++Lmld_rej_uniform_eta2_avx2_asm_scalar:+ cmpl $0x100, %eax # imm = 0x100+ jae Lmld_rej_uniform_eta2_avx2_asm_done+ cmpl $0x88, %ecx+ jae Lmld_rej_uniform_eta2_avx2_asm_done+ movzbl (%rsi,%rcx), %r11d+ incl %ecx+ movl %r11d, %r10d+ andl $0xf, %r10d+ cmpl $0xf, %r10d+ jae Lmld_rej_uniform_eta2_avx2_asm_high_nibble+ movl %r10d, %r11d+ imull $0xcd, %r11d, %r11d+ shrl $0xa, %r11d+ imull $0x5, %r11d, %r11d+ subl %r11d, %r10d+ movl $0x2, %r11d+ subl %r10d, %r11d+ movl %r11d, (%rdi,%rax,4)+ incl %eax+ cmpl $0x100, %eax # imm = 0x100+ jae Lmld_rej_uniform_eta2_avx2_asm_done++Lmld_rej_uniform_eta2_avx2_asm_high_nibble:+ movzbl -0x1(%rsi,%rcx), %r11d+ shrl $0x4, %r11d+ andl $0xf, %r11d+ cmpl $0xf, %r11d+ jae Lmld_rej_uniform_eta2_avx2_asm_scalar+ movl %r11d, %r10d+ imull $0xcd, %r10d, %r10d+ shrl $0xa, %r10d+ imull $0x5, %r10d, %r10d+ subl %r10d, %r11d+ movl $0x2, %r10d+ subl %r11d, %r10d+ movl %r10d, (%rdi,%rax,4)+ incl %eax+ jmp Lmld_rej_uniform_eta2_avx2_asm_scalar++Lmld_rej_uniform_eta2_avx2_asm_done:+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(rej_uniform_eta2_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S view
@@ -0,0 +1,176 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: rej_uniform_eta4_avx2_asm+ Description: x86_64 AVX2 rejection sampling of eta=4 secret coefficients.+ Extracts 4-bit nibbles from the input byte buffer, accepts those < 9,+ and maps an accepted nibble n to the coefficient 4 - n, producing+ values in [-4, 4].+ Signature: unsigned mld_rej_uniform_eta4_avx2_asm(int32_t *r, const uint8_t *buf, const uint8_t table[256][8])+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *r+ description: Output coefficient buffer (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 272+ permissions: read-only+ c_parameter: const uint8_t *buf+ description: Input byte buffer (MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN = 272)+ rdx:+ type: buffer+ size_bytes: 2048+ permissions: read-only+ c_parameter: const uint8_t *table+ description: Lookup table (256 x 8 uint8_t = mld_rej_uniform_table)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(rej_uniform_eta4_avx2_asm)+MLD_ASM_FN_SYMBOL(rej_uniform_eta4_avx2_asm)++ .cfi_startproc+ movl $0xf0f0f0f, %r8d # imm = 0xF0F0F0F+ vmovd %r8d, %xmm2+ vpbroadcastd %xmm2, %ymm2+ movl $0x4040404, %r8d # imm = 0x4040404+ vmovd %r8d, %xmm3+ vpbroadcastd %xmm3, %ymm3+ movl $0x9090909, %r8d # imm = 0x9090909+ vmovd %r8d, %xmm4+ vpbroadcastd %xmm4, %ymm4+ xorl %eax, %eax+ xorl %ecx, %ecx++Lmld_rej_uniform_eta4_avx2_asm_loop:+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta4_avx2_asm_scalar+ cmpl $0x100, %ecx # imm = 0x100+ ja Lmld_rej_uniform_eta4_avx2_asm_scalar+ vpmovzxbw (%rsi,%rcx), %ymm0+ vpsllw $0x4, %ymm0, %ymm1+ vpor %ymm1, %ymm0, %ymm0+ vpand %ymm2, %ymm0, %ymm0+ vpsubb %ymm4, %ymm0, %ymm1+ vpsubb %ymm0, %ymm3, %ymm0+ vpmovmskb %ymm1, %r8d+ vextracti128 $0x0, %ymm0, %xmm5+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm6+ vpshufb %xmm6, %xmm5, %xmm6+ vpmovsxbd %xmm6, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ shrl $0x8, %r8d+ addl $0x4, %ecx+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta4_avx2_asm_scalar+ vpsrldq $0x8, %xmm5, %xmm5 # xmm5 = xmm5[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm6+ vpshufb %xmm6, %xmm5, %xmm6+ vpmovsxbd %xmm6, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ shrl $0x8, %r8d+ addl $0x4, %ecx+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta4_avx2_asm_scalar+ vextracti128 $0x1, %ymm0, %xmm5+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm6+ vpshufb %xmm6, %xmm5, %xmm6+ vpmovsxbd %xmm6, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ shrl $0x8, %r8d+ addl $0x4, %ecx+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta4_avx2_asm_scalar+ vpsrldq $0x8, %xmm5, %xmm5 # xmm5 = xmm5[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm6+ vpshufb %xmm6, %xmm5, %xmm6+ vpmovsxbd %xmm6, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ addl $0x4, %ecx+ jmp Lmld_rej_uniform_eta4_avx2_asm_loop++Lmld_rej_uniform_eta4_avx2_asm_scalar:+ cmpl $0x100, %eax # imm = 0x100+ jae Lmld_rej_uniform_eta4_avx2_asm_done+ cmpl $0x110, %ecx # imm = 0x110+ jae Lmld_rej_uniform_eta4_avx2_asm_done+ movzbl (%rsi,%rcx), %r11d+ incl %ecx+ movl %r11d, %r10d+ andl $0xf, %r10d+ cmpl $0x9, %r10d+ jae Lmld_rej_uniform_eta4_avx2_asm_high_nibble+ movl $0x4, %r9d+ subl %r10d, %r9d+ movl %r9d, (%rdi,%rax,4)+ incl %eax+ cmpl $0x100, %eax # imm = 0x100+ jae Lmld_rej_uniform_eta4_avx2_asm_done++Lmld_rej_uniform_eta4_avx2_asm_high_nibble:+ shrl $0x4, %r11d+ andl $0xf, %r11d+ cmpl $0x9, %r11d+ jae Lmld_rej_uniform_eta4_avx2_asm_scalar+ movl $0x4, %r10d+ subl %r11d, %r10d+ movl %r10d, (%rdi,%rax,4)+ incl %eax+ jmp Lmld_rej_uniform_eta4_avx2_asm_scalar++Lmld_rej_uniform_eta4_avx2_asm_done:+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(rej_uniform_eta4_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mldsa/src/native/x86_64/src/rej_uniform_table.c view
@@ -0,0 +1,161 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include "arith_native_x86_64.h"++/*+ * Lookup table used by rejection sampling.+ * See autogen for details.+ */+MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint8_t+ mld_rej_uniform_table[256][8] = {+ {0, 0, 0, 0, 0, 0, 0, 0}, {0, 0, 0, 0, 0, 0, 0, 0},+ {1, 0, 0, 0, 0, 0, 0, 0}, {0, 1, 0, 0, 0, 0, 0, 0},+ {2, 0, 0, 0, 0, 0, 0, 0}, {0, 2, 0, 0, 0, 0, 0, 0},+ {1, 2, 0, 0, 0, 0, 0, 0}, {0, 1, 2, 0, 0, 0, 0, 0},+ {3, 0, 0, 0, 0, 0, 0, 0}, {0, 3, 0, 0, 0, 0, 0, 0},+ {1, 3, 0, 0, 0, 0, 0, 0}, {0, 1, 3, 0, 0, 0, 0, 0},+ {2, 3, 0, 0, 0, 0, 0, 0}, {0, 2, 3, 0, 0, 0, 0, 0},+ {1, 2, 3, 0, 0, 0, 0, 0}, {0, 1, 2, 3, 0, 0, 0, 0},+ {4, 0, 0, 0, 0, 0, 0, 0}, {0, 4, 0, 0, 0, 0, 0, 0},+ {1, 4, 0, 0, 0, 0, 0, 0}, {0, 1, 4, 0, 0, 0, 0, 0},+ {2, 4, 0, 0, 0, 0, 0, 0}, {0, 2, 4, 0, 0, 0, 0, 0},+ {1, 2, 4, 0, 0, 0, 0, 0}, {0, 1, 2, 4, 0, 0, 0, 0},+ {3, 4, 0, 0, 0, 0, 0, 0}, {0, 3, 4, 0, 0, 0, 0, 0},+ {1, 3, 4, 0, 0, 0, 0, 0}, {0, 1, 3, 4, 0, 0, 0, 0},+ {2, 3, 4, 0, 0, 0, 0, 0}, {0, 2, 3, 4, 0, 0, 0, 0},+ {1, 2, 3, 4, 0, 0, 0, 0}, {0, 1, 2, 3, 4, 0, 0, 0},+ {5, 0, 0, 0, 0, 0, 0, 0}, {0, 5, 0, 0, 0, 0, 0, 0},+ {1, 5, 0, 0, 0, 0, 0, 0}, {0, 1, 5, 0, 0, 0, 0, 0},+ {2, 5, 0, 0, 0, 0, 0, 0}, {0, 2, 5, 0, 0, 0, 0, 0},+ {1, 2, 5, 0, 0, 0, 0, 0}, {0, 1, 2, 5, 0, 0, 0, 0},+ {3, 5, 0, 0, 0, 0, 0, 0}, {0, 3, 5, 0, 0, 0, 0, 0},+ {1, 3, 5, 0, 0, 0, 0, 0}, {0, 1, 3, 5, 0, 0, 0, 0},+ {2, 3, 5, 0, 0, 0, 0, 0}, {0, 2, 3, 5, 0, 0, 0, 0},+ {1, 2, 3, 5, 0, 0, 0, 0}, {0, 1, 2, 3, 5, 0, 0, 0},+ {4, 5, 0, 0, 0, 0, 0, 0}, {0, 4, 5, 0, 0, 0, 0, 0},+ {1, 4, 5, 0, 0, 0, 0, 0}, {0, 1, 4, 5, 0, 0, 0, 0},+ {2, 4, 5, 0, 0, 0, 0, 0}, {0, 2, 4, 5, 0, 0, 0, 0},+ {1, 2, 4, 5, 0, 0, 0, 0}, {0, 1, 2, 4, 5, 0, 0, 0},+ {3, 4, 5, 0, 0, 0, 0, 0}, {0, 3, 4, 5, 0, 0, 0, 0},+ {1, 3, 4, 5, 0, 0, 0, 0}, {0, 1, 3, 4, 5, 0, 0, 0},+ {2, 3, 4, 5, 0, 0, 0, 0}, {0, 2, 3, 4, 5, 0, 0, 0},+ {1, 2, 3, 4, 5, 0, 0, 0}, {0, 1, 2, 3, 4, 5, 0, 0},+ {6, 0, 0, 0, 0, 0, 0, 0}, {0, 6, 0, 0, 0, 0, 0, 0},+ {1, 6, 0, 0, 0, 0, 0, 0}, {0, 1, 6, 0, 0, 0, 0, 0},+ {2, 6, 0, 0, 0, 0, 0, 0}, {0, 2, 6, 0, 0, 0, 0, 0},+ {1, 2, 6, 0, 0, 0, 0, 0}, {0, 1, 2, 6, 0, 0, 0, 0},+ {3, 6, 0, 0, 0, 0, 0, 0}, {0, 3, 6, 0, 0, 0, 0, 0},+ {1, 3, 6, 0, 0, 0, 0, 0}, {0, 1, 3, 6, 0, 0, 0, 0},+ {2, 3, 6, 0, 0, 0, 0, 0}, {0, 2, 3, 6, 0, 0, 0, 0},+ {1, 2, 3, 6, 0, 0, 0, 0}, {0, 1, 2, 3, 6, 0, 0, 0},+ {4, 6, 0, 0, 0, 0, 0, 0}, {0, 4, 6, 0, 0, 0, 0, 0},+ {1, 4, 6, 0, 0, 0, 0, 0}, {0, 1, 4, 6, 0, 0, 0, 0},+ {2, 4, 6, 0, 0, 0, 0, 0}, {0, 2, 4, 6, 0, 0, 0, 0},+ {1, 2, 4, 6, 0, 0, 0, 0}, {0, 1, 2, 4, 6, 0, 0, 0},+ {3, 4, 6, 0, 0, 0, 0, 0}, {0, 3, 4, 6, 0, 0, 0, 0},+ {1, 3, 4, 6, 0, 0, 0, 0}, {0, 1, 3, 4, 6, 0, 0, 0},+ {2, 3, 4, 6, 0, 0, 0, 0}, {0, 2, 3, 4, 6, 0, 0, 0},+ {1, 2, 3, 4, 6, 0, 0, 0}, {0, 1, 2, 3, 4, 6, 0, 0},+ {5, 6, 0, 0, 0, 0, 0, 0}, {0, 5, 6, 0, 0, 0, 0, 0},+ {1, 5, 6, 0, 0, 0, 0, 0}, {0, 1, 5, 6, 0, 0, 0, 0},+ {2, 5, 6, 0, 0, 0, 0, 0}, {0, 2, 5, 6, 0, 0, 0, 0},+ {1, 2, 5, 6, 0, 0, 0, 0}, {0, 1, 2, 5, 6, 0, 0, 0},+ {3, 5, 6, 0, 0, 0, 0, 0}, {0, 3, 5, 6, 0, 0, 0, 0},+ {1, 3, 5, 6, 0, 0, 0, 0}, {0, 1, 3, 5, 6, 0, 0, 0},+ {2, 3, 5, 6, 0, 0, 0, 0}, {0, 2, 3, 5, 6, 0, 0, 0},+ {1, 2, 3, 5, 6, 0, 0, 0}, {0, 1, 2, 3, 5, 6, 0, 0},+ {4, 5, 6, 0, 0, 0, 0, 0}, {0, 4, 5, 6, 0, 0, 0, 0},+ {1, 4, 5, 6, 0, 0, 0, 0}, {0, 1, 4, 5, 6, 0, 0, 0},+ {2, 4, 5, 6, 0, 0, 0, 0}, {0, 2, 4, 5, 6, 0, 0, 0},+ {1, 2, 4, 5, 6, 0, 0, 0}, {0, 1, 2, 4, 5, 6, 0, 0},+ {3, 4, 5, 6, 0, 0, 0, 0}, {0, 3, 4, 5, 6, 0, 0, 0},+ {1, 3, 4, 5, 6, 0, 0, 0}, {0, 1, 3, 4, 5, 6, 0, 0},+ {2, 3, 4, 5, 6, 0, 0, 0}, {0, 2, 3, 4, 5, 6, 0, 0},+ {1, 2, 3, 4, 5, 6, 0, 0}, {0, 1, 2, 3, 4, 5, 6, 0},+ {7, 0, 0, 0, 0, 0, 0, 0}, {0, 7, 0, 0, 0, 0, 0, 0},+ {1, 7, 0, 0, 0, 0, 0, 0}, {0, 1, 7, 0, 0, 0, 0, 0},+ {2, 7, 0, 0, 0, 0, 0, 0}, {0, 2, 7, 0, 0, 0, 0, 0},+ {1, 2, 7, 0, 0, 0, 0, 0}, {0, 1, 2, 7, 0, 0, 0, 0},+ {3, 7, 0, 0, 0, 0, 0, 0}, {0, 3, 7, 0, 0, 0, 0, 0},+ {1, 3, 7, 0, 0, 0, 0, 0}, {0, 1, 3, 7, 0, 0, 0, 0},+ {2, 3, 7, 0, 0, 0, 0, 0}, {0, 2, 3, 7, 0, 0, 0, 0},+ {1, 2, 3, 7, 0, 0, 0, 0}, {0, 1, 2, 3, 7, 0, 0, 0},+ {4, 7, 0, 0, 0, 0, 0, 0}, {0, 4, 7, 0, 0, 0, 0, 0},+ {1, 4, 7, 0, 0, 0, 0, 0}, {0, 1, 4, 7, 0, 0, 0, 0},+ {2, 4, 7, 0, 0, 0, 0, 0}, {0, 2, 4, 7, 0, 0, 0, 0},+ {1, 2, 4, 7, 0, 0, 0, 0}, {0, 1, 2, 4, 7, 0, 0, 0},+ {3, 4, 7, 0, 0, 0, 0, 0}, {0, 3, 4, 7, 0, 0, 0, 0},+ {1, 3, 4, 7, 0, 0, 0, 0}, {0, 1, 3, 4, 7, 0, 0, 0},+ {2, 3, 4, 7, 0, 0, 0, 0}, {0, 2, 3, 4, 7, 0, 0, 0},+ {1, 2, 3, 4, 7, 0, 0, 0}, {0, 1, 2, 3, 4, 7, 0, 0},+ {5, 7, 0, 0, 0, 0, 0, 0}, {0, 5, 7, 0, 0, 0, 0, 0},+ {1, 5, 7, 0, 0, 0, 0, 0}, {0, 1, 5, 7, 0, 0, 0, 0},+ {2, 5, 7, 0, 0, 0, 0, 0}, {0, 2, 5, 7, 0, 0, 0, 0},+ {1, 2, 5, 7, 0, 0, 0, 0}, {0, 1, 2, 5, 7, 0, 0, 0},+ {3, 5, 7, 0, 0, 0, 0, 0}, {0, 3, 5, 7, 0, 0, 0, 0},+ {1, 3, 5, 7, 0, 0, 0, 0}, {0, 1, 3, 5, 7, 0, 0, 0},+ {2, 3, 5, 7, 0, 0, 0, 0}, {0, 2, 3, 5, 7, 0, 0, 0},+ {1, 2, 3, 5, 7, 0, 0, 0}, {0, 1, 2, 3, 5, 7, 0, 0},+ {4, 5, 7, 0, 0, 0, 0, 0}, {0, 4, 5, 7, 0, 0, 0, 0},+ {1, 4, 5, 7, 0, 0, 0, 0}, {0, 1, 4, 5, 7, 0, 0, 0},+ {2, 4, 5, 7, 0, 0, 0, 0}, {0, 2, 4, 5, 7, 0, 0, 0},+ {1, 2, 4, 5, 7, 0, 0, 0}, {0, 1, 2, 4, 5, 7, 0, 0},+ {3, 4, 5, 7, 0, 0, 0, 0}, {0, 3, 4, 5, 7, 0, 0, 0},+ {1, 3, 4, 5, 7, 0, 0, 0}, {0, 1, 3, 4, 5, 7, 0, 0},+ {2, 3, 4, 5, 7, 0, 0, 0}, {0, 2, 3, 4, 5, 7, 0, 0},+ {1, 2, 3, 4, 5, 7, 0, 0}, {0, 1, 2, 3, 4, 5, 7, 0},+ {6, 7, 0, 0, 0, 0, 0, 0}, {0, 6, 7, 0, 0, 0, 0, 0},+ {1, 6, 7, 0, 0, 0, 0, 0}, {0, 1, 6, 7, 0, 0, 0, 0},+ {2, 6, 7, 0, 0, 0, 0, 0}, {0, 2, 6, 7, 0, 0, 0, 0},+ {1, 2, 6, 7, 0, 0, 0, 0}, {0, 1, 2, 6, 7, 0, 0, 0},+ {3, 6, 7, 0, 0, 0, 0, 0}, {0, 3, 6, 7, 0, 0, 0, 0},+ {1, 3, 6, 7, 0, 0, 0, 0}, {0, 1, 3, 6, 7, 0, 0, 0},+ {2, 3, 6, 7, 0, 0, 0, 0}, {0, 2, 3, 6, 7, 0, 0, 0},+ {1, 2, 3, 6, 7, 0, 0, 0}, {0, 1, 2, 3, 6, 7, 0, 0},+ {4, 6, 7, 0, 0, 0, 0, 0}, {0, 4, 6, 7, 0, 0, 0, 0},+ {1, 4, 6, 7, 0, 0, 0, 0}, {0, 1, 4, 6, 7, 0, 0, 0},+ {2, 4, 6, 7, 0, 0, 0, 0}, {0, 2, 4, 6, 7, 0, 0, 0},+ {1, 2, 4, 6, 7, 0, 0, 0}, {0, 1, 2, 4, 6, 7, 0, 0},+ {3, 4, 6, 7, 0, 0, 0, 0}, {0, 3, 4, 6, 7, 0, 0, 0},+ {1, 3, 4, 6, 7, 0, 0, 0}, {0, 1, 3, 4, 6, 7, 0, 0},+ {2, 3, 4, 6, 7, 0, 0, 0}, {0, 2, 3, 4, 6, 7, 0, 0},+ {1, 2, 3, 4, 6, 7, 0, 0}, {0, 1, 2, 3, 4, 6, 7, 0},+ {5, 6, 7, 0, 0, 0, 0, 0}, {0, 5, 6, 7, 0, 0, 0, 0},+ {1, 5, 6, 7, 0, 0, 0, 0}, {0, 1, 5, 6, 7, 0, 0, 0},+ {2, 5, 6, 7, 0, 0, 0, 0}, {0, 2, 5, 6, 7, 0, 0, 0},+ {1, 2, 5, 6, 7, 0, 0, 0}, {0, 1, 2, 5, 6, 7, 0, 0},+ {3, 5, 6, 7, 0, 0, 0, 0}, {0, 3, 5, 6, 7, 0, 0, 0},+ {1, 3, 5, 6, 7, 0, 0, 0}, {0, 1, 3, 5, 6, 7, 0, 0},+ {2, 3, 5, 6, 7, 0, 0, 0}, {0, 2, 3, 5, 6, 7, 0, 0},+ {1, 2, 3, 5, 6, 7, 0, 0}, {0, 1, 2, 3, 5, 6, 7, 0},+ {4, 5, 6, 7, 0, 0, 0, 0}, {0, 4, 5, 6, 7, 0, 0, 0},+ {1, 4, 5, 6, 7, 0, 0, 0}, {0, 1, 4, 5, 6, 7, 0, 0},+ {2, 4, 5, 6, 7, 0, 0, 0}, {0, 2, 4, 5, 6, 7, 0, 0},+ {1, 2, 4, 5, 6, 7, 0, 0}, {0, 1, 2, 4, 5, 6, 7, 0},+ {3, 4, 5, 6, 7, 0, 0, 0}, {0, 3, 4, 5, 6, 7, 0, 0},+ {1, 3, 4, 5, 6, 7, 0, 0}, {0, 1, 3, 4, 5, 6, 7, 0},+ {2, 3, 4, 5, 6, 7, 0, 0}, {0, 2, 3, 4, 5, 6, 7, 0},+ {1, 2, 3, 4, 5, 6, 7, 0}, {0, 1, 2, 3, 4, 5, 6, 7},+};++#else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++MLD_EMPTY_CU(avx2_rej_uniform_table)++#endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mldsa/src/packing.c view
@@ -0,0 +1,213 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#include <string.h>++#include "common.h"+#include "packing.h"+#include "poly.h"+#include "polyvec.h"+#include "rounding.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mldsa-native (e.g. with varying parameter sets)+ * within a single compilation unit. */+/* End of parameter set namespacing */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_INTERNAL_API+void mld_unpack_pk_t1(mld_poly *t1,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ unsigned int i)+{+ mld_polyt1_unpack(t1, pk + MLDSA_PK_T1_OFFSET + i * MLDSA_POLYT1_PACKEDBYTES);+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_INTERNAL_API+void mld_pack_sk_s1(uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ const mld_polyvecl *s1)+{+ mld_polyvecl_pack_eta(sk + MLDSA_SK_S1_OFFSET, s1);+}++MLD_INTERNAL_API+void mld_pack_sk_rho_key_tr_s2(uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ const uint8_t rho[MLDSA_SEEDBYTES],+ const uint8_t tr[MLDSA_TRBYTES],+ const uint8_t key[MLDSA_SEEDBYTES],+ const mld_polyveck *s2)+{+ mld_memcpy(sk + MLDSA_SK_RHO_OFFSET, rho, MLDSA_SEEDBYTES);+ mld_memcpy(sk + MLDSA_SK_KEY_OFFSET, key, MLDSA_SEEDBYTES);+ mld_memcpy(sk + MLDSA_SK_TR_OFFSET, tr, MLDSA_TRBYTES);+ /* s1 already packed via mld_pack_sk_s1 */+ mld_polyveck_pack_eta(sk + MLDSA_SK_S2_OFFSET, s2);+ /* t0 already packed via mld_compute_pack_t0_t1 */+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_INTERNAL_API+void mld_unpack_sk(uint8_t rho[MLDSA_SEEDBYTES], uint8_t tr[MLDSA_TRBYTES],+ uint8_t key[MLDSA_SEEDBYTES], mld_sk_t0hat *t0,+ mld_sk_s1hat *s1, mld_sk_s2hat *s2,+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES])+{+ mld_memcpy(rho, sk + MLDSA_SK_RHO_OFFSET, MLDSA_SEEDBYTES);+ mld_memcpy(key, sk + MLDSA_SK_KEY_OFFSET, MLDSA_SEEDBYTES);+ mld_memcpy(tr, sk + MLDSA_SK_TR_OFFSET, MLDSA_TRBYTES);+ mld_unpack_sk_s1hat(s1, sk + MLDSA_SK_S1_OFFSET);+ mld_unpack_sk_s2hat(s2, sk + MLDSA_SK_S2_OFFSET);+ mld_unpack_sk_t0hat(t0, sk + MLDSA_SK_T0_OFFSET);+}++MLD_INTERNAL_API+void mld_pack_sig_c(uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t c[MLDSA_CTILDEBYTES])+{+ mld_memcpy(sig, c, MLDSA_CTILDEBYTES);+}++MLD_INTERNAL_API+int mld_pack_sig_h(uint8_t sig[MLDSA_CRYPTO_BYTES], const mld_polyveck *w0,+ const mld_polyveck *w1)+{+ unsigned int j, k, n;++ /* The hint section of sig[] is MLDSA_POLYVECH_PACKEDBYTES long, where+ * MLDSA_POLYVECH_PACKEDBYTES = MLDSA_OMEGA + MLDSA_K.+ *+ * The first OMEGA bytes record the index numbers of the coefficients+ * that are not equal to 0.+ *+ * The final K bytes record a running tally of the number of hints+ * coming from each of the K polynomials. */+ uint8_t *sig_h = sig + MLDSA_SIG_H_OFFSET;++ mld_memset(sig_h, 0, MLDSA_POLYVECH_PACKEDBYTES);+ n = 0;++ /* For each coefficient of each polynomial, compute its hint bit and, if+ * non-zero, record the index in the hint section of sig. If recording the+ * hint would overflow the OMEGA-sized index array, abort early and return+ * MLD_ERR_FAIL. The caller is expected to reject the signature in that case.+ *+ * Constant time: At this point w0/w1 are public (see comment in sign.c+ * before the call), so a data-dependent early return is fine. */+ for (k = 0; k < MLDSA_K; k++)+ __loop__(+ assigns(k, j, n, memory_slice(sig_h, MLDSA_POLYVECH_PACKEDBYTES))+ invariant(k <= MLDSA_K && n <= MLDSA_OMEGA)+ decreases(MLDSA_K - k)+ )+ {+ for (j = 0; j < MLDSA_N; j++)+ __loop__(+ assigns(j, n, memory_slice(sig_h, MLDSA_POLYVECH_PACKEDBYTES))+ invariant(j <= MLDSA_N && n <= MLDSA_OMEGA)+ decreases(MLDSA_N - j)+ )+ {+ const unsigned int hint_bit =+ mld_make_hint(w0->vec[k].coeffs[j], w1->vec[k].coeffs[j]);+ if (hint_bit)+ {+ if (n == MLDSA_OMEGA)+ {+ return MLD_ERR_FAIL;+ }+ /* Safety: branch above ensures n < MLDSA_OMEGA so n is a valid index+ * into the OMEGA-sized index array; j < MLDSA_N <= 256 fits in+ * uint8_t. */+ sig_h[n] = (uint8_t)j;+ n++;+ }+ }+ /* Record the running tally into the correct slot for this polynomial.+ * Safety: k < MLDSA_K, so MLDSA_OMEGA + k is a valid index into the+ * K-byte tally tail; n <= MLDSA_OMEGA fits in uint8_t. */+ sig_h[MLDSA_OMEGA + k] = (uint8_t)n;+ }+ return 0;+}++MLD_INTERNAL_API+void mld_pack_sig_z(uint8_t sig[MLDSA_CRYPTO_BYTES], const mld_poly *zi,+ unsigned i)+{+ mld_polyz_pack(sig + MLDSA_SIG_Z_OFFSET + i * MLDSA_POLYZ_PACKEDBYTES, zi);+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_INTERNAL_API+int mld_sig_unpack_hints(mld_poly *h, const uint8_t sig[MLDSA_CRYPTO_BYTES],+ unsigned int i)+{+ const uint8_t *packed_hints = sig + MLDSA_SIG_H_OFFSET;+ const unsigned int old_hint_count =+ (i == 0) ? 0 : packed_hints[MLDSA_OMEGA + i - 1];+ const unsigned int new_hint_count = packed_hints[MLDSA_OMEGA + i];+ unsigned int j;++ if (new_hint_count < old_hint_count || new_hint_count > MLDSA_OMEGA)+ {+ return MLD_ERR_FAIL;+ }++ mld_memset(h, 0, sizeof(mld_poly));++ for (j = old_hint_count; j < new_hint_count; ++j)+ __loop__(+ invariant(j >= old_hint_count && j <= new_hint_count &&+ new_hint_count <= MLDSA_OMEGA)+ invariant(array_bound(h->coeffs, 0, MLDSA_N, 0, 2))+ invariant(forall(p, 0, MLDSA_N,+ (h->coeffs[p] == 1) ==+ exists(hj, old_hint_count, j, packed_hints[hj] == p)))+ decreases(new_hint_count - j)+ )+ {+ if (j > old_hint_count && packed_hints[j] <= packed_hints[j - 1])+ {+ return MLD_ERR_FAIL;+ }+ /* Safety: packed_hints[j] is uint8_t (<= 255) and MLDSA_N == 256. */+ h->coeffs[packed_hints[j]] = 1;+ }++ /* On the last row, also verify that the trailing index slots are zero. */+ if (i == MLDSA_K - 1)+ {+ for (j = new_hint_count; j < MLDSA_OMEGA; ++j)+ __loop__(+ invariant(j <= MLDSA_OMEGA)+ decreases(MLDSA_OMEGA - j)+ )+ {+ if (packed_hints[j] != 0)+ {+ return MLD_ERR_FAIL;+ }+ }+ }++ /* On success, h->coeffs[p] is 1 exactly for the hint indices decoded for this+ * row, i.e. packed_hints[old_hint_count, new_hint_count). Asserted here+ * rather than posted as a contract to not unnecessarily increase proof+ * complexity at callers that do not need the functional description. */+ cassert(forall(+ p, 0, MLDSA_N,+ (h->coeffs[p] == 1) ==+ exists(hj, old_hint_count, new_hint_count, packed_hints[hj] == p)));++ return 0;+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */
+ cbits/mldsa/src/packing.h view
@@ -0,0 +1,277 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_PACKING_H+#define MLD_PACKING_H++#include "polyvec.h"+#include "polyvec_lazy.h"++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_pack_sk_s1 MLD_NAMESPACE_KL(pack_sk_s1)+/**+ * Bit-pack the s1 component into the secret key.+ *+ * @spec{Partially implements @[FIPS204, Algorithm 24, skEncode] (s1+ * component).}+ *+ * @param[out] sk Output byte array.+ * @param[in] s1 Pointer to vector s1.+ */+MLD_INTERNAL_API+void mld_pack_sk_s1(uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ const mld_polyvecl *s1)+__contract__(+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ requires(memory_no_alias(s1, sizeof(mld_polyvecl)))+ requires(forall(k1, 0, MLDSA_L,+ array_abs_bound(s1->vec[k1].coeffs, 0, MLDSA_N, MLDSA_ETA + 1)))+ assigns(memory_slice(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+);++#define mld_pack_sk_rho_key_tr_s2 MLD_NAMESPACE_KL(pack_sk_rho_key_tr_s2)+/**+ * Bit-pack rho, key, tr, s2 into the secret key.+ *+ * s1 must already be packed via mld_pack_sk_s1, and t0 via+ * mld_compute_pack_t0_t1.+ *+ * @spec{Partially implements @[FIPS204, Algorithm 24, skEncode] (rho, key, tr,+ * s2 components).}+ *+ * @param[out] sk Output byte array.+ * @param[in] rho Byte array containing rho.+ * @param[in] tr Byte array containing tr.+ * @param[in] key Byte array containing key.+ * @param[in] s2 Pointer to vector s2.+ */+MLD_INTERNAL_API+void mld_pack_sk_rho_key_tr_s2(uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ const uint8_t rho[MLDSA_SEEDBYTES],+ const uint8_t tr[MLDSA_TRBYTES],+ const uint8_t key[MLDSA_SEEDBYTES],+ const mld_polyveck *s2)+__contract__(+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ requires(memory_no_alias(rho, MLDSA_SEEDBYTES))+ requires(memory_no_alias(tr, MLDSA_TRBYTES))+ requires(memory_no_alias(key, MLDSA_SEEDBYTES))+ requires(memory_no_alias(s2, sizeof(mld_polyveck)))+ requires(forall(k2, 0, MLDSA_K,+ array_abs_bound(s2->vec[k2].coeffs, 0, MLDSA_N, MLDSA_ETA + 1)))+ assigns(memory_slice(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */+++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_pack_sig_c MLD_NAMESPACE_KL(pack_sig_c)+/**+ * Bit-pack challenge c into sig = (c, z, h).+ *+ * @spec{Partially implements @[FIPS204, Algorithm 26, sigEncode] (c+ * component).}+ *+ * @param[out] sig Output byte array.+ * @param[in] c Pointer to challenge hash.+ */+MLD_INTERNAL_API+void mld_pack_sig_c(uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t c[MLDSA_CTILDEBYTES])+__contract__(+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(c, MLDSA_CTILDEBYTES))+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+);++#define mld_pack_sig_h MLD_NAMESPACE_KL(pack_sig_h)+/**+ * Compute hints from (w0, w1) and pack them into the hint section of sig.+ *+ * @spec{Combines the hint computation @[FIPS204, Algorithm 39, MakeHint] with+ * the packing @[FIPS204, Algorithm 20, HintBitPack] (the h component of+ * @[FIPS204, Algorithm 26, sigEncode]): it computes the hint vector h from+ * (w0, w1) and packs it, rather than receiving a ready-made h as HintBitPack+ * does. The hints are computed via mld_make_hint (rounding.h), a specialized+ * MakeHint valid only for the values arising during signing; see the block+ * comment in mld_attempt_signature_generation (sign.c).}+ *+ * @param[in,out] sig Byte array containing signature.+ * @param[in] w0 Pointer to low part of input vector.+ * @param[in] w1 Pointer to high part of input vector.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_FAIL The total number of hints exceeds MLDSA_OMEGA. In this+ * case the hint section of sig is left in a+ * partially-written state and the caller must reject the+ * signature.+ */+MLD_INTERNAL_API+MLD_MUST_CHECK_RETURN_VALUE+int mld_pack_sig_h(uint8_t sig[MLDSA_CRYPTO_BYTES], const mld_polyveck *w0,+ const mld_polyveck *w1)+__contract__(+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(w0, sizeof(mld_polyveck)))+ requires(memory_no_alias(w1, sizeof(mld_polyveck)))+ assigns(memory_slice(sig + MLDSA_SIG_H_OFFSET, MLDSA_POLYVECH_PACKEDBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_FAIL)+);++#define mld_pack_sig_z MLD_NAMESPACE_KL(pack_sig_z)+/**+ * Bit-pack single polynomial of z component of sig = (c, z, h).+ *+ * The c and h components are packed separately using mld_pack_sig_c and+ * mld_pack_sig_h.+ *+ * @spec{Partially implements @[FIPS204, Algorithm 26, sigEncode] (one+ * polynomial of the z component).}+ *+ * @param[in,out] sig Output byte array.+ * @param[in] zi Pointer to a single polynomial in z.+ * @param i Index of zi in vector z.+ */+MLD_INTERNAL_API+void mld_pack_sig_z(uint8_t sig[MLDSA_CRYPTO_BYTES], const mld_poly *zi,+ unsigned i)+__contract__(+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(zi, sizeof(mld_poly)))+ requires(i < MLDSA_L)+ requires(array_bound(zi->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_unpack_pk_t1 MLD_NAMESPACE_KL(unpack_pk_t1)+/**+ * Unpack a single polynomial of the t1 component of a public key+ * pk = (rho, t1).+ *+ * @spec{Partially implements @[FIPS204, Algorithm 23, pkDecode] (one polynomial+ * of t1).}+ *+ * @param[out] t1 Pointer to output polynomial t1[i].+ * @param[in] pk Byte array containing bit-packed pk.+ * @param i Row index, must be < MLDSA_K.+ */+MLD_INTERNAL_API+void mld_unpack_pk_t1(mld_poly *t1,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ unsigned int i)+__contract__(+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ requires(memory_no_alias(t1, sizeof(mld_poly)))+ requires(i < MLDSA_K)+ assigns(memory_slice(t1, sizeof(mld_poly)))+ ensures(array_bound(t1->coeffs, 0, MLDSA_N, 0, 1 << 10))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_unpack_sk MLD_NAMESPACE_KL(unpack_sk)+/**+ * Unpack secret key sk = (rho, tr, key, t0, s1, s2).+ *+ * NOTE: In REDUCE_RAM mode, s1/s2/t0 borrow from sk rather than copying.+ *+ * @spec{Implements @[FIPS204, Algorithm 25, skDecode].}+ *+ * @param[out] rho Output byte array for rho.+ * @param[out] tr Output byte array for tr.+ * @param[out] key Output byte array for key.+ * @param[out] t0 Pointer to output vector t0.+ * @param[out] s1 Pointer to output vector s1.+ * @param[out] s2 Pointer to output vector s2.+ * @param[in] sk Byte array containing bit-packed sk.+ */+MLD_INTERNAL_API+void mld_unpack_sk(uint8_t rho[MLDSA_SEEDBYTES], uint8_t tr[MLDSA_TRBYTES],+ uint8_t key[MLDSA_SEEDBYTES], mld_sk_t0hat *t0,+ mld_sk_s1hat *s1, mld_sk_s2hat *s2,+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES])+__contract__(+ requires(memory_no_alias(rho, MLDSA_SEEDBYTES))+ requires(memory_no_alias(tr, MLDSA_TRBYTES))+ requires(memory_no_alias(key, MLDSA_SEEDBYTES))+ requires(memory_no_alias(t0, sizeof(mld_sk_t0hat)))+ requires(memory_no_alias(s1, sizeof(mld_sk_s1hat)))+ requires(memory_no_alias(s2, sizeof(mld_sk_s2hat)))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ assigns(memory_slice(rho, MLDSA_SEEDBYTES))+ assigns(memory_slice(tr, MLDSA_TRBYTES))+ assigns(memory_slice(key, MLDSA_SEEDBYTES))+ assigns(memory_slice(t0, sizeof(mld_sk_t0hat)))+ assigns(memory_slice(s1, sizeof(mld_sk_s1hat)))+ assigns(memory_slice(s2, sizeof(mld_sk_s2hat)))+ MLD_IF_NOT_REDUCE_RAM(+ ensures(forall(k0, 0, MLDSA_K,+ array_abs_bound(t0->vec.vec[k0].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ ensures(forall(k1, 0, MLDSA_L,+ array_abs_bound(s1->vec.vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ ensures(forall(k2, 0, MLDSA_K,+ array_abs_bound(s2->vec.vec[k2].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ )+ MLD_IF_REDUCE_RAM(+ ensures(s1->packed == old(sk) + MLDSA_SK_S1_OFFSET)+ ensures(s2->packed == old(sk) + MLDSA_SK_S2_OFFSET)+ ensures(t0->packed == old(sk) + MLDSA_SK_T0_OFFSET)+ )+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_sig_unpack_hints MLD_NAMESPACE_KL(sig_unpack_hints)+/**+ * Decode and validate a single row of the hint vector h from a signature+ * buffer.+ *+ * The hint encoding is shared across all rows (a count array followed by a+ * single index list), so this function performs the validation relevant to+ * row i:+ * - the i'th hint count is non-decreasing and bounded by MLDSA_OMEGA;+ * - the indices for row i are strictly ascending;+ * - on i == MLDSA_K - 1, the trailing index slots are zero.+ *+ * Callers must invoke this for every i in [0, 1, .., MLDSA_K - 1]; if any+ * call returns MLD_ERR_FAIL the encoding is malformed and the signature must+ * be rejected.+ *+ * @spec{Implements @[FIPS204, Algorithm 21, HintBitUnpack] (one row; part of+ * @[FIPS204, Algorithm 27, sigDecode]).}+ *+ * @param[out] h Pointer to output polynomial h[i].+ * @param[in] sig Signature buffer.+ * @param i Row index, must be < MLDSA_K.+ *+ * @retval 0 Hints were decoded successfully.+ * @retval MLD_ERR_FAIL Hints are malformed.+ */+MLD_INTERNAL_API+MLD_MUST_CHECK_RETURN_VALUE+int mld_sig_unpack_hints(mld_poly *h, const uint8_t sig[MLDSA_CRYPTO_BYTES],+ unsigned int i)+__contract__(+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(h, sizeof(mld_poly)))+ requires(i < MLDSA_K)+ assigns(memory_slice(h, sizeof(mld_poly)))+ ensures(return_value == 0 || return_value == MLD_ERR_FAIL)+ ensures(return_value == 0 ==> array_bound(h->coeffs, 0, MLDSA_N, 0, 2))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#endif /* !MLD_PACKING_H */
+ cbits/mldsa/src/params.h view
@@ -0,0 +1,153 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_PARAMS_H+#define MLD_PARAMS_H++#define MLDSA_SEEDBYTES 32+#define MLDSA_CRHBYTES 64+#define MLDSA_TRBYTES 64+#define MLDSA_RNDBYTES 32+#define MLDSA_N 256+#define MLDSA_Q 8380417+#define MLDSA_Q_HALF ((MLDSA_Q + 1) / 2)+#define MLDSA_D 13++#define MLDSA_GAMMA2_88 ((MLDSA_Q - 1) / 88)+#define MLDSA_GAMMA2_32 ((MLDSA_Q - 1) / 32)+#define MLDSA_POLYW1_PACKEDBYTES_88 192+#define MLDSA_POLYW1_PACKEDBYTES_32 128++#if MLD_CONFIG_PARAMETER_SET == 44++#define MLDSA_K 4+#define MLDSA_L 4+#define MLDSA_ETA 2+#define MLDSA_TAU 39+#define MLDSA_BETA 78+#define MLDSA_GAMMA1 ((int32_t)1 << 17)+#define MLDSA_GAMMA2 MLDSA_GAMMA2_88+#define MLDSA_OMEGA 80+#define MLDSA_CTILDEBYTES 32+#define MLDSA_POLYZ_PACKEDBYTES 576+#define MLDSA_POLYW1_PACKEDBYTES MLDSA_POLYW1_PACKEDBYTES_88+#define MLDSA_POLYETA_PACKEDBYTES 96++#elif MLD_CONFIG_PARAMETER_SET == 65++#define MLDSA_K 6+#define MLDSA_L 5+#define MLDSA_ETA 4+#define MLDSA_TAU 49+#define MLDSA_BETA 196+#define MLDSA_GAMMA1 ((int32_t)1 << 19)+#define MLDSA_GAMMA2 MLDSA_GAMMA2_32+#define MLDSA_OMEGA 55+#define MLDSA_CTILDEBYTES 48+#define MLDSA_POLYZ_PACKEDBYTES 640+#define MLDSA_POLYW1_PACKEDBYTES MLDSA_POLYW1_PACKEDBYTES_32+#define MLDSA_POLYETA_PACKEDBYTES 128++#elif MLD_CONFIG_PARAMETER_SET == 87++#define MLDSA_K 8+#define MLDSA_L 7+#define MLDSA_ETA 2+#define MLDSA_TAU 60+#define MLDSA_BETA 120+#define MLDSA_GAMMA1 ((int32_t)1 << 19)+#define MLDSA_GAMMA2 MLDSA_GAMMA2_32+#define MLDSA_OMEGA 75+#define MLDSA_CTILDEBYTES 64+#define MLDSA_POLYZ_PACKEDBYTES 640+#define MLDSA_POLYW1_PACKEDBYTES MLDSA_POLYW1_PACKEDBYTES_32+#define MLDSA_POLYETA_PACKEDBYTES 96++#endif /* MLD_CONFIG_PARAMETER_SET == 87 */++#define MLDSA_POLYT1_PACKEDBYTES 320+#define MLDSA_POLYT0_PACKEDBYTES 416+#define MLDSA_POLYVECH_PACKEDBYTES (MLDSA_OMEGA + MLDSA_K)++/* Sampling y from counter kappa uses nonces kappa, ..., kappa+L-1, which fit in+ * uint16_t iff kappa <= UINT16_MAX - MLDSA_L. With kappa = attempt*MLDSA_L this+ * bounds the number of signing attempts by MLD_MAX_KAPPA / MLDSA_L; see+ * MLD_MAX_SIGNING_ATTEMPTS in sign.c. */+#define MLD_MAX_KAPPA (UINT16_MAX - MLDSA_L)++/* Layout of the packed public key pk[MLDSA_CRYPTO_PUBLICKEYBYTES] = (rho, t1):+ *+ * +-------------+--------------------------++ * | rho | t1 |+ * +-------------+--------------------------++ * | SEEDBYTES | K * POLYT1_PACKEDBYTES |+ * +-------------+--------------------------++ */+#define MLDSA_PK_RHO_OFFSET 0+#define MLDSA_PK_RHO_BYTES MLDSA_SEEDBYTES++#define MLDSA_PK_T1_OFFSET (MLDSA_PK_RHO_OFFSET + MLDSA_PK_RHO_BYTES)+#define MLDSA_PK_T1_BYTES (MLDSA_K * MLDSA_POLYT1_PACKEDBYTES)++#define MLDSA_PK_END (MLDSA_PK_T1_OFFSET + MLDSA_PK_T1_BYTES)++#define MLDSA_CRYPTO_PUBLICKEYBYTES MLDSA_PK_END++/* Layout of the packed secret key+ * sk[MLDSA_CRYPTO_SECRETKEYBYTES] = (rho, key, tr, s1, s2, t0):+ *+ * +-----------+-----------+-----------+-----------+-----------+-----------++ * | rho | key | tr | s1 | s2 | t0 |+ * +-----------+-----------+-----------+-----------+-----------+-----------++ * | SEEDBYTES | SEEDBYTES | TRBYTES | L * | K * | K * |+ * | | | | POLYETA_ | POLYETA_ | POLYT0_ |+ * | | | | PACKED- | PACKED- | PACKED- |+ * | | | | BYTES | BYTES | BYTES |+ * +-----------+-----------+-----------+-----------+-----------+-----------++ */+#define MLDSA_SK_RHO_OFFSET 0+#define MLDSA_SK_RHO_BYTES MLDSA_SEEDBYTES++#define MLDSA_SK_KEY_OFFSET (MLDSA_SK_RHO_OFFSET + MLDSA_SK_RHO_BYTES)+#define MLDSA_SK_KEY_BYTES MLDSA_SEEDBYTES++#define MLDSA_SK_TR_OFFSET (MLDSA_SK_KEY_OFFSET + MLDSA_SK_KEY_BYTES)+#define MLDSA_SK_TR_BYTES MLDSA_TRBYTES++#define MLDSA_SK_S1_OFFSET (MLDSA_SK_TR_OFFSET + MLDSA_SK_TR_BYTES)+#define MLDSA_SK_S1_BYTES (MLDSA_L * MLDSA_POLYETA_PACKEDBYTES)++#define MLDSA_SK_S2_OFFSET (MLDSA_SK_S1_OFFSET + MLDSA_SK_S1_BYTES)+#define MLDSA_SK_S2_BYTES (MLDSA_K * MLDSA_POLYETA_PACKEDBYTES)++#define MLDSA_SK_T0_OFFSET (MLDSA_SK_S2_OFFSET + MLDSA_SK_S2_BYTES)+#define MLDSA_SK_T0_BYTES (MLDSA_K * MLDSA_POLYT0_PACKEDBYTES)++#define MLDSA_SK_END (MLDSA_SK_T0_OFFSET + MLDSA_SK_T0_BYTES)++#define MLDSA_CRYPTO_SECRETKEYBYTES MLDSA_SK_END++/* Layout of the packed signature sig[MLDSA_CRYPTO_BYTES] = (c, z, h):+ *+ * +----------------+-------------------+----------------------++ * | c (challenge) | z | h (hints) |+ * +----------------+-------------------+----------------------++ * | CTILDEBYTES | L * | POLYVECH_PACKEDBYTES |+ * | | POLYZ_PACKEDBYTES | (= OMEGA + K) |+ * +----------------+-------------------+----------------------++ */+#define MLDSA_SIG_C_OFFSET 0+#define MLDSA_SIG_C_BYTES MLDSA_CTILDEBYTES++#define MLDSA_SIG_Z_OFFSET (MLDSA_SIG_C_OFFSET + MLDSA_SIG_C_BYTES)+#define MLDSA_SIG_Z_BYTES (MLDSA_L * MLDSA_POLYZ_PACKEDBYTES)++#define MLDSA_SIG_H_OFFSET (MLDSA_SIG_Z_OFFSET + MLDSA_SIG_Z_BYTES)+#define MLDSA_SIG_H_BYTES MLDSA_POLYVECH_PACKEDBYTES++#define MLDSA_SIG_END (MLDSA_SIG_H_OFFSET + MLDSA_SIG_H_BYTES)++#define MLDSA_CRYPTO_BYTES MLDSA_SIG_END++#endif /* !MLD_PARAMS_H */
+ cbits/mldsa/src/poly.c view
@@ -0,0 +1,1066 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ *+ * - [REF]+ * CRYSTALS-Dilithium reference implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/ref+ */++#include "poly.h"++#include "common.h"+#include "ct.h"+#include "debug.h"+#include "reduce.h"+#include "rounding.h"+#include "symmetric.h"++#if !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)+#include "zetas.inc"++MLD_INTERNAL_API+void mld_poly_reduce(mld_poly *a)+{+ unsigned int i;+ mld_assert_bound(a->coeffs, MLDSA_N, INT32_MIN, MLD_REDUCE32_DOMAIN_MAX);++ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ invariant(i <= MLDSA_N)+ invariant(forall(k0, i, MLDSA_N, a->coeffs[k0] == loop_entry(*a).coeffs[k0]))+ invariant(array_bound(a->coeffs, 0, i, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX))+ decreases(MLDSA_N - i))+ {+ a->coeffs[i] = mld_reduce32(a->coeffs[i]);+ }++ mld_assert_bound(a->coeffs, MLDSA_N, -MLD_REDUCE32_RANGE_MAX,+ MLD_REDUCE32_RANGE_MAX);+}++MLD_STATIC_TESTABLE void mld_poly_caddq_c(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+)+{+ unsigned int i;+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);++ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ invariant(i <= MLDSA_N)+ invariant(forall(k0, i, MLDSA_N, a->coeffs[k0] == loop_entry(*a).coeffs[k0]))+ invariant(array_bound(a->coeffs, 0, i, 0, MLDSA_Q))+ decreases(MLDSA_N - i)+ )+ {+ a->coeffs[i] = mld_caddq(a->coeffs[i]);+ }++ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);+}++MLD_INTERNAL_API+void mld_poly_caddq(mld_poly *a)+{+#if defined(MLD_USE_NATIVE_POLY_CADDQ)+ int ret;+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);+ ret = mld_poly_caddq_native(a->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);+ return;+ }+#endif /* MLD_USE_NATIVE_POLY_CADDQ */+ mld_poly_caddq_c(a);+}++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_SIGN_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+/* Reference: We use destructive version (output=first input) to avoid+ * reasoning about aliasing in the CBMC specification */+MLD_INTERNAL_API+void mld_poly_add(mld_poly *r, const mld_poly *b)+{+ unsigned int i;+ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ assigns(i, memory_slice(r, sizeof(mld_poly)))+ invariant(i <= MLDSA_N)+ invariant(forall(k0, i, MLDSA_N, r->coeffs[k0] == loop_entry(*r).coeffs[k0]))+ invariant(forall(k1, 0, i, r->coeffs[k1] == loop_entry(*r).coeffs[k1] + b->coeffs[k1]))+ invariant(forall(k2, 0, i, r->coeffs[k2] < MLD_REDUCE32_DOMAIN_MAX))+ invariant(forall(k2, 0, i, r->coeffs[k2] >= INT32_MIN))+ decreases(MLDSA_N - i)+ )+ {+ r->coeffs[i] = r->coeffs[i] + b->coeffs[i];+ }+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+/* Reference: We use destructive version (output=first input) to avoid+ * reasoning about aliasing in the CBMC specification */+MLD_INTERNAL_API+void mld_poly_sub(mld_poly *r, const mld_poly *b)+{+ unsigned int i;+ mld_assert_abs_bound(b->coeffs, MLDSA_N, MLDSA_Q);+ mld_assert_abs_bound(r->coeffs, MLDSA_N, MLDSA_Q);++ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ invariant(i <= MLDSA_N)+ invariant(array_bound(r->coeffs, 0, i, INT32_MIN, MLD_REDUCE32_DOMAIN_MAX))+ invariant(forall(k0, i, MLDSA_N, r->coeffs[k0] == loop_entry(*r).coeffs[k0]))+ decreases(MLDSA_N - i)+ )+ {+ r->coeffs[i] = r->coeffs[i] - b->coeffs[i];+ }++ mld_assert_bound(r->coeffs, MLDSA_N, INT32_MIN, MLD_REDUCE32_DOMAIN_MAX);+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_INTERNAL_API+void mld_poly_shiftl(mld_poly *a)+{+ unsigned int i;+ mld_assert_bound(a->coeffs, MLDSA_N, 0, 1 << 10);++ for (i = 0; i < MLDSA_N; i++)+ __loop__(+ invariant(i <= MLDSA_N)+ invariant(array_bound(a->coeffs, 0, i, 0, MLDSA_Q))+ invariant(forall(k0, i, MLDSA_N, a->coeffs[k0] == loop_entry(*a).coeffs[k0]))+ decreases(MLDSA_N - i))+ {+ /* Reference: uses a left shift by MLDSA_D which is undefined behaviour in+ * C90/C99+ */+ a->coeffs[i] *= (1 << MLDSA_D);+ }+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++static MLD_INLINE int32_t mld_fqmul(int32_t a, int32_t b)+__contract__(+ requires(b > -MLDSA_Q_HALF && b < MLDSA_Q_HALF)+ ensures(return_value > -MLD_FQMUL_BOUND && return_value < MLD_FQMUL_BOUND)+)+{+ /* Bounds: We argue in mld_montgomery_reduce() that the result+ * of Montgomery reduction is < MLDSA_Q if the input is smaller+ * than 2^31 * MLDSA_Q in absolute value. Indeed, we have:+ *+ * |a * b| = |a| * |b|+ * < 2^31 * MLDSA_Q_HALF+ * < 2^31 * MLDSA_Q+ *+ * So the output is < MLDSA_Q < MLD_FQMUL_BOUND.+ */+ return mld_montgomery_reduce((int64_t)a * (int64_t)b);+}++/* mld_ntt_butterfly_block()+ *+ * Computes a block CT butterflies with a fixed twiddle factor,+ * using Montgomery multiplication.+ *+ * Parameters:+ * - r: Pointer to base of polynomial (_not_ the base of butterfly block)+ * - zeta: Twiddle factor to use for the butterfly. This must be in+ * Montgomery form and signed canonical.+ * - start: Offset to the beginning of the butterfly block+ * - len: Index difference between coefficients subject to a butterfly+ * - bound: Ghost variable describing coefficient bound: Prior to `start`,+ * coefficients must be bound by `bound + MLDSA_Q`. Post `start`,+ * they must be bound by `bound`.+ * When this function returns, output coefficients in the index range+ * [start, start+2*len) have bound bumped to `bound + MLDSA_Q`.+ * Example:+ * - start=8, len=4+ * This would compute the following four butterflies+ * 8 -- 12+ * 9 -- 13+ * 10 -- 14+ * 11 -- 15+ * - start=4, len=2+ * This would compute the following two butterflies+ * 4 -- 6+ * 5 -- 7+ */++/* Reference: Embedded in `ntt()` in the reference implementation @[REF]. */+static MLD_INLINE void mld_ntt_butterfly_block(int32_t r[MLDSA_N],+ const int32_t zeta,+ const unsigned start,+ const unsigned len,+ const uint32_t bound)+__contract__(+ requires(start < MLDSA_N)+ requires(1 <= len && len <= MLDSA_N / 2 && start + 2 * len <= MLDSA_N)+ requires(0 <= bound && bound < INT32_MAX - MLD_FQMUL_BOUND)+ requires(-MLDSA_Q_HALF < zeta && zeta < MLDSA_Q_HALF)+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(r, 0, start, bound + MLD_FQMUL_BOUND))+ requires(array_abs_bound(r, start, MLDSA_N, bound))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(r, 0, start + 2*len, bound + MLD_FQMUL_BOUND))+ ensures(array_abs_bound(r, start + 2 * len, MLDSA_N, bound)))+{+ /* `bound` is a ghost variable only needed in the CBMC specification */+ unsigned j;+ ((void)bound);+ for (j = start; j < start + len; j++)+ __loop__(+ invariant(start <= j && j <= start + len)+ /*+ * Coefficients are updated in strided pairs, so the bounds for the+ * intermediate states alternate twice between the old and new bound+ */+ invariant(array_abs_bound(r, 0, j, bound + MLD_FQMUL_BOUND))+ invariant(array_abs_bound(r, j, start + len, bound))+ invariant(array_abs_bound(r, start + len, j + len, bound + MLD_FQMUL_BOUND))+ invariant(array_abs_bound(r, j + len, MLDSA_N, bound))+ decreases(start + len - j))+ {+ int32_t t;+ t = mld_fqmul(r[j + len], zeta);+ r[j + len] = r[j] - t;+ r[j] = r[j] + t;+ }+}++/* mld_ntt_layer()+ *+ * Compute one layer of forward NTT+ *+ * Parameters:+ * - r: Pointer to base of polynomial+ * - layer: Indicates which layer is being applied.+ */++/* Reference: Embedded in `ntt()` in the reference implementation @[REF]. */+static MLD_INLINE void mld_ntt_layer(int32_t r[MLDSA_N], const unsigned layer)+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(1 <= layer && layer <= 8)+ requires(array_abs_bound(r, 0, MLDSA_N, layer * MLD_FQMUL_BOUND))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(r, 0, MLDSA_N, (layer + 1) * MLD_FQMUL_BOUND)))+{+ unsigned start, k, len;+ /* Twiddle factors for layer n are at indices 2^(n-1)..2^n-1. */+ k = 1u << (layer - 1);+ len = (unsigned)MLDSA_N >> layer;+ for (start = 0; start < MLDSA_N; start += 2 * len)+ __loop__(+ invariant(start < MLDSA_N + 2 * len)+ invariant(k <= MLDSA_N)+ invariant(2 * len * k == start + MLDSA_N)+ invariant(array_abs_bound(r, 0, start, layer * MLD_FQMUL_BOUND + MLD_FQMUL_BOUND))+ invariant(array_abs_bound(r, start, MLDSA_N, layer * MLD_FQMUL_BOUND))+ decreases(MLDSA_N - start))+ {+ int32_t zeta = mld_zetas[k++];+ mld_ntt_butterfly_block(r, zeta, start, len, layer * MLD_FQMUL_BOUND);+ }+}++MLD_STATIC_TESTABLE void mld_poly_ntt_c(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_abs_bound(a->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+)+{+ unsigned int layer;+ int32_t *r;+++ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);+ r = a->coeffs;++ for (layer = 1; layer < 9; layer++)+ __loop__(+ invariant(1 <= layer && layer <= 9)+ invariant(array_abs_bound(r, 0, MLDSA_N, layer * MLD_FQMUL_BOUND))+ decreases(9 - layer)+ )+ {+ mld_ntt_layer(r, layer);+ }++ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_NTT_BOUND);+}++MLD_INTERNAL_API+void mld_poly_ntt(mld_poly *a)+{+#if defined(MLD_USE_NATIVE_NTT)+ int ret;+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);+ ret = mld_ntt_native(a->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_NTT_BOUND);+ return;+ }+#endif /* MLD_USE_NATIVE_NTT */+ mld_poly_ntt_c(a);+}++/**+ * Scale a field element by mont/256, i.e., perform Montgomery multiplication+ * by mont^2/256.+ *+ * Input is expected to have absolute value smaller than 256 * MLDSA_Q. Output+ * has absolute value smaller than MLD_INTT_BOUND.+ *+ * @param a Field element to be scaled.+ */+static MLD_INLINE int32_t mld_fqscale(int32_t a)+__contract__(+ requires(a > -256*MLDSA_Q && a < 256*MLDSA_Q)+ ensures(return_value > -MLD_INTT_BOUND && return_value < MLD_INTT_BOUND)+)+{+ /* check-magic: 41978 == pow(2,64-8,MLDSA_Q) */+ const int32_t f = 41978;+ /* Bounds: MLD_INTT_BOUND is MLDSA_Q, so the bounds reasoning is just+ * a special case of that in mld_fqmul(). */+ return mld_montgomery_reduce((int64_t)a * f);+}++/* Reference: Embedded into `invntt_tomont()` in the reference implementation+ * @[REF] */+static MLD_INLINE void mld_invntt_layer(int32_t r[MLDSA_N], unsigned layer)+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(1 <= layer && layer <= 8)+ requires(array_abs_bound(r, 0, MLDSA_N, (MLDSA_N >> layer) * MLDSA_Q))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(r, 0, MLDSA_N, (MLDSA_N >> (layer - 1)) * MLDSA_Q)))+{+ unsigned start, k, len;+ len = (unsigned)MLDSA_N >> layer;+ k = (1u << layer) - 1;+ for (start = 0; start < MLDSA_N; start += 2 * len)+ __loop__(+ invariant(start <= MLDSA_N && k <= 255)+ invariant(2 * len * k + start == 2 * MLDSA_N - 2 * len)+ invariant(array_abs_bound(r, 0, start, (MLDSA_N >> (layer - 1)) * MLDSA_Q))+ invariant(array_abs_bound(r, start, MLDSA_N, (MLDSA_N >> layer) * MLDSA_Q))+ decreases(MLDSA_N - start))+ {+ unsigned j;+ int32_t zeta = -mld_zetas[k--];++ /* The bound `(MLDSA_N >> (layer - 1)) * MLDSA_Q` is loose enough to+ * cover both the input bound `(MLDSA_N >> layer) * MLDSA_Q`+ * (for layers >= 1) and the fqmul output bound `MLD_FQMUL_BOUND`+ * (which is < 2 * MLDSA_Q <= (MLDSA_N >> (layer - 1)) * MLDSA_Q). */+ for (j = start; j < start + len; j++)+ __loop__(+ invariant(start <= j && j <= start + len)+ invariant(array_abs_bound(r, 0, start, (MLDSA_N >> (layer - 1)) * MLDSA_Q))+ invariant(array_abs_bound(r, start, j, (MLDSA_N >> (layer - 1)) * MLDSA_Q))+ invariant(array_abs_bound(r, j, start + len, (MLDSA_N >> layer) * MLDSA_Q))+ invariant(array_abs_bound(r, start + len, j + len, (MLDSA_N >> (layer - 1)) * MLDSA_Q))+ invariant(array_abs_bound(r, j + len, MLDSA_N, (MLDSA_N >> layer) * MLDSA_Q))+ decreases(start + len - j))+ {+ int32_t t = r[j];+ r[j] = t + r[j + len];+ r[j + len] = t - r[j + len];+ r[j + len] = mld_fqmul(r[j + len], zeta);+ }+ }+}++MLD_STATIC_TESTABLE void mld_poly_invntt_tomont_c(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_abs_bound(a->coeffs, 0, MLDSA_N, MLD_INTT_BOUND))+)+{+ unsigned int layer, j;+ int32_t *r;++ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);++ r = a->coeffs;+ for (layer = 8; layer >= 1; layer--)+ __loop__(+ invariant(layer <= 8)+ /* Absolute bounds increase from 1Q before layer 8 */+ /* up to 256Q after layer 1 */+ invariant(array_abs_bound(r, 0, MLDSA_N, (MLDSA_N >> layer) * MLDSA_Q))+ decreases(layer))+ {+ mld_invntt_layer(r, layer);+ }++ /* Coefficient bounds are now at 256Q. We now scale by mont / 256,+ * i.e., compute the Montgomery multiplication by mont^2 / 256.+ * mont corrects the mont^-1 factor introduced in the basemul.+ * 1/256 performs that scaling of the inverse NTT.+ * The reduced value is bounded by MLD_INTT_BOUND in absolute+ * value.*/+ for (j = 0; j < MLDSA_N; ++j)+ __loop__(+ invariant(j <= MLDSA_N)+ invariant(array_abs_bound(r, 0, j, MLD_INTT_BOUND))+ invariant(array_abs_bound(r, j, MLDSA_N, MLDSA_N * MLDSA_Q))+ decreases(MLDSA_N - j)+ )+ {+ r[j] = mld_fqscale(r[j]);+ }++ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_INTT_BOUND);+}+++MLD_INTERNAL_API+void mld_poly_invntt_tomont(mld_poly *a)+{+#if defined(MLD_USE_NATIVE_INTT)+ int ret;+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);+ ret = mld_intt_native(a->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_INTT_BOUND);+ return;+ }+#endif /* MLD_USE_NATIVE_INTT */+ mld_poly_invntt_tomont_c(a);+}++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+MLD_STATIC_TESTABLE void mld_poly_pointwise_montgomery_c(mld_poly *a,+ const mld_poly *b)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(memory_no_alias(b, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ requires(array_abs_bound(b->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+)+{+ unsigned int i;+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_NTT_BOUND);+ mld_assert_abs_bound(b->coeffs, MLDSA_N, MLD_NTT_BOUND);++ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ invariant(i <= MLDSA_N)+ invariant(array_abs_bound(a->coeffs, 0, i, MLDSA_Q))+ invariant(array_abs_bound(a->coeffs, i, MLDSA_N, MLD_NTT_BOUND))+ decreases(MLDSA_N - i)+ )+ {+ a->coeffs[i] = mld_montgomery_reduce((int64_t)a->coeffs[i] * b->coeffs[i]);+ }+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);+}++MLD_INTERNAL_API+void mld_poly_pointwise_montgomery(mld_poly *a, const mld_poly *b)+{+#if defined(MLD_USE_NATIVE_POINTWISE_MONTGOMERY)+ int ret;+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_NTT_BOUND);+ mld_assert_abs_bound(b->coeffs, MLDSA_N, MLD_NTT_BOUND);+ ret = mld_poly_pointwise_montgomery_native(a->coeffs, b->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);+ return;+ }+#endif /* MLD_USE_NATIVE_POINTWISE_MONTGOMERY */+ mld_poly_pointwise_montgomery_c(a, b);+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_INTERNAL_API+void mld_poly_power2round(mld_poly *a1, mld_poly *a0, const mld_poly *a)+{+ unsigned int i;+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);++ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ assigns(i, memory_slice(a0, sizeof(mld_poly)), memory_slice(a1, sizeof(mld_poly)))+ invariant(i <= MLDSA_N)+ invariant(forall(k0, i, MLDSA_N, a->coeffs[k0] == loop_entry(*a).coeffs[k0]))+ invariant(array_bound(a0->coeffs, 0, i, -(MLD_2_POW_D/2)+1, (MLD_2_POW_D/2)+1))+ invariant(array_bound(a1->coeffs, 0, i, 0, ((MLDSA_Q - 1) / MLD_2_POW_D) + 1))+ decreases(MLDSA_N - i)+ )+ {+ mld_power2round(&a0->coeffs[i], &a1->coeffs[i], a->coeffs[i]);+ }++ mld_assert_bound(a0->coeffs, MLDSA_N, -(MLD_2_POW_D / 2) + 1,+ (MLD_2_POW_D / 2) + 1);+ mld_assert_bound(a1->coeffs, MLDSA_N, 0, ((MLDSA_Q - 1) / MLD_2_POW_D) + 1);+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#ifndef MLD_POLY_UNIFORM_NBLOCKS+#define MLD_POLY_UNIFORM_NBLOCKS \+ ((768 + MLD_STREAM128_BLOCKBYTES - 1) / MLD_STREAM128_BLOCKBYTES)+#endif+/* Reference: `mld_rej_uniform()` in the reference implementation @[REF].+ * - Our signature differs from the reference implementation+ * in that it adds the offset and always expects the base of the+ * target buffer. This avoids shifting the buffer base in the+ * caller, which appears tricky to reason about. */+MLD_STATIC_TESTABLE unsigned int mld_rej_uniform_c(int32_t *a,+ unsigned int target,+ unsigned int offset,+ const uint8_t *buf,+ unsigned int buflen)+__contract__(+ requires(offset <= target && target <= MLDSA_N)+ requires(buflen <= (MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES) && buflen % 3 == 0)+ requires(memory_no_alias(a, sizeof(int32_t) * target))+ requires(memory_no_alias(buf, buflen))+ requires(array_bound(a, 0, offset, 0, MLDSA_Q))+ assigns(memory_slice(a, sizeof(int32_t) * target))+ ensures(offset <= return_value && return_value <= target)+ ensures(array_bound(a, 0, return_value, 0, MLDSA_Q))+)+{+ unsigned int ctr, pos;+ uint32_t t;+ mld_assert_bound(a, offset, 0, MLDSA_Q);++ ctr = offset;+ pos = 0;+ /* pos + 3 cannot overflow due to the assumption+ buflen <= (MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES) */+ while (ctr < target && pos + 3 <= buflen)+ __loop__(+ invariant(offset <= ctr && ctr <= target && pos <= buflen)+ invariant(array_bound(a, 0, ctr, 0, MLDSA_Q))+ decreases(buflen - pos))+ {+ t = buf[pos++];+ t |= (uint32_t)buf[pos++] << 8;+ t |= (uint32_t)buf[pos++] << 16;+ t &= 0x7FFFFF;++ if (t < MLDSA_Q)+ {+ a[ctr++] = (int32_t)t;+ }+ }++ mld_assert_bound(a, ctr, 0, MLDSA_Q);++ return ctr;+}+/**+ * Sample uniformly random coefficients in [0, MLDSA_Q-1] by performing+ * rejection sampling on an array of random bytes.+ *+ * @param[out] a Pointer to output array (allocated).+ * @param target Requested number of coefficients to sample.+ * @param offset Number of coefficients already sampled.+ * @param[in] buf Array of random bytes to sample from.+ * @param buflen Length of array of random bytes (must be multiple of 3).+ *+ * @return Number of sampled coefficients. Can be smaller than len if not+ * enough random bytes were given.+ */++/* Reference: `mld_rej_uniform()` in the reference implementation @[REF].+ * - Our signature differs from the reference implementation+ * in that it adds the offset and always expects the base of the+ * target buffer. This avoids shifting the buffer base in the+ * caller, which appears tricky to reason about. */+static unsigned int mld_rej_uniform(int32_t *a, unsigned int target,+ unsigned int offset, const uint8_t *buf,+ unsigned int buflen)+__contract__(+ requires(offset <= target && target <= MLDSA_N)+ requires(buflen <= (MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES) && buflen % 3 == 0)+ requires(memory_no_alias(a, sizeof(int32_t) * target))+ requires(memory_no_alias(buf, buflen))+ requires(array_bound(a, 0, offset, 0, MLDSA_Q))+ assigns(memory_slice(a, sizeof(int32_t) * target))+ ensures(offset <= return_value && return_value <= target)+ ensures(array_bound(a, 0, return_value, 0, MLDSA_Q))+)+{+#if defined(MLD_USE_NATIVE_REJ_UNIFORM)+ int ret;+ mld_assert_bound(a, offset, 0, MLDSA_Q);+ if (offset == 0)+ {+ ret = mld_rej_uniform_native(a, target, buf, buflen);+ if (ret != MLD_NATIVE_FUNC_FALLBACK)+ {+ unsigned res = (unsigned)ret;+ mld_assert_bound(a, res, 0, MLDSA_Q);+ return res;+ }+ }+#endif /* MLD_USE_NATIVE_REJ_UNIFORM */++ return mld_rej_uniform_c(a, target, offset, buf, buflen);+}++/* Reference: poly_uniform() in the reference implementation @[REF].+ * - Simplified from reference by removing buffer tail handling+ * since buflen % 3 = 0 always holds true (MLD_STREAM128_BLOCKBYTES+ * = 168).+ * - Modified rej_uniform interface to track offset directly.+ * - Pass nonce packed in the extended seed array instead of a third+ * argument.+ * */+MLD_INTERNAL_API+void mld_poly_uniform(mld_poly *a, const uint8_t seed[MLDSA_SEEDBYTES + 2])+{+ unsigned int ctr;+ unsigned int buflen = MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES;+ MLD_ALIGN uint8_t buf[MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES];+ mld_xof128_ctx state;++ mld_xof128_init(&state);+ mld_xof128_absorb_once(&state, seed, MLDSA_SEEDBYTES + 2);+ mld_xof128_squeezeblocks(buf, MLD_POLY_UNIFORM_NBLOCKS, &state);++ ctr = mld_rej_uniform(a->coeffs, MLDSA_N, 0, buf, buflen);+ buflen = MLD_STREAM128_BLOCKBYTES;+ while (ctr < MLDSA_N)+ __loop__(+ assigns(ctr, state, memory_slice(a, sizeof(mld_poly)), object_whole(buf))+ invariant(ctr <= MLDSA_N)+ invariant(array_bound(a->coeffs, 0, ctr, 0, MLDSA_Q))+ invariant(state.pos <= SHAKE128_RATE)+ )+ {+ mld_xof128_squeezeblocks(buf, 1, &state);+ ctr = mld_rej_uniform(a->coeffs, MLDSA_N, ctr, buf, buflen);+ }+ mld_xof128_release(&state);+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(buf, sizeof(buf));+}++#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+MLD_INTERNAL_API+void mld_poly_uniform_4x(mld_poly *vec0, mld_poly *vec1, mld_poly *vec2,+ mld_poly *vec3,+ uint8_t seed[4][MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)])+{+ /* Temporary buffers for XOF output before rejection sampling */+ MLD_ALIGN uint8_t+ buf[4][MLD_ALIGN_UP(MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES)];++ /* Tracks the number of coefficients we have already sampled */+ unsigned ctr[4];+ mld_xof128_x4_ctx state;+ unsigned buflen;++ mld_xof128_x4_init(&state);+ mld_xof128_x4_absorb(&state, seed, MLDSA_SEEDBYTES + 2);++ /*+ * Initially, squeeze heuristic number of MLD_POLY_UNIFORM_NBLOCKS.+ * This should generate the matrix entries with high probability.+ */++ mld_xof128_x4_squeezeblocks(buf, MLD_POLY_UNIFORM_NBLOCKS, &state);+ buflen = MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES;+ ctr[0] = mld_rej_uniform(vec0->coeffs, MLDSA_N, 0, buf[0], buflen);+ ctr[1] = mld_rej_uniform(vec1->coeffs, MLDSA_N, 0, buf[1], buflen);+ ctr[2] = mld_rej_uniform(vec2->coeffs, MLDSA_N, 0, buf[2], buflen);+ ctr[3] = mld_rej_uniform(vec3->coeffs, MLDSA_N, 0, buf[3], buflen);++ /*+ * So long as not all matrix entries have been generated, squeeze+ * one more block a time until we're done.+ */+ buflen = MLD_STREAM128_BLOCKBYTES;+ while (ctr[0] < MLDSA_N || ctr[1] < MLDSA_N || ctr[2] < MLDSA_N ||+ ctr[3] < MLDSA_N)+ __loop__(+ assigns(ctr, state, object_whole(buf),+ memory_slice(vec0, sizeof(mld_poly)), memory_slice(vec1, sizeof(mld_poly)),+ memory_slice(vec2, sizeof(mld_poly)), memory_slice(vec3, sizeof(mld_poly)))+ invariant(ctr[0] <= MLDSA_N && ctr[1] <= MLDSA_N)+ invariant(ctr[2] <= MLDSA_N && ctr[3] <= MLDSA_N)+ invariant(array_bound(vec0->coeffs, 0, ctr[0], 0, MLDSA_Q))+ invariant(array_bound(vec1->coeffs, 0, ctr[1], 0, MLDSA_Q))+ invariant(array_bound(vec2->coeffs, 0, ctr[2], 0, MLDSA_Q))+ invariant(array_bound(vec3->coeffs, 0, ctr[3], 0, MLDSA_Q)))+ {+ mld_xof128_x4_squeezeblocks(buf, 1, &state);+ ctr[0] = mld_rej_uniform(vec0->coeffs, MLDSA_N, ctr[0], buf[0], buflen);+ ctr[1] = mld_rej_uniform(vec1->coeffs, MLDSA_N, ctr[1], buf[1], buflen);+ ctr[2] = mld_rej_uniform(vec2->coeffs, MLDSA_N, ctr[2], buf[2], buflen);+ ctr[3] = mld_rej_uniform(vec3->coeffs, MLDSA_N, ctr[3], buf[3], buflen);+ }+ mld_xof128_x4_release(&state);++ mld_assert_bound(vec0->coeffs, MLDSA_N, 0, MLDSA_Q);+ mld_assert_bound(vec1->coeffs, MLDSA_N, 0, MLDSA_Q);+ mld_assert_bound(vec2->coeffs, MLDSA_N, 0, MLDSA_Q);+ mld_assert_bound(vec3->coeffs, MLDSA_N, 0, MLDSA_Q);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(buf, sizeof(buf));+}++#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY && (!MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST) */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_INTERNAL_API+void mld_polyt1_pack(uint8_t r[MLDSA_POLYT1_PACKEDBYTES], const mld_poly *a)+{+ unsigned int i;+ mld_assert_bound(a->coeffs, MLDSA_N, 0, 1 << 10);++ for (i = 0; i < MLDSA_N / 4; ++i)+ __loop__(+ invariant(i <= MLDSA_N/4)+ decreases(MLDSA_N / 4 - i))+ {+ r[5 * i + 0] = (uint8_t)((a->coeffs[4 * i + 0] >> 0) & 0xFF);+ r[5 * i + 1] =+ (uint8_t)(((a->coeffs[4 * i + 0] >> 8) | (a->coeffs[4 * i + 1] << 2)) &+ 0xFF);+ r[5 * i + 2] =+ (uint8_t)(((a->coeffs[4 * i + 1] >> 6) | (a->coeffs[4 * i + 2] << 4)) &+ 0xFF);+ r[5 * i + 3] =+ (uint8_t)(((a->coeffs[4 * i + 2] >> 4) | (a->coeffs[4 * i + 3] << 6)) &+ 0xFF);+ r[5 * i + 4] = (uint8_t)((a->coeffs[4 * i + 3] >> 2) & 0xFF);+ }+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_INTERNAL_API+void mld_polyt1_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYT1_PACKEDBYTES])+{+ unsigned int i;++ for (i = 0; i < MLDSA_N / 4; ++i)+ __loop__(+ invariant(i <= MLDSA_N/4)+ invariant(array_bound(r->coeffs, 0, i*4, 0, 1 << 10))+ decreases(MLDSA_N / 4 - i))+ {+ r->coeffs[4 * i + 0] =+ ((a[5 * i + 0] >> 0) | ((int32_t)a[5 * i + 1] << 8)) & 0x3FF;+ r->coeffs[4 * i + 1] =+ ((a[5 * i + 1] >> 2) | ((int32_t)a[5 * i + 2] << 6)) & 0x3FF;+ r->coeffs[4 * i + 2] =+ ((a[5 * i + 2] >> 4) | ((int32_t)a[5 * i + 3] << 4)) & 0x3FF;+ r->coeffs[4 * i + 3] =+ ((a[5 * i + 3] >> 6) | ((int32_t)a[5 * i + 4] << 2)) & 0x3FF;+ }++ mld_assert_bound(r->coeffs, MLDSA_N, 0, 1 << 10);+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_INTERNAL_API+void mld_polyt0_pack(uint8_t r[MLDSA_POLYT0_PACKEDBYTES], const mld_poly *a)+{+ unsigned int i;+ uint32_t t[8];++ mld_assert_bound(a->coeffs, MLDSA_N, -(1 << (MLDSA_D - 1)) + 1,+ (1 << (MLDSA_D - 1)) + 1);++ for (i = 0; i < MLDSA_N / 8; ++i)+ __loop__(+ invariant(i <= MLDSA_N/8)+ decreases(MLDSA_N / 8 - i))+ {+ /* Safety: a->coeffs[i] <= (1 << (MLDSA_D - 1) as they are output of+ * power2round, hence, these casts are safe. */+ t[0] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 0]);+ t[1] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 1]);+ t[2] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 2]);+ t[3] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 3]);+ t[4] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 4]);+ t[5] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 5]);+ t[6] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 6]);+ t[7] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 7]);++ r[13 * i + 0] = (uint8_t)((t[0]) & 0xFF);+ r[13 * i + 1] = (uint8_t)((t[0] >> 8) & 0xFF);+ r[13 * i + 1] |= (uint8_t)((t[1] << 5) & 0xFF);+ r[13 * i + 2] = (uint8_t)((t[1] >> 3) & 0xFF);+ r[13 * i + 3] = (uint8_t)((t[1] >> 11) & 0xFF);+ r[13 * i + 3] |= (uint8_t)((t[2] << 2) & 0xFF);+ r[13 * i + 4] = (uint8_t)((t[2] >> 6) & 0xFF);+ r[13 * i + 4] |= (uint8_t)((t[3] << 7) & 0xFF);+ r[13 * i + 5] = (uint8_t)((t[3] >> 1) & 0xFF);+ r[13 * i + 6] = (uint8_t)((t[3] >> 9) & 0xFF);+ r[13 * i + 6] |= (uint8_t)((t[4] << 4) & 0xFF);+ r[13 * i + 7] = (uint8_t)((t[4] >> 4) & 0xFF);+ r[13 * i + 8] = (uint8_t)((t[4] >> 12) & 0xFF);+ r[13 * i + 8] |= (uint8_t)((t[5] << 1) & 0xFF);+ r[13 * i + 9] = (uint8_t)((t[5] >> 7) & 0xFF);+ r[13 * i + 9] |= (uint8_t)((t[6] << 6) & 0xFF);+ r[13 * i + 10] = (uint8_t)((t[6] >> 2) & 0xFF);+ r[13 * i + 11] = (uint8_t)((t[6] >> 10) & 0xFF);+ r[13 * i + 11] |= (uint8_t)((t[7] << 3) & 0xFF);+ r[13 * i + 12] = (uint8_t)((t[7] >> 5) & 0xFF);+ }+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+void mld_polyt0_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYT0_PACKEDBYTES])+{+ unsigned int i;++ for (i = 0; i < MLDSA_N / 8; ++i)+ __loop__(+ invariant(i <= MLDSA_N/8)+ invariant(array_bound(r->coeffs, 0, i*8, -(1<<(MLDSA_D-1)) + 1, (1<<(MLDSA_D-1)) + 1))+ decreases(MLDSA_N / 8 - i))+ {+ r->coeffs[8 * i + 0] = a[13 * i + 0];+ r->coeffs[8 * i + 0] |= (int32_t)a[13 * i + 1] << 8;+ r->coeffs[8 * i + 0] &= 0x1FFF;++ r->coeffs[8 * i + 1] = a[13 * i + 1] >> 5;+ r->coeffs[8 * i + 1] |= (int32_t)a[13 * i + 2] << 3;+ r->coeffs[8 * i + 1] |= (int32_t)a[13 * i + 3] << 11;+ r->coeffs[8 * i + 1] &= 0x1FFF;++ r->coeffs[8 * i + 2] = a[13 * i + 3] >> 2;+ r->coeffs[8 * i + 2] |= (int32_t)a[13 * i + 4] << 6;+ r->coeffs[8 * i + 2] &= 0x1FFF;++ r->coeffs[8 * i + 3] = a[13 * i + 4] >> 7;+ r->coeffs[8 * i + 3] |= (int32_t)a[13 * i + 5] << 1;+ r->coeffs[8 * i + 3] |= (int32_t)a[13 * i + 6] << 9;+ r->coeffs[8 * i + 3] &= 0x1FFF;++ r->coeffs[8 * i + 4] = a[13 * i + 6] >> 4;+ r->coeffs[8 * i + 4] |= (int32_t)a[13 * i + 7] << 4;+ r->coeffs[8 * i + 4] |= (int32_t)a[13 * i + 8] << 12;+ r->coeffs[8 * i + 4] &= 0x1FFF;++ r->coeffs[8 * i + 5] = a[13 * i + 8] >> 1;+ r->coeffs[8 * i + 5] |= (int32_t)a[13 * i + 9] << 7;+ r->coeffs[8 * i + 5] &= 0x1FFF;++ r->coeffs[8 * i + 6] = a[13 * i + 9] >> 6;+ r->coeffs[8 * i + 6] |= (int32_t)a[13 * i + 10] << 2;+ r->coeffs[8 * i + 6] |= (int32_t)a[13 * i + 11] << 10;+ r->coeffs[8 * i + 6] &= 0x1FFF;++ r->coeffs[8 * i + 7] = a[13 * i + 11] >> 3;+ r->coeffs[8 * i + 7] |= (int32_t)a[13 * i + 12] << 5;+ r->coeffs[8 * i + 7] &= 0x1FFF;++ r->coeffs[8 * i + 0] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 0];+ r->coeffs[8 * i + 1] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 1];+ r->coeffs[8 * i + 2] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 2];+ r->coeffs[8 * i + 3] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 3];+ r->coeffs[8 * i + 4] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 4];+ r->coeffs[8 * i + 5] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 5];+ r->coeffs[8 * i + 6] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 6];+ r->coeffs[8 * i + 7] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 7];+ }++ mld_assert_bound(r->coeffs, MLDSA_N, -(1 << (MLDSA_D - 1)) + 1,+ (1 << (MLDSA_D - 1)) + 1);+}+#endif /* !MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST */++MLD_STATIC_TESTABLE uint32_t mld_poly_chknorm_c(const mld_poly *a, int32_t B)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(0 <= B && B <= MLDSA_Q - MLD_REDUCE32_RANGE_MAX)+ requires(array_bound(a->coeffs, 0, MLDSA_N, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX))+ ensures(return_value == 0 || return_value == 0xFFFFFFFF)+ ensures((return_value == 0) == array_abs_bound(a->coeffs, 0, MLDSA_N, B))+)+{+ unsigned int i;+ uint32_t t = 0;+ mld_assert_bound(a->coeffs, MLDSA_N, -MLD_REDUCE32_RANGE_MAX,+ MLD_REDUCE32_RANGE_MAX);+ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ invariant(i <= MLDSA_N)+ invariant(t == 0 || t == 0xFFFFFFFF)+ invariant((t == 0) == array_abs_bound(a->coeffs, 0, i, B))+ decreases(MLDSA_N - i)+ )+ {+ /*+ * Since we know that -MLD_REDUCE32_RANGE_MAX <= a < MLD_REDUCE32_RANGE_MAX,+ * and B <= MLDSA_Q - MLD_REDUCE32_RANGE_MAX, to check if+ * -B < (a mod± MLDSA_Q) < B, it suffices to check if -B < a < B.+ *+ * We prove this to be true using the following CBMC assertions.+ * a ==> b expressed as !a || b to also allow run-time assertion.+ */+ mld_assert(a->coeffs[i] < B || a->coeffs[i] - MLDSA_Q <= -B);+ mld_assert(a->coeffs[i] > -B || a->coeffs[i] + MLDSA_Q >= B);++ /* Reference: Leaks which coefficient violates the bound via a conditional.+ * We are more conservative to reduce the number of declassifications in+ * constant-time testing.+ */++ /* if (abs(a[i]) >= B) */+ t |= mld_ct_cmask_neg_i32(B - 1 - mld_ct_abs_i32(a->coeffs[i]));+ }++ return t;+}++/* Reference: explicitly checks the bound B to be <= (MLDSA_Q - 1) / 8).+ * This is unnecessary as it's always a compile-time constant.+ * We instead model it as a precondition.+ * Checking the bound is performed using a conditional arguing+ * that it is okay to leak which coefficient violates the bound (while the+ * coefficient itself must remain secret).+ * We instead perform everything in constant-time.+ * Also it is sufficient to check that it is smaller than+ * MLDSA_Q - MLD_REDUCE32_RANGE_MAX > (MLDSA_Q - 1) / 8).+ */+MLD_INTERNAL_API+uint32_t mld_poly_chknorm(const mld_poly *a, int32_t B)+{+#if defined(MLD_USE_NATIVE_POLY_CHKNORM)+ int ret;+ int success;+ mld_assert_bound(a->coeffs, MLDSA_N, -MLD_REDUCE32_RANGE_MAX,+ MLD_REDUCE32_RANGE_MAX);+ /* The native backend returns 0 if all coefficients are within the bound,+ * 1 if at least one coefficient exceeds the bound, and+ * -1 (MLD_NATIVE_FUNC_FALLBACK) if the platform does not have the+ * required capabilities to run the native function.+ */+ ret = mld_poly_chknorm_native(a->coeffs, B);++ success = (ret != MLD_NATIVE_FUNC_FALLBACK);+ /* Constant-time: It would be fine to leak the return value of chknorm+ * entirely (as it is fine to leak if any coefficient exceeded the bound or+ * not). However, it is cleaner to perform declassification in sign.c.+ * Hence, here we only declassify if the native function returned+ * MLD_NATIVE_FUNC_FALLBACK or not (which solely depends on system+ * capabilities).+ */+ MLD_CT_TESTING_DECLASSIFY(&success, sizeof(int));+ if (success)+ {+ /* Convert 0 / 1 to 0 / 0xFFFFFFFF here */+ return mld_ct_cmask_nonzero_u32((uint32_t)ret);+ }+#endif /* MLD_USE_NATIVE_POLY_CHKNORM */+ return mld_poly_chknorm_c(a, B);+}++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ MLD_CONFIG_PARAMETER_SET == 44+MLD_INTERNAL_API+void mld_polyw1_pack_88(uint8_t r[MLDSA_POLYW1_PACKEDBYTES_88],+ const mld_poly *a)+{+ unsigned int i;++ mld_assert_bound(a->coeffs, MLDSA_N, 0,+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2_88));++ for (i = 0; i < MLDSA_N / 4; ++i)+ __loop__(+ invariant(i <= MLDSA_N/4)+ decreases(MLDSA_N / 4 - i))+ {+ r[3 * i + 0] = (uint8_t)((a->coeffs[4 * i + 0]) & 0xFF);+ r[3 * i + 0] |= (uint8_t)((a->coeffs[4 * i + 1] << 6) & 0xFF);+ r[3 * i + 1] = (uint8_t)((a->coeffs[4 * i + 1] >> 2) & 0xFF);+ r[3 * i + 1] |= (uint8_t)((a->coeffs[4 * i + 2] << 4) & 0xFF);+ r[3 * i + 2] = (uint8_t)((a->coeffs[4 * i + 2] >> 4) & 0xFF);+ r[3 * i + 2] |= (uint8_t)((a->coeffs[4 * i + 3] << 2) & 0xFF);+ }+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_INTERNAL_API+void mld_polyw1_pack_32(uint8_t r[MLDSA_POLYW1_PACKEDBYTES_32],+ const mld_poly *a)+{+ unsigned int i;++ mld_assert_bound(a->coeffs, MLDSA_N, 0,+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2_32));++ for (i = 0; i < MLDSA_N / 2; ++i)+ __loop__(+ invariant(i <= MLDSA_N/2)+ decreases(MLDSA_N / 2 - i))+ {+ r[i] =+ (uint8_t)((a->coeffs[2 * i + 0] | (a->coeffs[2 * i + 1] << 4)) & 0xFF);+ }+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#else /* !MLD_CONFIG_MULTILEVEL_NO_SHARED */+MLD_EMPTY_CU(mld_poly)+#endif /* MLD_CONFIG_MULTILEVEL_NO_SHARED */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef MLD_POLY_UNIFORM_NBLOCKS
+ cbits/mldsa/src/poly.h view
@@ -0,0 +1,464 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_POLY_H+#define MLD_POLY_H++#include "cbmc.h"+#include "common.h"+#include "reduce.h"+#include "rounding.h"++/* Absolute exclusive upper bound for the output of fqmul */+#define MLD_FQMUL_BOUND ((5 * MLDSA_Q + 3) / 4)+/* Absolute exclusive upper bound for the output of the forward NTT */+#define MLD_NTT_BOUND (9 * MLD_FQMUL_BOUND)+/* Absolute exclusive upper bound for the output of the inverse NTT*/+#define MLD_INTT_BOUND MLDSA_Q++/**+ * Element of R_q = Z_q[X]/(X^n + 1). Represents polynomial+ * coeffs[0] + X*coeffs[1] + X^2*coeffs[2] + ... + X^{n-1}*coeffs[n-1].+ */+typedef struct+{+ int32_t coeffs[MLDSA_N]; /**< Polynomial coefficients. */+} MLD_ALIGN mld_poly;++#define mld_poly_reduce MLD_NAMESPACE(poly_reduce)+/**+ * In-place reduction of all coefficients of polynomial to representative in+ * [-MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX].+ *+ * @param[in,out] a Pointer to input/output polynomial.+ */+MLD_INTERNAL_API+void mld_poly_reduce(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, INT32_MIN, MLD_REDUCE32_DOMAIN_MAX))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX))+);++#define mld_poly_caddq MLD_NAMESPACE(poly_caddq)+/**+ * For all coefficients of in/out polynomial add MLDSA_Q if coefficient is+ * negative.+ *+ * @param[in,out] a Pointer to input/output polynomial.+ */+MLD_INTERNAL_API+void mld_poly_caddq(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+);++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_SIGN_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+#define mld_poly_add MLD_NAMESPACE(poly_add)+/**+ * Add polynomials. No modular reduction is performed.+ *+ * @spec{Implements @[FIPS204, Algorithm 44, AddNTT] (coefficientwise+ * polynomial addition; also used for addition in the normal domain).}+ *+ * @param[in,out] r Pointer to input-output polynomial to be added to.+ * @param[in] b Pointer to input polynomial that should be added to r.+ * Must be disjoint from r.+ */++/*+ * NOTE: The reference implementation uses a 3-argument poly_add.+ * We specialize to the accumulator form to avoid reasoning about aliasing.+ */+MLD_INTERNAL_API+void mld_poly_add(mld_poly *r, const mld_poly *b)+__contract__(+ requires(memory_no_alias(b, sizeof(mld_poly)))+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(forall(k0, 0, MLDSA_N, (int64_t) r->coeffs[k0] + b->coeffs[k0] < MLD_REDUCE32_DOMAIN_MAX))+ requires(forall(k1, 0, MLDSA_N, (int64_t) r->coeffs[k1] + b->coeffs[k1] >= INT32_MIN))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(forall(k2, 0, MLDSA_N, r->coeffs[k2] == old(*r).coeffs[k2] + b->coeffs[k2]))+ ensures(forall(k3, 0, MLDSA_N, r->coeffs[k3] < MLD_REDUCE32_DOMAIN_MAX))+ ensures(forall(k4, 0, MLDSA_N, r->coeffs[k4] >= INT32_MIN))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_poly_sub MLD_NAMESPACE(poly_sub)+/**+ * Subtract polynomials. No modular reduction is performed.+ *+ * @param[in,out] r Pointer to input-output polynomial.+ * @param[in] b Pointer to input polynomial that should be subtracted from+ * r. Must be disjoint from r.+ */+/*+ * NOTE: The reference implementation uses a 3-argument poly_sub.+ * We specialize to the accumulator form to avoid reasoning about aliasing.+ */+MLD_INTERNAL_API+void mld_poly_sub(mld_poly *r, const mld_poly *b)+__contract__(+ requires(memory_no_alias(b, sizeof(mld_poly)))+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(array_abs_bound(r->coeffs, 0, MLDSA_N, MLDSA_Q))+ requires(array_abs_bound(b->coeffs, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(array_bound(r->coeffs, 0, MLDSA_N, INT32_MIN, MLD_REDUCE32_DOMAIN_MAX))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_poly_shiftl MLD_NAMESPACE(poly_shiftl)+/**+ * Multiply polynomial by 2^MLDSA_D without modular reduction. Assumes input+ * coefficients to be less than 2^{31-MLDSA_D} in absolute value.+ *+ * @param[in,out] a Pointer to input/output polynomial.+ */+MLD_INTERNAL_API+void mld_poly_shiftl(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, 1 << 10))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#define mld_poly_ntt MLD_NAMESPACE(poly_ntt)+/**+ * In-place forward NTT. Output coefficients are bounded by MLD_NTT_BOUND in+ * absolute value.+ *+ * @spec{Implements @[FIPS204, Algorithm 41, NTT].}+ *+ * @param[in,out] a Pointer to input/output polynomial.+ */+MLD_INTERNAL_API+void mld_poly_ntt(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_abs_bound(a->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+);+++#define mld_poly_invntt_tomont MLD_NAMESPACE(poly_invntt_tomont)+/**+ * In-place inverse NTT.+ *+ * Input coefficients need to be less than MLDSA_Q in absolute value and+ * output coefficients are bounded by MLD_INTT_BOUND.+ *+ * @spec{Implements @[FIPS204, Algorithm 42, NTT^{-1}] up to scaling:+ * The input is scaled by 2^{-32} as a result of the Montgomery base+ * multiplication. The output is in normal domain. In other words, this+ * function implements `NTT^{-1} o mult(2^32)`.}+ *+ * @param[in,out] a Pointer to input/output polynomial.+ */+MLD_INTERNAL_API+void mld_poly_invntt_tomont(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_abs_bound(a->coeffs, 0, MLDSA_N, MLD_INTT_BOUND))+);++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+#define mld_poly_pointwise_montgomery MLD_NAMESPACE(poly_pointwise_montgomery)+/**+ * Pointwise multiplication of polynomials. Destructive in the first argument.+ *+ * @spec{Implements @[FIPS204, Algorithm 45, MultiplyNTT], up to scaling: The+ * input is in normal domain, the output is scaled by 2^{-32} as a result of+ * the use of Montgomery multiplication. In other words, this function+ * implements `mult(2^{-32}) o MultiplyNTT`.}+ *+ * @param[in,out] a Pointer to first input/output polynomial. On entry, holds+ * the first multiplicand; on exit, holds the product+ * a * b * 2^{-32}.+ * @param[in] b Pointer to second input polynomial.+ */+MLD_INTERNAL_API+void mld_poly_pointwise_montgomery(mld_poly *a, const mld_poly *b)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(memory_no_alias(b, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ requires(array_abs_bound(b->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_poly_power2round MLD_NAMESPACE(poly_power2round)+/**+ * For all coefficients c of the input polynomial, compute c0, c1 such that+ * c mod MLDSA_Q = c1*2^MLDSA_D + c0 with -2^{MLDSA_D-1} < c0 <= 2^{MLDSA_D-1}.+ * Assumes coefficients to be standard representatives.+ *+ * @param[out] a1 Pointer to output polynomial with coefficients c1.+ * @param[out] a0 Pointer to output polynomial with coefficients c0; may alias+ * the input polynomial a.+ * @param[in] a Pointer to input polynomial.+ */+MLD_INTERNAL_API+void mld_poly_power2round(mld_poly *a1, mld_poly *a0, const mld_poly *a)+__contract__(+ requires(memory_no_alias(a0, sizeof(mld_poly)))+ requires(memory_no_alias(a1, sizeof(mld_poly)))+ /* The implementation does not require a0 == a, but the single call site+ * aliases them and asserting equality simplifies the proof. */+ requires(a0 == a)+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(mld_poly)))+ assigns(memory_slice(a0, sizeof(mld_poly)))+ ensures(array_bound(a0->coeffs, 0, MLDSA_N, -(MLD_2_POW_D/2)+1, (MLD_2_POW_D/2)+1))+ ensures(array_bound(a1->coeffs, 0, MLDSA_N, 0, ((MLDSA_Q - 1) / MLD_2_POW_D) + 1))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#define mld_poly_uniform MLD_NAMESPACE(poly_uniform)+/**+ * Sample polynomial with uniformly random coefficients in [0, MLDSA_Q-1] by+ * performing rejection sampling on the output stream of SHAKE128(seed|nonce).+ *+ * @spec{Implements @[FIPS204, Algorithm 30, RejNTTPoly].}+ *+ * @param[out] a Pointer to output polynomial.+ * @param[in] seed Byte array with seed of length MLDSA_SEEDBYTES and the+ * packed 2-byte nonce.+ */+MLD_INTERNAL_API+void mld_poly_uniform(mld_poly *a, const uint8_t seed[MLDSA_SEEDBYTES + 2])+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(memory_no_alias(seed, MLDSA_SEEDBYTES + 2))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+);++#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+#define mld_poly_uniform_4x MLD_NAMESPACE(poly_uniform_4x)+/**+ * Generate four polynomials using rejection sampling on (pseudo-)uniformly+ * random bytes sampled from a seed.+ *+ * @spec{Implements @[FIPS204, Algorithm 30, RejNTTPoly] (four-way batched).}+ *+ * @param[out] vec0 Pointer to first polynomial to be sampled.+ * @param[out] vec1 Pointer to second polynomial to be sampled.+ * @param[out] vec2 Pointer to third polynomial to be sampled.+ * @param[out] vec3 Pointer to fourth polynomial to be sampled.+ * @param[in] seed Pointer to consecutive array of seed buffers of size+ * MLDSA_SEEDBYTES + 2 each, plus padding for alignment.+ */+MLD_INTERNAL_API+void mld_poly_uniform_4x(mld_poly *vec0, mld_poly *vec1, mld_poly *vec2,+ mld_poly *vec3,+ uint8_t seed[4][MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)])+__contract__(+ requires(memory_no_alias(vec0, sizeof(mld_poly)))+ requires(memory_no_alias(vec1, sizeof(mld_poly)))+ requires(memory_no_alias(vec2, sizeof(mld_poly)))+ requires(memory_no_alias(vec3, sizeof(mld_poly)))+ requires(memory_no_alias(seed, 4 * MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)))+ assigns(memory_slice(vec0, sizeof(mld_poly)))+ assigns(memory_slice(vec1, sizeof(mld_poly)))+ assigns(memory_slice(vec2, sizeof(mld_poly)))+ assigns(memory_slice(vec3, sizeof(mld_poly)))+ ensures(array_bound(vec0->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ ensures(array_bound(vec1->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ ensures(array_bound(vec2->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ ensures(array_bound(vec3->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+);+#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY && (!MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST) */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_polyt1_pack MLD_NAMESPACE(polyt1_pack)+/**+ * Bit-pack polynomial t1 with coefficients fitting in 10 bits. Input+ * coefficients are assumed to be standard representatives.+ *+ * @spec{Implements @[FIPS204, Algorithm 16, SimpleBitPack].}+ *+ * @param[out] r Pointer to output byte array with at least+ * MLDSA_POLYT1_PACKEDBYTES bytes.+ * @param[in] a Pointer to input polynomial.+ */+MLD_INTERNAL_API+void mld_polyt1_pack(uint8_t r[MLDSA_POLYT1_PACKEDBYTES], const mld_poly *a)+__contract__(+ requires(memory_no_alias(r, MLDSA_POLYT1_PACKEDBYTES))+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, 1 << 10))+ assigns(memory_slice(r, MLDSA_POLYT1_PACKEDBYTES))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_polyt1_unpack MLD_NAMESPACE(polyt1_unpack)+/**+ * Unpack polynomial t1 with 10-bit coefficients. Output coefficients are+ * standard representatives.+ *+ * @spec{Implements @[FIPS204, Algorithm 18, SimpleBitUnpack].}+ *+ * @param[out] r Pointer to output polynomial.+ * @param[in] a Byte array with bit-packed polynomial.+ */+MLD_INTERNAL_API+void mld_polyt1_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYT1_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(memory_no_alias(a, MLDSA_POLYT1_PACKEDBYTES))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(array_bound(r->coeffs, 0, MLDSA_N, 0, 1 << 10))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_polyt0_pack MLD_NAMESPACE(polyt0_pack)+/**+ * Bit-pack polynomial t0 with coefficients in ]-2^{MLDSA_D-1}, 2^{MLDSA_D-1}].+ *+ * @spec{Implements @[FIPS204, Algorithm 17, BitPack].}+ *+ * @param[out] r Pointer to output byte array with at least+ * MLDSA_POLYT0_PACKEDBYTES bytes.+ * @param[in] a Pointer to input polynomial.+ */+MLD_INTERNAL_API+void mld_polyt0_pack(uint8_t r[MLDSA_POLYT0_PACKEDBYTES], const mld_poly *a)+__contract__(+ requires(memory_no_alias(r, MLDSA_POLYT0_PACKEDBYTES))+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, -(1<<(MLDSA_D-1)) + 1, (1<<(MLDSA_D-1)) + 1))+ assigns(memory_slice(r, MLDSA_POLYT0_PACKEDBYTES))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)+#define mld_polyt0_unpack MLD_NAMESPACE(polyt0_unpack)+/**+ * Unpack polynomial t0 with coefficients in ]-2^{MLDSA_D-1}, 2^{MLDSA_D-1}].+ *+ * @spec{Implements @[FIPS204, Algorithm 19, BitUnpack].}+ *+ * @param[out] r Pointer to output polynomial.+ * @param[in] a Byte array with bit-packed polynomial.+ */+MLD_INTERNAL_API+void mld_polyt0_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYT0_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(memory_no_alias(a, MLDSA_POLYT0_PACKEDBYTES))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(array_bound(r->coeffs, 0, MLDSA_N, -(1<<(MLDSA_D-1)) + 1, (1<<(MLDSA_D-1)) + 1))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST */++#define mld_poly_chknorm MLD_NAMESPACE(poly_chknorm)+/**+ * Check infinity norm of polynomial against given bound. Assumes input+ * coefficients were reduced by mld_reduce32().+ *+ * @spec{@[FIPS204] defines the infinity norm via signed canonical reduction+ * (mod± MLDSA_Q) prior to applying the bounds check. However,+ * `-B < (a mod± MLDSA_Q) < B` is equivalent to `-B < a < B` under the+ * assumption that `B <= MLDSA_Q - MLD_REDUCE32_RANGE_MAX` (cf. the assertion in+ * the code). Hence, this contract and implementation are correct without+ * reduction.}+ *+ * @param[in] a Pointer to polynomial.+ * @param B Norm bound.+ *+ * @return 0 if norm is strictly smaller than+ * B <= (MLDSA_Q - MLD_REDUCE32_RANGE_MAX) and 0xFFFFFFFF otherwise.+ */+MLD_INTERNAL_API+MLD_MUST_CHECK_RETURN_VALUE+uint32_t mld_poly_chknorm(const mld_poly *a, int32_t B)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(0 <= B && B <= MLDSA_Q - MLD_REDUCE32_RANGE_MAX)+ requires(array_bound(a->coeffs, 0, MLDSA_N, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX))+ ensures(return_value == 0 || return_value == 0xFFFFFFFF)+ ensures((return_value == 0) == array_abs_bound(a->coeffs, 0, MLDSA_N, B))+);++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+#define mld_polyw1_pack_88 MLD_NAMESPACE(polyw1_pack_88)+/**+ * Bit-pack polynomial w1, using 6 bits per coefficient.+ * This is the variant for parameter sets with MLDSA_GAMMA2 = (MLDSA_Q-1)/88+ * (ML-DSA-44), for which w1 coefficients lie in [0, 43].+ *+ * @param[out] r Pointer to output byte array (MLDSA_POLYW1_PACKEDBYTES_88).+ * @param[in] a Pointer to input polynomial.+ */+MLD_INTERNAL_API+void mld_polyw1_pack_88(uint8_t r[MLDSA_POLYW1_PACKEDBYTES_88],+ const mld_poly *a)+__contract__(+ requires(memory_no_alias(r, MLDSA_POLYW1_PACKEDBYTES_88))+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2_88)))+ assigns(memory_slice(r, MLDSA_POLYW1_PACKEDBYTES_88))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+#define mld_polyw1_pack_32 MLD_NAMESPACE(polyw1_pack_32)+/**+ * Bit-pack polynomial w1, using 4 bits per coefficient.+ * This is the variant for parameter sets with MLDSA_GAMMA2 = (MLDSA_Q-1)/32+ * (ML-DSA-65 and ML-DSA-87), for which w1 coefficients lie in [0, 15].+ *+ * @param[out] r Pointer to output byte array (MLDSA_POLYW1_PACKEDBYTES_32).+ * @param[in] a Pointer to input polynomial.+ */+MLD_INTERNAL_API+void mld_polyw1_pack_32(uint8_t r[MLDSA_POLYW1_PACKEDBYTES_32],+ const mld_poly *a)+__contract__(+ requires(memory_no_alias(r, MLDSA_POLYW1_PACKEDBYTES_32))+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2_32)))+ assigns(memory_slice(r, MLDSA_POLYW1_PACKEDBYTES_32))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#endif /* !MLD_POLY_H */
+ cbits/mldsa/src/poly_kl.c view
@@ -0,0 +1,910 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ *+ * - [REF]+ * CRYSTALS-Dilithium reference implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/ref+ */++#include "poly_kl.h"++#include "ct.h"+#include "debug.h"+#include "rounding.h"+#include "symmetric.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mldsa-native (e.g. with varying parameter sets)+ * within a single compilation unit. */+#define mld_rej_eta MLD_ADD_PARAM_SET(mld_rej_eta)+#define mld_rej_eta_c MLD_ADD_PARAM_SET(mld_rej_eta_c)+#define mld_poly_decompose_c MLD_ADD_PARAM_SET(mld_poly_decompose_c)+#define mld_poly_use_hint_c MLD_ADD_PARAM_SET(mld_poly_use_hint_c)+#define mld_polyz_unpack_c MLD_ADD_PARAM_SET(mld_polyz_unpack_c)+/* End of parameter set namespacing */+++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_STATIC_TESTABLE+void mld_poly_decompose_c(mld_poly *a1, mld_poly *a0)+__contract__(+ requires(memory_no_alias(a1, sizeof(mld_poly)))+ requires(memory_no_alias(a0, sizeof(mld_poly)))+ requires(array_bound(a0->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(mld_poly)))+ assigns(memory_slice(a0, sizeof(mld_poly)))+ ensures(array_bound(a1->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ ensures(array_abs_bound(a0->coeffs, 0, MLDSA_N, MLDSA_GAMMA2+1))+)+{+ unsigned int i;+ mld_assert_bound(a0->coeffs, MLDSA_N, 0, MLDSA_Q);+ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ assigns(i, memory_slice(a0, sizeof(mld_poly)), memory_slice(a1, sizeof(mld_poly)))+ invariant(i <= MLDSA_N)+ invariant(array_bound(a0->coeffs, i, MLDSA_N, 0, MLDSA_Q))+ invariant(array_bound(a1->coeffs, 0, i, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ invariant(array_abs_bound(a0->coeffs, 0, i, MLDSA_GAMMA2+1))+ decreases(MLDSA_N - i)+ )+ {+ mld_decompose(&a0->coeffs[i], &a1->coeffs[i], a0->coeffs[i]);+ }++ mld_assert_abs_bound(a0->coeffs, MLDSA_N, MLDSA_GAMMA2 + 1);+ mld_assert_bound(a1->coeffs, MLDSA_N, 0, (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));+}++MLD_INTERNAL_API+void mld_poly_decompose(mld_poly *a1, mld_poly *a0)+{+#if defined(MLD_USE_NATIVE_POLY_DECOMPOSE_88) && MLD_CONFIG_PARAMETER_SET == 44+ int ret;+ mld_assert_bound(a0->coeffs, MLDSA_N, 0, MLDSA_Q);+ ret = mld_poly_decompose_88_native(a1->coeffs, a0->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(a0->coeffs, MLDSA_N, MLDSA_GAMMA2 + 1);+ mld_assert_bound(a1->coeffs, MLDSA_N, 0,+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));+ return;+ }+#elif defined(MLD_USE_NATIVE_POLY_DECOMPOSE_32) && \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+ int ret;+ mld_assert_bound(a0->coeffs, MLDSA_N, 0, MLDSA_Q);+ ret = mld_poly_decompose_32_native(a1->coeffs, a0->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(a0->coeffs, MLDSA_N, MLDSA_GAMMA2 + 1);+ mld_assert_bound(a1->coeffs, MLDSA_N, 0,+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));+ return;+ }+#endif /* !(MLD_USE_NATIVE_POLY_DECOMPOSE_88 && MLD_CONFIG_PARAMETER_SET == \+ 44) && MLD_USE_NATIVE_POLY_DECOMPOSE_32 && (MLD_CONFIG_PARAMETER_SET \+ == 65 || MLD_CONFIG_PARAMETER_SET == 87) */+ mld_poly_decompose_c(a1, a0);+}++#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_STATIC_TESTABLE void mld_poly_use_hint_c(mld_poly *a, const mld_poly *h)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(memory_no_alias(h, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h->coeffs, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+)+{+ unsigned int i;+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);+ mld_assert_bound(h->coeffs, MLDSA_N, 0, 2);++ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ invariant(i <= MLDSA_N)+ invariant(array_bound(a->coeffs, 0, i, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ invariant(array_bound(a->coeffs, i, MLDSA_N, 0, MLDSA_Q))+ decreases(MLDSA_N - i)+ )+ {+ a->coeffs[i] = mld_use_hint(a->coeffs[i], h->coeffs[i]);+ }+ mld_assert_bound(a->coeffs, MLDSA_N, 0, (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));+}++MLD_INTERNAL_API+void mld_poly_use_hint(mld_poly *a, const mld_poly *h)+{+#if defined(MLD_USE_NATIVE_POLY_USE_HINT_88) && MLD_CONFIG_PARAMETER_SET == 44+ int ret;+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);+ mld_assert_bound(h->coeffs, MLDSA_N, 0, 2);+ ret = mld_poly_use_hint_88_native(a->coeffs, h->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_bound(a->coeffs, MLDSA_N, 0, (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));+ return;+ }+#elif defined(MLD_USE_NATIVE_POLY_USE_HINT_32) && \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+ int ret;+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);+ mld_assert_bound(h->coeffs, MLDSA_N, 0, 2);+ ret = mld_poly_use_hint_32_native(a->coeffs, h->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_bound(a->coeffs, MLDSA_N, 0, (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));+ return;+ }+#endif /* !(MLD_USE_NATIVE_POLY_USE_HINT_88 && MLD_CONFIG_PARAMETER_SET == 44) \+ && MLD_USE_NATIVE_POLY_USE_HINT_32 && (MLD_CONFIG_PARAMETER_SET == \+ 65 || MLD_CONFIG_PARAMETER_SET == 87) */+ mld_poly_use_hint_c(a, h);+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+/**+ * Sample uniformly random coefficients in [-MLDSA_ETA, MLDSA_ETA] by+ * performing rejection sampling on an array of random bytes.+ *+ * @param[out] a Pointer to output array (allocated).+ * @param target Requested number of coefficients to sample.+ * @param offset Number of coefficients already sampled.+ * @param[in] buf Array of random bytes to sample from.+ * @param buflen Length of array of random bytes.+ *+ * @return Number of sampled coefficients. Can be smaller than target if not+ * enough random bytes were given.+ */++/* Reference: `mld_rej_eta()` in the reference implementation @[REF].+ * - Our signature differs from the reference implementation+ * in that it adds the offset and always expects the base of the+ * target buffer. This avoids shifting the buffer base in the+ * caller, which appears tricky to reason about. */+#if MLDSA_ETA == 2+/*+ * Sampling 256 coefficients mod 15 using rejection sampling from 4 bits.+ * Expected number of required bytes: (256 * (16/15))/2 = 136.5 bytes.+ * We sample 1 block (=136 bytes) of SHAKE256_RATE output initially.+ * Sampling 2 blocks initially results in slightly worse performance.+ */+#define MLD_POLY_UNIFORM_ETA_NBLOCKS 1+#elif MLDSA_ETA == 4+/*+ * Sampling 256 coefficients mod 9 using rejection sampling from 4 bits.+ * Expected number of required bytes: (256 * (16/9))/2 = 227.5 bytes.+ * We sample 2 blocks (=272 bytes) of SHAKE256_RATE output initially.+ */+#define MLD_POLY_UNIFORM_ETA_NBLOCKS 2+#else /* MLDSA_ETA == 4 */+#error "Invalid value of MLDSA_ETA"+#endif /* MLDSA_ETA != 2 && MLDSA_ETA != 4 */++MLD_STATIC_TESTABLE unsigned int mld_rej_eta_c(int32_t *a, unsigned int target,+ unsigned int offset,+ const uint8_t *buf,+ unsigned int buflen)+__contract__(+ requires(offset <= target && target <= MLDSA_N)+ requires(buflen <= (MLD_POLY_UNIFORM_ETA_NBLOCKS * MLD_STREAM256_BLOCKBYTES))+ requires(memory_no_alias(a, sizeof(int32_t) * target))+ requires(memory_no_alias(buf, buflen))+ requires(array_abs_bound(a, 0, offset, MLDSA_ETA + 1))+ assigns(memory_slice(a, sizeof(int32_t) * target))+ ensures(offset <= return_value && return_value <= target)+ ensures(array_abs_bound(a, 0, return_value, MLDSA_ETA + 1))+)+{+ unsigned int ctr, pos;+ int t_valid;+ uint32_t t0, t1;+ mld_assert_abs_bound(a, offset, MLDSA_ETA + 1);+ ctr = offset;+ pos = 0;+ while (ctr < target && pos < buflen)+ __loop__(+ invariant(offset <= ctr && ctr <= target && pos <= buflen)+ invariant(array_abs_bound(a, 0, ctr, MLDSA_ETA + 1))+ decreases(buflen - pos)+ )+ {+ t0 = buf[pos] & 0x0F;+ t1 = buf[pos++] >> 4;++ /* Constant time: The inputs and outputs to the rejection sampling are+ * secret. However, it is fine to leak which coefficients have been+ * rejected. For constant-time testing, we declassify the result of+ * the comparison.+ */+#if MLDSA_ETA == 2+ t_valid = t0 < 15;+ MLD_CT_TESTING_DECLASSIFY(&t_valid, sizeof(int));+ if (t_valid) /* t0 < 15 */+ {+ t0 = t0 - (205 * t0 >> 10) * 5;+ a[ctr++] = 2 - (int32_t)t0;+ }+ t_valid = t1 < 15;+ MLD_CT_TESTING_DECLASSIFY(&t_valid, sizeof(int));+ if (t_valid && ctr < target) /* t1 < 15 */+ {+ t1 = t1 - (205 * t1 >> 10) * 5;+ a[ctr++] = 2 - (int32_t)t1;+ }+#elif MLDSA_ETA == 4+ t_valid = t0 < 9;+ MLD_CT_TESTING_DECLASSIFY(&t_valid, sizeof(int));+ if (t_valid) /* t0 < 9 */+ {+ a[ctr++] = 4 - (int32_t)t0;+ }+ t_valid = t1 < 9; /* t1 < 9 */+ MLD_CT_TESTING_DECLASSIFY(&t_valid, sizeof(int));+ if (t_valid && ctr < target)+ {+ a[ctr++] = 4 - (int32_t)t1;+ }+#else /* MLDSA_ETA == 4 */+#error "Invalid value of MLDSA_ETA"+#endif /* MLDSA_ETA != 2 && MLDSA_ETA != 4 */+ }++ mld_assert_abs_bound(a, ctr, MLDSA_ETA + 1);++ return ctr;+}++static unsigned int mld_rej_eta(int32_t *a, unsigned int target,+ unsigned int offset, const uint8_t *buf,+ unsigned int buflen)+__contract__(+ requires(offset <= target && target <= MLDSA_N)+ requires(buflen <= (MLD_POLY_UNIFORM_ETA_NBLOCKS * MLD_STREAM256_BLOCKBYTES))+ requires(memory_no_alias(a, sizeof(int32_t) * target))+ requires(memory_no_alias(buf, buflen))+ requires(array_abs_bound(a, 0, offset, MLDSA_ETA + 1))+ assigns(memory_slice(a, sizeof(int32_t) * target))+ ensures(offset <= return_value && return_value <= target)+ ensures(array_abs_bound(a, 0, return_value, MLDSA_ETA + 1))+)+{+#if MLDSA_ETA == 2 && defined(MLD_USE_NATIVE_REJ_UNIFORM_ETA2)+ int ret;+ mld_assert_abs_bound(a, offset, MLDSA_ETA + 1);+ if (offset == 0)+ {+ ret = mld_rej_uniform_eta2_native(a, target, buf, buflen);+ if (ret != MLD_NATIVE_FUNC_FALLBACK)+ {+ unsigned res = (unsigned)ret;+ mld_assert_abs_bound(a, res, MLDSA_ETA + 1);+ return res;+ }+ }+#elif MLDSA_ETA == 4 && defined(MLD_USE_NATIVE_REJ_UNIFORM_ETA4)+ int ret;+ mld_assert_abs_bound(a, offset, MLDSA_ETA + 1);+ if (offset == 0)+ {+ ret = mld_rej_uniform_eta4_native(a, target, buf, buflen);+ if (ret != MLD_NATIVE_FUNC_FALLBACK)+ {+ unsigned res = (unsigned)ret;+ mld_assert_abs_bound(a, res, MLDSA_ETA + 1);+ return res;+ }+ }+#endif /* !(MLDSA_ETA == 2 && MLD_USE_NATIVE_REJ_UNIFORM_ETA2) && MLDSA_ETA == \+ 4 && MLD_USE_NATIVE_REJ_UNIFORM_ETA4 */++ return mld_rej_eta_c(a, target, offset, buf, buflen);+}++#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+MLD_INTERNAL_API+void mld_poly_uniform_eta_4x(mld_poly *r0, mld_poly *r1, mld_poly *r2,+ mld_poly *r3, const uint8_t seed[MLDSA_CRHBYTES],+ uint8_t nonce0, uint8_t nonce1, uint8_t nonce2,+ uint8_t nonce3)+{+ /* Temporary buffers for XOF output before rejection sampling */+ MLD_ALIGN uint8_t buf[4][MLD_ALIGN_UP(MLD_POLY_UNIFORM_ETA_NBLOCKS *+ MLD_STREAM256_BLOCKBYTES)];++ MLD_ALIGN uint8_t extseed[4][MLD_ALIGN_UP(MLDSA_CRHBYTES + 2)];++ /* Tracks the number of coefficients we have already sampled */+ unsigned ctr[4];+ mld_xof256_x4_ctx state;+ unsigned buflen;++ mld_memcpy(extseed[0], seed, MLDSA_CRHBYTES);+ mld_memcpy(extseed[1], seed, MLDSA_CRHBYTES);+ mld_memcpy(extseed[2], seed, MLDSA_CRHBYTES);+ mld_memcpy(extseed[3], seed, MLDSA_CRHBYTES);+ extseed[0][MLDSA_CRHBYTES] = nonce0;+ extseed[1][MLDSA_CRHBYTES] = nonce1;+ extseed[2][MLDSA_CRHBYTES] = nonce2;+ extseed[3][MLDSA_CRHBYTES] = nonce3;+ extseed[0][MLDSA_CRHBYTES + 1] = 0;+ extseed[1][MLDSA_CRHBYTES + 1] = 0;+ extseed[2][MLDSA_CRHBYTES + 1] = 0;+ extseed[3][MLDSA_CRHBYTES + 1] = 0;++ mld_xof256_x4_init(&state);+ mld_xof256_x4_absorb(&state, extseed, MLDSA_CRHBYTES + 2);++ /*+ * Initially, squeeze heuristic number of MLD_POLY_UNIFORM_ETA_NBLOCKS.+ * This should generate the coefficients with high probability.+ */+ mld_xof256_x4_squeezeblocks(buf, MLD_POLY_UNIFORM_ETA_NBLOCKS, &state);+ buflen = MLD_POLY_UNIFORM_ETA_NBLOCKS * MLD_STREAM256_BLOCKBYTES;++ ctr[0] = mld_rej_eta(r0->coeffs, MLDSA_N, 0, buf[0], buflen);+ ctr[1] = mld_rej_eta(r1->coeffs, MLDSA_N, 0, buf[1], buflen);+ ctr[2] = mld_rej_eta(r2->coeffs, MLDSA_N, 0, buf[2], buflen);+ ctr[3] = mld_rej_eta(r3->coeffs, MLDSA_N, 0, buf[3], buflen);++ /*+ * So long as not all entries have been generated, squeeze+ * one more block at a time until we're done.+ */+ buflen = MLD_STREAM256_BLOCKBYTES;+ while (ctr[0] < MLDSA_N || ctr[1] < MLDSA_N || ctr[2] < MLDSA_N ||+ ctr[3] < MLDSA_N)+ __loop__(+ assigns(ctr, state, memory_slice(r0, sizeof(mld_poly)),+ memory_slice(r1, sizeof(mld_poly)), memory_slice(r2, sizeof(mld_poly)),+ memory_slice(r3, sizeof(mld_poly)), object_whole(buf[0]),+ object_whole(buf[1]), object_whole(buf[2]),+ object_whole(buf[3]))+ invariant(ctr[0] <= MLDSA_N && ctr[1] <= MLDSA_N)+ invariant(ctr[2] <= MLDSA_N && ctr[3] <= MLDSA_N)+ invariant(array_abs_bound(r0->coeffs, 0, ctr[0], MLDSA_ETA + 1))+ invariant(array_abs_bound(r1->coeffs, 0, ctr[1], MLDSA_ETA + 1))+ invariant(array_abs_bound(r2->coeffs, 0, ctr[2], MLDSA_ETA + 1))+ invariant(array_abs_bound(r3->coeffs, 0, ctr[3], MLDSA_ETA + 1)))+ {+ mld_xof256_x4_squeezeblocks(buf, 1, &state);+ ctr[0] = mld_rej_eta(r0->coeffs, MLDSA_N, ctr[0], buf[0], buflen);+ ctr[1] = mld_rej_eta(r1->coeffs, MLDSA_N, ctr[1], buf[1], buflen);+ ctr[2] = mld_rej_eta(r2->coeffs, MLDSA_N, ctr[2], buf[2], buflen);+ ctr[3] = mld_rej_eta(r3->coeffs, MLDSA_N, ctr[3], buf[3], buflen);+ }++ mld_xof256_x4_release(&state);++ mld_assert_abs_bound(r0->coeffs, MLDSA_N, MLDSA_ETA + 1);+ mld_assert_abs_bound(r1->coeffs, MLDSA_N, MLDSA_ETA + 1);+ mld_assert_abs_bound(r2->coeffs, MLDSA_N, MLDSA_ETA + 1);+ mld_assert_abs_bound(r3->coeffs, MLDSA_N, MLDSA_ETA + 1);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(buf, sizeof(buf));+ mld_zeroize(extseed, sizeof(extseed));+}+#else /* !MLD_CONFIG_SERIAL_FIPS202_ONLY */++MLD_INTERNAL_API+void mld_poly_uniform_eta(mld_poly *r, const uint8_t seed[MLDSA_CRHBYTES],+ uint8_t nonce)+{+ /* Temporary buffer for XOF output before rejection sampling */+ MLD_ALIGN uint8_t+ buf[MLD_POLY_UNIFORM_ETA_NBLOCKS * MLD_STREAM256_BLOCKBYTES];+ MLD_ALIGN uint8_t extseed[MLDSA_CRHBYTES + 2];++ /* Tracks the number of coefficients we have already sampled */+ unsigned ctr;+ mld_xof256_ctx state;+ unsigned buflen;++ mld_memcpy(extseed, seed, MLDSA_CRHBYTES);+ extseed[MLDSA_CRHBYTES] = nonce;+ extseed[MLDSA_CRHBYTES + 1] = 0;++ mld_xof256_init(&state);+ mld_xof256_absorb_once(&state, extseed, MLDSA_CRHBYTES + 2);++ /*+ * Initially, squeeze heuristic number of MLD_POLY_UNIFORM_ETA_NBLOCKS.+ * This should generate the coefficients with high probability.+ */+ mld_xof256_squeezeblocks(buf, MLD_POLY_UNIFORM_ETA_NBLOCKS, &state);+ buflen = MLD_POLY_UNIFORM_ETA_NBLOCKS * MLD_STREAM256_BLOCKBYTES;++ ctr = mld_rej_eta(r->coeffs, MLDSA_N, 0, buf, buflen);++ /*+ * So long as not all entries have been generated, squeeze+ * one more block at a time until we're done.+ */+ buflen = MLD_STREAM256_BLOCKBYTES;+ while (ctr < MLDSA_N)+ __loop__(+ assigns(ctr, object_whole(&state),+ object_whole(buf), memory_slice(r, sizeof(mld_poly)))+ invariant(ctr <= MLDSA_N)+ invariant(state.pos <= SHAKE256_RATE)+ invariant(array_abs_bound(r->coeffs, 0, ctr, MLDSA_ETA + 1)))+ {+ mld_xof256_squeezeblocks(buf, 1, &state);+ ctr = mld_rej_eta(r->coeffs, MLDSA_N, ctr, buf, buflen);+ }++ mld_xof256_release(&state);++ mld_assert_abs_bound(r->coeffs, MLDSA_N, MLDSA_ETA + 1);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(buf, sizeof(buf));+ mld_zeroize(extseed, sizeof(extseed));+}+#endif /* MLD_CONFIG_SERIAL_FIPS202_ONLY */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define MLD_POLY_UNIFORM_GAMMA1_NBLOCKS \+ ((MLDSA_POLYZ_PACKEDBYTES + MLD_STREAM256_BLOCKBYTES - 1) / \+ MLD_STREAM256_BLOCKBYTES)++#if MLD_CONFIG_PARAMETER_SET == 65 || \+ defined(MLD_CONFIG_SERIAL_FIPS202_ONLY) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+void mld_poly_uniform_gamma1(mld_poly *a, const uint8_t seed[MLDSA_CRHBYTES],+ uint16_t nonce)+{+ MLD_ALIGN uint8_t+ buf[MLD_POLY_UNIFORM_GAMMA1_NBLOCKS * MLD_STREAM256_BLOCKBYTES];+ MLD_ALIGN uint8_t extseed[MLDSA_CRHBYTES + 2];+ mld_xof256_ctx state;++ mld_memcpy(extseed, seed, MLDSA_CRHBYTES);+ extseed[MLDSA_CRHBYTES] = (uint8_t)(nonce & 0xFF);+ extseed[MLDSA_CRHBYTES + 1] = (uint8_t)(nonce >> 8);++ mld_xof256_init(&state);+ mld_xof256_absorb_once(&state, extseed, MLDSA_CRHBYTES + 2);++ mld_xof256_squeezeblocks(buf, MLD_POLY_UNIFORM_GAMMA1_NBLOCKS, &state);+ mld_polyz_unpack(a, buf);++ mld_xof256_release(&state);++ mld_assert_bound(a->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(buf, sizeof(buf));+ mld_zeroize(extseed, sizeof(extseed));+}+#endif /* MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_SERIAL_FIPS202_ONLY || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */+++#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+MLD_INTERNAL_API+void mld_poly_uniform_gamma1_4x(mld_poly *r0, mld_poly *r1, mld_poly *r2,+ mld_poly *r3,+ const uint8_t seed[MLDSA_CRHBYTES],+ uint16_t nonce0, uint16_t nonce1,+ uint16_t nonce2, uint16_t nonce3)+{+ /* Temporary buffers for XOF output before rejection sampling */+ MLD_ALIGN uint8_t buf[4][MLD_ALIGN_UP(MLD_POLY_UNIFORM_GAMMA1_NBLOCKS *+ MLD_STREAM256_BLOCKBYTES)];++ MLD_ALIGN uint8_t extseed[4][MLD_ALIGN_UP(MLDSA_CRHBYTES + 2)];++ /* Tracks the number of coefficients we have already sampled */+ mld_xof256_x4_ctx state;++ mld_memcpy(extseed[0], seed, MLDSA_CRHBYTES);+ mld_memcpy(extseed[1], seed, MLDSA_CRHBYTES);+ mld_memcpy(extseed[2], seed, MLDSA_CRHBYTES);+ mld_memcpy(extseed[3], seed, MLDSA_CRHBYTES);+ extseed[0][MLDSA_CRHBYTES] = (uint8_t)(nonce0 & 0xFF);+ extseed[1][MLDSA_CRHBYTES] = (uint8_t)(nonce1 & 0xFF);+ extseed[2][MLDSA_CRHBYTES] = (uint8_t)(nonce2 & 0xFF);+ extseed[3][MLDSA_CRHBYTES] = (uint8_t)(nonce3 & 0xFF);+ extseed[0][MLDSA_CRHBYTES + 1] = (uint8_t)(nonce0 >> 8);+ extseed[1][MLDSA_CRHBYTES + 1] = (uint8_t)(nonce1 >> 8);+ extseed[2][MLDSA_CRHBYTES + 1] = (uint8_t)(nonce2 >> 8);+ extseed[3][MLDSA_CRHBYTES + 1] = (uint8_t)(nonce3 >> 8);++ mld_xof256_x4_init(&state);+ mld_xof256_x4_absorb(&state, extseed, MLDSA_CRHBYTES + 2);+ mld_xof256_x4_squeezeblocks(buf, MLD_POLY_UNIFORM_GAMMA1_NBLOCKS, &state);++ mld_polyz_unpack(r0, buf[0]);+ mld_polyz_unpack(r1, buf[1]);+ mld_polyz_unpack(r2, buf[2]);+ mld_polyz_unpack(r3, buf[3]);+ mld_xof256_x4_release(&state);++ mld_assert_bound(r0->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);+ mld_assert_bound(r1->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);+ mld_assert_bound(r2->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);+ mld_assert_bound(r3->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(buf, sizeof(buf));+ mld_zeroize(extseed, sizeof(extseed));+}+#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY && (!MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST) */+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_INTERNAL_API+void mld_poly_challenge(mld_poly *c, const uint8_t seed[MLDSA_CTILDEBYTES])+{+ unsigned int i, j, pos;+ uint64_t signs;+ uint64_t offset;+ MLD_ALIGN uint8_t buf[SHAKE256_RATE];+ mld_shake256ctx state;++ mld_shake256_init(&state);+ mld_shake256_absorb(&state, seed, MLDSA_CTILDEBYTES);+ mld_shake256_finalize(&state);+ mld_shake256_squeeze(buf, SHAKE256_RATE, &state);++ /* Convert the first 8 bytes of buf[] into an unsigned 64-bit value. */+ /* Each bit of that dictates the sign of the resulting challenge value */+ signs = 0;+ for (i = 0; i < 8; ++i)+ __loop__(+ assigns(i, signs)+ invariant(i <= 8)+ decreases(8 - i)+ )+ {+ signs |= (uint64_t)buf[i] << 8 * i;+ }+ pos = 8;++ mld_memset(c, 0, sizeof(mld_poly));++ for (i = MLDSA_N - MLDSA_TAU; i < MLDSA_N; ++i)+ __loop__(+ assigns(i, j, object_whole(buf), state, pos, memory_slice(c, sizeof(mld_poly)), signs)+ invariant(i >= MLDSA_N - MLDSA_TAU)+ invariant(i <= MLDSA_N)+ invariant(pos >= 1)+ invariant(pos <= SHAKE256_RATE)+ invariant(array_bound(c->coeffs, 0, MLDSA_N, -1, 2))+ invariant(state.pos <= SHAKE256_RATE)+ decreases(MLDSA_N - i)+ )+ {+ /* This loop terminates only probabilistically, hence no decreases+ * clause. */+ do+ __loop__(+ assigns(j, object_whole(buf), state, pos)+ invariant(state.pos <= SHAKE256_RATE)+ )+ {+ if (pos >= SHAKE256_RATE)+ {+ mld_shake256_squeeze(buf, SHAKE256_RATE, &state);+ pos = 0;+ }+ j = buf[pos++];+ } while (j > i);++ c->coeffs[i] = c->coeffs[j];++ /* Reference: Compute coefficient value here in two steps to */+ /* avoid mixing unsigned and signed arithmetic with implicit */+ /* conversions, and so that CBMC can keep track of ranges */+ /* to complete type-safety proof here. */++ /* The least-significant bit of signs tells us if we want -1 or +1 */+ offset = 2 * (signs & 1);++ /* offset has value 0 or 2 here, so (1 - (int32_t) offset) has+ * value -1 or +1 */+ c->coeffs[j] = 1 - (int32_t)offset;++ /* Move to the next bit of signs for next time */+ signs >>= 1;+ }++ mld_assert_bound(c->coeffs, MLDSA_N, -1, 2);+ mld_shake256_release(&state);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(buf, sizeof(buf));+ mld_zeroize(&signs, sizeof(signs));+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_INTERNAL_API+void mld_polyeta_pack(uint8_t r[MLDSA_POLYETA_PACKEDBYTES], const mld_poly *a)+{+ unsigned int i;+ uint8_t t[8];++ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_ETA + 1);++#if MLDSA_ETA == 2+ for (i = 0; i < MLDSA_N / 8; ++i)+ __loop__(+ invariant(i <= MLDSA_N/8)+ decreases(MLDSA_N / 8 - i))+ {+ /* The casts are safe since we assume that the coefficients+ * of a are <= MLDSA_ETA in absolute value. */+ t[0] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 0]);+ t[1] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 1]);+ t[2] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 2]);+ t[3] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 3]);+ t[4] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 4]);+ t[5] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 5]);+ t[6] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 6]);+ t[7] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 7]);++ r[3 * i + 0] = (uint8_t)(((t[0] >> 0) | (t[1] << 3) | (t[2] << 6)) & 0xFF);+ r[3 * i + 1] =+ (uint8_t)(((t[2] >> 2) | (t[3] << 1) | (t[4] << 4) | (t[5] << 7)) &+ 0xFF);+ r[3 * i + 2] = (uint8_t)(((t[5] >> 1) | (t[6] << 2) | (t[7] << 5)) & 0xFF);+ }+#elif MLDSA_ETA == 4+ for (i = 0; i < MLDSA_N / 2; ++i)+ __loop__(+ invariant(i <= MLDSA_N/2)+ decreases(MLDSA_N / 2 - i))+ {+ /* The casts are safe since we assume that the coefficients+ * of a are <= MLDSA_ETA in absolute value. */+ t[0] = (uint8_t)(MLDSA_ETA - a->coeffs[2 * i + 0]);+ t[1] = (uint8_t)(MLDSA_ETA - a->coeffs[2 * i + 1]);+ r[i] = (uint8_t)(t[0] | (t[1] << 4));+ }+#else /* MLDSA_ETA == 4 */+#error "Invalid value of MLDSA_ETA"+#endif /* MLDSA_ETA != 2 && MLDSA_ETA != 4 */+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_SIGN_API)+MLD_INTERNAL_API+void mld_polyeta_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYETA_PACKEDBYTES])+{+ unsigned int i;++#if MLDSA_ETA == 2+ for (i = 0; i < MLDSA_N / 8; ++i)+ __loop__(+ invariant(i <= MLDSA_N/8)+ invariant(array_bound(r->coeffs, 0, i*8, -5, MLDSA_ETA + 1))+ decreases(MLDSA_N / 8 - i))+ {+ r->coeffs[8 * i + 0] = (a[3 * i + 0] >> 0) & 7;+ r->coeffs[8 * i + 1] = (a[3 * i + 0] >> 3) & 7;+ r->coeffs[8 * i + 2] = ((a[3 * i + 0] >> 6) | (a[3 * i + 1] << 2)) & 7;+ r->coeffs[8 * i + 3] = (a[3 * i + 1] >> 1) & 7;+ r->coeffs[8 * i + 4] = (a[3 * i + 1] >> 4) & 7;+ r->coeffs[8 * i + 5] = ((a[3 * i + 1] >> 7) | (a[3 * i + 2] << 1)) & 7;+ r->coeffs[8 * i + 6] = (a[3 * i + 2] >> 2) & 7;+ r->coeffs[8 * i + 7] = (a[3 * i + 2] >> 5) & 7;++ r->coeffs[8 * i + 0] = MLDSA_ETA - r->coeffs[8 * i + 0];+ r->coeffs[8 * i + 1] = MLDSA_ETA - r->coeffs[8 * i + 1];+ r->coeffs[8 * i + 2] = MLDSA_ETA - r->coeffs[8 * i + 2];+ r->coeffs[8 * i + 3] = MLDSA_ETA - r->coeffs[8 * i + 3];+ r->coeffs[8 * i + 4] = MLDSA_ETA - r->coeffs[8 * i + 4];+ r->coeffs[8 * i + 5] = MLDSA_ETA - r->coeffs[8 * i + 5];+ r->coeffs[8 * i + 6] = MLDSA_ETA - r->coeffs[8 * i + 6];+ r->coeffs[8 * i + 7] = MLDSA_ETA - r->coeffs[8 * i + 7];+ }+#elif MLDSA_ETA == 4+ for (i = 0; i < MLDSA_N / 2; ++i)+ __loop__(+ invariant(i <= MLDSA_N/2)+ invariant(array_bound(r->coeffs, 0, i*2, -11, MLDSA_ETA + 1))+ decreases(MLDSA_N / 2 - i))+ {+ r->coeffs[2 * i + 0] = a[i] & 0x0F;+ r->coeffs[2 * i + 1] = a[i] >> 4;+ r->coeffs[2 * i + 0] = MLDSA_ETA - r->coeffs[2 * i + 0];+ r->coeffs[2 * i + 1] = MLDSA_ETA - r->coeffs[2 * i + 1];+ }+#else /* MLDSA_ETA == 4 */+#error "Invalid value of MLDSA_ETA"+#endif /* MLDSA_ETA != 2 && MLDSA_ETA != 4 */++ mld_assert_bound(r->coeffs, MLDSA_N, MLD_POLYETA_UNPACK_LOWER_BOUND,+ MLDSA_ETA + 1);+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_INTERNAL_API+void mld_polyz_pack(uint8_t r[MLDSA_POLYZ_PACKEDBYTES], const mld_poly *a)+{+ unsigned int i;+ uint32_t t[4];++ mld_assert_bound(a->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);++#if MLD_CONFIG_PARAMETER_SET == 44+ for (i = 0; i < MLDSA_N / 4; ++i)+ __loop__(+ invariant(i <= MLDSA_N/4)+ decreases(MLDSA_N / 4 - i))+ {+ /* Safety: a->coeffs[i] <= MLDSA_GAMMA1, hence, these casts are safe. */+ t[0] = (uint32_t)(MLDSA_GAMMA1 - a->coeffs[4 * i + 0]);+ t[1] = (uint32_t)(MLDSA_GAMMA1 - a->coeffs[4 * i + 1]);+ t[2] = (uint32_t)(MLDSA_GAMMA1 - a->coeffs[4 * i + 2]);+ t[3] = (uint32_t)(MLDSA_GAMMA1 - a->coeffs[4 * i + 3]);++ r[9 * i + 0] = (uint8_t)((t[0]) & 0xFF);+ r[9 * i + 1] = (uint8_t)((t[0] >> 8) & 0xFF);+ r[9 * i + 2] = (uint8_t)((t[0] >> 16) & 0xFF);+ r[9 * i + 2] |= (uint8_t)((t[1] << 2) & 0xFF);+ r[9 * i + 3] = (uint8_t)((t[1] >> 6) & 0xFF);+ r[9 * i + 4] = (uint8_t)((t[1] >> 14) & 0xFF);+ r[9 * i + 4] |= (uint8_t)((t[2] << 4) & 0xFF);+ r[9 * i + 5] = (uint8_t)((t[2] >> 4) & 0xFF);+ r[9 * i + 6] = (uint8_t)((t[2] >> 12) & 0xFF);+ r[9 * i + 6] |= (uint8_t)((t[3] << 6) & 0xFF);+ r[9 * i + 7] = (uint8_t)((t[3] >> 2) & 0xFF);+ r[9 * i + 8] = (uint8_t)((t[3] >> 10) & 0xFF);+ }+#else /* MLD_CONFIG_PARAMETER_SET == 44 */+ for (i = 0; i < MLDSA_N / 2; ++i)+ __loop__(+ invariant(i <= MLDSA_N/2)+ decreases(MLDSA_N / 2 - i))+ {+ /* Safety: a->coeffs[i] <= MLDSA_GAMMA1, hence, these casts are safe. */+ t[0] = (uint32_t)(MLDSA_GAMMA1 - a->coeffs[2 * i + 0]);+ t[1] = (uint32_t)(MLDSA_GAMMA1 - a->coeffs[2 * i + 1]);++ r[5 * i + 0] = (uint8_t)((t[0]) & 0xFF);+ r[5 * i + 1] = (uint8_t)((t[0] >> 8) & 0xFF);+ r[5 * i + 2] = (uint8_t)((t[0] >> 16) & 0xFF);+ r[5 * i + 2] |= (uint8_t)((t[1] << 4) & 0xFF);+ r[5 * i + 3] = (uint8_t)((t[1] >> 4) & 0xFF);+ r[5 * i + 4] = (uint8_t)((t[1] >> 12) & 0xFF);+ }+#endif /* MLD_CONFIG_PARAMETER_SET != 44 */+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_STATIC_TESTABLE void mld_polyz_unpack_c(+ mld_poly *r, const uint8_t a[MLDSA_POLYZ_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(memory_no_alias(a, MLDSA_POLYZ_PACKEDBYTES))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(array_bound(r->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+)+{+ unsigned int i;+#if MLD_CONFIG_PARAMETER_SET == 44+ for (i = 0; i < MLDSA_N / 4; ++i)+ __loop__(+ invariant(i <= MLDSA_N/4)+ invariant(array_bound(r->coeffs, 0, i*4, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ decreases(MLDSA_N / 4 - i))+ {+ r->coeffs[4 * i + 0] = a[9 * i + 0];+ r->coeffs[4 * i + 0] |= (int32_t)a[9 * i + 1] << 8;+ r->coeffs[4 * i + 0] |= (int32_t)a[9 * i + 2] << 16;+ r->coeffs[4 * i + 0] &= 0x3FFFF;++ r->coeffs[4 * i + 1] = a[9 * i + 2] >> 2;+ r->coeffs[4 * i + 1] |= (int32_t)a[9 * i + 3] << 6;+ r->coeffs[4 * i + 1] |= (int32_t)a[9 * i + 4] << 14;+ r->coeffs[4 * i + 1] &= 0x3FFFF;++ r->coeffs[4 * i + 2] = a[9 * i + 4] >> 4;+ r->coeffs[4 * i + 2] |= (int32_t)a[9 * i + 5] << 4;+ r->coeffs[4 * i + 2] |= (int32_t)a[9 * i + 6] << 12;+ r->coeffs[4 * i + 2] &= 0x3FFFF;++ r->coeffs[4 * i + 3] = a[9 * i + 6] >> 6;+ r->coeffs[4 * i + 3] |= (int32_t)a[9 * i + 7] << 2;+ r->coeffs[4 * i + 3] |= (int32_t)a[9 * i + 8] << 10;+ r->coeffs[4 * i + 3] &= 0x3FFFF;++ r->coeffs[4 * i + 0] = MLDSA_GAMMA1 - r->coeffs[4 * i + 0];+ r->coeffs[4 * i + 1] = MLDSA_GAMMA1 - r->coeffs[4 * i + 1];+ r->coeffs[4 * i + 2] = MLDSA_GAMMA1 - r->coeffs[4 * i + 2];+ r->coeffs[4 * i + 3] = MLDSA_GAMMA1 - r->coeffs[4 * i + 3];+ }+#else /* MLD_CONFIG_PARAMETER_SET == 44 */+ for (i = 0; i < MLDSA_N / 2; ++i)+ __loop__(+ invariant(i <= MLDSA_N/2)+ invariant(array_bound(r->coeffs, 0, i*2, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ decreases(MLDSA_N / 2 - i))+ {+ r->coeffs[2 * i + 0] = a[5 * i + 0];+ r->coeffs[2 * i + 0] |= (int32_t)a[5 * i + 1] << 8;+ r->coeffs[2 * i + 0] |= (int32_t)a[5 * i + 2] << 16;+ r->coeffs[2 * i + 0] &= 0xFFFFF;++ r->coeffs[2 * i + 1] = a[5 * i + 2] >> 4;+ r->coeffs[2 * i + 1] |= (int32_t)a[5 * i + 3] << 4;+ r->coeffs[2 * i + 1] |= (int32_t)a[5 * i + 4] << 12;+ /* r->coeffs[2*i+1] &= 0xFFFFF; */ /* No effect, since we're anyway at 20+ bits */++ r->coeffs[2 * i + 0] = MLDSA_GAMMA1 - r->coeffs[2 * i + 0];+ r->coeffs[2 * i + 1] = MLDSA_GAMMA1 - r->coeffs[2 * i + 1];+ }+#endif /* MLD_CONFIG_PARAMETER_SET != 44 */+ mld_assert_bound(r->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);+}++MLD_INTERNAL_API+void mld_polyz_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYZ_PACKEDBYTES])+{+#if defined(MLD_USE_NATIVE_POLYZ_UNPACK_17) && MLD_CONFIG_PARAMETER_SET == 44+ int ret;+ ret = mld_polyz_unpack_17_native(r->coeffs, a);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_bound(r->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);+ return;+ }+#elif defined(MLD_USE_NATIVE_POLYZ_UNPACK_19) && \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+ int ret;+ ret = mld_polyz_unpack_19_native(r->coeffs, a);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_bound(r->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);+ return;+ }+#endif /* !(MLD_USE_NATIVE_POLYZ_UNPACK_17 && MLD_CONFIG_PARAMETER_SET == 44) \+ && MLD_USE_NATIVE_POLYZ_UNPACK_19 && (MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87) */++ mld_polyz_unpack_c(r, a);+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros. */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef mld_rej_eta+#undef mld_rej_eta_c+#undef mld_poly_decompose_c+#undef mld_poly_use_hint_c+#undef mld_polyz_unpack_c+#undef MLD_POLY_UNIFORM_ETA_NBLOCKS+#undef MLD_POLY_UNIFORM_GAMMA1_NBLOCKS
+ cbits/mldsa/src/poly_kl.h view
@@ -0,0 +1,367 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_POLY_KL_H+#define MLD_POLY_KL_H++#include "cbmc.h"+#include "common.h"+#include "poly.h"++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_poly_decompose MLD_NAMESPACE_KL(poly_decompose)+/**+ * For all coefficients c of the input polynomial, compute high and low bits+ * c0, c1 such c mod MLDSA_Q = c1*ALPHA + c0 with -ALPHA/2 < c0 <= ALPHA/2+ * except c1 = (MLDSA_Q-1)/ALPHA where we set c1 = 0 and+ * -ALPHA/2 <= c0 = c mod MLDSA_Q - MLDSA_Q < 0. Assumes coefficients to be+ * standard representatives.+ *+ * @reference{The reference implementation has the input polynomial as a+ * separate argument that may be aliased with either of the outputs. Removing+ * the aliasing eases CBMC proofs.}+ *+ * @param[out] a1 Pointer to output polynomial with coefficients c1.+ * @param[in,out] a0 Pointer to input/output polynomial. Output polynomial has+ * coefficients c0.+ */+MLD_INTERNAL_API+void mld_poly_decompose(mld_poly *a1, mld_poly *a0)+__contract__(+ requires(memory_no_alias(a1, sizeof(mld_poly)))+ requires(memory_no_alias(a0, sizeof(mld_poly)))+ requires(array_bound(a0->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(mld_poly)))+ assigns(memory_slice(a0, sizeof(mld_poly)))+ ensures(array_bound(a1->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ ensures(array_abs_bound(a0->coeffs, 0, MLDSA_N, MLDSA_GAMMA2+1))+);++#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_poly_use_hint MLD_NAMESPACE_KL(poly_use_hint)+/**+ * Use hint polynomial h to correct the high bits of a in-place.+ *+ * @param[in,out] a Input/output polynomial.+ * @param[in] h Hint polynomial.+ */+MLD_INTERNAL_API+void mld_poly_use_hint(mld_poly *a, const mld_poly *h)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(memory_no_alias(h, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h->coeffs, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+#define mld_poly_uniform_eta_4x MLD_NAMESPACE_KL(poly_uniform_eta_4x)+/**+ * Sample four polynomials with uniformly random coefficients in+ * [-MLDSA_ETA, MLDSA_ETA] by performing rejection sampling on the output+ * stream from SHAKE256(seed|nonce_i).+ *+ * @spec{Implements @[FIPS204, Algorithm 31, RejBoundedPoly] (four-way+ * batched).}+ *+ * @param[out] r0 Pointer to first output polynomial.+ * @param[out] r1 Pointer to second output polynomial.+ * @param[out] r2 Pointer to third output polynomial.+ * @param[out] r3 Pointer to fourth output polynomial.+ * @param[in] seed Byte array with seed of length MLDSA_CRHBYTES.+ * @param nonce0 First nonce.+ * @param nonce1 Second nonce.+ * @param nonce2 Third nonce.+ * @param nonce3 Fourth nonce.+ */+MLD_INTERNAL_API+void mld_poly_uniform_eta_4x(mld_poly *r0, mld_poly *r1, mld_poly *r2,+ mld_poly *r3, const uint8_t seed[MLDSA_CRHBYTES],+ uint8_t nonce0, uint8_t nonce1, uint8_t nonce2,+ uint8_t nonce3)+__contract__(+ requires(memory_no_alias(r0, sizeof(mld_poly)))+ requires(memory_no_alias(r1, sizeof(mld_poly)))+ requires(memory_no_alias(r2, sizeof(mld_poly)))+ requires(memory_no_alias(r3, sizeof(mld_poly)))+ requires(memory_no_alias(seed, MLDSA_CRHBYTES))+ assigns(memory_slice(r0, sizeof(mld_poly)))+ assigns(memory_slice(r1, sizeof(mld_poly)))+ assigns(memory_slice(r2, sizeof(mld_poly)))+ assigns(memory_slice(r3, sizeof(mld_poly)))+ ensures(array_abs_bound(r0->coeffs, 0, MLDSA_N, MLDSA_ETA + 1))+ ensures(array_abs_bound(r1->coeffs, 0, MLDSA_N, MLDSA_ETA + 1))+ ensures(array_abs_bound(r2->coeffs, 0, MLDSA_N, MLDSA_ETA + 1))+ ensures(array_abs_bound(r3->coeffs, 0, MLDSA_N, MLDSA_ETA + 1))+);+#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY */++#if defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+#define mld_poly_uniform_eta MLD_NAMESPACE_KL(poly_uniform_eta)+/**+ * Sample polynomial with uniformly random coefficients in+ * [-MLDSA_ETA, MLDSA_ETA] by performing rejection sampling on the output+ * stream from SHAKE256(seed|nonce).+ *+ * @spec{Implements @[FIPS204, Algorithm 31, RejBoundedPoly].}+ *+ * @param[out] r Pointer to output polynomial.+ * @param[in] seed Byte array with seed of length MLDSA_CRHBYTES.+ * @param nonce Nonce.+ */+MLD_INTERNAL_API+void mld_poly_uniform_eta(mld_poly *r, const uint8_t seed[MLDSA_CRHBYTES],+ uint8_t nonce)+__contract__(+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(memory_no_alias(seed, MLDSA_CRHBYTES))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(array_abs_bound(r->coeffs, 0, MLDSA_N, MLDSA_ETA + 1))+);+#endif /* MLD_CONFIG_SERIAL_FIPS202_ONLY */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#if MLD_CONFIG_PARAMETER_SET == 65 || \+ defined(MLD_CONFIG_SERIAL_FIPS202_ONLY) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+#define mld_poly_uniform_gamma1 MLD_NAMESPACE_KL(poly_uniform_gamma1)+/**+ * Sample polynomial with uniformly random coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1] by unpacking output stream of+ * SHAKE256(seed|nonce).+ *+ * @spec{Partially implements @[FIPS204, Algorithm 34, ExpandMask] (one+ * polynomial, i.e. the loop body of lines 3-5).}+ *+ * @param[out] a Pointer to output polynomial.+ * @param[in] seed Byte array with seed of length MLDSA_CRHBYTES.+ * @param nonce 16-bit nonce.+ */+MLD_INTERNAL_API+void mld_poly_uniform_gamma1(mld_poly *a, const uint8_t seed[MLDSA_CRHBYTES],+ uint16_t nonce)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(memory_no_alias(seed, MLDSA_CRHBYTES))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+);+#endif /* MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_SERIAL_FIPS202_ONLY || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+#define mld_poly_uniform_gamma1_4x MLD_NAMESPACE_KL(poly_uniform_gamma1_4x)+/**+ * Sample four polynomials with uniformly random coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1] by unpacking output streams of+ * SHAKE256(seed|nonce_i).+ *+ * @spec{Partially implements @[FIPS204, Algorithm 34, ExpandMask] (four-way+ * batched, i.e. four iterations of the loop body of lines 3-5).}+ *+ * @param[out] r0 Pointer to first output polynomial.+ * @param[out] r1 Pointer to second output polynomial.+ * @param[out] r2 Pointer to third output polynomial.+ * @param[out] r3 Pointer to fourth output polynomial.+ * @param[in] seed Byte array with seed of length MLDSA_CRHBYTES.+ * @param nonce0 First 16-bit nonce.+ * @param nonce1 Second 16-bit nonce.+ * @param nonce2 Third 16-bit nonce.+ * @param nonce3 Fourth 16-bit nonce.+ */+MLD_INTERNAL_API+void mld_poly_uniform_gamma1_4x(mld_poly *r0, mld_poly *r1, mld_poly *r2,+ mld_poly *r3,+ const uint8_t seed[MLDSA_CRHBYTES],+ uint16_t nonce0, uint16_t nonce1,+ uint16_t nonce2, uint16_t nonce3)+__contract__(+ requires(memory_no_alias(r0, sizeof(mld_poly)))+ requires(memory_no_alias(r1, sizeof(mld_poly)))+ requires(memory_no_alias(r2, sizeof(mld_poly)))+ requires(memory_no_alias(r3, sizeof(mld_poly)))+ requires(memory_no_alias(seed, MLDSA_CRHBYTES))+ assigns(memory_slice(r0, sizeof(mld_poly)))+ assigns(memory_slice(r1, sizeof(mld_poly)))+ assigns(memory_slice(r2, sizeof(mld_poly)))+ assigns(memory_slice(r3, sizeof(mld_poly)))+ ensures(array_bound(r0->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ ensures(array_bound(r1->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ ensures(array_bound(r2->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ ensures(array_bound(r3->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+);+#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY && (!MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST) */+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_poly_challenge MLD_NAMESPACE_KL(poly_challenge)+/**+ * Samples polynomial with MLDSA_TAU nonzero coefficients in {-1, 1} using the+ * output stream of SHAKE256(seed).+ *+ * @spec{Implements @[FIPS204, Algorithm 29, SampleInBall].}+ *+ * @param[out] c Pointer to output polynomial.+ * @param[in] seed Byte array containing seed of length MLDSA_CTILDEBYTES.+ */+MLD_INTERNAL_API+void mld_poly_challenge(mld_poly *c, const uint8_t seed[MLDSA_CTILDEBYTES])+__contract__(+ requires(memory_no_alias(c, sizeof(mld_poly)))+ requires(memory_no_alias(seed, MLDSA_CTILDEBYTES))+ assigns(memory_slice(c, sizeof(mld_poly)))+ /* All coefficients of c are -1, 0 or +1 */+ ensures(array_bound(c->coeffs, 0, MLDSA_N, -1, 2))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_polyeta_pack MLD_NAMESPACE_KL(polyeta_pack)+/**+ * Bit-pack polynomial with coefficients in [-MLDSA_ETA, MLDSA_ETA].+ *+ * @spec{Implements @[FIPS204, Algorithm 17, BitPack].}+ *+ * @param[out] r Pointer to output byte array with at least+ * MLDSA_POLYETA_PACKEDBYTES bytes.+ * @param[in] a Pointer to input polynomial.+ */+MLD_INTERNAL_API+void mld_polyeta_pack(uint8_t r[MLDSA_POLYETA_PACKEDBYTES], const mld_poly *a)+__contract__(+ requires(memory_no_alias(r, MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_ETA + 1))+ assigns(memory_slice(r, MLDSA_POLYETA_PACKEDBYTES))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_SIGN_API)+/*+ * polyeta_unpack produces coefficients in [-MLDSA_ETA, MLDSA_ETA] for+ * well-formed inputs (i.e., those produced by polyeta_pack).+ * However, when passed an arbitrary byte array, it may produce smaller values,+ * i.e., values in [MLD_POLYETA_UNPACK_LOWER_BOUND, MLDSA_ETA].+ * Even though this should never happen, we use use the bound for arbitrary+ * inputs in the CBMC proofs.+ */+#if MLDSA_ETA == 2+#define MLD_POLYETA_UNPACK_LOWER_BOUND (-5)+#elif MLDSA_ETA == 4+#define MLD_POLYETA_UNPACK_LOWER_BOUND (-11)+#else+#error "Invalid value of MLDSA_ETA"+#endif++#define mld_polyeta_unpack MLD_NAMESPACE_KL(polyeta_unpack)+/**+ * Unpack polynomial with coefficients in [-MLDSA_ETA, MLDSA_ETA].+ *+ * @spec{Implements @[FIPS204, Algorithm 19, BitUnpack].}+ *+ * @param[out] r Pointer to output polynomial.+ * @param[in] a Byte array with bit-packed polynomial.+ */+MLD_INTERNAL_API+void mld_polyeta_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYETA_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(memory_no_alias(a, MLDSA_POLYETA_PACKEDBYTES))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(array_bound(r->coeffs, 0, MLDSA_N, MLD_POLYETA_UNPACK_LOWER_BOUND, MLDSA_ETA + 1))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_polyz_pack MLD_NAMESPACE_KL(polyz_pack)+/**+ * Bit-pack polynomial with coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1].+ *+ * @spec{Implements @[FIPS204, Algorithm 17, BitPack].}+ *+ * @param[out] r Pointer to output byte array with at least+ * MLDSA_POLYZ_PACKEDBYTES bytes.+ * @param[in] a Pointer to input polynomial.+ */+MLD_INTERNAL_API+void mld_polyz_pack(uint8_t r[MLDSA_POLYZ_PACKEDBYTES], const mld_poly *a)+__contract__(+ requires(memory_no_alias(r, MLDSA_POLYZ_PACKEDBYTES))+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ assigns(memory_slice(r, MLDSA_POLYZ_PACKEDBYTES))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_polyz_unpack MLD_NAMESPACE_KL(polyz_unpack)+/**+ * Unpack polynomial z with coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1].+ *+ * @spec{Implements @[FIPS204, Algorithm 19, BitUnpack].}+ *+ * @param[out] r Pointer to output polynomial.+ * @param[in] a Byte array with bit-packed polynomial.+ */+MLD_INTERNAL_API+void mld_polyz_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYZ_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(memory_no_alias(a, MLDSA_POLYZ_PACKEDBYTES))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(array_bound(r->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+);++#define mld_polyw1_pack MLD_NAMESPACE_KL(polyw1_pack)+/**+ * Bit-pack polynomial w1. Input coefficients must be in+ * [0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)), i.e. [0, 43] for ML-DSA-44 and [0, 15]+ * for ML-DSA-65/87. Dispatches to the value-specialized variant for the+ * selected parameter set.+ *+ * @spec{Implements @[FIPS204, Algorithm 16, SimpleBitPack].}+ *+ * @param[out] r Pointer to output byte array with at least+ * MLDSA_POLYW1_PACKEDBYTES bytes.+ * @param[in] a Pointer to input polynomial.+ */+static MLD_INLINE void mld_polyw1_pack(uint8_t r[MLDSA_POLYW1_PACKEDBYTES],+ const mld_poly *a)+__contract__(+ requires(memory_no_alias(r, MLDSA_POLYW1_PACKEDBYTES))+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ assigns(memory_slice(r, MLDSA_POLYW1_PACKEDBYTES))+)+{+#if MLD_CONFIG_PARAMETER_SET == 44+ mld_polyw1_pack_88(r, a);+#else+ mld_polyw1_pack_32(r, a);+#endif+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#endif /* !MLD_POLY_KL_H */
+ cbits/mldsa/src/polyvec.c view
@@ -0,0 +1,509 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#include "polyvec.h"++#include "debug.h"+#include "polyvec_lazy.h"++/* This namespacing is not done at the top to avoid a naming conflict+ * with native backends, which are currently not yet namespaced. */+#define mld_polyvecl_pointwise_acc_montgomery_c \+ MLD_ADD_PARAM_SET(mld_polyvecl_pointwise_acc_montgomery_c)++/**************************************************************/+/************ Vectors of polynomials of length MLDSA_L **************/+/**************************************************************/+#if !defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+MLD_INTERNAL_API+void mld_polyvecl_uniform_gamma1(mld_polyvecl *v,+ const uint8_t seed[MLDSA_CRHBYTES],+ uint16_t kappa)+{+#if defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+ int i;+#endif++ /* The caller passes the base counter kappa; component i is sampled from+ * kappa + i. Safety: kappa <= MLD_MAX_KAPPA and i < MLDSA_L, so the+ * casts below are safe. See MLD_MAX_KAPPA comment in params.h. */+#if defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+ for (i = 0; i < MLDSA_L; i++)+ {+ mld_poly_uniform_gamma1(&v->vec[i], seed, (uint16_t)(kappa + i));+ }+#else /* MLD_CONFIG_SERIAL_FIPS202_ONLY */+#if MLDSA_L == 4+ mld_poly_uniform_gamma1_4x(&v->vec[0], &v->vec[1], &v->vec[2], &v->vec[3],+ seed, kappa, (uint16_t)(kappa + 1),+ (uint16_t)(kappa + 2), (uint16_t)(kappa + 3));+#elif MLDSA_L == 5+ mld_poly_uniform_gamma1_4x(&v->vec[0], &v->vec[1], &v->vec[2], &v->vec[3],+ seed, kappa, (uint16_t)(kappa + 1),+ (uint16_t)(kappa + 2), (uint16_t)(kappa + 3));+ mld_poly_uniform_gamma1(&v->vec[4], seed, (uint16_t)(kappa + 4));+#elif MLDSA_L == 7+ mld_poly_uniform_gamma1_4x(&v->vec[0], &v->vec[1], &v->vec[2],+ &v->vec[3 /* irrelevant */], seed, kappa,+ (uint16_t)(kappa + 1), (uint16_t)(kappa + 2),+ 0xFF /* irrelevant */);+ mld_poly_uniform_gamma1_4x(&v->vec[3], &v->vec[4], &v->vec[5], &v->vec[6],+ seed, (uint16_t)(kappa + 3), (uint16_t)(kappa + 4),+ (uint16_t)(kappa + 5), (uint16_t)(kappa + 6));+#endif /* MLDSA_L == 7 */+#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY */++ mld_assert_bound_2d(v->vec, MLDSA_L, MLDSA_N, -(MLDSA_GAMMA1 - 1),+ MLDSA_GAMMA1 + 1);+}+#endif /* !MLD_CONFIG_NO_SIGN_API && (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) \+ */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ !defined(MLD_CONFIG_NO_VERIFY_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+MLD_INTERNAL_API+void mld_polyvecl_ntt(mld_polyvecl *v)+{+ unsigned int i;+ mld_assert_abs_bound_2d(v->vec, MLDSA_L, MLDSA_N, MLDSA_Q);++ for (i = 0; i < MLDSA_L; ++i)+ __loop__(+ assigns(i, memory_slice(v, sizeof(mld_polyvecl)))+ invariant(i <= MLDSA_L)+ invariant(forall(k0, i, MLDSA_L, forall(k1, 0, MLDSA_N, v->vec[k0].coeffs[k1] == loop_entry(*v).vec[k0].coeffs[k1])))+ invariant(forall(k1, 0, i, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ decreases(MLDSA_L - i))+ {+ mld_poly_ntt(&v->vec[i]);+ }++ mld_assert_abs_bound_2d(v->vec, MLDSA_L, MLDSA_N, MLD_NTT_BOUND);+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_VERIFY_API || \+ (!MLD_CONFIG_NO_SIGN_API && (!MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST)) */++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+MLD_STATIC_TESTABLE void mld_polyvecl_pointwise_acc_montgomery_c(+ mld_poly *w, const mld_polyvecl *u, const mld_polyvecl *v)+__contract__(+ requires(memory_no_alias(w, sizeof(mld_poly)))+ requires(memory_no_alias(u, sizeof(mld_polyvecl)))+ requires(memory_no_alias(v, sizeof(mld_polyvecl)))+ requires(forall(l0, 0, MLDSA_L,+ array_bound(u->vec[l0].coeffs, 0, MLDSA_N, 0, MLDSA_Q)))+ requires(forall(l1, 0, MLDSA_L,+ array_abs_bound(v->vec[l1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ assigns(memory_slice(w, sizeof(mld_poly)))+ ensures(array_abs_bound(w->coeffs, 0, MLDSA_N, MLDSA_Q))+)+{+ unsigned int i, j;+ mld_assert_bound_2d(u->vec, MLDSA_L, MLDSA_N, 0, MLDSA_Q);+ mld_assert_abs_bound_2d(v->vec, MLDSA_L, MLDSA_N, MLD_NTT_BOUND);+ for (i = 0; i < MLDSA_N; i++)+ __loop__(+ assigns(i, j, memory_slice(w, sizeof(mld_poly)))+ invariant(i <= MLDSA_N)+ invariant(array_abs_bound(w->coeffs, 0, i, MLDSA_Q))+ decreases(MLDSA_N - i)+ )+ {+ int64_t t = 0;+ int32_t r;+ for (j = 0; j < MLDSA_L; j++)+ __loop__(+ assigns(j, t)+ invariant(j <= MLDSA_L)+ invariant(t >= -(int64_t)j*(MLDSA_Q - 1)*(MLD_NTT_BOUND - 1))+ invariant(t <= (int64_t)j*(MLDSA_Q - 1)*(MLD_NTT_BOUND - 1))+ decreases(MLDSA_L - j)+ )+ {+ t += (int64_t)u->vec[j].coeffs[i] * v->vec[j].coeffs[i];+ }++ r = mld_montgomery_reduce(t);+ w->coeffs[i] = r;+ }++ mld_assert_abs_bound(w->coeffs, MLDSA_N, MLDSA_Q);+}++MLD_INTERNAL_API+void mld_polyvecl_pointwise_acc_montgomery(mld_poly *w, const mld_polyvecl *u,+ const mld_polyvecl *v)+{+#if defined(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4) && \+ MLD_CONFIG_PARAMETER_SET == 44+ int ret;+ mld_assert_bound_2d(u->vec, MLDSA_L, MLDSA_N, 0, MLDSA_Q);+ mld_assert_abs_bound_2d(v->vec, MLDSA_L, MLDSA_N, MLD_NTT_BOUND);+ ret = mld_polyvecl_pointwise_acc_montgomery_l4_native(+ w->coeffs, (const int32_t (*)[MLDSA_N])u->vec,+ (const int32_t (*)[MLDSA_N])v->vec);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(w->coeffs, MLDSA_N, MLDSA_Q);+ return;+ }+#elif defined(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5) && \+ MLD_CONFIG_PARAMETER_SET == 65+ int ret;+ mld_assert_bound_2d(u->vec, MLDSA_L, MLDSA_N, 0, MLDSA_Q);+ mld_assert_abs_bound_2d(v->vec, MLDSA_L, MLDSA_N, MLD_NTT_BOUND);+ ret = mld_polyvecl_pointwise_acc_montgomery_l5_native(+ w->coeffs, (const int32_t (*)[MLDSA_N])u->vec,+ (const int32_t (*)[MLDSA_N])v->vec);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(w->coeffs, MLDSA_N, MLDSA_Q);+ return;+ }+#elif defined(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7) && \+ MLD_CONFIG_PARAMETER_SET == 87+ int ret;+ mld_assert_bound_2d(u->vec, MLDSA_L, MLDSA_N, 0, MLDSA_Q);+ mld_assert_abs_bound_2d(v->vec, MLDSA_L, MLDSA_N, MLD_NTT_BOUND);+ ret = mld_polyvecl_pointwise_acc_montgomery_l7_native(+ w->coeffs, (const int32_t (*)[MLDSA_N])u->vec,+ (const int32_t (*)[MLDSA_N])v->vec);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(w->coeffs, MLDSA_N, MLDSA_Q);+ return;+ }+#endif /* !(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4 && \+ MLD_CONFIG_PARAMETER_SET == 44) && \+ !(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5 && \+ MLD_CONFIG_PARAMETER_SET == 65) && \+ MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7 && \+ MLD_CONFIG_PARAMETER_SET == 87 */+ /* The first input is bounded by [0, MLDSA_Q-1] inclusive.+ * The second input is bounded by [-(MLD_NTT_BOUND-1), MLD_NTT_BOUND-1].+ * Hence, we can safely accumulate in 64-bits without intermediate reductions+ * as MLDSA_L * (MLD_NTT_BOUND-1) * (MLDSA_Q-1) < INT64_MAX.+ *+ * The worst case is ML-DSA-87: 7 * (MLD_NTT_BOUND-1) * (MLDSA_Q-1) < 2**53+ * (and likewise for negative values).+ */+ mld_polyvecl_pointwise_acc_montgomery_c(w, u, v);+}+#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+uint32_t mld_polyvecl_chknorm(const mld_polyvecl *v, int32_t bound)+{+ unsigned int i;+ uint32_t t = 0;+ mld_assert_bound_2d(v->vec, MLDSA_L, MLDSA_N, -MLD_REDUCE32_RANGE_MAX,+ MLD_REDUCE32_RANGE_MAX);++ for (i = 0; i < MLDSA_L; ++i)+ __loop__(+ invariant(i <= MLDSA_L)+ invariant(t == 0 || t == 0xFFFFFFFF)+ invariant((t == 0) == forall(k1, 0, i, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, bound)))+ decreases(MLDSA_L - i)+ )+ {+ /* Reference: Leaks which polynomial violates the bound via a conditional.+ * We are more conservative to reduce the number of declassifications in+ * constant-time testing.+ */+ t |= mld_poly_chknorm(&v->vec[i], bound);+ }+ return t;+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_VERIFY_API || \+ MLD_UNIT_TEST */++/**************************************************************/+/************ Vectors of polynomials of length MLDSA_K **************/+/**************************************************************/+#if (!defined(MLD_CONFIG_NO_SIGN_API) && \+ defined(MLD_CONFIG_REDUCE_RAM)) || \+ defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+void mld_polyveck_reduce(mld_polyveck *v)+{+ unsigned int i;+ mld_assert_bound_2d(v->vec, MLDSA_K, MLDSA_N, INT32_MIN,+ MLD_REDUCE32_DOMAIN_MAX);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(v, sizeof(mld_polyveck)))+ invariant(i <= MLDSA_K)+ invariant(forall(k0, i, MLDSA_K, forall(k1, 0, MLDSA_N, v->vec[k0].coeffs[k1] == loop_entry(*v).vec[k0].coeffs[k1])))+ invariant(forall(k2, 0, i,+ array_bound(v->vec[k2].coeffs, 0, MLDSA_N, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX)))+ decreases(MLDSA_K - i)+ )+ {+ mld_poly_reduce(&v->vec[i]);+ }++ mld_assert_bound_2d(v->vec, MLDSA_K, MLDSA_N, -MLD_REDUCE32_RANGE_MAX,+ MLD_REDUCE32_RANGE_MAX);+}+#endif /* (!MLD_CONFIG_NO_SIGN_API && MLD_CONFIG_REDUCE_RAM) || MLD_UNIT_TEST \+ */++#if !defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+void mld_polyveck_caddq(mld_polyveck *v)+{+ unsigned int i;+ mld_assert_abs_bound_2d(v->vec, MLDSA_K, MLDSA_N, MLDSA_Q);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(v, sizeof(mld_polyveck)))+ invariant(i <= MLDSA_K)+ invariant(forall(k0, i, MLDSA_K, forall(k1, 0, MLDSA_N, v->vec[k0].coeffs[k1] == loop_entry(*v).vec[k0].coeffs[k1])))+ invariant(forall(k1, 0, i, array_bound(v->vec[k1].coeffs, 0, MLDSA_N, 0, MLDSA_Q)))+ decreases(MLDSA_K - i))+ {+ mld_poly_caddq(&v->vec[i]);+ }++ mld_assert_bound_2d(v->vec, MLDSA_K, MLDSA_N, 0, MLDSA_Q);+}+#endif /* !MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST */++#if (!defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+MLD_INTERNAL_API+void mld_polyveck_ntt(mld_polyveck *v)+{+ unsigned int i;+ mld_assert_abs_bound_2d(v->vec, MLDSA_K, MLDSA_N, MLDSA_Q);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(v, sizeof(mld_polyveck)))+ invariant(i <= MLDSA_K)+ invariant(forall(k0, i, MLDSA_K, forall(k1, 0, MLDSA_N, v->vec[k0].coeffs[k1] == loop_entry(*v).vec[k0].coeffs[k1])))+ invariant(forall(k1, 0, i, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ decreases(MLDSA_K - i))+ {+ mld_poly_ntt(&v->vec[i]);+ }+ mld_assert_abs_bound_2d(v->vec, MLDSA_K, MLDSA_N, MLD_NTT_BOUND);+}+#endif /* (!MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST) && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) */++#if !defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+void mld_polyveck_invntt_tomont(mld_polyveck *v)+{+ unsigned int i;+ mld_assert_abs_bound_2d(v->vec, MLDSA_K, MLDSA_N, MLDSA_Q);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(v, sizeof(mld_polyveck)))+ invariant(i <= MLDSA_K)+ invariant(forall(k0, i, MLDSA_K, forall(k1, 0, MLDSA_N, v->vec[k0].coeffs[k1] == loop_entry(*v).vec[k0].coeffs[k1])))+ invariant(forall(k1, 0, i, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, MLD_INTT_BOUND)))+ decreases(MLDSA_K - i))+ {+ mld_poly_invntt_tomont(&v->vec[i]);+ }++ mld_assert_abs_bound_2d(v->vec, MLDSA_K, MLDSA_N, MLD_INTT_BOUND);+}+#endif /* !MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_INTERNAL_API+uint32_t mld_polyveck_chknorm(const mld_polyveck *v, int32_t bound)+{+ unsigned int i;+ uint32_t t = 0;+ mld_assert_bound_2d(v->vec, MLDSA_K, MLDSA_N, -MLD_REDUCE32_RANGE_MAX,+ MLD_REDUCE32_RANGE_MAX);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ invariant(i <= MLDSA_K)+ invariant(t == 0 || t == 0xFFFFFFFF)+ invariant((t == 0) == forall(k1, 0, i, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, bound)))+ decreases(MLDSA_K - i)+ )+ {+ /* Reference: Leaks which polynomial violates the bound via a conditional.+ * We are more conservative to reduce the number of declassifications in+ * constant-time testing.+ */+ t |= mld_poly_chknorm(&v->vec[i], bound);+ }++ return t;+}++#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_INTERNAL_API+void mld_polyveck_decompose(mld_polyveck *v1, mld_polyveck *v0)+{+ unsigned int i;+ mld_assert_bound_2d(v0->vec, MLDSA_K, MLDSA_N, 0, MLDSA_Q);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(v0, sizeof(mld_polyveck)), memory_slice(v1, sizeof(mld_polyveck)))+ invariant(i <= MLDSA_K)+ invariant(forall(k1, 0, i,+ array_bound(v1->vec[k1].coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2))))+ invariant(forall(k2, 0, i,+ array_abs_bound(v0->vec[k2].coeffs, 0, MLDSA_N, MLDSA_GAMMA2+1)))+ invariant(forall(k3, i, MLDSA_K,+ array_bound(v0->vec[k3].coeffs, 0, MLDSA_N, 0, MLDSA_Q)))+ decreases(MLDSA_K - i)+ )+ {+ mld_poly_decompose(&v1->vec[i], &v0->vec[i]);+ }++ mld_assert_bound_2d(v1->vec, MLDSA_K, MLDSA_N, 0,+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));+ mld_assert_abs_bound_2d(v0->vec, MLDSA_K, MLDSA_N, MLDSA_GAMMA2 + 1);+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_INTERNAL_API+void mld_polyveck_pack_w1(uint8_t r[MLDSA_K * MLDSA_POLYW1_PACKEDBYTES],+ const mld_polyveck *w1)+{+ unsigned int i;+ mld_assert_bound_2d(w1->vec, MLDSA_K, MLDSA_N, 0,+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(r, MLDSA_K * MLDSA_POLYW1_PACKEDBYTES))+ invariant(i <= MLDSA_K)+ decreases(MLDSA_K - i)+ )+ {+ mld_polyw1_pack(&r[i * MLDSA_POLYW1_PACKEDBYTES], &w1->vec[i]);+ }+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_INTERNAL_API+void mld_polyveck_pack_eta(uint8_t r[MLDSA_K * MLDSA_POLYETA_PACKEDBYTES],+ const mld_polyveck *p)+{+ unsigned int i;+ mld_assert_abs_bound_2d(p->vec, MLDSA_K, MLDSA_N, MLDSA_ETA + 1);+ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(r, MLDSA_K * MLDSA_POLYETA_PACKEDBYTES))+ invariant(i <= MLDSA_K)+ decreases(MLDSA_K - i)+ )+ {+ mld_polyeta_pack(&r[i * MLDSA_POLYETA_PACKEDBYTES], &p->vec[i]);+ }+}++MLD_INTERNAL_API+void mld_polyvecl_pack_eta(uint8_t r[MLDSA_L * MLDSA_POLYETA_PACKEDBYTES],+ const mld_polyvecl *p)+{+ unsigned int i;+ mld_assert_abs_bound_2d(p->vec, MLDSA_L, MLDSA_N, MLDSA_ETA + 1);+ for (i = 0; i < MLDSA_L; ++i)+ __loop__(+ assigns(i, memory_slice(r, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+ invariant(i <= MLDSA_L)+ decreases(MLDSA_L - i)+ )+ {+ mld_polyeta_pack(&r[i * MLDSA_POLYETA_PACKEDBYTES], &p->vec[i]);+ }+}++#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+MLD_INTERNAL_API+void mld_polyvecl_unpack_eta(+ mld_polyvecl *p, const uint8_t r[MLDSA_L * MLDSA_POLYETA_PACKEDBYTES])+{+ unsigned int i;+ for (i = 0; i < MLDSA_L; ++i)+ {+ mld_polyeta_unpack(&p->vec[i], r + i * MLDSA_POLYETA_PACKEDBYTES);+ }++ mld_assert_bound_2d(p->vec, MLDSA_L, MLDSA_N, MLD_POLYETA_UNPACK_LOWER_BOUND,+ MLDSA_ETA + 1);+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || (!MLD_CONFIG_NO_SIGN_API && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST)) */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_INTERNAL_API+void mld_polyvecl_unpack_z(mld_polyvecl *z,+ const uint8_t r[MLDSA_L * MLDSA_POLYZ_PACKEDBYTES])+{+ unsigned int i;+ for (i = 0; i < MLDSA_L; ++i)+ {+ mld_polyz_unpack(&z->vec[i], r + i * MLDSA_POLYZ_PACKEDBYTES);+ }++ mld_assert_bound_2d(z->vec, MLDSA_L, MLDSA_N, -(MLDSA_GAMMA1 - 1),+ MLDSA_GAMMA1 + 1);+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+MLD_INTERNAL_API+void mld_polyveck_unpack_eta(+ mld_polyveck *p, const uint8_t r[MLDSA_K * MLDSA_POLYETA_PACKEDBYTES])+{+ unsigned int i;+ for (i = 0; i < MLDSA_K; ++i)+ {+ mld_polyeta_unpack(&p->vec[i], r + i * MLDSA_POLYETA_PACKEDBYTES);+ }++ mld_assert_bound_2d(p->vec, MLDSA_K, MLDSA_N, MLD_POLYETA_UNPACK_LOWER_BOUND,+ MLDSA_ETA + 1);+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || (!MLD_CONFIG_NO_SIGN_API && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST)) */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef mld_polyvecl_pointwise_acc_montgomery_c
+ cbits/mldsa/src/polyvec.h view
@@ -0,0 +1,435 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_POLYVEC_H+#define MLD_POLYVEC_H++#include "cbmc.h"+#include "common.h"+#include "poly.h"+#include "poly_kl.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mldsa-native (e.g. with varying parameter sets)+ * within a single compilation unit. */+#define mld_polyvecl MLD_ADD_PARAM_SET(mld_polyvecl)+#define mld_polyveck MLD_ADD_PARAM_SET(mld_polyveck)+/* End of parameter set namespacing */++/** Vector of MLDSA_L polynomials. */+typedef struct+{+ mld_poly vec[MLDSA_L]; /**< Component polynomials. */+} mld_polyvecl;+++#if !defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+#define mld_polyvecl_uniform_gamma1 MLD_NAMESPACE_KL(polyvecl_uniform_gamma1)+/**+ * Sample vector of polynomials with uniformly random coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1] by unpacking output stream of+ * SHAKE256(seed|kappa+i) for component i.+ *+ * @spec{Implements @[FIPS204, Algorithm 34, ExpandMask].}+ *+ * @param[out] v Pointer to output vector.+ * @param[in] seed Byte array with seed of length MLDSA_CRHBYTES.+ * @param kappa Base counter; component i uses kappa + i.+ */+MLD_INTERNAL_API+void mld_polyvecl_uniform_gamma1(mld_polyvecl *v,+ const uint8_t seed[MLDSA_CRHBYTES],+ uint16_t kappa)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyvecl)))+ requires(memory_no_alias(seed, MLDSA_CRHBYTES))+ requires(kappa <= MLD_MAX_KAPPA)+ assigns(memory_slice(v, sizeof(mld_polyvecl)))+ ensures(forall(k0, 0, MLDSA_L,+ array_bound(v->vec[k0].coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1)))+);+#endif /* !MLD_CONFIG_NO_SIGN_API && (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) \+ */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ !defined(MLD_CONFIG_NO_VERIFY_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+#define mld_polyvecl_ntt MLD_NAMESPACE_KL(polyvecl_ntt)+/**+ * Forward NTT of all polynomials in vector of length MLDSA_L. Output+ * coefficients are bounded by MLD_NTT_BOUND in absolute value.+ *+ * @param[in,out] v Pointer to input/output vector.+ */+MLD_INTERNAL_API+void mld_polyvecl_ntt(mld_polyvecl *v)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyvecl)))+ requires(forall(k0, 0, MLDSA_L, array_abs_bound(v->vec[k0].coeffs, 0, MLDSA_N, MLDSA_Q)))+ assigns(memory_slice(v, sizeof(mld_polyvecl)))+ ensures(forall(k1, 0, MLDSA_L, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_VERIFY_API || \+ (!MLD_CONFIG_NO_SIGN_API && (!MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST)) */++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+#define mld_polyvecl_pointwise_acc_montgomery \+ MLD_NAMESPACE_KL(polyvecl_pointwise_acc_montgomery)+/**+ * Pointwise multiply vectors of polynomials of length MLDSA_L, multiply+ * resulting vector by 2^{-32} and add (accumulate) polynomials in it.+ * Input/output vectors are in NTT domain representation.+ *+ * The first input "u" must be the output of polyvec_matrix_expand() and so+ * have coefficients in [0, MLDSA_Q-1] inclusive.+ *+ * The second input "v" is assumed to be output of an NTT, and hence must have+ * coefficients bounded by [-(MLD_NTT_BOUND-1), MLD_NTT_BOUND-1] inclusive.+ *+ * @spec{Partially implements @[FIPS204, Algorithm 48, MatrixVectorNTT]+ * (one output polynomial; multiply-accumulate of two NTT-domain vectors).}+ *+ * @param[out] w Output polynomial.+ * @param[in] u Pointer to first input vector.+ * @param[in] v Pointer to second input vector.+ */+MLD_INTERNAL_API+void mld_polyvecl_pointwise_acc_montgomery(mld_poly *w, const mld_polyvecl *u,+ const mld_polyvecl *v)+__contract__(+ requires(memory_no_alias(w, sizeof(mld_poly)))+ requires(memory_no_alias(u, sizeof(mld_polyvecl)))+ requires(memory_no_alias(v, sizeof(mld_polyvecl)))+ requires(forall(l0, 0, MLDSA_L,+ array_bound(u->vec[l0].coeffs, 0, MLDSA_N, 0, MLDSA_Q)))+ requires(forall(l1, 0, MLDSA_L,+ array_abs_bound(v->vec[l1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ assigns(memory_slice(w, sizeof(mld_poly)))+ ensures(array_abs_bound(w->coeffs, 0, MLDSA_N, MLDSA_Q))+);+#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_polyvecl_chknorm MLD_NAMESPACE_KL(polyvecl_chknorm)+/**+ * Check infinity norm of polynomials in vector of length MLDSA_L. Assumes+ * input mld_polyvecl to be reduced by polyvecl_reduce().+ *+ * @param[in] v Pointer to vector.+ * @param B Norm bound.+ *+ * @return 0 if norm of all polynomials is strictly smaller than+ * B <= (MLDSA_Q-1)/8 and 0xFFFFFFFF otherwise.+ */+MLD_INTERNAL_API+MLD_MUST_CHECK_RETURN_VALUE+uint32_t mld_polyvecl_chknorm(const mld_polyvecl *v, int32_t B)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyvecl)))+ requires(0 <= B && B <= (MLDSA_Q - 1) / 8)+ requires(forall(k0, 0, MLDSA_L,+ array_bound(v->vec[k0].coeffs, 0, MLDSA_N, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX)))+ ensures(return_value == 0 || return_value == 0xFFFFFFFF)+ ensures((return_value == 0) == forall(k1, 0, MLDSA_L, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, B)))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_VERIFY_API */++/** Vector of MLDSA_K polynomials. */+typedef struct+{+ mld_poly vec[MLDSA_K]; /**< Component polynomials. */+} mld_polyveck;++#if (!defined(MLD_CONFIG_NO_SIGN_API) && defined(MLD_CONFIG_REDUCE_RAM)) || \+ defined(MLD_UNIT_TEST)+#define mld_polyveck_reduce MLD_NAMESPACE_KL(polyveck_reduce)+/**+ * Reduce coefficients of polynomials in vector of length MLDSA_K to+ * representatives in [-MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX].+ *+ * @param[in,out] v Pointer to input/output vector.+ */+MLD_INTERNAL_API+void mld_polyveck_reduce(mld_polyveck *v)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyveck)))+ requires(forall(k0, 0, MLDSA_K,+ array_bound(v->vec[k0].coeffs, 0, MLDSA_N, INT32_MIN, MLD_REDUCE32_DOMAIN_MAX)))+ assigns(memory_slice(v, sizeof(mld_polyveck)))+ ensures(forall(k1, 0, MLDSA_K,+ array_bound(v->vec[k1].coeffs, 0, MLDSA_N, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX)))+);+#endif /* (!MLD_CONFIG_NO_SIGN_API && MLD_CONFIG_REDUCE_RAM) || MLD_UNIT_TEST \+ */++#if !defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)+#define mld_polyveck_caddq MLD_NAMESPACE_KL(polyveck_caddq)+/**+ * For all coefficients of polynomials in vector of length MLDSA_K add MLDSA_Q+ * if coefficient is negative.+ *+ * @param[in,out] v Pointer to input/output vector.+ */+MLD_INTERNAL_API+void mld_polyveck_caddq(mld_polyveck *v)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyveck)))+ requires(forall(k0, 0, MLDSA_K,+ array_abs_bound(v->vec[k0].coeffs, 0, MLDSA_N, MLDSA_Q)))+ assigns(memory_slice(v, sizeof(mld_polyveck)))+ ensures(forall(k1, 0, MLDSA_K,+ array_bound(v->vec[k1].coeffs, 0, MLDSA_N, 0, MLDSA_Q)))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST */++#if (!defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+#define mld_polyveck_ntt MLD_NAMESPACE_KL(polyveck_ntt)+/**+ * Forward NTT of all polynomials in vector of length MLDSA_K. Output+ * coefficients are bounded by MLD_NTT_BOUND in absolute value.+ *+ * @param[in,out] v Pointer to input/output vector.+ */+MLD_INTERNAL_API+void mld_polyveck_ntt(mld_polyveck *v)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyveck)))+ requires(forall(k0, 0, MLDSA_K, array_abs_bound(v->vec[k0].coeffs, 0, MLDSA_N, MLDSA_Q)))+ assigns(memory_slice(v, sizeof(mld_polyveck)))+ ensures(forall(k1, 0, MLDSA_K, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+);+#endif /* (!MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST) && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) */++#if !defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)+#define mld_polyveck_invntt_tomont MLD_NAMESPACE_KL(polyveck_invntt_tomont)+/**+ * Inverse NTT and multiplication by 2^{32} of polynomials in vector of+ * length MLDSA_K.+ *+ * Input coefficients need to be less than MLDSA_Q, and output coefficients+ * are bounded by MLD_INTT_BOUND.+ *+ * @param[in,out] v Pointer to input/output vector.+ */+MLD_INTERNAL_API+void mld_polyveck_invntt_tomont(mld_polyveck *v)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyveck)))+ requires(forall(k0, 0, MLDSA_K, array_abs_bound(v->vec[k0].coeffs, 0, MLDSA_N, MLDSA_Q)))+ assigns(memory_slice(v, sizeof(mld_polyveck)))+ ensures(forall(k1, 0, MLDSA_K, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, MLD_INTT_BOUND)))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_polyveck_chknorm MLD_NAMESPACE_KL(polyveck_chknorm)+/**+ * Check infinity norm of polynomials in vector of length MLDSA_K. Assumes+ * input mld_polyveck to be reduced by polyveck_reduce().+ *+ * @param[in] v Pointer to vector.+ * @param B Norm bound.+ *+ * @return 0 if norm of all polynomials are strictly smaller than+ * B <= (MLDSA_Q-1)/8 and 0xFFFFFFFF otherwise.+ */+MLD_INTERNAL_API+MLD_MUST_CHECK_RETURN_VALUE+uint32_t mld_polyveck_chknorm(const mld_polyveck *v, int32_t B)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyveck)))+ requires(0 <= B && B <= (MLDSA_Q - 1) / 8)+ requires(forall(k0, 0, MLDSA_K,+ array_bound(v->vec[k0].coeffs, 0, MLDSA_N,+ -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX)))+ ensures(return_value == 0 || return_value == 0xFFFFFFFF)+ ensures((return_value == 0) == forall(k1, 0, MLDSA_K, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, B)))+);++#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_polyveck_decompose MLD_NAMESPACE_KL(polyveck_decompose)+/**+ * For all coefficients a of polynomials in vector of length MLDSA_K, compute+ * high and low bits a0, a1 such a mod^+ MLDSA_Q = a1*ALPHA + a0 with+ * -ALPHA/2 < a0 <= ALPHA/2 except a1 = (MLDSA_Q-1)/ALPHA where we set+ * a1 = 0 and -ALPHA/2 <= a0 = a mod MLDSA_Q - MLDSA_Q < 0. Assumes+ * coefficients to be standard representatives.+ *+ * @reference{The reference implementation has the input polynomial as a+ * separate argument that may be aliased with either of the outputs. Removing+ * the aliasing eases CBMC proofs.}+ *+ * @param[out] v1 Pointer to output vector of polynomials with+ * coefficients a1.+ * @param[in,out] v0 Pointer to input/output vector of polynomials. Output+ * polynomial has coefficients a0.+ */+MLD_INTERNAL_API+void mld_polyveck_decompose(mld_polyveck *v1, mld_polyveck *v0)+__contract__(+ requires(memory_no_alias(v1, sizeof(mld_polyveck)))+ requires(memory_no_alias(v0, sizeof(mld_polyveck)))+ requires(forall(k0, 0, MLDSA_K,+ array_bound(v0->vec[k0].coeffs, 0, MLDSA_N, 0, MLDSA_Q)))+ assigns(memory_slice(v1, sizeof(mld_polyveck)))+ assigns(memory_slice(v0, sizeof(mld_polyveck)))+ ensures(forall(k1, 0, MLDSA_K,+ array_bound(v1->vec[k1].coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2))))+ ensures(forall(k2, 0, MLDSA_K,+ array_abs_bound(v0->vec[k2].coeffs, 0, MLDSA_N, MLDSA_GAMMA2+1)))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_polyveck_pack_w1 MLD_NAMESPACE_KL(polyveck_pack_w1)+/**+ * Bit-pack polynomial vector w1 with coefficients in [0, 15] or [0, 43]. Input+ * coefficients are assumed to be standard representatives.+ *+ * @spec{Implements @[FIPS204, Algorithm 28, w1Encode].}+ *+ * @param[out] r Pointer to output byte array with at least+ * MLDSA_K * MLDSA_POLYW1_PACKEDBYTES bytes.+ * @param[in] w1 Pointer to input polynomial vector.+ */+MLD_INTERNAL_API+void mld_polyveck_pack_w1(uint8_t r[MLDSA_K * MLDSA_POLYW1_PACKEDBYTES],+ const mld_polyveck *w1)+__contract__(+ requires(memory_no_alias(r, MLDSA_K * MLDSA_POLYW1_PACKEDBYTES))+ requires(memory_no_alias(w1, sizeof(mld_polyveck)))+ requires(forall(k1, 0, MLDSA_K,+ array_bound(w1->vec[k1].coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2))))+ assigns(memory_slice(r, MLDSA_K * MLDSA_POLYW1_PACKEDBYTES))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_polyveck_pack_eta MLD_NAMESPACE_KL(polyveck_pack_eta)+/**+ * Bit-pack polynomial vector with coefficients in [-MLDSA_ETA, MLDSA_ETA].+ *+ * @param[out] r Pointer to output byte array with+ * MLDSA_K * MLDSA_POLYETA_PACKEDBYTES bytes.+ * @param[in] p Pointer to input polynomial vector.+ */+MLD_INTERNAL_API+void mld_polyveck_pack_eta(uint8_t r[MLDSA_K * MLDSA_POLYETA_PACKEDBYTES],+ const mld_polyveck *p)+__contract__(+ requires(memory_no_alias(r, MLDSA_K * MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(p, sizeof(mld_polyveck)))+ requires(forall(k1, 0, MLDSA_K,+ array_abs_bound(p->vec[k1].coeffs, 0, MLDSA_N, MLDSA_ETA + 1)))+ assigns(memory_slice(r, MLDSA_K * MLDSA_POLYETA_PACKEDBYTES))+);++#define mld_polyvecl_pack_eta MLD_NAMESPACE_KL(polyvecl_pack_eta)+/**+ * Bit-pack polynomial vector with coefficients in [-MLDSA_ETA, MLDSA_ETA].+ *+ * @param[out] r Pointer to output byte array with+ * MLDSA_L * MLDSA_POLYETA_PACKEDBYTES bytes.+ * @param[in] p Pointer to input polynomial vector.+ */+MLD_INTERNAL_API+void mld_polyvecl_pack_eta(uint8_t r[MLDSA_L * MLDSA_POLYETA_PACKEDBYTES],+ const mld_polyvecl *p)+__contract__(+ requires(memory_no_alias(r, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(p, sizeof(mld_polyvecl)))+ requires(forall(k1, 0, MLDSA_L,+ array_abs_bound(p->vec[k1].coeffs, 0, MLDSA_N, MLDSA_ETA + 1)))+ assigns(memory_slice(r, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+);++#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+#define mld_polyvecl_unpack_eta MLD_NAMESPACE_KL(polyvecl_unpack_eta)+/**+ * Unpack polynomial vector with coefficients in [-MLDSA_ETA, MLDSA_ETA].+ *+ * @param[out] p Pointer to output polynomial vector.+ * @param[in] r Input byte array with bit-packed polynomial vector.+ */+MLD_INTERNAL_API+void mld_polyvecl_unpack_eta(+ mld_polyvecl *p, const uint8_t r[MLDSA_L * MLDSA_POLYETA_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(p, sizeof(mld_polyvecl)))+ assigns(memory_slice(p, sizeof(mld_polyvecl)))+ ensures(forall(k1, 0, MLDSA_L,+ array_bound(p->vec[k1].coeffs, 0, MLDSA_N, MLD_POLYETA_UNPACK_LOWER_BOUND, MLDSA_ETA + 1)))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || (!MLD_CONFIG_NO_SIGN_API && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST)) */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_polyvecl_unpack_z MLD_NAMESPACE_KL(polyvecl_unpack_z)+/**+ * Unpack polynomial vector with coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1].+ *+ * @param[out] z Pointer to output polynomial vector.+ * @param[in] r Input byte array with bit-packed polynomial vector.+ */+MLD_INTERNAL_API+void mld_polyvecl_unpack_z(mld_polyvecl *z,+ const uint8_t r[MLDSA_L * MLDSA_POLYZ_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, MLDSA_L * MLDSA_POLYZ_PACKEDBYTES))+ requires(memory_no_alias(z, sizeof(mld_polyvecl)))+ assigns(memory_slice(z, sizeof(mld_polyvecl)))+ ensures(forall(k1, 0, MLDSA_L,+ array_bound(z->vec[k1].coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1)))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+#define mld_polyveck_unpack_eta MLD_NAMESPACE_KL(polyveck_unpack_eta)+/**+ * Unpack polynomial vector with coefficients in [-MLDSA_ETA, MLDSA_ETA].+ *+ * @param[out] p Pointer to output polynomial vector.+ * @param[in] r Input byte array with bit-packed polynomial vector.+ */+MLD_INTERNAL_API+void mld_polyveck_unpack_eta(+ mld_polyveck *p, const uint8_t r[MLDSA_K * MLDSA_POLYETA_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, MLDSA_K * MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(p, sizeof(mld_polyveck)))+ assigns(memory_slice(p, sizeof(mld_polyveck)))+ ensures(forall(k1, 0, MLDSA_K,+ array_bound(p->vec[k1].coeffs, 0, MLDSA_N, MLD_POLYETA_UNPACK_LOWER_BOUND, MLDSA_ETA + 1)))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || (!MLD_CONFIG_NO_SIGN_API && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST)) */+++#endif /* !MLD_POLYVEC_H */
+ cbits/mldsa/src/polyvec_lazy.c view
@@ -0,0 +1,311 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#include "polyvec_lazy.h"++#include "debug.h"++/* This namespacing is not done at the top to avoid a naming conflict+ * with native backends, which are currently not yet namespaced. */+#define mld_polymat_expand_entry MLD_ADD_PARAM_SET(mld_polymat_expand_entry)++/**+ * Sample a single matrix entry A[k][l] of ExpandA(rho) by rejection sampling+ * from SHAKE128(rho|l|k), and apply the custom-order permutation when a+ * native NTT backend is in use.+ *+ * The caller is expected to have copied rho into the first MLDSA_SEEDBYTES+ * of seed_ext. This function writes the domain-separation bytes+ * seed_ext[SEEDBYTES..+2] = {l, k} before sampling.+ *+ * @spec{Partially implements @[FIPS204, Algorithm 32, ExpandA] (samples one+ * matrix entry via @[FIPS204, Algorithm 30, RejNTTPoly]).}+ *+ * @param[out] p Pointer to output polynomial.+ * @param[in,out] seed_ext Seed buffer pre-filled with rho in the first+ * MLDSA_SEEDBYTES; the final two bytes are+ * overwritten.+ * @param l Column index (inner, aka nonce low byte).+ * @param k Row index (outer, aka nonce high byte).+ */+static MLD_INLINE void mld_polymat_expand_entry(+ mld_poly *p, uint8_t seed_ext[MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)], uint8_t l,+ uint8_t k)+__contract__(+ requires(memory_no_alias(p, sizeof(mld_poly)))+ requires(memory_no_alias(seed_ext, MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)))+ assigns(memory_slice(p, sizeof(mld_poly)))+ assigns(memory_slice(seed_ext, MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)))+ ensures(array_bound(p->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+)+{+ seed_ext[MLDSA_SEEDBYTES + 0] = l;+ seed_ext[MLDSA_SEEDBYTES + 1] = k;+ mld_poly_uniform(p, seed_ext);+ mld_poly_permute_bitrev_to_custom_optional(p);+}++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)++MLD_INTERNAL_API+void mld_polyvec_matrix_expand_eager(mld_polymat_eager *mat,+ const uint8_t rho[MLDSA_SEEDBYTES])+{+ unsigned int i, j;+ MLD_ALIGN uint8_t seed_ext[4][MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)];++ for (j = 0; j < 4; j++)+ __loop__(+ assigns(j, object_whole(seed_ext))+ invariant(j <= 4)+ decreases(4 - j)+ )+ {+ mld_memcpy(seed_ext[j], rho, MLDSA_SEEDBYTES);+ }++#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+ /* Sample 4 matrix entries a time. */+ for (i = 0; i < (MLDSA_K * MLDSA_L / 4) * 4; i += 4)+ __loop__(+ assigns(i, j, object_whole(seed_ext), memory_slice(mat, sizeof(mld_polymat_eager)))+ invariant(i <= (MLDSA_K * MLDSA_L / 4) * 4 && i % 4 == 0)+ /* vectors 0 .. i / MLDSA_L are completely sampled */+ invariant(forall(k1, 0, i / MLDSA_L, forall(l1, 0, MLDSA_L,+ array_bound(mat->vec[k1].vec[l1].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+ /* last vector is sampled up to i % MLDSA_L */+ invariant(forall(k2, i / MLDSA_L, i / MLDSA_L + 1, forall(l2, 0, i % MLDSA_L,+ array_bound(mat->vec[k2].vec[l2].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+ decreases((MLDSA_K * MLDSA_L / 4) * 4 - i)+ )+ {+ for (j = 0; j < 4; j++)+ __loop__(+ assigns(j, object_whole(seed_ext))+ invariant(j <= 4)+ decreases(4 - j)+ )+ {+ uint8_t x = (uint8_t)((i + j) / MLDSA_L);+ uint8_t y = (uint8_t)((i + j) % MLDSA_L);++ seed_ext[j][MLDSA_SEEDBYTES + 0] = y;+ seed_ext[j][MLDSA_SEEDBYTES + 1] = x;+ }++ mld_poly_uniform_4x(&mat->vec[i / MLDSA_L].vec[i % MLDSA_L],+ &mat->vec[(i + 1) / MLDSA_L].vec[(i + 1) % MLDSA_L],+ &mat->vec[(i + 2) / MLDSA_L].vec[(i + 2) % MLDSA_L],+ &mat->vec[(i + 3) / MLDSA_L].vec[(i + 3) % MLDSA_L],+ seed_ext);+ mld_poly_permute_bitrev_to_custom_optional(+ &mat->vec[i / MLDSA_L].vec[i % MLDSA_L]);+ mld_poly_permute_bitrev_to_custom_optional(+ &mat->vec[(i + 1) / MLDSA_L].vec[(i + 1) % MLDSA_L]);+ mld_poly_permute_bitrev_to_custom_optional(+ &mat->vec[(i + 2) / MLDSA_L].vec[(i + 2) % MLDSA_L]);+ mld_poly_permute_bitrev_to_custom_optional(+ &mat->vec[(i + 3) / MLDSA_L].vec[(i + 3) % MLDSA_L]);+ }+#else /* !MLD_CONFIG_SERIAL_FIPS202_ONLY */+ i = 0;+#endif /* MLD_CONFIG_SERIAL_FIPS202_ONLY */++ /* Entries omitted by the batch-sampling are sampled individually. */+ while (i < MLDSA_K * MLDSA_L)+ __loop__(+ assigns(i, object_whole(seed_ext), memory_slice(mat, sizeof(mld_polymat_eager)))+ invariant(i <= MLDSA_K * MLDSA_L)+ /* vectors 0 .. i / MLDSA_L are completely sampled */+ invariant(forall(k1, 0, i / MLDSA_L, forall(l1, 0, MLDSA_L,+ array_bound(mat->vec[k1].vec[l1].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+ /* last vector is sampled up to i % MLDSA_L */+ invariant(forall(k2, i / MLDSA_L, i / MLDSA_L + 1, forall(l2, 0, i % MLDSA_L,+ array_bound(mat->vec[k2].vec[l2].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+ decreases(MLDSA_K * MLDSA_L - i)+ )+ {+ uint8_t x = (uint8_t)(i / MLDSA_L);+ uint8_t y = (uint8_t)(i % MLDSA_L);+ mld_polymat_expand_entry(&mat->vec[x].vec[y], seed_ext[0], y, x);+ i++;+ }++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(seed_ext, sizeof(seed_ext));+}++MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_row_eager(mld_poly *t_row,+ mld_polymat_eager *mat,+ const mld_polyvecl *v,+ unsigned int i)+{+ mld_polyvecl_pointwise_acc_montgomery(t_row, &mat->vec[i], v);+}++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_yvec_eager(mld_polyveck *w,+ mld_polymat_eager *mat,+ const mld_yvec_eager *y,+ mld_polyvecl *scratch)+{+ unsigned int i;+ *scratch = y->vec;+ mld_polyvecl_ntt(scratch);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(w, sizeof(mld_polyveck)))+ invariant(i <= MLDSA_K)+ invariant(forall(k0, 0, i,+ array_abs_bound(w->vec[k0].coeffs, 0, MLDSA_N, MLDSA_Q)))+ decreases(MLDSA_K - i)+ )+ {+ mld_polyvec_matrix_pointwise_montgomery_row_eager(&w->vec[i], mat, scratch,+ i);+ }++ mld_polyveck_invntt_tomont(w);+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)++MLD_INTERNAL_API+void mld_polyvec_matrix_expand_lazy(mld_polymat_lazy *mat,+ const uint8_t rho[MLDSA_SEEDBYTES])+{+ mld_memcpy(mat->rho, rho, MLDSA_SEEDBYTES);+}++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_row_lazy(mld_poly *t_row,+ mld_polymat_lazy *mat,+ const mld_polyvecl *v,+ unsigned int i)+{+ unsigned int l;+ MLD_ALIGN uint8_t seed_ext[MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)];+ mld_memcpy(seed_ext, mat->rho, MLDSA_SEEDBYTES);++ mld_polymat_expand_entry(t_row, seed_ext, 0, (uint8_t)i);+ mld_poly_pointwise_montgomery(t_row, &v->vec[0]);++ for (l = 1; l < MLDSA_L; ++l)+ __loop__(+ assigns(l, object_whole(seed_ext),+ memory_slice(t_row, sizeof(mld_poly)),+ memory_slice(mat, sizeof(mld_polymat_lazy)))+ invariant(l >= 1 && l <= MLDSA_L)+ invariant(array_abs_bound(t_row->coeffs, 0, MLDSA_N, l * MLDSA_Q))+ decreases(MLDSA_L - l)+ )+ {+ mld_polymat_expand_entry(&mat->cur, seed_ext, (uint8_t)l, (uint8_t)i);+ mld_poly_pointwise_montgomery(&mat->cur, &v->vec[l]);+ mld_poly_add(t_row, &mat->cur);+ }+ mld_poly_reduce(t_row);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(seed_ext, sizeof(seed_ext));+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_yvec_lazy(mld_polyveck *w,+ mld_polymat_lazy *mat,+ const mld_yvec_lazy *y,+ mld_polyvecl *scratch)+{+ unsigned int k, l;+ MLD_ALIGN uint8_t seed_ext[MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)];+ /* Only the first poly of the polyvecl scratch is used. The polyvecl type+ * matches the eager variant for API uniformity; in REDUCE_RAM mode the+ * polyvecl storage is provided "for free" by the caller's polyveck/polyvecl+ * union. */+ mld_poly *y_ntt = &scratch->vec[0];++ mld_memcpy(seed_ext, mat->rho, MLDSA_SEEDBYTES);++ /* Column-by-column: sample y[l], NTT, accumulate column l of A into w. */+ for (l = 0; l < MLDSA_L; l++)+ __loop__(+ assigns(k, l, object_whole(seed_ext),+ memory_slice(w, sizeof(mld_polyveck)),+ memory_slice(mat, sizeof(mld_polymat_lazy)),+ memory_slice(scratch, sizeof(mld_polyvecl)))+ invariant(l <= MLDSA_L)+ invariant(l == 0 ||+ forall(k0, 0, MLDSA_K,+ array_abs_bound(w->vec[k0].coeffs, 0, MLDSA_N,+ (int)l * MLDSA_Q)))+ decreases(MLDSA_L - l)+ )+ {+ mld_yvec_get_poly_lazy(y_ntt, y, l);+ mld_poly_ntt(y_ntt);+ for (k = 0; k < MLDSA_K; k++)+ __loop__(+ assigns(k, object_whole(seed_ext),+ memory_slice(w, sizeof(mld_polyveck)),+ memory_slice(mat, sizeof(mld_polymat_lazy)))+ invariant(k <= MLDSA_K)+ invariant(l != 0 ||+ forall(k1, 0, k,+ array_abs_bound(w->vec[k1].coeffs, 0, MLDSA_N, MLDSA_Q)))+ invariant(l == 0 ||+ forall(k2, 0, k,+ array_abs_bound(w->vec[k2].coeffs, 0, MLDSA_N,+ ((int)l + 1) * MLDSA_Q)))+ invariant(l == 0 ||+ forall(k3, k, MLDSA_K,+ array_abs_bound(w->vec[k3].coeffs, 0, MLDSA_N,+ (int)l * MLDSA_Q)))+ decreases(MLDSA_K - k)+ )+ {+ if (l == 0)+ {+ mld_polymat_expand_entry(&w->vec[k], seed_ext, 0, (uint8_t)k);+ mld_poly_pointwise_montgomery(&w->vec[k], y_ntt);+ }+ else+ {+ mld_polymat_expand_entry(&mat->cur, seed_ext, (uint8_t)l, (uint8_t)k);+ mld_poly_pointwise_montgomery(&mat->cur, y_ntt);+ mld_poly_add(&w->vec[k], &mat->cur);+ }+ }+ }++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(seed_ext, sizeof(seed_ext));+ mld_polyveck_reduce(w);+ mld_polyveck_invntt_tomont(w);+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#endif /* MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef mld_polymat_expand_entry
+ cbits/mldsa/src/polyvec_lazy.h view
@@ -0,0 +1,652 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++/*+ * Eager and lazy variants of polynomial vector types.+ *+ * In eager mode, full vectors are precomputed and stored in memory.+ * In lazy mode, data is stored in packed form and expanded on demand,+ * trading computation for reduced memory usage.+ *+ * MLD_CONFIG_REDUCE_RAM selects which variant is used.+ */++#ifndef MLD_POLYVEC_LAZY_H+#define MLD_POLYVEC_LAZY_H++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_SIGN_API) || \+ !defined(MLD_CONFIG_NO_VERIFY_API)++#include "poly.h"+#include "poly_kl.h"+#include "polyvec.h"++/* Parameter set namespacing */+#define mld_sk_s1hat_eager MLD_ADD_PARAM_SET(mld_sk_s1hat_eager)+#define mld_sk_s1hat_lazy MLD_ADD_PARAM_SET(mld_sk_s1hat_lazy)+#define mld_sk_s1hat MLD_ADD_PARAM_SET(mld_sk_s1hat)+#define mld_unpack_sk_s1hat_eager MLD_ADD_PARAM_SET(mld_unpack_sk_s1hat_eager)+#define mld_unpack_sk_s1hat_lazy MLD_ADD_PARAM_SET(mld_unpack_sk_s1hat_lazy)+#define mld_sk_s1hat_get_poly_eager \+ MLD_ADD_PARAM_SET(mld_sk_s1hat_get_poly_eager)+#define mld_sk_s1hat_get_poly_lazy MLD_ADD_PARAM_SET(mld_sk_s1hat_get_poly_lazy)+#define mld_sk_s2hat_eager MLD_ADD_PARAM_SET(mld_sk_s2hat_eager)+#define mld_sk_s2hat_lazy MLD_ADD_PARAM_SET(mld_sk_s2hat_lazy)+#define mld_sk_s2hat MLD_ADD_PARAM_SET(mld_sk_s2hat)+#define mld_unpack_sk_s2hat_eager MLD_ADD_PARAM_SET(mld_unpack_sk_s2hat_eager)+#define mld_unpack_sk_s2hat_lazy MLD_ADD_PARAM_SET(mld_unpack_sk_s2hat_lazy)+#define mld_sk_s2hat_get_poly_eager \+ MLD_ADD_PARAM_SET(mld_sk_s2hat_get_poly_eager)+#define mld_sk_s2hat_get_poly_lazy MLD_ADD_PARAM_SET(mld_sk_s2hat_get_poly_lazy)+#define mld_sk_t0hat_eager MLD_ADD_PARAM_SET(mld_sk_t0hat_eager)+#define mld_sk_t0hat_lazy MLD_ADD_PARAM_SET(mld_sk_t0hat_lazy)+#define mld_sk_t0hat MLD_ADD_PARAM_SET(mld_sk_t0hat)+#define mld_unpack_sk_t0hat_eager MLD_ADD_PARAM_SET(mld_unpack_sk_t0hat_eager)+#define mld_unpack_sk_t0hat_lazy MLD_ADD_PARAM_SET(mld_unpack_sk_t0hat_lazy)+#define mld_sk_t0hat_get_poly_eager \+ MLD_ADD_PARAM_SET(mld_sk_t0hat_get_poly_eager)+#define mld_sk_t0hat_get_poly_lazy MLD_ADD_PARAM_SET(mld_sk_t0hat_get_poly_lazy)+#define mld_polymat MLD_ADD_PARAM_SET(mld_polymat)+#define mld_polymat_eager MLD_ADD_PARAM_SET(mld_polymat_eager)+#define mld_polymat_lazy MLD_ADD_PARAM_SET(mld_polymat_lazy)+#define mld_poly_permute_bitrev_to_custom_optional \+ MLD_ADD_PARAM_SET(mld_poly_permute_bitrev_to_custom_optional)+#define mld_polyvec_matrix_expand_eager \+ MLD_NAMESPACE_KL(polyvec_matrix_expand_eager)+#define mld_polyvec_matrix_expand_lazy \+ MLD_NAMESPACE_KL(polyvec_matrix_expand_lazy)+#define mld_polyvec_matrix_pointwise_montgomery \+ MLD_NAMESPACE_KL(polyvec_matrix_pointwise_montgomery)+#define mld_polyvec_matrix_pointwise_montgomery_row_eager \+ MLD_NAMESPACE_KL(polyvec_matrix_pointwise_montgomery_row_eager)+#define mld_polyvec_matrix_pointwise_montgomery_row_lazy \+ MLD_NAMESPACE_KL(polyvec_matrix_pointwise_montgomery_row_lazy)+#define mld_polyvec_matrix_pointwise_montgomery_yvec_eager \+ MLD_NAMESPACE_KL(polyvec_matrix_pointwise_montgomery_yvec_eager)+#define mld_polyvec_matrix_pointwise_montgomery_yvec_lazy \+ MLD_NAMESPACE_KL(polyvec_matrix_pointwise_montgomery_yvec_lazy)+#define mld_yvec_eager MLD_ADD_PARAM_SET(mld_yvec_eager)+#define mld_yvec_lazy MLD_ADD_PARAM_SET(mld_yvec_lazy)+#define mld_yvec MLD_ADD_PARAM_SET(mld_yvec)+#define mld_yvec_init_eager MLD_ADD_PARAM_SET(mld_yvec_init_eager)+#define mld_yvec_init_lazy MLD_ADD_PARAM_SET(mld_yvec_init_lazy)+#define mld_yvec_get_poly_eager MLD_ADD_PARAM_SET(mld_yvec_get_poly_eager)+#define mld_yvec_get_poly_lazy MLD_ADD_PARAM_SET(mld_yvec_get_poly_lazy)+/* End of parameter set namespacing */++/** Eager s1hat: precomputed s1 vector in NTT domain. */+typedef struct+{+ mld_polyvecl vec; /**< s1 vector in NTT domain. */+} mld_sk_s1hat_eager;++/** Eager s2hat: precomputed s2 vector in NTT domain. */+typedef struct+{+ mld_polyveck vec; /**< s2 vector in NTT domain. */+} mld_sk_s2hat_eager;++/** Eager t0hat: precomputed t0 vector in NTT domain. */+typedef struct+{+ mld_polyveck vec; /**< t0 vector in NTT domain. */+} mld_sk_t0hat_eager;++/** Lazy s1hat: borrow packed s1, unpack and convert to NTT domain on demand. */+typedef struct+{+ const uint8_t *packed; /**< Pointer to packed s1 in the secret key. */+} mld_sk_s1hat_lazy;++/** Lazy s2hat: borrow packed s2, unpack and convert to NTT domain on demand. */+typedef struct+{+ const uint8_t *packed; /**< Pointer to packed s2 in the secret key. */+} mld_sk_s2hat_lazy;++/** Lazy t0hat: borrow packed t0, unpack and convert to NTT domain on demand. */+typedef struct+{+ const uint8_t *packed; /**< Pointer to packed t0 in the secret key. */+} mld_sk_t0hat_lazy;++/** Eager yvec: precomputed and stored full signing masking vector y. */+typedef struct+{+ mld_polyvecl vec; /**< Masking vector y. */+} mld_yvec_eager;++/** Lazy yvec: store seed and base counter kappa, regenerate y[i] on demand. */+typedef struct+{+ const uint8_t *rhoprime; /**< Pointer to seed used to derive y. */+ uint16_t kappa; /**< Base counter; component i uses kappa + i. */+} mld_yvec_lazy;++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_SIGN_API)+/* s1vec */++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+static MLD_INLINE void mld_unpack_sk_s1hat_eager(+ mld_sk_s1hat_eager *s1,+ const uint8_t packed_s1[MLDSA_L * MLDSA_POLYETA_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(s1, sizeof(mld_sk_s1hat_eager)))+ requires(memory_no_alias(packed_s1, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+ assigns(memory_slice(s1, sizeof(mld_sk_s1hat_eager)))+ ensures(forall(k1, 0, MLDSA_L,+ array_abs_bound(s1->vec.vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+)+{+ mld_polyvecl_unpack_eta(&s1->vec, packed_s1);+ mld_polyvecl_ntt(&s1->vec);+}++#if !defined(MLD_CONFIG_NO_SIGN_API)+static MLD_INLINE void mld_sk_s1hat_get_poly_eager(mld_poly *buf,+ const mld_sk_s1hat_eager *s1,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(s1, sizeof(mld_sk_s1hat_eager)))+ requires(i < MLDSA_L)+ requires(array_abs_bound(s1->vec.vec[i].coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_abs_bound(buf->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+) { *buf = s1->vec.vec[i]; }+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */+#if defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+static MLD_INLINE void mld_unpack_sk_s1hat_lazy(+ mld_sk_s1hat_lazy *s1,+ const uint8_t packed_s1[MLDSA_L * MLDSA_POLYETA_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(s1, sizeof(mld_sk_s1hat_lazy)))+ assigns(memory_slice(s1, sizeof(mld_sk_s1hat_lazy)))+ ensures(s1->packed == old(packed_s1))+) { s1->packed = packed_s1; }++#if !defined(MLD_CONFIG_NO_SIGN_API)+static MLD_INLINE void mld_sk_s1hat_get_poly_lazy(mld_poly *buf,+ const mld_sk_s1hat_lazy *s1,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(s1, sizeof(mld_sk_s1hat_lazy)))+ requires(i < MLDSA_L)+ requires(memory_no_alias(s1->packed, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_abs_bound(buf->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+)+{+ mld_polyeta_unpack(buf, s1->packed + i * MLDSA_POLYETA_PACKEDBYTES);+ mld_poly_ntt(buf);+}+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++/* s2vec */++#if (!defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+static MLD_INLINE void mld_unpack_sk_s2hat_eager(+ mld_sk_s2hat_eager *s2,+ const uint8_t packed_s2[MLDSA_K * MLDSA_POLYETA_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(s2, sizeof(mld_sk_s2hat_eager)))+ requires(memory_no_alias(packed_s2, MLDSA_K * MLDSA_POLYETA_PACKEDBYTES))+ assigns(memory_slice(s2, sizeof(mld_sk_s2hat_eager)))+ ensures(forall(k1, 0, MLDSA_K,+ array_abs_bound(s2->vec.vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+)+{+ mld_polyveck_unpack_eta(&s2->vec, packed_s2);+ mld_polyveck_ntt(&s2->vec);+}++#if !defined(MLD_CONFIG_NO_SIGN_API)+static MLD_INLINE void mld_sk_s2hat_get_poly_eager(mld_poly *buf,+ const mld_sk_s2hat_eager *s2,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(s2, sizeof(mld_sk_s2hat_eager)))+ requires(i < MLDSA_K)+ requires(array_abs_bound(s2->vec.vec[i].coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_abs_bound(buf->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+) { *buf = s2->vec.vec[i]; }+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* (!MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST) && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) */+#if defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+static MLD_INLINE void mld_unpack_sk_s2hat_lazy(+ mld_sk_s2hat_lazy *s2,+ const uint8_t packed_s2[MLDSA_K * MLDSA_POLYETA_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(s2, sizeof(mld_sk_s2hat_lazy)))+ assigns(memory_slice(s2, sizeof(mld_sk_s2hat_lazy)))+ ensures(s2->packed == old(packed_s2))+) { s2->packed = packed_s2; }++#if !defined(MLD_CONFIG_NO_SIGN_API)+static MLD_INLINE void mld_sk_s2hat_get_poly_lazy(mld_poly *buf,+ const mld_sk_s2hat_lazy *s2,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(s2, sizeof(mld_sk_s2hat_lazy)))+ requires(i < MLDSA_K)+ requires(memory_no_alias(s2->packed, MLDSA_K * MLDSA_POLYETA_PACKEDBYTES))+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_abs_bound(buf->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+)+{+ mld_polyeta_unpack(buf, s2->packed + i * MLDSA_POLYETA_PACKEDBYTES);+ mld_poly_ntt(buf);+}+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++/* t0vec */++#if (!defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+static MLD_INLINE void mld_unpack_sk_t0hat_eager(+ mld_sk_t0hat_eager *t0,+ const uint8_t packed_t0[MLDSA_K * MLDSA_POLYT0_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(t0, sizeof(mld_sk_t0hat_eager)))+ requires(memory_no_alias(packed_t0, MLDSA_K * MLDSA_POLYT0_PACKEDBYTES))+ assigns(memory_slice(t0, sizeof(mld_sk_t0hat_eager)))+ ensures(forall(k1, 0, MLDSA_K,+ array_abs_bound(t0->vec.vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+)+{+ unsigned int i;+ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(t0, sizeof(mld_sk_t0hat_eager)))+ invariant(i <= MLDSA_K)+ invariant(forall(k0, 0, i,+ array_bound(t0->vec.vec[k0].coeffs, 0, MLDSA_N,+ -(1 << (MLDSA_D - 1)) + 1, (1 << (MLDSA_D - 1)) + 1)))+ decreases(MLDSA_K - i)+ )+ {+ mld_polyt0_unpack(&t0->vec.vec[i],+ packed_t0 + i * MLDSA_POLYT0_PACKEDBYTES);+ }+ mld_polyveck_ntt(&t0->vec);+}++#if !defined(MLD_CONFIG_NO_SIGN_API)+static MLD_INLINE void mld_sk_t0hat_get_poly_eager(mld_poly *buf,+ const mld_sk_t0hat_eager *t0,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(t0, sizeof(mld_sk_t0hat_eager)))+ requires(i < MLDSA_K)+ requires(array_abs_bound(t0->vec.vec[i].coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_abs_bound(buf->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+) { *buf = t0->vec.vec[i]; }+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* (!MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST) && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) */+#if defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+static MLD_INLINE void mld_unpack_sk_t0hat_lazy(+ mld_sk_t0hat_lazy *t0,+ const uint8_t packed_t0[MLDSA_K * MLDSA_POLYT0_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(t0, sizeof(mld_sk_t0hat_lazy)))+ assigns(memory_slice(t0, sizeof(mld_sk_t0hat_lazy)))+ ensures(t0->packed == old(packed_t0))+) { t0->packed = packed_t0; }++#if !defined(MLD_CONFIG_NO_SIGN_API)+static MLD_INLINE void mld_sk_t0hat_get_poly_lazy(mld_poly *buf,+ const mld_sk_t0hat_lazy *t0,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(t0, sizeof(mld_sk_t0hat_lazy)))+ requires(i < MLDSA_K)+ requires(memory_no_alias(t0->packed, MLDSA_K * MLDSA_POLYT0_PACKEDBYTES))+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_abs_bound(buf->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+)+{+ mld_polyt0_unpack(buf, t0->packed + i * MLDSA_POLYT0_PACKEDBYTES);+ mld_poly_ntt(buf);+}+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API */++/* yvec */++#if !defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+static MLD_INLINE void mld_yvec_init_eager(+ mld_yvec_eager *y, const uint8_t rhoprime[MLDSA_CRHBYTES], uint16_t kappa)+__contract__(+ requires(memory_no_alias(y, sizeof(mld_yvec_eager)))+ requires(memory_no_alias(rhoprime, MLDSA_CRHBYTES))+ requires(kappa <= MLD_MAX_KAPPA)+ assigns(memory_slice(y, sizeof(mld_yvec_eager)))+ ensures(forall(k1, 0, MLDSA_L,+ array_bound(y->vec.vec[k1].coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1)))+)+{+ mld_polyvecl_uniform_gamma1(&y->vec, rhoprime, kappa);+}++static MLD_INLINE void mld_yvec_get_poly_eager(mld_poly *buf,+ const mld_yvec_eager *y,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(y, sizeof(mld_yvec_eager)))+ requires(i < MLDSA_L)+ requires(array_bound(y->vec.vec[i].coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_bound(buf->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+) { *buf = y->vec.vec[i]; }+#endif /* !MLD_CONFIG_NO_SIGN_API && (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) \+ */+#if !defined(MLD_CONFIG_NO_SIGN_API) && \+ (defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+static MLD_INLINE void mld_yvec_init_lazy(+ mld_yvec_lazy *y, const uint8_t rhoprime[MLDSA_CRHBYTES], uint16_t kappa)+__contract__(+ requires(memory_no_alias(y, sizeof(mld_yvec_lazy)))+ assigns(memory_slice(y, sizeof(mld_yvec_lazy)))+ ensures(y->rhoprime == old(rhoprime))+ ensures(y->kappa == old(kappa))+)+{+ y->rhoprime = rhoprime;+ y->kappa = kappa;+}++static MLD_INLINE void mld_yvec_get_poly_lazy(mld_poly *buf,+ const mld_yvec_lazy *y,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(y, sizeof(mld_yvec_lazy)))+ requires(i < MLDSA_L)+ requires(memory_no_alias(y->rhoprime, MLDSA_CRHBYTES))+ requires(y->kappa <= MLD_MAX_KAPPA)+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_bound(buf->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+)+{+ /* Safety: y->kappa <= MLD_MAX_KAPPA and i < MLDSA_L, so y->kappa + i+ * fits in uint16_t. See MLD_MAX_KAPPA comment in params.h. */+ mld_poly_uniform_gamma1(buf, y->rhoprime, (uint16_t)(y->kappa + i));+}+#endif /* !MLD_CONFIG_NO_SIGN_API && (MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) \+ */++/* polymat */++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+/** Eager polymat: precomputed and stored full MLDSA_K x MLDSA_L matrix. */+typedef struct+{+ mld_polyvecl vec[MLDSA_K]; /**< Rows of the matrix. */+} mld_polymat_eager;+#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++/** Lazy polymat: store seed, sample elements A[k][l] on demand. */+typedef struct+{+ mld_poly cur; /**< On-demand sampled matrix element A[k][l]. */+ uint8_t rho[MLDSA_SEEDBYTES]; /**< Public seed used to expand A. */+} mld_polymat_lazy;++static MLD_INLINE void mld_poly_permute_bitrev_to_custom_optional(mld_poly *p)+__contract__(+ /* We don't specify that this is a permutation, only that it preserves+ * the bounds.+ * When the native NTT backend does not use the custom order, this is a no-op. */+ requires(memory_no_alias(p, sizeof(mld_poly)))+ requires(array_bound(p->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(p, sizeof(mld_poly)))+ ensures(array_bound(p->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+)+{+#if defined(MLD_USE_NATIVE_NTT_CUSTOM_ORDER)+ mld_poly_permute_bitrev_to_custom(p->coeffs);+#else+ (void)p;+#endif+}++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+/**+ * Generates matrix A with uniformly random coefficients a_{i,j} by performing+ * rejection sampling on the output stream of SHAKE128(rho|j|i).+ *+ * @spec{Implements @[FIPS204, Algorithm 32, ExpandA].}+ *+ * @param[out] mat Pointer to output matrix.+ * @param[in] rho Byte array containing seed rho.+ */+MLD_INTERNAL_API+void mld_polyvec_matrix_expand_eager(mld_polymat_eager *mat,+ const uint8_t rho[MLDSA_SEEDBYTES])+__contract__(+ requires(memory_no_alias(mat, sizeof(mld_polymat_eager)))+ requires(memory_no_alias(rho, MLDSA_SEEDBYTES))+ assigns(memory_slice(mat, sizeof(mld_polymat_eager)))+ ensures(forall(k1, 0, MLDSA_K, forall(l1, 0, MLDSA_L,+ array_bound(mat->vec[k1].vec[l1].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+);++/**+ * Compute row i of matrix-vector multiplication in NTT domain with pointwise+ * multiplication and multiplication by 2^{-32}.+ *+ * Input matrix and vector must be in NTT domain representation. Output+ * coefficients are bounded by MLDSA_Q in absolute value.+ *+ * @param[out] t_row Pointer to output row polynomial.+ * @param[in] mat Pointer to input matrix.+ * @param[in] v Pointer to input vector v.+ * @param i Row index, 0 <= i < MLDSA_K.+ */+MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_row_eager(mld_poly *t_row,+ mld_polymat_eager *mat,+ const mld_polyvecl *v,+ unsigned int i)+__contract__(+ requires(memory_no_alias(t_row, sizeof(mld_poly)))+ requires(memory_no_alias(mat, sizeof(mld_polymat_eager)))+ requires(memory_no_alias(v, sizeof(mld_polyvecl)))+ requires(i < MLDSA_K)+ requires(forall(l1, 0, MLDSA_L,+ array_bound(mat->vec[i].vec[l1].coeffs, 0, MLDSA_N, 0, MLDSA_Q)))+ requires(forall(l2, 0, MLDSA_L,+ array_abs_bound(v->vec[l2].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ assigns(memory_slice(t_row, sizeof(mld_poly)))+ ensures(array_abs_bound(t_row->coeffs, 0, MLDSA_N, MLDSA_Q))+);++#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute w = invNTT(A * NTT(y)) for the signing y vector.+ *+ * The eager variant copies y into the scratch polyvecl, NTTs it in place,+ * calls the standard matrix-vector multiply, and finally inverse-NTTs the+ * result into w.+ *+ * @param[out] w Pointer to output vector.+ * @param[in] mat Pointer to input matrix.+ * @param[in] y Pointer to (non-NTT) y vector.+ * @param[out] scratch Scratch polyvecl for NTT'd copy of y.+ */+MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_yvec_eager(mld_polyveck *w,+ mld_polymat_eager *mat,+ const mld_yvec_eager *y,+ mld_polyvecl *scratch)+__contract__(+ requires(memory_no_alias(w, sizeof(mld_polyveck)))+ requires(memory_no_alias(mat, sizeof(mld_polymat_eager)))+ requires(memory_no_alias(y, sizeof(mld_yvec_eager)))+ requires(memory_no_alias(scratch, sizeof(mld_polyvecl)))+ requires(forall(k1, 0, MLDSA_K, forall(l1, 0, MLDSA_L,+ array_bound(mat->vec[k1].vec[l1].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+ requires(forall(l2, 0, MLDSA_L,+ array_bound(y->vec.vec[l2].coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1)))+ assigns(memory_slice(w, sizeof(mld_polyveck)))+ assigns(memory_slice(scratch, sizeof(mld_polyvecl)))+ ensures(forall(k0, 0, MLDSA_K,+ array_abs_bound(w->vec[k0].coeffs, 0, MLDSA_N, MLD_INTT_BOUND)))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+void mld_polyvec_matrix_expand_lazy(mld_polymat_lazy *mat,+ const uint8_t rho[MLDSA_SEEDBYTES])+__contract__(+ requires(memory_no_alias(mat, sizeof(mld_polymat_lazy)))+ requires(memory_no_alias(rho, MLDSA_SEEDBYTES))+ assigns(memory_slice(mat, sizeof(mld_polymat_lazy)))+);++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Compute row i of matrix-vector multiplication in NTT domain with pointwise+ * multiplication and multiplication by 2^{-32}.+ *+ * Input vector must be in NTT domain representation; the matrix entries are+ * sampled on demand from the seed stored in mat->rho, using mat->cur as+ * scratch. Output coefficients are bounded by MLDSA_Q in absolute value.+ *+ * @param[out] t_row Pointer to output row polynomial.+ * @param[in,out] mat Pointer to input matrix (seed + scratch).+ * @param[in] v Pointer to input vector v.+ * @param i Row index, 0 <= i < MLDSA_K.+ */+MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_row_lazy(mld_poly *t_row,+ mld_polymat_lazy *mat,+ const mld_polyvecl *v,+ unsigned int i)+__contract__(+ requires(memory_no_alias(t_row, sizeof(mld_poly)))+ requires(memory_no_alias(mat, sizeof(mld_polymat_lazy)))+ requires(memory_no_alias(v, sizeof(mld_polyvecl)))+ requires(i < MLDSA_K)+ requires(forall(l1, 0, MLDSA_L,+ array_abs_bound(v->vec[l1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ assigns(memory_slice(t_row, sizeof(mld_poly)))+ assigns(memory_slice(mat, sizeof(mld_polymat_lazy)))+ ensures(array_abs_bound(t_row->coeffs, 0, MLDSA_N, MLDSA_Q))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute w = invNTT(A * NTT(y)) for the signing y vector.+ *+ * The lazy variant samples one column of y at a time, NTTs it into+ * &scratch->vec[0], and accumulates the matrix-vector product+ * column-by-column with on-demand sampling of A[k][l]. Only the first poly of+ * the polyvecl scratch is used; the polyvecl type is shared with the eager+ * variant for API uniformity (the storage is provided "for free" by the+ * caller's polyveck/polyvecl union in REDUCE_RAM mode).+ *+ * @param[out] w Pointer to output vector.+ * @param[in,out] mat Pointer to input matrix.+ * @param[in] y Pointer to y seed/kappa.+ * @param[out] scratch Scratch (only &scratch->vec[0] used).+ */+MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_yvec_lazy(mld_polyveck *w,+ mld_polymat_lazy *mat,+ const mld_yvec_lazy *y,+ mld_polyvecl *scratch)+__contract__(+ requires(memory_no_alias(w, sizeof(mld_polyveck)))+ requires(memory_no_alias(mat, sizeof(mld_polymat_lazy)))+ requires(memory_no_alias(y, sizeof(mld_yvec_lazy)))+ requires(memory_no_alias(scratch, sizeof(mld_polyvecl)))+ requires(memory_no_alias(y->rhoprime, MLDSA_CRHBYTES))+ requires(y->kappa <= MLD_MAX_KAPPA)+ assigns(memory_slice(w, sizeof(mld_polyveck)))+ assigns(memory_slice(mat, sizeof(mld_polymat_lazy)))+ assigns(memory_slice(scratch, sizeof(mld_polyvecl)))+ ensures(forall(k0, 0, MLDSA_K,+ array_abs_bound(w->vec[k0].coeffs, 0, MLDSA_N, MLD_INTT_BOUND)))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++/* Dispatch: typedef and define based on MLD_CONFIG_REDUCE_RAM */+#if defined(MLD_CONFIG_REDUCE_RAM)+typedef mld_sk_s1hat_lazy mld_sk_s1hat;+typedef mld_sk_s2hat_lazy mld_sk_s2hat;+typedef mld_sk_t0hat_lazy mld_sk_t0hat;+typedef mld_polymat_lazy mld_polymat;+typedef mld_yvec_lazy mld_yvec;+#define mld_unpack_sk_s1hat mld_unpack_sk_s1hat_lazy+#define mld_unpack_sk_s2hat mld_unpack_sk_s2hat_lazy+#define mld_unpack_sk_t0hat mld_unpack_sk_t0hat_lazy+#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_sk_s1hat_get_poly mld_sk_s1hat_get_poly_lazy+#define mld_sk_s2hat_get_poly mld_sk_s2hat_get_poly_lazy+#define mld_sk_t0hat_get_poly mld_sk_t0hat_get_poly_lazy+#endif+#define mld_polyvec_matrix_expand mld_polyvec_matrix_expand_lazy+#define mld_polyvec_matrix_pointwise_montgomery_row \+ mld_polyvec_matrix_pointwise_montgomery_row_lazy+#define mld_yvec_init mld_yvec_init_lazy+#define mld_yvec_get_poly mld_yvec_get_poly_lazy+#define mld_polyvec_matrix_pointwise_montgomery_yvec \+ mld_polyvec_matrix_pointwise_montgomery_yvec_lazy+#else /* MLD_CONFIG_REDUCE_RAM */+typedef mld_sk_s1hat_eager mld_sk_s1hat;+typedef mld_sk_s2hat_eager mld_sk_s2hat;+typedef mld_sk_t0hat_eager mld_sk_t0hat;+typedef mld_polymat_eager mld_polymat;+typedef mld_yvec_eager mld_yvec;+#define mld_unpack_sk_s1hat mld_unpack_sk_s1hat_eager+#define mld_unpack_sk_s2hat mld_unpack_sk_s2hat_eager+#define mld_unpack_sk_t0hat mld_unpack_sk_t0hat_eager+#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_sk_s2hat_get_poly mld_sk_s2hat_get_poly_eager+#define mld_sk_s1hat_get_poly mld_sk_s1hat_get_poly_eager+#define mld_sk_t0hat_get_poly mld_sk_t0hat_get_poly_eager+#endif+#define mld_polyvec_matrix_expand mld_polyvec_matrix_expand_eager+#define mld_polyvec_matrix_pointwise_montgomery_row \+ mld_polyvec_matrix_pointwise_montgomery_row_eager+#define mld_yvec_init mld_yvec_init_eager+#define mld_yvec_get_poly mld_yvec_get_poly_eager+#define mld_polyvec_matrix_pointwise_montgomery_yvec \+ mld_polyvec_matrix_pointwise_montgomery_yvec_eager+#endif /* !MLD_CONFIG_REDUCE_RAM */++#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_NO_VERIFY_API */+#endif /* !MLD_POLYVEC_LAZY_H */
+ cbits/mldsa/src/randombytes.h view
@@ -0,0 +1,26 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_RANDOMBYTES_H+#define MLD_RANDOMBYTES_H++#include <stddef.h>++#include "cbmc.h"+#include "common.h"++#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+#if !defined(MLD_CONFIG_CUSTOM_RANDOMBYTES)+MLD_MUST_CHECK_RETURN_VALUE+int randombytes(uint8_t *out, size_t outlen);++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_randombytes(uint8_t *out, size_t outlen)+__contract__(+ requires(memory_no_alias(out, outlen))+ assigns(memory_slice(out, outlen))+) { return randombytes(out, outlen); }+#endif /* !MLD_CONFIG_CUSTOM_RANDOMBYTES */+#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLD_RANDOMBYTES_H */
+ cbits/mldsa/src/reduce.h view
@@ -0,0 +1,144 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_REDUCE_H+#define MLD_REDUCE_H++#include "cbmc.h"+#include "common.h"+#include "ct.h"+#include "debug.h"++/* check-magic: -4186625 == pow(2,32,MLDSA_Q) */+#define MLD_MONT (-4186625)++/* Upper bound for domain of mld_reduce32() */+#define MLD_REDUCE32_DOMAIN_MAX (INT32_MAX - ((int32_t)1 << 22))++/* Absolute bound for range of mld_reduce32() */+/* check-magic: 6283009 == (MLD_REDUCE32_DOMAIN_MAX - 255 * MLDSA_Q + 1) */+#define MLD_REDUCE32_RANGE_MAX 6283009++/**+ * Generic Montgomery reduction; given a 64-bit integer a, computes a 32-bit+ * integer congruent to a * R^-1 mod MLDSA_Q, where R=2^32.+ *+ * @spec{Implements @[FIPS204, Algorithm 49, MontgomeryReduce].}+ *+ * @param a Input integer to be reduced, of absolute value smaller or equal+ * to INT64_MAX - 2^31 * MLDSA_Q.+ *+ * @return Integer congruent to a * R^-1 modulo MLDSA_Q, with absolute value+ * <= |a| / 2^32 + MLDSA_Q / 2.+ * In particular, if |a| < 2^31 * MLDSA_Q, the absolute value of the+ * return value is < MLDSA_Q.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int32_t mld_montgomery_reduce(int64_t a)+__contract__(+ /* We don't attempt to express an input-dependent output bound+ * as the post-condition here, as all call-sites satisfy the+ * absolute input bound 2^31 * MLDSA_Q and higher-level+ * reasoning can be conducted using |return_value| < MLDSA_Q. */+ requires(a > -(((int64_t)1 << 31) * MLDSA_Q) &&+ a < (((int64_t)1 << 31) * MLDSA_Q))+ ensures(return_value > -MLDSA_Q && return_value < MLDSA_Q)+)+{+ /* check-magic: 58728449 == unsigned_mod(pow(MLDSA_Q, -1, 2^32), 2^32) */+ const uint64_t QINV = 58728449;++ /* Compute a*q^{-1} mod 2^32 in unsigned representatives */+ const uint32_t a_reduced = mld_cast_int64_to_uint32(a);+ const uint32_t a_inverted = (a_reduced * QINV) & UINT32_MAX;++ /* Lift to signed canonical representative mod 2^32. */+ const int32_t t = mld_cast_uint32_to_int32(a_inverted);++ int64_t r;++ mld_assert(a < +(INT64_MAX - (((int64_t)1 << 31) * MLDSA_Q)) &&+ a > -(INT64_MAX - (((int64_t)1 << 31) * MLDSA_Q)));++ r = a - (int64_t)t * MLDSA_Q;++ /*+ * PORTABILITY: Right-shift on a signed integer is, strictly-speaking,+ * implementation-defined for negative left argument. Here,+ * we assume it's sign-preserving "arithmetic" shift right. (C99 6.5.7 (5))+ */+ r = r >> 32;++ /* Bounds:+ *+ * By construction of the Montgomery multiplication, by the time we+ * compute r >> 32, r is divisible by 2^32, and hence+ *+ * |r >> 32| = |r| / 2^32+ * <= |a| / 2^32 + MLDSA_Q / 2+ *+ * (In general, we would only have |x >> n| <= ceil(|x| / 2^n)).+ *+ * In particular, if |a| < 2^31 * MLDSA_Q, then |return_value| < MLDSA_Q.+ */+ return (int32_t)r;+}++/**+ * For finite field element a with a <= 2^{31} - 2^{22} - 1, compute+ * r congruent to a (mod MLDSA_Q) such that+ * -MLD_REDUCE32_RANGE_MAX <= r < MLD_REDUCE32_RANGE_MAX.+ *+ * @param a Finite field element.+ *+ * @return r.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int32_t mld_reduce32(int32_t a)+__contract__(+ requires(a <= MLD_REDUCE32_DOMAIN_MAX)+ ensures(return_value >= -MLD_REDUCE32_RANGE_MAX)+ ensures(return_value < MLD_REDUCE32_RANGE_MAX)+)+{+ int32_t t;++ t = (a + ((int32_t)1 << 22)) >> 23;+ t = a - t * MLDSA_Q;+ mld_assert((t - a) % MLDSA_Q == 0);+ return t;+}++/**+ * Add MLDSA_Q if input coefficient is negative.+ *+ * @param a Finite field element.+ *+ * @return r.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int32_t mld_caddq(int32_t a)+__contract__(+ requires(a > -MLDSA_Q)+ requires(a < MLDSA_Q)+ ensures(return_value >= 0)+ ensures(return_value < MLDSA_Q)+ ensures(return_value == ((a >= 0) ? a : (a + MLDSA_Q)))+)+{+ return mld_ct_sel_int32(a + MLDSA_Q, a, mld_ct_cmask_neg_i32(a));+}+++#endif /* !MLD_REDUCE_H */
+ cbits/mldsa/src/rounding.h view
@@ -0,0 +1,265 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_ROUNDING_H+#define MLD_ROUNDING_H++#include "cbmc.h"+#include "common.h"+#include "ct.h"+#include "debug.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mldsa-native (e.g. with varying parameter sets)+ * within a single compilation unit. */+#define mld_power2round MLD_ADD_PARAM_SET(mld_power2round)+#define mld_decompose MLD_ADD_PARAM_SET(mld_decompose)+#define mld_make_hint MLD_ADD_PARAM_SET(mld_make_hint)+#define mld_use_hint MLD_ADD_PARAM_SET(mld_use_hint)+/* End of parameter set namespacing */++#define MLD_2_POW_D (1 << MLDSA_D)++/**+ * For finite field element a, compute a0, a1 such that+ * a mod^+ MLDSA_Q = a1*2^MLDSA_D + a0 with+ * -2^{MLDSA_D-1} < a0 <= 2^{MLDSA_D-1}. Assumes a to be standard+ * representative.+ *+ * @spec{Implements @[FIPS204, Algorithm 35, Power2Round].}+ *+ * @reference{In the reference implementation, a1 is passed as a return value+ * instead.}+ *+ * @param[out] a0 Pointer to output element a0.+ * @param[out] a1 Pointer to output element a1.+ * @param a Input element.+ */+static MLD_INLINE void mld_power2round(int32_t *a0, int32_t *a1, int32_t a)+__contract__(+ requires(memory_no_alias(a0, sizeof(int32_t)))+ requires(memory_no_alias(a1, sizeof(int32_t)))+ requires(a >= 0 && a < MLDSA_Q)+ assigns(memory_slice(a0, sizeof(int32_t)))+ assigns(memory_slice(a1, sizeof(int32_t)))+ ensures(*a0 > -(MLD_2_POW_D/2) && *a0 <= (MLD_2_POW_D/2))+ ensures(*a1 >= 0 && *a1 <= (MLDSA_Q - 1) / MLD_2_POW_D)+ ensures((*a1 * MLD_2_POW_D + *a0 - a) % MLDSA_Q == 0)+)+{+ *a1 = (a + (1 << (MLDSA_D - 1)) - 1) >> MLDSA_D;+ *a0 = a - (*a1 << MLDSA_D);+}++/**+ * For finite field element a, compute high and low bits a0, a1 such that+ * a mod^+ MLDSA_Q = a1 * 2 * MLDSA_GAMMA2 + a0 with+ * -MLDSA_GAMMA2 < a0 <= MLDSA_GAMMA2 except if+ * a1 = (MLDSA_Q-1)/(MLDSA_GAMMA2*2) where we set a1 = 0 and+ * -MLDSA_GAMMA2 <= a0 = a mod^+ MLDSA_Q - MLDSA_Q < 0. Assumes a to be+ * standard representative.+ *+ * @spec{Implements @[FIPS204, Algorithm 36, Decompose].}+ *+ * @reference{In the reference implementation, a1 is passed as a return value+ * instead.}+ *+ * @param[out] a0 Pointer to output element a0.+ * @param[out] a1 Pointer to output element a1.+ * @param a Input element.+ */+static MLD_INLINE void mld_decompose(int32_t *a0, int32_t *a1, int32_t a)+__contract__(+ requires(memory_no_alias(a0, sizeof(int32_t)))+ requires(memory_no_alias(a1, sizeof(int32_t)))+ requires(a >= 0 && a < MLDSA_Q)+ assigns(memory_slice(a0, sizeof(int32_t)))+ assigns(memory_slice(a1, sizeof(int32_t)))+ /* a0 = -MLDSA_GAMMA2 occurs exactly when a = MLDSA_Q - MLDSA_GAMMA2: the+ * border case of Decompose where a1 = (MLDSA_Q-1)/(2*MLDSA_GAMMA2) is+ * wrapped to 0 and a0 = a - MLDSA_Q (@[FIPS204, Algorithm 36, Decompose]) */+ ensures(*a0 >= -MLDSA_GAMMA2 && *a0 <= MLDSA_GAMMA2)+ ensures(*a1 >= 0 && *a1 < (MLDSA_Q-1)/(2*MLDSA_GAMMA2))+ ensures((*a1 * 2 * MLDSA_GAMMA2 + *a0 - a) % MLDSA_Q == 0)+)+{+ /*+ * The goal is to compute f1 = round-(f / (2*GAMMA2)), which can be computed+ * alternatively as round-(f / (128B)) = round-(ceil(f / 128) / B) where+ * B = 2*GAMMA2 / 128. Here round-() denotes "round half down".+ *+ * The equality round-(f / (128B)) = round-(ceil(f / 128) / B) can deduced+ * as follows. Since changing f to align-up(f, 128) can move f onto but not+ * across a rounding boundary for division by 128*B (note that we need B to be+ * even for this to work), and round- rounds down on the boundary, we have+ *+ * round-(f / (128B)) = round-(align-up(f, 128) / (128B))+ * = round-((align-up(f, 128) / 128) / B)+ * = round-(ceil(f / 128) / B).+ */+ *a1 = (a + 127) >> 7;+ /* We know a >= 0 and a < MLDSA_Q, so... */+ /* check-magic: 65472 == round((MLDSA_Q-1)/128) */+ mld_assert(*a1 >= 0 && *a1 <= 65472);++#if MLD_CONFIG_PARAMETER_SET == 44+ /* check-magic: 1488 == 2 * intdiv(intdiv(MLDSA_Q - 1, 88), 128) */+ /* check-magic: 11275 == floor(2**24 / 1488) */+ /* check-magic: 1560281088 == 1 / (1 / 1488 - 11275 / 2**24) */+ /*+ * Compute f1 = round-(f1' / B) ≈ round(f1' * 11275 / 2^24). This is exact for+ * 0 <= f1' < 2^16.+ *+ * To see this, consider the (signed) error f1' * (1 / B - 11275 / 2^24)+ * between f1' / B and the (under-)approximation f1' * 11275 / 2^24. Because+ * eps := 1 / B - 11275 / 2^24 is 1 / 1560281088 ≈ 2^(-30.54) < 2^(-30), we+ * have 0 <= f1' * eps < 2^16 * 2^(-30) = 1 / 2^14 < 1 / 2^11 < 1 / B (note+ * that f1' is non-negative).+ *+ * On the other hand, 1 / B is the spacing between the integral multiples+ * of 1 / B, which includes all rounding boundaries n + 0.5 (since B is even).+ * Hence, if f1' / B is not of the form n + 0.5, then it is at least 1 / B+ * away from the nearest rounding boundary, so moving from f1' / B to+ * f1' * 11275 / 2^24 does not affect the rounding result, no matter the type+ * of rounding used in either side. In particular, we have round-(f1' / B) =+ * round(f1' * 11275 / 2^24) as claimed.+ *+ * As for the remaining case where f1' / B _is_ of the form n + 0.5, because+ * f1' * 11275 / 2^24 is slightly but strictly below f1' / B = n + 0.5 (note+ * that f1' and thus the error f1' * eps cannot be 0 here), it is always+ * rounded down to n. More precisely, we have round-(f1' / B) =+ * round(f1' * 11275 / 2^24), where the round-down on the LHS is essential,+ * and on the RHS the type of rounding again does not matter. This concludes+ * the proof.+ *+ * See proofs/isabelle/compress for a formalization of the above argument.+ */+ *a1 = (*a1 * 11275 + ((int32_t)1 << 23)) >> 24;+ mld_assert(*a1 >= 0 && *a1 <= 44);++ *a1 = mld_ct_sel_int32(0, *a1, mld_ct_cmask_neg_i32(43 - *a1));+ mld_assert(*a1 >= 0 && *a1 <= 43);+#else /* MLD_CONFIG_PARAMETER_SET == 44 */+ /* check-magic: 4092 == 2 * intdiv(intdiv(MLDSA_Q - 1, 32), 128) */+ /* check-magic: 1025 == floor(2**22 / 4092) */+ /* check-magic: 4290772992 == 1 / (1 / 4092 - 1025 / 2**22) */+ /*+ * Compute f1 = round-(f1' / B) ≈ round(f1' * 1025 / 2^22). This is exact for+ * 0 <= f1' < 2^16. Following the same argument above, it suffices to show+ * that f1' * eps < 1 / B, where eps := 1 / B - 1025 / 2^22. Indeed, we have+ * eps = 1 / 4290772992 ≈ 2^(-31.99) < 2^(-31), therefore f1' * eps <+ * 2^16 * 2^(-31) = 1 / 2^15 < 1 / 2^12 < 1 / B.+ */+ *a1 = (*a1 * 1025 + ((int32_t)1 << 21)) >> 22;+ mld_assert(*a1 >= 0 && *a1 <= 16);++ *a1 &= 15;+ mld_assert(*a1 >= 0 && *a1 <= 15);++#endif /* MLD_CONFIG_PARAMETER_SET != 44 */++ *a0 = a - *a1 * 2 * MLDSA_GAMMA2;+ *a0 = mld_ct_sel_int32(*a0 - MLDSA_Q, *a0,+ mld_ct_cmask_neg_i32((MLDSA_Q - 1) / 2 - *a0));+}++/**+ * Decide a single hint bit from the low part a0 and high part a1 of a+ * coefficient: return 1 unless a0 lies in the range (-GAMMA2, GAMMA2] that+ * LowBits would produce, with the boundary value -GAMMA2 also admitted when+ * a1 == 0 (the Decompose border case).+ *+ * @note This is not a line-for-line implementation of FIPS 204's MakeHint(z, r)+ * (@[FIPS204, Algorithm 39, MakeHint]), which takes two ring elements and+ * returns [[HighBits(r) != HighBits(r + z)]]. Instead, it takes the already+ * decomposed low/high parts (a0, a1) of a coefficient and decides the hint bit+ * from them directly. As explained in the block comment of+ * mld_attempt_signature_generation (sign.c), for the specific values that arise+ * during signing -- a0 = w0 - cs2 + ct0 and a1 = w1 = HighBits(w) -- this is+ * equivalent to the spec's MakeHint(-ct0, w - cs2 + ct0) coefficient-wise.+ * Because it consumes (a0, a1) rather than (z, r), it relies on the caller+ * having computed a compatible decomposition.+ *+ * @param a0 Low bits of input element.+ * @param a1 High bits of input element.+ *+ * @return 1 if overflow, 0 otherwise.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE unsigned int mld_make_hint(int32_t a0, int32_t a1)+__contract__(+ ensures(return_value >= 0 && return_value <= 1)+ ensures(return_value == (a0 > MLDSA_GAMMA2 || a0 < -MLDSA_GAMMA2 ||+ (a0 == -MLDSA_GAMMA2 && a1 != 0)))+)+{+ if (a0 > MLDSA_GAMMA2 || a0 < -MLDSA_GAMMA2 ||+ (a0 == -MLDSA_GAMMA2 && a1 != 0))+ {+ return 1;+ }++ return 0;+}++/**+ * Correct high bits according to hint.+ *+ * @spec{Implements @[FIPS204, Algorithm 40, UseHint].}+ *+ * @param a Input element.+ * @param hint Hint bit.+ *+ * @return Corrected high bits.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int32_t mld_use_hint(int32_t a, int32_t hint)+__contract__(+ requires(hint >= 0 && hint <= 1)+ requires(a >= 0 && a < MLDSA_Q)+ ensures(return_value >= 0 && return_value < (MLDSA_Q-1)/(2*MLDSA_GAMMA2))+)+{+ int32_t a0, a1;++ mld_decompose(&a0, &a1, a);+ if (hint == 0)+ {+ return a1;+ }++#if MLD_CONFIG_PARAMETER_SET == 44+ if (a0 > 0)+ {+ return (a1 == 43) ? 0 : a1 + 1;+ }+ else+ {+ return (a1 == 0) ? 43 : a1 - 1;+ }+#else /* MLD_CONFIG_PARAMETER_SET == 44 */+ if (a0 > 0)+ {+ return (a1 + 1) & 15;+ }+ else+ {+ return (a1 - 1) & 15;+ }+#endif /* MLD_CONFIG_PARAMETER_SET != 44 */+}+++#endif /* !MLD_ROUNDING_H */
+ cbits/mldsa/src/sign.c view
@@ -0,0 +1,1720 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS140_3_IG]+ * Implementation Guidance for FIPS 140-3 and the Cryptographic Module+ * Validation Program+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/projects/cryptographic-module-validation-program/fips-140-3-ig-announcements+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ *+ * - [FIPS204_UPDATES]+ * FIPS 204 Potential Updates (Errata)+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/files/pubs/fips/204/final/docs/fips-204-potential-updates.xlsx+ *+ * - [Round3_Spec]+ * CRYSTALS-Dilithium Algorithm Specifications and Supporting Documentation+ * (Version 3.1)+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://pq-crystals.org/dilithium/data/dilithium-specification-round3-20210208.pdf+ */++#include "sign.h"++#include "cbmc.h"+#include "ct.h"+#include "debug.h"+#include "packing.h"+#include "poly.h"+#include "poly_kl.h"+#include "polyvec.h"+#include "randombytes.h"+#include "symmetric.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mldsa-native (e.g. with varying parameter sets)+ * within a single compilation unit. */+#define mld_check_pct MLD_ADD_PARAM_SET(mld_check_pct) MLD_CONTEXT_PARAMETERS_2+#define mld_sample_s1_s2 MLD_ADD_PARAM_SET(mld_sample_s1_s2)+#define mld_validate_hash_length MLD_ADD_PARAM_SET(mld_validate_hash_length)+#define mld_get_hash_oid MLD_ADD_PARAM_SET(mld_get_hash_oid)+#define mld_H MLD_ADD_PARAM_SET(mld_H)+#define mld_compute_pack_z MLD_ADD_PARAM_SET(mld_compute_pack_z)+#define mld_attempt_signature_generation \+ MLD_ADD_PARAM_SET(mld_attempt_signature_generation) MLD_CONTEXT_PARAMETERS_8+#define mld_compute_pack_t0_t1 \+ MLD_ADD_PARAM_SET(mld_compute_pack_t0_t1) MLD_CONTEXT_PARAMETERS_5+#define mld_get_max_signing_attempts \+ MLD_ADD_PARAM_SET(mld_get_max_signing_attempts)++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+static int mld_check_pct(uint8_t const pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ uint8_t const sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY ||+ return_value == MLD_ERR_RNG_FAIL ||+ return_value == MLD_ERR_SIGNING_PAUSED ||+ return_value == MLD_ERR_PCT_FAIL)+);++#if defined(MLD_CONFIG_KEYGEN_PCT)+/**+ * Pair-wise Consistency Test (PCT) for DSA keypairs.+ *+ * @[FIPS140_3_IG] TE10.35.02+ * (https://csrc.nist.gov/csrc/media/Projects/cryptographic-module-validation-program/documents/fips%20140-3/FIPS%20140-3%20IG.pdf).+ *+ * Validates that a generated public/private key pair can correctly sign and+ * verify data. Performs signature generation using the private key (sk),+ * followed by signature verification using the public key (pk).+ *+ * @note @[FIPS204] requires that public/private key pairs are to be used+ * only for the calculation and/or verification of digital signatures.+ *+ * @param[in] pk Public key.+ * @param[in] sk Secret key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was used and an+ * allocation via MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed.+ * @retval MLD_ERR_SIGNING_PAUSED The PCT's signing step was paused by a+ * MLD_CONFIG_SIGN_HOOK_ATTEMPT hook.+ * @retval MLD_ERR_PCT_FAIL The consistency check failed.+ */+static int mld_check_pct(uint8_t const pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ uint8_t const sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ MLD_ALIGN uint8_t message[1] = {0};+ int ret;+ MLD_ALLOC(signature, uint8_t, MLDSA_CRYPTO_BYTES, context);+ MLD_ALLOC(pk_test, uint8_t, MLDSA_CRYPTO_PUBLICKEYBYTES, context);++ if (signature == NULL || pk_test == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Copy public key for testing */+ mld_memcpy(pk_test, pk, MLDSA_CRYPTO_PUBLICKEYBYTES);++ /* Sign a test message using the original secret key */+ ret = mld_sign_signature(signature, message, sizeof(message), NULL, 0, sk,+ context);+ if (ret != 0)+ {+ goto cleanup;+ }++#if defined(MLD_CONFIG_KEYGEN_PCT_BREAKAGE_TEST)+ /* Deliberately break public key for testing purposes */+ if (mld_break_pct())+ {+ pk_test[0] = ~pk_test[0];+ }+#endif /* MLD_CONFIG_KEYGEN_PCT_BREAKAGE_TEST */++ /* Verify the signature using the (potentially corrupted) public key. */+ ret = mld_sign_verify(signature, message, sizeof(message), NULL, 0, pk_test,+ context);+ if (ret != 0)+ {+ goto cleanup;+ }++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(pk_test, uint8_t, MLDSA_CRYPTO_PUBLICKEYBYTES, context);+ MLD_FREE(signature, uint8_t, MLDSA_CRYPTO_BYTES, context);++ /* A failed signing operation or an invalid signature hint at a faulty+ * implementation and map to a dedicated error code for PCT failure. */+ if (ret == MLD_ERR_INVALID_SIGNATURE ||+ ret == MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED)+ {+ ret = MLD_ERR_PCT_FAIL;+ }++ /* Other error codes, e.g. platform failures like out of memory or+ * randomness failure, are passed on unmodified. */++ return ret;+}+#else /* MLD_CONFIG_KEYGEN_PCT */+static int mld_check_pct(uint8_t const pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ uint8_t const sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ /* Skip PCT */+ ((void)pk);+ ((void)sk);+ MLD_CONTEXT_UNUSED(context);+ return 0;+}+#endif /* !MLD_CONFIG_KEYGEN_PCT */++/**+ * Sample the short secret vectors s1 (length MLDSA_L) and s2 (length MLDSA_K)+ * with coefficients in [-MLDSA_ETA, MLDSA_ETA] from the seed.+ *+ * @spec{Implements @[FIPS204, Algorithm 33, ExpandS].}+ *+ * @param[out] s1 Output vector s1.+ * @param[out] s2 Output vector s2.+ * @param[in] seed Byte array with seed of length MLDSA_CRHBYTES.+ */+static void mld_sample_s1_s2(mld_polyvecl *s1, mld_polyveck *s2,+ const uint8_t seed[MLDSA_CRHBYTES])+__contract__(+ requires(memory_no_alias(s1, sizeof(mld_polyvecl)))+ requires(memory_no_alias(s2, sizeof(mld_polyveck)))+ requires(memory_no_alias(seed, MLDSA_CRHBYTES))+ assigns(object_whole(s1), object_whole(s2))+ ensures(forall(l0, 0, MLDSA_L, array_abs_bound(s1->vec[l0].coeffs, 0, MLDSA_N, MLDSA_ETA + 1)))+ ensures(forall(k0, 0, MLDSA_K, array_abs_bound(s2->vec[k0].coeffs, 0, MLDSA_N, MLDSA_ETA + 1)))+)+{+/* Sample short vectors s1 and s2 */+#if defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+ int i;+ uint16_t nonce = 0;+ /* Safety: The nonces are at most 14 (MLDSA_L + MLDSA_K - 1), and, hence, the+ * casts are safe. */+ for (i = 0; i < MLDSA_L; i++)+ {+ mld_poly_uniform_eta(&s1->vec[i], seed, (uint8_t)(nonce + i));+ }+ for (i = 0; i < MLDSA_K; i++)+ {+ mld_poly_uniform_eta(&s2->vec[i], seed, (uint8_t)(nonce + MLDSA_L + i));+ }+#else /* MLD_CONFIG_SERIAL_FIPS202_ONLY */+#if MLD_CONFIG_PARAMETER_SET == 44+ mld_poly_uniform_eta_4x(&s1->vec[0], &s1->vec[1], &s1->vec[2], &s1->vec[3],+ seed, 0, 1, 2, 3);+ mld_poly_uniform_eta_4x(&s2->vec[0], &s2->vec[1], &s2->vec[2], &s2->vec[3],+ seed, 4, 5, 6, 7);+#elif MLD_CONFIG_PARAMETER_SET == 65+ mld_poly_uniform_eta_4x(&s1->vec[0], &s1->vec[1], &s1->vec[2], &s1->vec[3],+ seed, 0, 1, 2, 3);+ mld_poly_uniform_eta_4x(&s1->vec[4], &s2->vec[0], &s2->vec[1],+ &s2->vec[2] /* irrelevant */, seed, 4, 5, 6,+ 0xFF /* irrelevant */);+ mld_poly_uniform_eta_4x(&s2->vec[2], &s2->vec[3], &s2->vec[4], &s2->vec[5],+ seed, 7, 8, 9, 10);+#elif MLD_CONFIG_PARAMETER_SET == 87+ mld_poly_uniform_eta_4x(&s1->vec[0], &s1->vec[1], &s1->vec[2], &s1->vec[3],+ seed, 0, 1, 2, 3);+ mld_poly_uniform_eta_4x(&s1->vec[4], &s1->vec[5], &s1->vec[6],+ &s2->vec[0] /* irrelevant */, seed, 4, 5, 6,+ 0xFF /* irrelevant */);+ mld_poly_uniform_eta_4x(&s2->vec[0], &s2->vec[1], &s2->vec[2], &s2->vec[3],+ seed, 7, 8, 9, 10);+ mld_poly_uniform_eta_4x(&s2->vec[4], &s2->vec[5], &s2->vec[6], &s2->vec[7],+ seed, 11, 12, 13, 14);+#endif /* MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY */+}++/**+ * Compute t = A*s1hat + s2 row by row, decompose each row into t0[k] and+ * t1[k] via power2round, and bit-pack t1[k] into pk_t1 and t0[k] into the+ * t0_packed buffer. Used by both keygen and pk_from_sk.+ *+ * @spec{Partially implements @[FIPS204, Algorithm 22, pkEncode] (t1) and+ * @[FIPS204, Algorithm 24, skEncode] (t0).}+ *+ * @param[out] pk_t1 Output buffer for packed t1 (size+ * MLDSA_K * MLDSA_POLYT1_PACKEDBYTES; i.e. the t1+ * region of pk).+ * @param[out] t0_packed Output buffer for packed t0 (size+ * MLDSA_K * MLDSA_POLYT0_PACKEDBYTES).+ * @param[in] s1hat s1 in NTT domain.+ * @param[in] s2 s2.+ * @param[in] rho Byte array containing seed rho.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @return - 0: Success.+ * - MLD_ERR_OUT_OF_MEMORY: If MLD_CONFIG_CUSTOM_ALLOC_FREE is used and+ * an allocation via MLD_CUSTOM_ALLOC returned NULL.+ */+MLD_MUST_CHECK_RETURN_VALUE+static int mld_compute_pack_t0_t1(+ uint8_t pk_t1[MLDSA_K * MLDSA_POLYT1_PACKEDBYTES],+ uint8_t t0_packed[MLDSA_K * MLDSA_POLYT0_PACKEDBYTES],+ const mld_polyvecl *s1hat, const mld_polyveck *s2,+ const uint8_t rho[MLDSA_SEEDBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk_t1, MLDSA_K * MLDSA_POLYT1_PACKEDBYTES))+ requires(memory_no_alias(t0_packed, MLDSA_K * MLDSA_POLYT0_PACKEDBYTES))+ requires(memory_no_alias(s1hat, sizeof(mld_polyvecl)))+ requires(memory_no_alias(s2, sizeof(mld_polyveck)))+ requires(memory_no_alias(rho, MLDSA_SEEDBYTES))+ requires(forall(l1, 0, MLDSA_L,+ array_abs_bound(s1hat->vec[l1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ requires(forall(k2, 0, MLDSA_K,+ array_bound(s2->vec[k2].coeffs, 0, MLDSA_N,+ MLD_POLYETA_UNPACK_LOWER_BOUND, MLDSA_ETA + 1)))+ assigns(memory_slice(pk_t1, MLDSA_K * MLDSA_POLYT1_PACKEDBYTES))+ assigns(memory_slice(t0_packed, MLDSA_K * MLDSA_POLYT0_PACKEDBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY))+{+ unsigned int k;+ int ret;+ MLD_ALLOC(mat, mld_polymat, 1, context);+ MLD_ALLOC(t0k, mld_poly, 1, context);+ MLD_ALLOC(t1k, mld_poly, 1, context);++ if (mat == NULL || t0k == NULL || t1k == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Expand matrix */+ mld_polyvec_matrix_expand(mat, rho);++ for (k = 0; k < MLDSA_K; k++)+ __loop__(+ assigns(k, memory_slice(pk_t1, MLDSA_K * MLDSA_POLYT1_PACKEDBYTES),+ memory_slice(t0_packed, MLDSA_K * MLDSA_POLYT0_PACKEDBYTES),+ memory_slice(t0k, sizeof(mld_poly)),+ memory_slice(t1k, sizeof(mld_poly))+ MLD_IF_REDUCE_RAM(, memory_slice(mat, sizeof(mld_polymat))))+ invariant(k <= MLDSA_K)+ decreases(MLDSA_K - k)+ )+ {+ /* t0k = (A * s1hat)_k in NTT domain */+ mld_polyvec_matrix_pointwise_montgomery_row(t0k, mat, s1hat, k);++ /* t0k = invNTT(t0k) */+ mld_poly_invntt_tomont(t0k);++ /* t0k += s2[k] */+ mld_poly_add(t0k, &s2->vec[k]);++ /* Reference: The following reduction is not present in the reference+ * implementation. Omitting this reduction requires the output+ * of the invntt to be small enough such that the addition of+ * s2 does not result in absolute values >= MLDSA_Q. While our+ * C, x86_64, and AArch64 invntt implementations produce small+ * enough values for this to work out, it complicates the+ * bounds reasoning. We instead add an additional reduction,+ * and can consequently, relax the bounds requirements for the+ * invntt.+ */+ mld_poly_reduce(t0k);++ /* Decompose into t1[k] and t0[k] (in place into t0k). */+ mld_poly_caddq(t0k);+ mld_poly_power2round(t1k, t0k, t0k);++ /* Pack t1[k] into pk and t0[k] into the t0 output buffer. */+ mld_polyt1_pack(pk_t1 + k * MLDSA_POLYT1_PACKEDBYTES, t1k);+ mld_polyt0_pack(t0_packed + k * MLDSA_POLYT0_PACKEDBYTES, t0k);+ }++ ret = 0;+cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(t1k, mld_poly, 1, context);+ MLD_FREE(t0k, mld_poly, 1, context);+ MLD_FREE(mat, mld_polymat, 1, context);+ return ret;+}++MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_keypair_internal(uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ const uint8_t seed[MLDSA_SEEDBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret;+ const uint8_t *rho, *rhoprime, *key;++ MLD_ALLOC(seedbuf, uint8_t, 2 * MLDSA_SEEDBYTES + MLDSA_CRHBYTES, context);+ MLD_ALLOC(inbuf, uint8_t, MLDSA_SEEDBYTES + 2, context);+ MLD_ALLOC(tr, uint8_t, MLDSA_TRBYTES, context);+ MLD_ALLOC(s1, mld_polyvecl, 1, context);+ MLD_ALLOC(s2, mld_polyveck, 1, context);++ if (seedbuf == NULL || inbuf == NULL || tr == NULL || s1 == NULL ||+ s2 == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Get randomness for rho, rhoprime and key */+ mld_memcpy(inbuf, seed, MLDSA_SEEDBYTES);+ inbuf[MLDSA_SEEDBYTES + 0] = MLDSA_K;+ inbuf[MLDSA_SEEDBYTES + 1] = MLDSA_L;+ mld_shake256(seedbuf, 2 * MLDSA_SEEDBYTES + MLDSA_CRHBYTES, inbuf,+ MLDSA_SEEDBYTES + 2);+ rho = seedbuf;+ rhoprime = rho + MLDSA_SEEDBYTES;+ key = rhoprime + MLDSA_CRHBYTES;++ /* Constant time: rho is part of the public key and, hence, public. */+ MLD_CT_TESTING_DECLASSIFY(rho, MLDSA_SEEDBYTES);++ /* Sample s1 and s2 */+ mld_sample_s1_s2(s1, s2, rhoprime);++ /* Pack s1 into sk before NTT */+ mld_pack_sk_s1(sk, s1);++ /* NTT s1 in place to use as s1hat */+ mld_polyvecl_ntt(s1);++ /* Pack rho into pk */+ mld_memcpy(pk + MLDSA_PK_RHO_OFFSET, rho, MLDSA_SEEDBYTES);++ /* Compute t = A*s1hat + s2 row by row, decompose into t1/t0, and pack+ * t1 into pk and t0 directly into the t0 region of sk. */+ ret = mld_compute_pack_t0_t1(pk + MLDSA_PK_T1_OFFSET, sk + MLDSA_SK_T0_OFFSET,+ s1, s2, rho, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ /* Compute tr = H(pk) */+ mld_shake256(tr, MLDSA_TRBYTES, pk, MLDSA_CRYPTO_PUBLICKEYBYTES);++ /* Pack remaining secret key components (s1 and t0 already packed) */+ mld_pack_sk_rho_key_tr_s2(sk, rho, tr, key, s2);++ /* Constant time: pk is the public key, inherently public data */+ MLD_CT_TESTING_DECLASSIFY(pk, MLDSA_CRYPTO_PUBLICKEYBYTES);++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(s2, mld_polyveck, 1, context);+ MLD_FREE(s1, mld_polyvecl, 1, context);+ MLD_FREE(tr, uint8_t, MLDSA_TRBYTES, context);+ MLD_FREE(inbuf, uint8_t, MLDSA_SEEDBYTES + 2, context);+ MLD_FREE(seedbuf, uint8_t, 2 * MLDSA_SEEDBYTES + MLDSA_CRHBYTES, context);++ /* Pairwise Consistency Test (PCT) @[FIPS140_3_IG, p.87] */+ /* Do this after freeing all temporaries. */+ if (ret == 0)+ {+ ret = mld_check_pct(pk, sk, context);+ }++ if (ret != 0)+ {+ /* Clear caller outputs on failure. */+ mld_zeroize(pk, MLDSA_CRYPTO_PUBLICKEYBYTES);+ mld_zeroize(sk, MLDSA_CRYPTO_SECRETKEYBYTES);+ }++ return ret;+}++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_keypair(uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret;+ MLD_ALLOC(seed, uint8_t, MLDSA_SEEDBYTES, context);++ if (seed == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ if (mld_randombytes(seed, MLDSA_SEEDBYTES) != 0)+ {+ ret = MLD_ERR_RNG_FAIL;+ goto cleanup;+ }+ MLD_CT_TESTING_SECRET(seed, MLDSA_SEEDBYTES);+ ret = mld_sign_keypair_internal(pk, sk, seed, context);++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(seed, uint8_t, MLDSA_SEEDBYTES, context);+ return ret;+}+#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Abstracts application of SHAKE256 to one, two or three blocks of data,+ * yielding a user-requested size of output.+ *+ * @param[out] out Pointer to output.+ * @param outlen Requested output length in bytes.+ * @param[in] in1 Pointer to input block 1. Must NOT be NULL.+ * @param in1len Length of input in1 in bytes.+ * @param[in] in2 Pointer to input block 2. May be NULL if in2len == 0,+ * in which case this block is ignored.+ * @param in2len Length of input in2 in bytes.+ * @param[in] in3 Pointer to input block 3. May be NULL if in3len == 0,+ * in which case this block is ignored.+ * @param in3len Length of input in3 in bytes.+ */+static void mld_H(uint8_t *out, size_t outlen, const uint8_t *in1,+ size_t in1len, const uint8_t *in2, size_t in2len,+ const uint8_t *in3, size_t in3len)+__contract__(+ requires(in1len <= MLD_MAX_BUFFER_SIZE)+ requires(in2len <= MLD_MAX_BUFFER_SIZE)+ requires(in3len <= MLD_MAX_BUFFER_SIZE)+ requires(outlen <= 8 * SHAKE256_RATE /* somewhat arbitrary bound */)+ requires(memory_no_alias(in1, in1len))+ requires(in2len == 0 || memory_no_alias(in2, in2len))+ requires(in3len == 0 || memory_no_alias(in3, in3len))+ requires(memory_no_alias(out, outlen))+ assigns(memory_slice(out, outlen))+)+{+ mld_shake256ctx state;+ mld_shake256_init(&state);+ mld_shake256_absorb(&state, in1, in1len);+ if (in2len != 0)+ {+ mld_shake256_absorb(&state, in2, in2len);+ }+ if (in3len != 0)+ {+ mld_shake256_absorb(&state, in3, in3len);+ }+ mld_shake256_finalize(&state);+ mld_shake256_squeeze(out, outlen, &state);+ mld_shake256_release(&state);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(&state, sizeof(state));+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+/* MLD_MAX_KAPPA (see params.h) bounds the rejection-sampling counter kappa;+ * MLD_MAX_SIGNING_ATTEMPTS below turns that into a bound on attempts. */++/**+ * Compute z = y + s1*c, check that z has coefficients smaller than+ * MLDSA_GAMMA1 - MLDSA_BETA, and pack z into the signature buffer.+ *+ * @reference{This function is inlined into mld_sign_signature in the+ * reference implementation.}+ *+ * @param[in,out] sig Output signature.+ * @param[in] cp Challenge polynomial.+ * @param[in] s1hat Secret vector s1 in NTT domain.+ * @param[in] y Masking vector y (or seed in REDUCE_RAM mode).+ * @param[out] z Scratch polynomial for z computation.+ * @param[out] tmp Scratch polynomial.+ *+ * @return - 0: Success (z has coefficients smaller than+ * MLDSA_GAMMA1 - MLDSA_BETA).+ * - MLD_ERR_FAIL: z rejected (norm check failed).+ * - MLD_ERR_OUT_OF_MEMORY: If MLD_CONFIG_CUSTOM_ALLOC_FREE is used and+ * an allocation via MLD_CUSTOM_ALLOC returned NULL.+ */+MLD_MUST_CHECK_RETURN_VALUE+static int mld_compute_pack_z(uint8_t sig[MLDSA_CRYPTO_BYTES],+ const mld_poly *cp, const mld_sk_s1hat *s1hat,+ const mld_yvec *y, mld_poly *z, mld_poly *tmp)+__contract__(+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(cp, sizeof(mld_poly)))+ requires(memory_no_alias(s1hat, sizeof(mld_sk_s1hat)))+ requires(memory_no_alias(y, sizeof(mld_yvec)))+ requires(memory_no_alias(z, sizeof(mld_poly)))+ requires(memory_no_alias(tmp, sizeof(mld_poly)))+ requires(array_abs_bound(cp->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ MLD_IF_NOT_REDUCE_RAM(+ requires(forall(k0, 0, MLDSA_L,+ array_bound(y->vec.vec[k0].coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1)))+ requires(forall(k1, 0, MLDSA_L, array_abs_bound(s1hat->vec.vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ )+ MLD_IF_REDUCE_RAM(+ requires(memory_no_alias(s1hat->packed, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(y->rhoprime, MLDSA_CRHBYTES))+ requires(y->kappa <= MLD_MAX_KAPPA)+ )+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+ assigns(memory_slice(z, sizeof(mld_poly)))+ assigns(memory_slice(tmp, sizeof(mld_poly)))+ ensures(return_value == 0 || return_value == MLD_ERR_FAIL ||+ return_value == MLD_ERR_OUT_OF_MEMORY)+)+{+ unsigned int i;+ uint32_t z_invalid;+ for (i = 0; i < MLDSA_L; i++)+ __loop__(+ assigns(i, memory_slice(z, sizeof(mld_poly)),+ memory_slice(tmp, sizeof(mld_poly)),+ memory_slice(sig, MLDSA_CRYPTO_BYTES))+ invariant(i <= MLDSA_L)+ decreases(MLDSA_L - i)+ )+ {+ mld_sk_s1hat_get_poly(z, s1hat, i);+ mld_poly_pointwise_montgomery(z, cp);+ mld_poly_invntt_tomont(z);+ mld_yvec_get_poly(tmp, y, i);+ mld_poly_add(z, tmp);+ mld_poly_reduce(z);++ z_invalid = mld_poly_chknorm(z, MLDSA_GAMMA1 - MLDSA_BETA);+ /* Constant time: It is fine (and prohibitively expensive to avoid)+ * to leak the result of the norm check and which polynomial in z caused a+ * rejection. It would even be okay to leak which coefficient led to+ * rejection as the candidate signature will be discarded anyway.+ * See Section 5.5 of @[Round3_Spec]. */+ MLD_CT_TESTING_DECLASSIFY(&z_invalid, sizeof(uint32_t));+ if (z_invalid)+ {+ return MLD_ERR_FAIL; /* reject */+ }+ /* If z is valid, then its coefficients are bounded by+ * MLDSA_GAMMA1 - MLDSA_BETA. This will be needed below+ * to prove the pre-condition of pack_sig_z() */+ mld_assert_abs_bound(z, MLDSA_N, (MLDSA_GAMMA1 - MLDSA_BETA));++ /* After the norm check, the distribution of each coefficient of z is+ * independent of the secret key and it can, hence, be considered+ * public. It is, hence, okay to immediately pack it into the user-provided+ * signature buffer. */+ mld_pack_sig_z(sig, z, i);+ }+ return 0;+}++/* Effective bound on signing attempts: the configured bound+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS (see mldsa_native_config.h) if set, otherwise+ * the hard type-safety bound MLD_MAX_KAPPA / MLDSA_L (see MLD_MAX_KAPPA in+ * params.h). */+#if defined(MLD_CONFIG_MAX_SIGNING_ATTEMPTS)++#if !defined(MLD_ALLOW_NONCOMPLIANT_SIGNING_BOUND) && \+ MLD_CONFIG_MAX_SIGNING_ATTEMPTS < 821+#error Bad configuration: MLD_CONFIG_MAX_SIGNING_ATTEMPTS must be >= 821 for FIPS 204 compliance @[FIPS204, Appendix C] @[FIPS204_UPDATES]+#endif++#if MLD_CONFIG_MAX_SIGNING_ATTEMPTS < 1+#error Bad configuration: MLD_CONFIG_MAX_SIGNING_ATTEMPTS must be >= 1+#endif++#if MLD_CONFIG_MAX_SIGNING_ATTEMPTS > MLD_MAX_KAPPA / MLDSA_L+#error Bad configuration: MLD_CONFIG_MAX_SIGNING_ATTEMPTS exceeds the maximum allowed value.+#endif++#define MLD_MAX_SIGNING_ATTEMPTS MLD_CONFIG_MAX_SIGNING_ATTEMPTS+#else /* MLD_CONFIG_MAX_SIGNING_ATTEMPTS */+#define MLD_MAX_SIGNING_ATTEMPTS (MLD_MAX_KAPPA / MLDSA_L)+#endif /* !MLD_CONFIG_MAX_SIGNING_ATTEMPTS */++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint16_t mld_get_max_signing_attempts(void)+__contract__(+ ensures(return_value >= 1)+ ensures(return_value <= MLD_MAX_KAPPA / MLDSA_L)+)+{+ /* cassert(0) ensures CBMC uses the contract rather than inlining the body,+ * keeping proofs agnostic of the configured value. */+ cassert(0);+ return MLD_MAX_SIGNING_ATTEMPTS;+}++/**+ * Attempt to generate a single signature: one iteration of the+ * ML-DSA.Sign_internal rejection-sampling loop.+ *+ * @spec{Implements one iteration of the rejection-sampling loop body of+ * @[FIPS204, Algorithm 7, ML-DSA.Sign_internal] (lines 11-30) plus, on success,+ * the sigEncode step (line 33). The per-signature setup (Algorithm 7 lines 1-7:+ * skDecode, NTT of s1/s2/t0, ExpandA, and computation of mu and rhoprime) and+ * the loop itself (lines 8-10, 31-32) live in the caller+ * mld_sign_signature_internal; kappa is this iteration's counter, used to+ * sample y.}+ *+ * @reference{This code differs from the reference implementation in that it+ * factors out the core signature generation step into a distinct function+ * here in order to improve efficiency of CBMC proof.}+ *+ * @param[out] sig Pointer to output signature.+ * @param[in] mu Pointer to message or hash of exactly MLDSA_CRHBYTES+ * bytes.+ * @param[in] rhoprime Pointer to randomness seed.+ * @param kappa Counter for this iteration (= attempt*MLDSA_L).+ * @param[in] mat Expanded matrix.+ * @param[in] s1hat Secret vector s1 in NTT domain.+ * @param[in] s2hat Secret vector s2 in NTT domain.+ * @param[in] t0hat Vector t0 in NTT domain.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @return - 0: Signature generation succeeded.+ * - MLD_ERR_FAIL: Signature rejected (norm check failed).+ * - MLD_ERR_OUT_OF_MEMORY: If MLD_CONFIG_CUSTOM_ALLOC_FREE is used and+ * an allocation via MLD_CUSTOM_ALLOC returned NULL.+ */+MLD_MUST_CHECK_RETURN_VALUE+/* NOLINTNEXTLINE(readability-function-cognitive-complexity) */+static int mld_attempt_signature_generation(+ uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *mu,+ const uint8_t rhoprime[MLDSA_CRHBYTES], uint16_t kappa, mld_polymat *mat,+ const mld_sk_s1hat *s1hat, const mld_sk_s2hat *s2hat,+ const mld_sk_t0hat *t0hat, MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(mu, MLDSA_CRHBYTES))+ requires(memory_no_alias(rhoprime, MLDSA_CRHBYTES))+ requires(memory_no_alias(mat, sizeof(mld_polymat)))+ requires(memory_no_alias(s1hat, sizeof(mld_sk_s1hat)))+ requires(memory_no_alias(s2hat, sizeof(mld_sk_s2hat)))+ requires(memory_no_alias(t0hat, sizeof(mld_sk_t0hat)))+ requires(kappa <= MLD_MAX_KAPPA)+ MLD_IF_NOT_REDUCE_RAM(+ requires(forall(k1, 0, MLDSA_K, forall(l1, 0, MLDSA_L,+ array_bound(mat->vec[k1].vec[l1].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+ requires(forall(k2, 0, MLDSA_K, array_abs_bound(t0hat->vec.vec[k2].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ requires(forall(k3, 0, MLDSA_L, array_abs_bound(s1hat->vec.vec[k3].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ requires(forall(k4, 0, MLDSA_K, array_abs_bound(s2hat->vec.vec[k4].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ )+ MLD_IF_REDUCE_RAM(+ requires(memory_no_alias(s1hat->packed, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(s2hat->packed, MLDSA_K * MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(t0hat->packed, MLDSA_K * MLDSA_POLYT0_PACKEDBYTES))+ )+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+ MLD_IF_REDUCE_RAM(+ assigns(memory_slice(mat, sizeof(mld_polymat)))+ )+ ensures(return_value == 0 || return_value == MLD_ERR_FAIL ||+ return_value == MLD_ERR_OUT_OF_MEMORY)+)+{+ unsigned int k;+ uint32_t w0_invalid, h_invalid;+ int ret;++ typedef union+ {+ mld_polyveck w1;+ mld_polyvecl tmp;+ } w1tmp_u;+ mld_polyveck *w1;+ mld_polyvecl *tmp;++ MLD_ALLOC(challenge_bytes, uint8_t, MLDSA_CTILDEBYTES, context);+ MLD_ALLOC(y, mld_yvec, 1, context);+ MLD_ALLOC(z, mld_poly, 1, context);+ MLD_ALLOC(w1tmp, w1tmp_u, 1, context);+ MLD_ALLOC(w0, mld_polyveck, 1, context);+ MLD_ALLOC(cp, mld_poly, 1, context);+ MLD_ALLOC(t, mld_poly, 1, context);++ if (challenge_bytes == NULL || y == NULL || z == NULL || w1tmp == NULL ||+ w0 == NULL || cp == NULL || t == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }+ w1 = &w1tmp->w1;+ tmp = &w1tmp->tmp;++ /* @[FIPS204, Algorithm 7, line 11] y <- ExpandMask(rhoprime, kappa). */+ mld_yvec_init(y, rhoprime, kappa);++ /* @[FIPS204, Algorithm 7, line 12] w <- invNTT(A_hat o NTT(y)). This call+ * performs the whole line: it NTTs y, accumulates the pointwise product with+ * A_hat, and applies the inverse NTT. In REDUCE_RAM mode the y sampling is+ * fused into the same pass. */+ mld_polyvec_matrix_pointwise_montgomery_yvec(w0, mat, y, tmp);++ /* @[FIPS204, Algorithm 7, line 13] w1 <- HighBits(w), here together with the+ * low part: Decompose yields w = 2*GAMMA2*w1 + w0, keeping both w1 and w0+ * (w0 is reused below in the line-21/26 alternative, see further down). */+ mld_polyveck_caddq(w0);+ mld_polyveck_decompose(w1, w0);++ /* @[FIPS204, Algorithm 7, line 15] ctilde <- H(mu || w1Encode(w1), lambda/4).+ * w1Encode(w1) is packed into the w1 region of sig (mld_polyveck_pack_w1),+ * then absorbed by H together with mu. */+ mld_polyveck_pack_w1(sig, w1);++ mld_H(challenge_bytes, MLDSA_CTILDEBYTES, mu, MLDSA_CRHBYTES, sig,+ MLDSA_K * MLDSA_POLYW1_PACKEDBYTES, NULL, 0);+ /* Constant time: Leaking challenge_bytes does not reveal any information+ * about the secret key as H() is modelled as random oracle.+ * This also applies to challenges for rejected signatures.+ * See Section 5.5 of @[Round3_Spec]. */+ MLD_CT_TESTING_DECLASSIFY(challenge_bytes, MLDSA_CTILDEBYTES);+ /* @[FIPS204, Algorithm 7, line 16] c <- SampleInBall(ctilde) and+ * @[FIPS204, Algorithm 7, line 17] c_hat <- NTT(c). */+ mld_poly_challenge(cp, challenge_bytes);+ mld_poly_ntt(cp);++ /* @[FIPS204, Algorithm 7, lines 18+20] cs1 <- invNTT(c_hat o s1_hat) and+ * z <- y + cs1, followed by the line-23 norm check ||z||_inf >= GAMMA1 -+ * BETA. mld_compute_pack_z fuses all three per polynomial and, on success,+ * packs z into sig; it returns MLD_ERR_FAIL if the norm check rejects z. */+ ret = mld_compute_pack_z(sig, cp, s1hat, y, t, z);+ if (ret != 0)+ {+ goto cleanup;+ }++ /* The remaining steps realize @[FIPS204, Algorithm 7, lines 21-28] (the+ * low-bits norm check and the hint h) via the faster alternative formulation+ * of @[Round3_Spec, Section 5.1]. @[FIPS204] explicitly permits this: the+ * note accompanying Algorithm 7 states that the validity checks on z and the+ * computation of h may instead be implemented "as described in Section 5.1 of+ * [6]", and that reference is @[Round3_Spec, Section 5.1].+ *+ * The loop below builds w0 - cs2 + ct0 in place in w0; w1 is unmodified, and+ * is HighBits(w) from line 13. Those are the inputs to the streamlined+ * computation of MakeHint explained below.+ *+ * Low-bits norm check:+ * @[FIPS204, Algorithm 7, line 21] computes r0 = LowBits(w - cs2) and line+ * 23 rejects when ||r0||_inf >= GAMMA2 - BETA. By @[Round3_Spec, Section+ * 5.1] (Lemma 3), this line-23 check on r0 = LowBits(w - cs2) is implied by+ * ||w0 - cs2||_inf < GAMMA2 - BETA, where w0 is the low part of w. In our+ * context, w0 already holds the low part of w from the line-13 Decompose;+ * after subtracting cs2 from it in place, the mld_poly_chknorm(w0, GAMMA2 -+ * BETA) call below is exactly that check.+ *+ * Hint:+ * @[FIPS204, Algorithm 7, line 26] sets h = MakeHint(-ct0, w - cs2 + ct0),+ * and line 28 rejects when ||ct0||_inf >= GAMMA2 or h has more than OMEGA+ * nonzero coefficients. @[Round3_Spec, Section 5.1] provides the following+ * alternative description for MakeHint(-ct0, w - cs2 + ct0): a hint bit is+ * zero exactly when the coefficient of w0 - cs2 + ct0 lies in+ * (-GAMMA2, GAMMA2], or equals -GAMMA2 while the matching w1 coefficient is+ * zero (the Decompose border case), and is set otherwise. This equivalence+ * is precisely what mld_pack_sig_h -> mld_make_hint compute from w0+ * (= w0 - cs2 + ct0) and w1. The line-28 ||ct0||_inf >= GAMMA2 check is the+ * mld_poly_chknorm(z, GAMMA2) call on ct0 below; the weight bound is+ * enforced by mld_pack_sig_h.+ *+ * Building w0 per-component and checking norms incrementally also avoids+ * allocating a full polyveck for h. */+ for (k = 0; k < MLDSA_K; k++)+ __loop__(+ assigns(k,+ object_whole(z),+ object_whole(w0))+ invariant(k <= MLDSA_K)+ invariant(forall(k0, k, MLDSA_K,+ array_abs_bound(w0->vec[k0].coeffs, 0, MLDSA_N, MLDSA_GAMMA2 + 1)))+ decreases(MLDSA_K - k)+ )+ {+ /* @[FIPS204, Algorithm 7, line 19] cs2[k] <- invNTT(c_hat o s2_hat)[k],+ * then subtract from w0[k] to form (w0 - cs2)[k]. */+ mld_sk_s2hat_get_poly(z, s2hat, k);+ mld_poly_pointwise_montgomery(z, cp);+ mld_poly_invntt_tomont(z);++ mld_poly_sub(&w0->vec[k], z);+ mld_poly_reduce(&w0->vec[k]);++ /* Low-bits norm check (see block comment above): the line-23 check on+ * r0 = LowBits(w - cs2) holds via ||w0 - cs2||_inf < GAMMA2 - BETA. */+ w0_invalid = mld_poly_chknorm(&w0->vec[k], MLDSA_GAMMA2 - MLDSA_BETA);+ /* Constant time: w0_invalid may be leaked - see comment for z_invalid. */+ MLD_CT_TESTING_DECLASSIFY(&w0_invalid, sizeof(uint32_t));+ if (w0_invalid)+ {+ ret = MLD_ERR_FAIL; /* reject */+ goto cleanup;+ }++ /* @[FIPS204, Algorithm 7, line 25] ct0[k] <- invNTT(c_hat o t0_hat)[k]. */+ mld_sk_t0hat_get_poly(z, t0hat, k);+ mld_poly_pointwise_montgomery(z, cp);+ mld_poly_invntt_tomont(z);+ mld_poly_reduce(z);++ /* @[FIPS204, Algorithm 7, line 28] reject when ||ct0||_inf >= GAMMA2 (the+ * second part, the OMEGA weight bound, is enforced by mld_pack_sig_h). */+ h_invalid = mld_poly_chknorm(z, MLDSA_GAMMA2);+ /* Constant time: h_invalid may be leaked - see comment for z_invalid. */+ MLD_CT_TESTING_DECLASSIFY(&h_invalid, sizeof(uint32_t));+ if (h_invalid)+ {+ ret = MLD_ERR_FAIL; /* reject */+ goto cleanup;+ }++ /* Add ct0[k] to (w0 - cs2)[k], leaving (w0 - cs2 + ct0)[k] in w0[k] -- the+ * MakeHint input prepared for mld_pack_sig_h (see block comment above). */+ mld_poly_add(&w0->vec[k], z);+ }++ /* Constant time: At this point all norm checks have passed and we, hence,+ * know that the signature does not leak any secret information.+ * Consequently, any value that can be computed from the signature and public+ * key is considered public.+ * w0 and w1 are public as they can be computed from Az - ct = \alpha w1 + w0.+ * h=c*t0 is public as both c and t0 are considered public.+ * While t0 is not part of the public key, it can be reconstructed from+ * a small number of signatures and need not be regarded as secret+ * (see @[FIPS204, Section 6.1]).+ */+ MLD_CT_TESTING_DECLASSIFY(w0, sizeof(*w0));+ MLD_CT_TESTING_DECLASSIFY(w1, sizeof(*w1));++ /* @[FIPS204, Algorithm 7, line 33] sigEncode(ctilde, z mod+/- q, h) is split+ * across three calls: z was already packed by mld_compute_pack_z, this call+ * packs ctilde, and mld_pack_sig_h below packs the hint h. */+ mld_pack_sig_c(sig, challenge_bytes);++ /* @[FIPS204, Algorithm 7, line 26] h <- MakeHint(-ct0, w - cs2 + ct0),+ * computed from (w0 = w0 - cs2 + ct0, w1) as described in the block comment+ * above, and packed as the h component of the line-33 sigEncode. Returns+ * MLD_ERR_FAIL if h would exceed OMEGA nonzero coefficients (the remaining+ * part of the line-28 check), in which case we reject. */+ ret = mld_pack_sig_h(sig, w0, w1);+ if (ret != 0)+ {+ goto cleanup;+ }++ /* Constant time: At this point it is clear that the signature is valid - it+ * can, hence, be considered public. */+ MLD_CT_TESTING_DECLASSIFY(sig, MLDSA_CRYPTO_BYTES);+ ret = 0; /* success */++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(t, mld_poly, 1, context);+ MLD_FREE(cp, mld_poly, 1, context);+ MLD_FREE(w0, mld_polyveck, 1, context);+ MLD_FREE(w1tmp, w1tmp_u, 1, context);+ MLD_FREE(z, mld_poly, 1, context);+ MLD_FREE(y, mld_yvec, 1, context);+ MLD_FREE(challenge_bytes, uint8_t, MLDSA_CTILDEBYTES, context);++ return ret;+}+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_internal(uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t *m, size_t mlen,+ const uint8_t *pre, size_t prelen,+ const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ int externalmu,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret;+ uint8_t *rho, *tr, *key, *mu, *rhoprime;+ uint16_t attempt;+ const uint16_t max_signing_attempts = mld_get_max_signing_attempts();+ MLD_ALLOC(seedbuf, uint8_t,+ 2 * MLDSA_SEEDBYTES + MLDSA_TRBYTES + 2 * MLDSA_CRHBYTES, context);+ MLD_ALLOC(mat, mld_polymat, 1, context);+ MLD_ALLOC(s1hat, mld_sk_s1hat, 1, context);+ MLD_ALLOC(t0hat, mld_sk_t0hat, 1, context);+ MLD_ALLOC(s2hat, mld_sk_s2hat, 1, context);++ if (seedbuf == NULL || mat == NULL || s1hat == NULL || t0hat == NULL ||+ s2hat == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* If a resume hook is configured via MLD_CONFIG_SIGN_HOOK_RESUME, it provides+ * the attempt to resume from after an earlier pause. Otherwise, we start at+ * 0. Clamp to max_signing_attempts. */+ attempt = mld_sign_resume(context);+ if (attempt > max_signing_attempts)+ {+ attempt = max_signing_attempts;+ }++ rho = seedbuf;+ tr = rho + MLDSA_SEEDBYTES;+ key = tr + MLDSA_TRBYTES;+ mu = key + MLDSA_SEEDBYTES;+ rhoprime = mu + MLDSA_CRHBYTES;+ /* @[FIPS204, Algorithm 7, line 1] (rho, K, tr, s1, s2, t0) <- skDecode(sk)+ * and @[FIPS204, Algorithm 7, lines 2-4] s1_hat/s2_hat/t0_hat <- NTT(...):+ * mld_unpack_sk returns s1hat, s2hat, t0hat already in NTT domain. The spec's+ * private random seed K is held in the local variable key. */+ mld_unpack_sk(rho, tr, key, t0hat, s1hat, s2hat, sk);++ if (!externalmu)+ {+ /* @[FIPS204, Algorithm 7, line 6] mu <- H(BytesToBits(tr) || M', 64). */+ mld_H(mu, MLDSA_CRHBYTES, tr, MLDSA_TRBYTES, pre, prelen, m, mlen);+ }+ else+ {+ /* mu has been provided directly (external-mu variant; line 6 done by the+ * caller in a separate cryptographic module). */+ mld_memcpy(mu, m, MLDSA_CRHBYTES);+ }++ /* @[FIPS204, Algorithm 7, line 7] rhoprime <- H(K || rnd || mu, 64). */+ mld_H(rhoprime, MLDSA_CRHBYTES, key, MLDSA_SEEDBYTES, rnd, MLDSA_RNDBYTES, mu,+ MLDSA_CRHBYTES);++ /* Constant time: rho is part of the public key and, hence, public. */+ MLD_CT_TESTING_DECLASSIFY(rho, MLDSA_SEEDBYTES);+ /* @[FIPS204, Algorithm 7, line 5] A_hat <- ExpandA(rho). */+ mld_polyvec_matrix_expand(mat, rho);++ /* @[FIPS204, Algorithm 7, lines 8-10 and 31-32] the rejection-sampling loop,+ * tracked by attempt (kappa = attempt*MLDSA_L). Each iteration's body (lines+ * 11-30) plus, on success, the line-33 sigEncode are performed by+ * mld_attempt_signature_generation. */++ /* Reference: the reference loops unboundedly; we instead iterate over the+ * bounded range [0, max_signing_attempts) for predictable termination.+ * A success or fatal error exits via goto cleanup; running to completion+ * means every attempt was rejected; with a FIPS compliant choice of+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS, this should never happen. */+ for (; attempt < max_signing_attempts; attempt++)+ __loop__(+ MLD_IF_NOT_REDUCE_RAM(+ assigns(attempt, ret, memory_slice(sig, MLDSA_CRYPTO_BYTES))+ )+ MLD_IF_REDUCE_RAM(+ assigns(attempt, ret, memory_slice(sig, MLDSA_CRYPTO_BYTES),+ memory_slice(mat, sizeof(mld_polymat)))+ )+ invariant(attempt <= max_signing_attempts)++ /* t0, s1, s2, and mat are initialized above and are NOT changed by this */+ /* loop. We can therefore re-assert their bounds here as part of the */+ /* loop invariant. This makes proof noticeably faster with CBMC */+ MLD_IF_NOT_REDUCE_RAM(+ invariant(forall(k1, 0, MLDSA_K, forall(l1, 0, MLDSA_L,+ array_bound(mat->vec[k1].vec[l1].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+ invariant(forall(k2, 0, MLDSA_K, array_abs_bound(t0hat->vec.vec[k2].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ invariant(forall(k3, 0, MLDSA_L, array_abs_bound(s1hat->vec.vec[k3].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ invariant(forall(k4, 0, MLDSA_K, array_abs_bound(s2hat->vec.vec[k4].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ )+ decreases(max_signing_attempts - attempt)+ )+ {+ /* Safety: attempt < max_signing_attempts <= MLD_MAX_KAPPA / MLDSA_L, so+ * kappa <= MLD_MAX_KAPPA and the cast is safe. */+ const uint16_t kappa = (uint16_t)(attempt * MLDSA_L);++ /* Query configurable signing hook whether signing should be paused.+ * This is skipped by default and only used if the user sets the+ * configuration option MLD_CONFIG_SIGN_HOOK_ATTEMPT. */+ if (mld_sign_attempt(attempt, context) != 0)+ {+ ret = MLD_ERR_SIGNING_PAUSED;+ goto cleanup;+ }++ ret = mld_attempt_signature_generation(sig, mu, rhoprime, kappa, mat, s1hat,+ s2hat, t0hat, context);++ /* Decide whether to keep trying based on the return value:+ * - ret == 0: a valid signature was produced; we are done.+ * - ret == MLD_ERR_FAIL: this candidate was rejected by one of the norm+ * or hint checks. We continue the loop and try again with the next+ * nonce.+ * - any other value (e.g. MLD_ERR_OUT_OF_MEMORY): an unrecoverable error+ * occurred, so we propagate it to the caller. */+ if (ret == 0)+ {+ /* Signing succeeded: record the attempt that succeeded. No-op in the+ * default build. */+ mld_sign_finish(attempt, context);+ goto cleanup;+ }+ if (ret != MLD_ERR_FAIL)+ {+ goto cleanup;+ }+ }++ /* Loop ran to completion: all attempts rejected, budget exhausted.+ * This should never happen with a FIPS compliant choice of+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS. */+ ret = MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED;++cleanup:++ if (ret != 0)+ {+ /* To be on the safe-side, we zeroize the signature buffer. */+ mld_zeroize(sig, MLDSA_CRYPTO_BYTES);+ }++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(s2hat, mld_sk_s2hat, 1, context);+ MLD_FREE(t0hat, mld_sk_t0hat, 1, context);+ MLD_FREE(s1hat, mld_sk_s1hat, 1, context);+ MLD_FREE(mat, mld_polymat, 1, context);+ MLD_FREE(seedbuf, uint8_t,+ 2 * MLDSA_SEEDBYTES + MLDSA_TRBYTES + 2 * MLDSA_CRHBYTES, context);+ return ret;+}++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature(uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ size_t pre_len;+ int ret;+ MLD_ALLOC(pre, uint8_t, MLD_DOMAIN_SEPARATION_MAX_BYTES, context);+ MLD_ALLOC(rnd, uint8_t, MLDSA_RNDBYTES, context);++ if (pre == NULL || rnd == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Prepare domain separation prefix for pure ML-DSA */+ pre_len = mld_prepare_domain_separation_prefix(pre, NULL, 0, ctx, ctxlen,+ MLD_PREHASH_NONE);+ if (pre_len == 0)+ {+ ret = MLD_ERR_INVALID_ARG;+ goto cleanup;+ }++ /* Randomized variant of ML-DSA. If you need the deterministic variant,+ * call mld_sign_signature_internal directly with all-zero rnd. */+ if (mld_randombytes(rnd, MLDSA_RNDBYTES) != 0)+ {+ ret = MLD_ERR_RNG_FAIL;+ goto cleanup;+ }+ MLD_CT_TESTING_SECRET(rnd, MLDSA_RNDBYTES);++ ret = mld_sign_signature_internal(sig, m, mlen, pre, pre_len, rnd, sk, 0,+ context);++cleanup:+ if (ret != 0)+ {+ /* To be on the safe-side, make sure sig has a well-defined value, even in+ * the case of error.+ *+ * If we come from mld_sign_signature_internal, this is redundant, but the+ * error case should not be the norm, and the added cost of the zeroization+ * insignificant. */+ mld_zeroize(sig, MLDSA_CRYPTO_BYTES);+ }++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(rnd, uint8_t, MLDSA_RNDBYTES, context);+ MLD_FREE(pre, uint8_t, MLD_DOMAIN_SEPARATION_MAX_BYTES, context);++ return ret;+}+#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */++#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_extmu(uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t mu[MLDSA_CRHBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret;+ MLD_ALLOC(rnd, uint8_t, MLDSA_RNDBYTES, context);++ if (rnd == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Randomized variant of ML-DSA. If you need the deterministic variant,+ * call mld_sign_signature_internal directly with all-zero rnd. */+ if (mld_randombytes(rnd, MLDSA_RNDBYTES) != 0)+ {+ ret = MLD_ERR_RNG_FAIL;+ goto cleanup;+ }+ MLD_CT_TESTING_SECRET(rnd, MLDSA_RNDBYTES);++ ret = mld_sign_signature_internal(sig, mu, MLDSA_CRHBYTES, NULL, 0, rnd, sk,+ 1, context);++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(rnd, uint8_t, MLDSA_RNDBYTES, context);++ return ret;+}+#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_internal(const uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t *m, size_t mlen, const uint8_t *pre,+ size_t prelen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ int externalmu,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret, cmp;+ unsigned int i;++ MLD_ALLOC(buf, uint8_t, (MLDSA_K * MLDSA_POLYW1_PACKEDBYTES), context);+ MLD_ALLOC(mu, uint8_t, MLDSA_CRHBYTES, context);+ MLD_ALLOC(c, uint8_t, MLDSA_CTILDEBYTES, context);+ MLD_ALLOC(c2, uint8_t, MLDSA_CTILDEBYTES, context);+ MLD_ALLOC(z, mld_polyvecl, 1, context);+ MLD_ALLOC(cp, mld_poly, 1, context);+ MLD_ALLOC(mat, mld_polymat, 1, context);+ MLD_ALLOC(w1, mld_poly, 1, context);+ MLD_ALLOC(tmp, mld_poly, 1, context);++ if (buf == NULL || mu == NULL || c == NULL || c2 == NULL || z == NULL ||+ cp == NULL || mat == NULL || w1 == NULL || tmp == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ mld_memcpy(c, sig, MLDSA_CTILDEBYTES);+ mld_polyvecl_unpack_z(z, sig + MLDSA_SIG_Z_OFFSET);++ /* mld_polyvecl_chknorm signals failure through a single non-zero error code+ * that's not yet aligned with MLD_ERR_XXX. A norm-check failure here means+ * the signature is invalid, so map it to MLD_ERR_INVALID_SIGNATURE. */+ if (mld_polyvecl_chknorm(z, MLDSA_GAMMA1 - MLDSA_BETA))+ {+ ret = MLD_ERR_INVALID_SIGNATURE;+ goto cleanup;+ }++ if (!externalmu)+ {+ /* Compute CRH(H(rho, t1), pre, msg) */+ MLD_ALIGN uint8_t hpk[MLDSA_CRHBYTES];+ mld_H(hpk, MLDSA_TRBYTES, pk, MLDSA_CRYPTO_PUBLICKEYBYTES, NULL, 0, NULL,+ 0);+ mld_H(mu, MLDSA_CRHBYTES, hpk, MLDSA_TRBYTES, pre, prelen, m, mlen);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(hpk, sizeof(hpk));+ }+ else+ {+ /* mu has been provided directly */+ mld_memcpy(mu, m, MLDSA_CRHBYTES);+ }++ /* Matrix-vector multiplication and per-row reconstruction of w1. */+ mld_polyvecl_ntt(z);+ mld_polyvec_matrix_expand(mat, pk);+ mld_poly_challenge(cp, c);+ mld_poly_ntt(cp);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(MLD_IF_REDUCE_RAM(memory_slice(mat, sizeof(mld_polymat)),)+ i, ret,+ memory_slice(w1, sizeof(mld_poly)),+ memory_slice(tmp, sizeof(mld_poly)),+ memory_slice(buf, MLDSA_K * MLDSA_POLYW1_PACKEDBYTES)+ )+ invariant(i <= MLDSA_K)+ decreases(MLDSA_K - i)+ )+ {+ /* w1 = (A * z)_i in NTT domain */+ mld_polyvec_matrix_pointwise_montgomery_row(w1, mat, z, i);++ /* tmp = c * t1_i * 2^d in NTT domain */+ mld_unpack_pk_t1(tmp, pk, i);+ mld_poly_shiftl(tmp);+ mld_poly_ntt(tmp);+ mld_poly_pointwise_montgomery(tmp, cp);++ /* w1 = invNTT(w1 - c * t1_i * 2^d) */+ mld_poly_sub(w1, tmp);+ mld_poly_reduce(w1);+ mld_poly_invntt_tomont(w1);+ mld_poly_caddq(w1);++ /* tmp = h_i (decoded and validated from signature). A non-zero return+ * means the hint encoding is malformed, i.e. the signature is invalid. */+ if (mld_sig_unpack_hints(tmp, sig, i) != 0)+ {+ ret = MLD_ERR_INVALID_SIGNATURE;+ goto cleanup;+ }++ /* w1 = use_hint(w1, tmp), then pack into buf[i] */+ mld_poly_use_hint(w1, tmp);+ mld_polyw1_pack(buf + i * MLDSA_POLYW1_PACKEDBYTES, w1);+ }++ /* Call random oracle and verify challenge */+ mld_H(c2, MLDSA_CTILDEBYTES, mu, MLDSA_CRHBYTES, buf,+ MLDSA_K * MLDSA_POLYW1_PACKEDBYTES, NULL, 0);++ cmp = mld_ct_memcmp(c, c2, MLDSA_CTILDEBYTES);++ /* Declassify the result of the verification. */+ MLD_CT_TESTING_DECLASSIFY(&cmp, sizeof(cmp));++ ret = cmp == 0 ? 0 : MLD_ERR_INVALID_SIGNATURE;++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(tmp, mld_poly, 1, context);+ MLD_FREE(w1, mld_poly, 1, context);+ MLD_FREE(mat, mld_polymat, 1, context);+ MLD_FREE(cp, mld_poly, 1, context);+ MLD_FREE(z, mld_polyvecl, 1, context);+ MLD_FREE(c2, uint8_t, MLDSA_CTILDEBYTES, context);+ MLD_FREE(c, uint8_t, MLDSA_CTILDEBYTES, context);+ MLD_FREE(mu, uint8_t, MLDSA_CRHBYTES, context);+ MLD_FREE(buf, uint8_t, (MLDSA_K * MLDSA_POLYW1_PACKEDBYTES), context);+ return ret;+}++#if !defined(MLD_CONFIG_CORE_API_ONLY)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify(const uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ MLD_ALIGN uint8_t pre[MLD_DOMAIN_SEPARATION_MAX_BYTES];+ size_t pre_len;+ int ret;++ pre_len = mld_prepare_domain_separation_prefix(pre, NULL, 0, ctx, ctxlen,+ MLD_PREHASH_NONE);+ if (pre_len == 0)+ {+ ret = MLD_ERR_INVALID_ARG;+ goto cleanup;+ }++ ret = mld_sign_verify_internal(sig, m, mlen, pre, pre_len, pk, 0, context);++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(pre, sizeof(pre));++ return ret;+}++MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_extmu(const uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t mu[MLDSA_CRHBYTES],+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ return mld_sign_verify_internal(sig, mu, MLDSA_CRHBYTES, NULL, 0, pk, 1,+ context);+}+#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_pre_hash_internal(+ uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *ph, size_t phlen,+ const uint8_t *ctx, size_t ctxlen, const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES], int hashalg,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ MLD_ALIGN uint8_t pre[MLD_DOMAIN_SEPARATION_MAX_BYTES];+ size_t pre_len;+ int ret;++ if (hashalg == MLD_PREHASH_NONE)+ {+ ret = MLD_ERR_INVALID_ARG;+ goto cleanup;+ }++ pre_len = mld_prepare_domain_separation_prefix(pre, ph, phlen, ctx, ctxlen,+ hashalg);+ if (pre_len == 0)+ {+ ret = MLD_ERR_INVALID_ARG;+ goto cleanup;+ }++ ret = mld_sign_signature_internal(sig, pre, pre_len, NULL, 0, rnd, sk, 0,+ context);+cleanup:+ if (ret != 0)+ {+ /* To be on the safe-side, make sure sig has a well-defined value, even in+ * the case of error.+ *+ * If we come from mld_sign_signature_internal, this is redundant, but the+ * error case should not be the norm, and the added cost of the zeroization+ * insignificant. */+ mld_zeroize(sig, MLDSA_CRYPTO_BYTES);+ }++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(pre, sizeof(pre));+ return ret;+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_pre_hash_internal(+ const uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *ph, size_t phlen,+ const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES], int hashalg,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ MLD_ALIGN uint8_t pre[MLD_DOMAIN_SEPARATION_MAX_BYTES];+ size_t pre_len;+ int ret;++ if (hashalg == MLD_PREHASH_NONE)+ {+ ret = MLD_ERR_INVALID_ARG;+ goto cleanup;+ }++ pre_len = mld_prepare_domain_separation_prefix(pre, ph, phlen, ctx, ctxlen,+ hashalg);+ if (pre_len == 0)+ {+ ret = MLD_ERR_INVALID_ARG;+ goto cleanup;+ }++ ret = mld_sign_verify_internal(sig, pre, pre_len, NULL, 0, pk, 0, context);++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(pre, sizeof(pre));+ return ret;+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_pre_hash_shake256(+ uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m, size_t mlen,+ const uint8_t *ctx, size_t ctxlen, const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ MLD_ALIGN uint8_t ph[64];+ int ret;+ mld_shake256(ph, sizeof(ph), m, mlen);+ ret = mld_sign_signature_pre_hash_internal(sig, ph, sizeof(ph), ctx, ctxlen,+ rnd, sk, MLD_PREHASH_SHAKE_256,+ context);+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(ph, sizeof(ph));+ return ret;+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_pre_hash_shake256(+ const uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m, size_t mlen,+ const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ MLD_ALIGN uint8_t ph[64];+ int ret;+ mld_shake256(ph, sizeof(ph), m, mlen);+ ret = mld_sign_verify_pre_hash_internal(sig, ph, sizeof(ph), ctx, ctxlen, pk,+ MLD_PREHASH_SHAKE_256, context);+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(ph, sizeof(ph));+ return ret;+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#define MLD_PRE_HASH_OID_LEN 11++/**+ * Return the OID of a given SHA-2/SHA-3 hash function.+ *+ * @param[out] oid Pointer to output OID.+ * @param hashalg Hash algorithm constant (MLD_PREHASH_*).+ */+static void mld_get_hash_oid(uint8_t oid[MLD_PRE_HASH_OID_LEN], int hashalg)+{+ unsigned int i;+ static const struct+ {+ int alg;+ uint8_t oid[MLD_PRE_HASH_OID_LEN];+ } oid_map[] = {+ {MLD_PREHASH_SHA2_224,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x04}},+ {MLD_PREHASH_SHA2_256,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x01}},+ {MLD_PREHASH_SHA2_384,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x02}},+ {MLD_PREHASH_SHA2_512,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x03}},+ {MLD_PREHASH_SHA2_512_224,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x05}},+ {MLD_PREHASH_SHA2_512_256,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x06}},+ {MLD_PREHASH_SHA3_224,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x07}},+ {MLD_PREHASH_SHA3_256,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x08}},+ {MLD_PREHASH_SHA3_384,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x09}},+ {MLD_PREHASH_SHA3_512,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x0A}},+ {MLD_PREHASH_SHAKE_128,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x0B}},+ {MLD_PREHASH_SHAKE_256,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x0C}}};++ for (i = 0; i < sizeof(oid_map) / sizeof(oid_map[0]); i++)+ __loop__(+ invariant(i <= sizeof(oid_map) / sizeof(oid_map[0]))+ decreases(sizeof(oid_map) / sizeof(oid_map[0]) - i)+ )+ {+ if (oid_map[i].alg == hashalg)+ {+ mld_memcpy(oid, oid_map[i].oid, MLD_PRE_HASH_OID_LEN);+ return;+ }+ }+}++static int mld_validate_hash_length(int hashalg, size_t len)+{+ switch (hashalg)+ {+ case MLD_PREHASH_SHA2_224:+ return (len == 224 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA2_256:+ return (len == 256 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA2_384:+ return (len == 384 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA2_512:+ return (len == 512 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA2_512_224:+ return (len == 224 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA2_512_256:+ return (len == 256 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA3_224:+ return (len == 224 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA3_256:+ return (len == 256 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA3_384:+ return (len == 384 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA3_512:+ return (len == 512 / 8) ? 0 : -1;+ case MLD_PREHASH_SHAKE_128:+ return (len == 256 / 8) ? 0 : -1;+ case MLD_PREHASH_SHAKE_256:+ return (len == 512 / 8) ? 0 : -1;+ default:+ return -1;+ }+}++MLD_EXTERNAL_API+size_t mld_prepare_domain_separation_prefix(+ uint8_t prefix[MLD_DOMAIN_SEPARATION_MAX_BYTES], const uint8_t *ph,+ size_t phlen, const uint8_t *ctx, size_t ctxlen, int hashalg)+{+ if (ctxlen > 255)+ {+ return 0;+ }++ if (hashalg != MLD_PREHASH_NONE)+ {+ if (ph == NULL || mld_validate_hash_length(hashalg, phlen) != 0)+ {+ return 0;+ }+ }++ /* Common prefix: 0x00/0x01 || ctxlen || ctx */+ prefix[0] = (hashalg == MLD_PREHASH_NONE) ? 0 : 1;+ prefix[1] = (uint8_t)ctxlen;+ if (ctxlen > 0)+ {+ mld_memcpy(prefix + 2, ctx, ctxlen);+ }++ if (hashalg == MLD_PREHASH_NONE)+ {+ return 2 + ctxlen;+ }++ /* HashML-DSA: append oid || ph */+ mld_get_hash_oid(prefix + 2 + ctxlen, hashalg);+ mld_memcpy(prefix + 2 + ctxlen + MLD_PRE_HASH_OID_LEN, ph, phlen);+ return 2 + ctxlen + MLD_PRE_HASH_OID_LEN + phlen;+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_EXTERNAL_API+int mld_sign_pk_from_sk(uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ uint8_t check, cmp0, cmp1, chk1, chk2;+ int ret;+ MLD_ALLOC(rho, uint8_t, MLDSA_SEEDBYTES, context);+ MLD_ALLOC(tr, uint8_t, MLDSA_TRBYTES, context);+ MLD_ALLOC(tr_computed, uint8_t, MLDSA_TRBYTES, context);+ MLD_ALLOC(key, uint8_t, MLDSA_SEEDBYTES, context);+ MLD_ALLOC(s1, mld_polyvecl, 1, context);+ MLD_ALLOC(s2, mld_polyveck, 1, context);+ MLD_ALLOC(t0_packed, uint8_t, MLDSA_K *MLDSA_POLYT0_PACKEDBYTES, context);++ if (rho == NULL || tr == NULL || tr_computed == NULL || key == NULL ||+ s1 == NULL || s2 == NULL || t0_packed == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Inline unpack_sk: mld_unpack_sk uses lazy types for s1/s2/t0 which+ * we cannot use here. t0 stays in packed form -- we compare it against+ * the recomputed value below. */+ mld_memcpy(rho, sk + MLDSA_SK_RHO_OFFSET, MLDSA_SEEDBYTES);+ mld_memcpy(key, sk + MLDSA_SK_KEY_OFFSET, MLDSA_SEEDBYTES);+ mld_memcpy(tr, sk + MLDSA_SK_TR_OFFSET, MLDSA_TRBYTES);+ mld_polyvecl_unpack_eta(s1, sk + MLDSA_SK_S1_OFFSET);+ mld_polyveck_unpack_eta(s2, sk + MLDSA_SK_S2_OFFSET);++ /* Validate s1 and s2 coefficients are within [-MLDSA_ETA, MLDSA_ETA] */+ chk1 = mld_polyvecl_chknorm(s1, MLDSA_ETA + 1) & 0xFF;+ chk2 = mld_polyveck_chknorm(s2, MLDSA_ETA + 1) & 0xFF;++ /* NTT s1 in place to use as s1hat */+ mld_polyvecl_ntt(s1);++ /* Pack rho into pk */+ mld_memcpy(pk + MLDSA_PK_RHO_OFFSET, rho, MLDSA_SEEDBYTES);++ /* Recompute t row by row, decompose, and pack t1 into pk and t0 into+ * t0_packed. */+ ret = mld_compute_pack_t0_t1(pk + MLDSA_PK_T1_OFFSET, t0_packed, s1, s2, rho,+ context);+ if (ret != 0)+ {+ goto cleanup;+ }++ /* Compare recomputed packed t0 against the t0 region of sk. */+ cmp0 = mld_ct_memcmp(t0_packed, sk + MLDSA_SK_T0_OFFSET,+ MLDSA_K * MLDSA_POLYT0_PACKEDBYTES);++ /* Compute tr_computed = H(pk) and compare to the stored tr */+ mld_shake256(tr_computed, MLDSA_TRBYTES, pk, MLDSA_CRYPTO_PUBLICKEYBYTES);+ cmp1 = mld_ct_memcmp((const uint8_t *)tr, (const uint8_t *)tr_computed,+ MLDSA_TRBYTES);+ check = mld_value_barrier_u8(cmp0 | cmp1 | chk1 | chk2);++ /* Declassify the final result of the validity check. */+ MLD_CT_TESTING_DECLASSIFY(&check, sizeof(check));+ ret = (check != 0) ? MLD_ERR_INVALID_KEY : 0;++cleanup:++ if (ret != 0)+ {+ mld_zeroize(pk, MLDSA_CRYPTO_PUBLICKEYBYTES);+ }++ /* Constant time: pk is either the valid public key or zeroed on error */+ MLD_CT_TESTING_DECLASSIFY(pk, MLDSA_CRYPTO_PUBLICKEYBYTES);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(t0_packed, uint8_t, MLDSA_K *MLDSA_POLYT0_PACKEDBYTES, context);+ MLD_FREE(s2, mld_polyveck, 1, context);+ MLD_FREE(s1, mld_polyvecl, 1, context);+ MLD_FREE(key, uint8_t, MLDSA_SEEDBYTES, context);+ MLD_FREE(tr_computed, uint8_t, MLDSA_TRBYTES, context);+ MLD_FREE(tr, uint8_t, MLDSA_TRBYTES, context);+ MLD_FREE(rho, uint8_t, MLDSA_SEEDBYTES, context);++ return ret;+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef mld_check_pct+#undef mld_sample_s1_s2+#undef mld_validate_hash_length+#undef mld_get_hash_oid+#undef mld_H+#undef mld_compute_pack_z+#undef mld_attempt_signature_generation+#undef mld_compute_pack_t0_t1+#undef mld_get_max_signing_attempts+#undef MLD_MAX_SIGNING_ATTEMPTS+#undef MLD_PRE_HASH_OID_LEN
+ cbits/mldsa/src/sign.h view
@@ -0,0 +1,850 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_SIGN_H+#define MLD_SIGN_H++#include <stddef.h>+#include "cbmc.h"+#include "common.h"+#include "poly.h"+#include "polyvec.h"+#include "sys.h"++#if defined(MLD_CHECK_APIS)+/* Include to ensure consistency between internal sign.h+ * and external mldsa_native.h. */+#include "mldsa_native.h"++#if MLDSA_CRYPTO_SECRETKEYBYTES != \+ MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)+#error Mismatch for SECRETKEYBYTES between sign.h and mldsa_native.h+#endif++#if MLDSA_CRYPTO_PUBLICKEYBYTES != \+ MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)+#error Mismatch for PUBLICKEYBYTES between sign.h and mldsa_native.h+#endif++#if MLDSA_CRYPTO_BYTES != MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)+#error Mismatch for BYTES between sign.h and mldsa_native.h+#endif++#endif /* MLD_CHECK_APIS */++#define mld_sign_keypair_internal \+ MLD_NAMESPACE_KL(keypair_internal) MLD_CONTEXT_PARAMETERS_3+#define mld_sign_keypair MLD_NAMESPACE_KL(keypair) MLD_CONTEXT_PARAMETERS_2+#define mld_sign_signature_internal \+ MLD_NAMESPACE_KL(signature_internal) MLD_CONTEXT_PARAMETERS_8+#define mld_sign_signature MLD_NAMESPACE_KL(signature) MLD_CONTEXT_PARAMETERS_6+#define mld_sign_signature_extmu \+ MLD_NAMESPACE_KL(signature_extmu) MLD_CONTEXT_PARAMETERS_3+#define mld_sign_verify_internal \+ MLD_NAMESPACE_KL(verify_internal) MLD_CONTEXT_PARAMETERS_7+#define mld_sign_verify MLD_NAMESPACE_KL(verify) MLD_CONTEXT_PARAMETERS_6+#define mld_sign_verify_extmu \+ MLD_NAMESPACE_KL(verify_extmu) MLD_CONTEXT_PARAMETERS_3+#define mld_sign_signature_pre_hash_internal \+ MLD_NAMESPACE_KL(signature_pre_hash_internal) MLD_CONTEXT_PARAMETERS_8+#define mld_sign_verify_pre_hash_internal \+ MLD_NAMESPACE_KL(verify_pre_hash_internal) MLD_CONTEXT_PARAMETERS_7+#define mld_sign_signature_pre_hash_shake256 \+ MLD_NAMESPACE_KL(signature_pre_hash_shake256) MLD_CONTEXT_PARAMETERS_7+#define mld_sign_verify_pre_hash_shake256 \+ MLD_NAMESPACE_KL(verify_pre_hash_shake256) MLD_CONTEXT_PARAMETERS_6+#define mld_prepare_domain_separation_prefix \+ MLD_NAMESPACE_KL(prepare_domain_separation_prefix)+#define mld_sign_pk_from_sk \+ MLD_NAMESPACE_KL(pk_from_sk) MLD_CONTEXT_PARAMETERS_2++/* Hash algorithm constants for domain separation */+#define MLD_PREHASH_NONE 0+#define MLD_PREHASH_SHA2_224 1+#define MLD_PREHASH_SHA2_256 2+#define MLD_PREHASH_SHA2_384 3+#define MLD_PREHASH_SHA2_512 4+#define MLD_PREHASH_SHA2_512_224 5+#define MLD_PREHASH_SHA2_512_256 6+#define MLD_PREHASH_SHA3_224 7+#define MLD_PREHASH_SHA3_256 8+#define MLD_PREHASH_SHA3_384 9+#define MLD_PREHASH_SHA3_512 10+#define MLD_PREHASH_SHAKE_128 11+#define MLD_PREHASH_SHAKE_256 12++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+/**+ * Generate a public-private key pair from a seed.+ *+ * When MLD_CONFIG_KEYGEN_PCT is set, performs a Pairwise Consistency Test+ * (PCT) as required by FIPS 140-3 IG.+ *+ * @spec{Implements @[FIPS204, Algorithm 6, ML-DSA.KeyGen_internal].}+ *+ * @param[out] pk Output public key.+ * @param[out] sk Output private key.+ * @param[in] seed Input random seed.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed+ * during the PCT. Only possible when+ * MLD_CONFIG_KEYGEN_PCT is enabled.+ * @retval MLD_ERR_SIGNING_PAUSED The PCT's signing step was paused by+ * a MLD_CONFIG_SIGN_HOOK_ATTEMPT hook.+ * This should currently never happen:+ * signing hooks require+ * MLD_CONFIG_NO_RANDOMIZED_API, which+ * is incompatible with+ * MLD_CONFIG_KEYGEN_PCT, so the two+ * cannot be enabled simultaneously.+ * @retval MLD_ERR_PCT_FAIL MLD_CONFIG_KEYGEN_PCT is enabled and+ * the PCT check failed.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_keypair_internal(uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ const uint8_t seed[MLDSA_SEEDBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ requires(memory_no_alias(seed, MLDSA_SEEDBYTES))+ assigns(object_whole(pk))+ assigns(object_whole(sk))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY ||+ return_value == MLD_ERR_RNG_FAIL ||+ return_value == MLD_ERR_SIGNING_PAUSED ||+ return_value == MLD_ERR_PCT_FAIL)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+);++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+/**+ * Generate a public-private key pair.+ *+ * When MLD_CONFIG_KEYGEN_PCT is set, performs a Pairwise Consistency Test+ * (PCT) as required by FIPS 140-3 IG.+ *+ * @spec{Implements @[FIPS204, Algorithm 1, ML-DSA.KeyGen].}+ *+ * @param[out] pk Output public key.+ * @param[out] sk Output private key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed.+ * @retval MLD_ERR_SIGNING_PAUSED The PCT's signing step was paused by+ * a MLD_CONFIG_SIGN_HOOK_ATTEMPT hook.+ * This should currently never happen:+ * signing hooks require+ * MLD_CONFIG_NO_RANDOMIZED_API, which+ * is incompatible with+ * MLD_CONFIG_KEYGEN_PCT, so the two+ * cannot be enabled simultaneously.+ * @retval MLD_ERR_PCT_FAIL MLD_CONFIG_KEYGEN_PCT is enabled and+ * the PCT check failed.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_keypair(uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ assigns(object_whole(pk))+ assigns(object_whole(sk))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY ||+ return_value == MLD_ERR_RNG_FAIL ||+ return_value == MLD_ERR_SIGNING_PAUSED ||+ return_value == MLD_ERR_PCT_FAIL)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+);+#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute signature using a caller-supplied random seed and prefix.+ *+ * On error (non-zero return value), the signature buffer sig is zeroized.+ *+ * @spec{Implements @[FIPS204, Algorithm 7, ML-DSA.Sign_internal].}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use mld_sign_pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Pointer to buffer to hold the generated signature of+ * MLDSA_CRYPTO_BYTES bytes.+ * @param[in] m Pointer to message to be signed (when+ * externalmu == 0), or to a precomputed+ * message representative mu (when externalmu != 0).+ * @param mlen Length of m. Must equal MLDSA_CRHBYTES when+ * externalmu != 0.+ * @param[in] pre Pointer to prefix string. Ignored when+ * externalmu != 0.+ * @param prelen Length of prefix string. Ignored when+ * externalmu != 0.+ * @param[in] rnd Random seed.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param externalmu 0: m/mlen is the raw message; mu = H(tr, pre, m) is+ * computed internally.+ * non-zero: m points to a precomputed mu of+ * MLDSA_CRHBYTES bytes; pre/prelen unused.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_internal(uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t *m, size_t mlen,+ const uint8_t *pre, size_t prelen,+ const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ int externalmu,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(mlen <= MLD_MAX_BUFFER_SIZE)+ requires(prelen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(m, mlen))+ requires(memory_no_alias(rnd, MLDSA_RNDBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ requires((externalmu == 0) ==> ((prelen == 0) || memory_no_alias(pre, prelen)))+ requires((externalmu != 0) ==> (mlen == MLDSA_CRHBYTES))+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+ ensures(return_value == 0 ||+ return_value == MLD_ERR_OUT_OF_MEMORY ||+ return_value == MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED ||+ return_value == MLD_ERR_SIGNING_PAUSED)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sig, MLDSA_CRYPTO_BYTES)));++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+/**+ * Compute signature. This function implements the randomized variant of+ * ML-DSA. If you require the deterministic variant, use+ * mld_sign_signature_internal directly.+ *+ * @spec{Implements @[FIPS204, Algorithm 2, ML-DSA.Sign].}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use mld_sign_pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Output signature.+ * @param[in] m Pointer to message to be signed.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string. Should be <= 255.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255+ * bytes.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature(uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(mlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(m, mlen))+ requires(ctxlen <= MLD_MAX_BUFFER_SIZE)+ requires(ctxlen == 0 || memory_no_alias(ctx, ctxlen))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY ||+ return_value == MLD_ERR_RNG_FAIL ||+ return_value == MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED ||+ return_value == MLD_ERR_SIGNING_PAUSED ||+ return_value == MLD_ERR_INVALID_ARG)+ ensures((return_value == MLD_ERR_INVALID_ARG) ==> (ctxlen > 255))+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sig, MLDSA_CRYPTO_BYTES))+);++/**+ * Compute signature in "external mu" mode: the caller has already computed+ * the message representative mu = SHAKE256(tr || M', 64), where+ * tr = SHAKE256(pk, 64) and M' is the FIPS 204 formatted message (e.g.+ * 0x00 || ctxlen || ctx || msg for pure ML-DSA). This is the randomized+ * variant; for the deterministic variant, use mld_sign_signature_internal+ * directly with externalmu set to non-zero and an all-zero rnd.+ *+ * @spec{Implements @[FIPS204, Algorithm 2, ML-DSA.Sign external mu variant].}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use mld_sign_pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Output signature.+ * @param[in] mu Precomputed message representative.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_extmu(uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t mu[MLDSA_CRHBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(mu, MLDSA_CRHBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY ||+ return_value == MLD_ERR_RNG_FAIL ||+ return_value == MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED ||+ return_value == MLD_ERR_SIGNING_PAUSED)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sig, MLDSA_CRYPTO_BYTES))+);++#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Verify signature.+ *+ * @spec{Implements @[FIPS204, Algorithm 8, ML-DSA.Verify_internal].}+ *+ * @param[in] sig Pointer to input signature of+ * MLDSA_CRYPTO_BYTES bytes.+ * @param[in] m Pointer to message (when externalmu == 0), or to a+ * precomputed message representative mu (when+ * externalmu != 0).+ * @param mlen Length of m. Must equal MLDSA_CRHBYTES when+ * externalmu != 0.+ * @param[in] pre Pointer to prefix string. Ignored when externalmu != 0.+ * @param prelen Length of prefix string. Ignored when externalmu != 0.+ * @param[in] pk Bit-packed public key.+ * @param externalmu 0: m/mlen is the raw message; mu = H(H(pk), pre, m) is+ * computed internally.+ * non-zero: m points to a precomputed mu of+ * MLDSA_CRHBYTES bytes; pre/prelen unused.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_internal(const uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t *m, size_t mlen, const uint8_t *pre,+ size_t prelen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ int externalmu,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(prelen <= MLD_MAX_BUFFER_SIZE)+ requires(mlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(m, mlen))+ requires((externalmu == 0) ==> ((prelen == 0) || memory_no_alias(pre, prelen)))+ requires((externalmu != 0) ==> (mlen == MLDSA_CRHBYTES))+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_INVALID_SIGNATURE || return_value == MLD_ERR_OUT_OF_MEMORY)+);++#if !defined(MLD_CONFIG_CORE_API_ONLY)+/**+ * Verify signature.+ *+ * @spec{Implements @[FIPS204, Algorithm 3, ML-DSA.Verify].}+ *+ * @param[in] sig Pointer to input signature.+ * @param[in] m Pointer to message.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] pk Bit-packed public key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255 bytes.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify(const uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(mlen <= MLD_MAX_BUFFER_SIZE)+ requires(ctxlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(m, mlen))+ requires(ctxlen == 0 || memory_no_alias(ctx, ctxlen))+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_INVALID_SIGNATURE || return_value == MLD_ERR_INVALID_ARG || return_value == MLD_ERR_OUT_OF_MEMORY)+ ensures((return_value == MLD_ERR_INVALID_ARG) ==> (ctxlen > 255))+);++/**+ * Verify signature in "external mu" mode: the caller has already computed+ * the message representative mu = SHAKE256(tr || M', 64), where+ * tr = SHAKE256(pk, 64) and M' is the FIPS 204 formatted message (e.g.+ * 0x00 || ctxlen || ctx || msg for pure ML-DSA). The same mu must have been+ * used at signing time.+ *+ * @spec{Implements @[FIPS204, Algorithm 3, ML-DSA.Verify external mu variant].}+ *+ * @param[in] sig Pointer to input signature.+ * @param[in] mu Precomputed message representative.+ * @param[in] pk Bit-packed public key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_extmu(const uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t mu[MLDSA_CRHBYTES],+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__( requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(mu, MLDSA_CRHBYTES))+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_INVALID_SIGNATURE || return_value == MLD_ERR_OUT_OF_MEMORY)+);++#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute signature with pre-hashed message.+ *+ * @spec{Implements @[FIPS204, Algorithm 4, HashML-DSA.Sign].}+ *+ * Supported hash algorithm constants:+ * MLD_PREHASH_SHA2_224, MLD_PREHASH_SHA2_256, MLD_PREHASH_SHA2_384,+ * MLD_PREHASH_SHA2_512, MLD_PREHASH_SHA2_512_224, MLD_PREHASH_SHA2_512_256,+ * MLD_PREHASH_SHA3_224, MLD_PREHASH_SHA3_256, MLD_PREHASH_SHA3_384,+ * MLD_PREHASH_SHA3_512, MLD_PREHASH_SHAKE_128, MLD_PREHASH_SHAKE_256.+ *+ * MLD_PREHASH_NONE is rejected by this API.+ *+ * @warning This is an unstable API that may change in the future. If you need+ * a stable API use mld_sign_signature_pre_hash_shake256.+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use mld_sign_pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Output signature.+ * @param[in] ph Pointer to pre-hashed message.+ * @param phlen Length of pre-hashed message. Must match the output+ * length of hashalg (the digest size for SHA-2/SHA-3,+ * 32 bytes for MLD_PREHASH_SHAKE_128, 64 bytes for+ * MLD_PREHASH_SHAKE_256).+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] rnd Random seed.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param hashalg Hash algorithm constant (one of MLD_PREHASH_*).+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ * @retval MLD_ERR_INVALID_ARG The pre-hash algorithm was+ * MLD_PREHASH_NONE or unsupported,+ * phlen did not match the output+ * length of hashalg, or the context+ * string exceeded 255 bytes.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_pre_hash_internal(+ uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *ph, size_t phlen,+ const uint8_t *ctx, size_t ctxlen, const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES], int hashalg,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(ctxlen <= MLD_MAX_BUFFER_SIZE)+ requires(phlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(ph, phlen))+ requires(ctxlen == 0 || memory_no_alias(ctx, ctxlen))+ requires(memory_no_alias(rnd, MLDSA_RNDBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY || return_value == MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED || return_value == MLD_ERR_SIGNING_PAUSED || return_value == MLD_ERR_INVALID_ARG)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sig, MLDSA_CRYPTO_BYTES))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Verify signature with pre-hashed message.+ *+ * @spec{Implements @[FIPS204, Algorithm 5, HashML-DSA.Verify].}+ *+ * Supported hash algorithm constants:+ * MLD_PREHASH_SHA2_224, MLD_PREHASH_SHA2_256, MLD_PREHASH_SHA2_384,+ * MLD_PREHASH_SHA2_512, MLD_PREHASH_SHA2_512_224, MLD_PREHASH_SHA2_512_256,+ * MLD_PREHASH_SHA3_224, MLD_PREHASH_SHA3_256, MLD_PREHASH_SHA3_384,+ * MLD_PREHASH_SHA3_512, MLD_PREHASH_SHAKE_128, MLD_PREHASH_SHAKE_256.+ *+ * MLD_PREHASH_NONE is rejected by this API.+ *+ * @warning This is an unstable API that may change in the future. If you need+ * a stable API use mld_sign_verify_pre_hash_shake256.+ *+ * @param[in] sig Pointer to input signature.+ * @param[in] ph Pointer to pre-hashed message.+ * @param phlen Length of pre-hashed message. Must match the output+ * length of hashalg (the digest size for SHA-2/SHA-3,+ * 32 bytes for MLD_PREHASH_SHAKE_128, 64 bytes for+ * MLD_PREHASH_SHAKE_256).+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] pk Bit-packed public key.+ * @param hashalg Hash algorithm constant (one of MLD_PREHASH_*).+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ * @retval MLD_ERR_INVALID_ARG The pre-hash algorithm was+ * MLD_PREHASH_NONE or unsupported, phlen+ * did not match the output length of+ * hashalg, or the context string exceeded+ * 255 bytes.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_pre_hash_internal(+ const uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *ph, size_t phlen,+ const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES], int hashalg,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(phlen <= MLD_MAX_BUFFER_SIZE)+ requires(ctxlen <= MLD_MAX_BUFFER_SIZE - 77)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(ph, phlen))+ requires(ctxlen == 0 || memory_no_alias(ctx, ctxlen))+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_INVALID_SIGNATURE || return_value == MLD_ERR_INVALID_ARG || return_value == MLD_ERR_OUT_OF_MEMORY)+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute signature with pre-hashed message using SHAKE256. This function+ * computes the SHAKE256 hash of the message internally.+ *+ * @spec{Implements @[FIPS204, Algorithm 4, HashML-DSA.Sign] with SHAKE256 as+ * the pre-hash.}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use mld_sign_pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Output signature.+ * @param[in] m Pointer to message to be hashed and signed.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] rnd Random seed.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255+ * bytes.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_pre_hash_shake256(+ uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m, size_t mlen,+ const uint8_t *ctx, size_t ctxlen, const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(mlen <= MLD_MAX_BUFFER_SIZE)+ requires(ctxlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(m, mlen))+ requires(ctxlen == 0 || memory_no_alias(ctx, ctxlen))+ requires(memory_no_alias(rnd, MLDSA_RNDBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY || return_value == MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED || return_value == MLD_ERR_SIGNING_PAUSED || return_value == MLD_ERR_INVALID_ARG)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sig, MLDSA_CRYPTO_BYTES))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Verify signature with pre-hashed message using SHAKE256. This function+ * computes the SHAKE256 hash of the message internally.+ *+ * @spec{Implements @[FIPS204, Algorithm 5, HashML-DSA.Verify] with SHAKE256 as+ * the pre-hash.}+ *+ * @param[in] sig Pointer to input signature.+ * @param[in] m Pointer to message to be hashed and verified.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] pk Bit-packed public key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255 bytes.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_pre_hash_shake256(+ const uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m, size_t mlen,+ const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(mlen <= MLD_MAX_BUFFER_SIZE)+ requires(ctxlen <= MLD_MAX_BUFFER_SIZE - 77)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(m, mlen))+ requires(ctxlen == 0 || memory_no_alias(ctx, ctxlen))+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_INVALID_SIGNATURE || return_value == MLD_ERR_INVALID_ARG || return_value == MLD_ERR_OUT_OF_MEMORY)+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+/* Maximum formatted domain separation message length:+ * - Pure ML-DSA: 0x00 || ctxlen || ctx (max 255)+ * - HashML-DSA: 0x01 || ctxlen || ctx (max 255) || oid (11) || ph (max 64) */+#define MLD_DOMAIN_SEPARATION_MAX_BYTES (2 + 255 + 11 + 64)++/**+ * Prepare domain separation prefix for ML-DSA signing.+ *+ * For pure ML-DSA (hashalg == MLD_PREHASH_NONE):+ * Format: 0x00 || ctxlen (1 byte) || ctx.+ *+ * For HashML-DSA (hashalg != MLD_PREHASH_NONE):+ * Format: 0x01 || ctxlen (1 byte) || ctx || oid (11 bytes) || ph.+ *+ * This function is useful for building incremental signing APIs.+ *+ * @spec{For HashML-DSA (hashalg != MLD_PREHASH_NONE), implements+ * @[FIPS204, Algorithm 4, line 23]. For Pure ML-DSA+ * (hashalg == MLD_PREHASH_NONE), implements+ * ```+ * M' <- BytesToBits(IntegerToBytes(0, 1)+ * || IntegerToBytes(|ctx|, 1)+ * || ctx+ * ```+ * which is part of @[FIPS204, Algorithm 2, ML-DSA.Sign, line 10] and+ * @[FIPS204, Algorithm 3, ML-DSA.Verify, line 5].}+ *+ * @param[out] prefix Output domain separation prefix buffer.+ * @param[in] ph Pointer to pre-hashed message (ignored for pure+ * ML-DSA).+ * @param phlen Length of pre-hashed message; must match the output+ * length of hashalg (ignored for pure ML-DSA).+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param hashalg Hash algorithm constant (MLD_PREHASH_NONE for pure+ * ML-DSA, or MLD_PREHASH_* for HashML-DSA).+ *+ * @return The total length of the formatted prefix, or 0 on error.+ * Errors are:+ * - The context string exceeded 255 bytes.+ * - For HashML-DSA: hashalg was unsupported, ph was NULL, or phlen+ * did not match the output length of hashalg.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+size_t mld_prepare_domain_separation_prefix(+ uint8_t prefix[MLD_DOMAIN_SEPARATION_MAX_BYTES], const uint8_t *ph,+ size_t phlen, const uint8_t *ctx, size_t ctxlen, int hashalg)+__contract__(+ requires(ctxlen <= 255)+ requires(phlen <= MLD_MAX_BUFFER_SIZE)+ requires(ctxlen == 0 || memory_no_alias(ctx, ctxlen))+ requires(hashalg == MLD_PREHASH_NONE || memory_no_alias(ph, phlen))+ requires(memory_no_alias(prefix, MLD_DOMAIN_SEPARATION_MAX_BYTES))+ assigns(memory_slice(prefix, MLD_DOMAIN_SEPARATION_MAX_BYTES))+ ensures(return_value <= MLD_DOMAIN_SEPARATION_MAX_BYTES)+);+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+/**+ * Perform basic validity checks on secret key, and derive public key.+ *+ * Referring to the decoding of the secret key `sk=(rho, K, tr, s1, s2, t0)`+ * (cf. @[FIPS204, Algorithm 25, skDecode]), the following checks are+ * performed:+ * - Check that s1 and s2 have coefficients in [-MLDSA_ETA, MLDSA_ETA].+ * - Check that t0 and tr stored in sk match recomputed values.+ *+ * @note This function leaks whether the secret key is valid or invalid+ * through its return value and timing.+ *+ * @param[out] pk Output public key.+ * @param[in] sk Input secret key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was used and an+ * allocation via MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_KEY Secret key validation failed.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_pk_from_sk(uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ assigns(memory_slice(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_INVALID_KEY || return_value == MLD_ERR_OUT_OF_MEMORY)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */++#endif /* !MLD_SIGN_H */
+ cbits/mldsa/src/symmetric.h view
@@ -0,0 +1,68 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_SYMMETRIC_H+#define MLD_SYMMETRIC_H++#include "cbmc.h"+#include "common.h"++#include MLD_FIPS202_HEADER_FILE+#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+#include MLD_FIPS202X4_HEADER_FILE+#endif++#define MLD_STREAM128_BLOCKBYTES SHAKE128_RATE+#define MLD_STREAM256_BLOCKBYTES SHAKE256_RATE++#define mld_xof256_ctx mld_shake256ctx+#define mld_xof256_init(CTX) mld_shake256_init(CTX)++#define mld_xof256_absorb_once(CTX, IN, INBYTES) \+ do \+ { \+ mld_shake256_absorb(CTX, IN, INBYTES); \+ mld_shake256_finalize(CTX); \+ } while (0)+++#define mld_xof256_release(CTX) mld_shake256_release(CTX)+#define mld_xof256_squeezeblocks(OUT, OUTBLOCKS, STATE) \+ mld_shake256_squeeze(OUT, (OUTBLOCKS) * SHAKE256_RATE, STATE)++#define mld_xof128_ctx mld_shake128ctx+#define mld_xof128_init(CTX) mld_shake128_init(CTX)++#define mld_xof128_absorb_once(CTX, IN, INBYTES) \+ do \+ { \+ mld_shake128_absorb(CTX, IN, INBYTES); \+ mld_shake128_finalize(CTX); \+ } while (0)++#define mld_xof128_release(CTX) mld_shake128_release(CTX)+#define mld_xof128_squeezeblocks(OUT, OUTBLOCKS, STATE) \+ mld_shake128_squeeze(OUT, (OUTBLOCKS) * SHAKE128_RATE, STATE)++#define mld_xof256_x4_ctx mld_shake256x4ctx+#define mld_xof256_x4_init(CTX) mld_shake256x4_init((CTX))+#define mld_xof256_x4_absorb(CTX, IN, INBYTES) \+ mld_shake256x4_absorb_once((CTX), (IN)[0], (IN)[1], (IN)[2], (IN)[3], \+ (INBYTES))+#define mld_xof256_x4_squeezeblocks(BUF, NBLOCKS, CTX) \+ mld_shake256x4_squeezeblocks((BUF)[0], (BUF)[1], (BUF)[2], (BUF)[3], \+ (NBLOCKS), (CTX))+#define mld_xof256_x4_release(CTX) mld_shake256x4_release((CTX))++#define mld_xof128_x4_ctx mld_shake128x4ctx+#define mld_xof128_x4_init(CTX) mld_shake128x4_init((CTX))+#define mld_xof128_x4_absorb(CTX, IN, INBYTES) \+ mld_shake128x4_absorb_once((CTX), (IN)[0], (IN)[1], (IN)[2], (IN)[3], \+ (INBYTES))+#define mld_xof128_x4_squeezeblocks(BUF, NBLOCKS, CTX) \+ mld_shake128x4_squeezeblocks((BUF)[0], (BUF)[1], (BUF)[2], (BUF)[3], \+ (NBLOCKS), (CTX))+#define mld_xof128_x4_release(CTX) mld_shake128x4_release((CTX))++#endif /* !MLD_SYMMETRIC_H */
+ cbits/mldsa/src/sys.h view
@@ -0,0 +1,327 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_SYS_H+#define MLD_SYS_H++#if !defined(MLD_CONFIG_NO_ASM) && (defined(__GNUC__) || defined(__clang__))+#define MLD_HAVE_INLINE_ASM+#endif++/* Try to find endianness, if not forced through CFLAGS already */+#if !defined(MLD_SYS_LITTLE_ENDIAN) && !defined(MLD_SYS_BIG_ENDIAN)+#if defined(__BYTE_ORDER__)+#if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__+#define MLD_SYS_LITTLE_ENDIAN+#elif __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__+#define MLD_SYS_BIG_ENDIAN+#else+#error "__BYTE_ORDER__ defined, but don't recognize value."+#endif+#endif /* __BYTE_ORDER__ */++/* MSVC does not define __BYTE_ORDER__. However, MSVC only supports+ * little endian x86, x86_64, and AArch64. It is, hence, safe to assume+ * little endian. */+#if defined(_MSC_VER) && (defined(_M_X64) || defined(_M_AMD64) || \+ defined(_M_IX86) || defined(_M_ARM64))+#define MLD_SYS_LITTLE_ENDIAN+#endif++#endif /* !MLD_SYS_LITTLE_ENDIAN && !MLD_SYS_BIG_ENDIAN */++/* Check if we're running on an AArch64 little endian system. _M_ARM64 is set by+ * MSVC. */+#if defined(__AARCH64EL__) || defined(_M_ARM64)+#define MLD_SYS_AARCH64+#endif++/* Check if the AArch64 compilation target supports NEON (Advanced SIMD).+ *+ * Some compilers also define __ARM_NEON__, but __ARM_NEON is the most reliable+ * signal. Specifically, clang on Apple appears to keep __ARM_NEON__ set even if+ * -march=armv8-a+nosimd is set.+ *+ * gcc 4.8 -- the first gcc version introducing Neon support -- sets neither+ * __ARM_NEON nor __ARM_NEON__; in fact, there is no preprocessor signal that+ * Neon is enabled. If you use gcc 4.8 and need Neon, you should set+ * MLD_SYS_AARCH64_NEON manually. gcc 4.9 onwards do set __ARM_NEON.+ */+#if defined(MLD_SYS_AARCH64) && defined(__ARM_NEON)+#define MLD_SYS_AARCH64_NEON+#endif++/* Check if we're running on an AArch64 big endian system. */+#if defined(__AARCH64EB__)+#define MLD_SYS_AARCH64_EB+#endif++/* Check if we're running on an Armv8.1-M system with MVE */+#if defined(__ARM_ARCH_8_1M_MAIN__) || defined(__ARM_FEATURE_MVE)+#define MLD_SYS_ARMV81M_MVE+#endif++/* Check if we're running on an x86_64 system. */+#if defined(__x86_64__) || defined(_M_X64) || defined(_M_AMD64)+#define MLD_SYS_X86_64+#if defined(__AVX2__)+#define MLD_SYS_X86_64_AVX2+#endif+#endif /* __x86_64__ || _M_X64 || _M_AMD64 */++#if defined(MLD_SYS_LITTLE_ENDIAN) && defined(__powerpc64__)+#define MLD_SYS_PPC64LE+#endif++#if defined(__riscv) && defined(__riscv_xlen) && __riscv_xlen == 64+#define MLD_SYS_RISCV64+#endif++#if defined(MLD_SYS_RISCV64) && defined(__riscv_vector) && \+ defined(__riscv_v_intrinsic)+#define MLD_SYS_RISCV64_RVV+#endif++#if defined(__riscv) && defined(__riscv_xlen) && __riscv_xlen == 32+#define MLD_SYS_RISCV32+#endif++#if defined(_WIN64) || defined(_WIN32)+#define MLD_SYS_WINDOWS+#endif++#if defined(__linux__)+#define MLD_SYS_LINUX+#endif++#if defined(__APPLE__)+#define MLD_SYS_APPLE+#endif++/* If MLD_FORCE_AARCH64 is set, assert that we're indeed on an AArch64 system.+ */+#if defined(MLD_FORCE_AARCH64) && !defined(MLD_SYS_AARCH64)+#error "MLD_FORCE_AARCH64 is set, but we don't seem to be on an AArch64 system."+#endif++/* If MLD_FORCE_AARCH64_EB is set, assert that we're indeed on a big endian+ * AArch64 system. */+#if defined(MLD_FORCE_AARCH64_EB) && !defined(MLD_SYS_AARCH64_EB)+#error \+ "MLD_FORCE_AARCH64_EB is set, but we don't seem to be on an AArch64 system."+#endif++/* If MLD_FORCE_X86_64 is set, assert that we're indeed on an X86_64 system. */+#if defined(MLD_FORCE_X86_64) && !defined(MLD_SYS_X86_64)+#error "MLD_FORCE_X86_64 is set, but we don't seem to be on an X86_64 system."+#endif++#if defined(MLD_FORCE_PPC64LE) && !defined(MLD_SYS_PPC64LE)+#error "MLD_FORCE_PPC64LE is set, but we don't seem to be on a PPC64LE system."+#endif++#if defined(MLD_FORCE_RISCV64) && !defined(MLD_SYS_RISCV64)+#error "MLD_FORCE_RISCV64 is set, but we don't seem to be on a RISCV64 system."+#endif++#if defined(MLD_FORCE_RISCV32) && !defined(MLD_SYS_RISCV32)+#error "MLD_FORCE_RISCV32 is set, but we don't seem to be on a RISCV32 system."+#endif++/*+ * MLD_INLINE: Hint for inlining.+ * - MSVC: __inline+ * - C99+: inline+ * - GCC/Clang C90: __attribute__((unused)) to silence warnings+ * - Other C90: empty+ */+#if !defined(MLD_INLINE)+#if defined(_MSC_VER)+#define MLD_INLINE __inline+#elif defined(inline) || \+ (defined(__STDC_VERSION__) && __STDC_VERSION__ >= 199901L)+#define MLD_INLINE inline+#elif defined(__GNUC__) || defined(__clang__)+#define MLD_INLINE __attribute__((unused))+#else+#define MLD_INLINE+#endif+#endif /* !MLD_INLINE */++/*+ * MLD_ALWAYS_INLINE: Force inlining.+ * - MSVC: __forceinline+ * - GCC/Clang C99+: MLD_INLINE __attribute__((always_inline))+ * - Other: MLD_INLINE (no forced inlining)+ */+#if !defined(MLD_ALWAYS_INLINE)+#if defined(_MSC_VER)+#define MLD_ALWAYS_INLINE __forceinline+#elif (defined(__GNUC__) || defined(__clang__)) && \+ (defined(inline) || \+ (defined(__STDC_VERSION__) && __STDC_VERSION__ >= 199901L))+#define MLD_ALWAYS_INLINE MLD_INLINE __attribute__((always_inline))+#else+#define MLD_ALWAYS_INLINE MLD_INLINE+#endif+#endif /* !MLD_ALWAYS_INLINE */++/*+ * MLD_NOINLINE: Prevent inlining.+ * - MSVC: __declspec(noinline)+ * - GCC/Clang: __attribute__((noinline))+ * - Other: empty+ */+#if !defined(MLD_NOINLINE)+#if defined(_MSC_VER)+#define MLD_NOINLINE __declspec(noinline)+#elif defined(__GNUC__) || defined(__clang__)+#define MLD_NOINLINE __attribute__((noinline))+#else+#define MLD_NOINLINE+#endif+#endif /* !MLD_NOINLINE */++#ifndef MLD_STATIC_TESTABLE+#define MLD_STATIC_TESTABLE static+#endif++/*+ * C90 does not have the restrict compiler directive yet.+ * We don't use it in C90 builds.+ */+#if !defined(restrict)+#if defined(__STDC_VERSION__) && __STDC_VERSION__ >= 199901L+#define MLD_RESTRICT restrict+#else+#define MLD_RESTRICT+#endif++#else /* !restrict */++#define MLD_RESTRICT restrict+#endif /* restrict */++#define MLD_DEFAULT_ALIGN 32+#define MLD_ALIGN_UP(N) \+ ((((N) + (MLD_DEFAULT_ALIGN - 1)) / MLD_DEFAULT_ALIGN) * MLD_DEFAULT_ALIGN)+#if defined(__GNUC__)+#define MLD_ALIGN __attribute__((aligned(MLD_DEFAULT_ALIGN)))+#elif defined(_MSC_VER)+#define MLD_ALIGN __declspec(align(MLD_DEFAULT_ALIGN))+#else+#define MLD_ALIGN /* No known support for alignment constraints */+#endif+++/* New X86_64 CPUs support control-flow protection using the CET instructions.+ * When enabled (through -fcf-protection=), all compilation units (including+ * empty ones) need to support CET for this to work.+ * For assembly, this means that source files need to signal support for+ * CET by setting the appropriate note.gnu.property section.+ * This can be achieved by including the <cet.h> header in all assembly file.+ * This file also provides the _CET_ENDBR macro which needs to be placed at+ * every potential target of an indirect branch.+ * If CET is enabled _CET_ENDBR maps to the endbr64 instruction, otherwise+ * it is empty.+ * In case the compiler does not support CET (e.g., <gcc8, <clang11),+ * the __CET__ macro is not set and we default to nothing.+ * Note that we only issue _CET_ENDBR instructions through the MLD_ASM_FN_SYMBOL+ * macro as the global symbols are the only possible targets of indirect+ * branches in our code.+ */+#if defined(MLD_SYS_X86_64)+#if defined(__CET__)+#include <cet.h>+#define MLD_CET_ENDBR _CET_ENDBR+#else+#define MLD_CET_ENDBR+#endif+#endif /* MLD_SYS_X86_64 */++#if defined(MLD_CONFIG_CT_TESTING_ENABLED) && !defined(__ASSEMBLER__)+#include <valgrind/memcheck.h>+#define MLD_CT_TESTING_SECRET(ptr, len) \+ VALGRIND_MAKE_MEM_UNDEFINED((ptr), (len))+#define MLD_CT_TESTING_DECLASSIFY(ptr, len) \+ VALGRIND_MAKE_MEM_DEFINED((ptr), (len))+#else /* MLD_CONFIG_CT_TESTING_ENABLED && !__ASSEMBLER__ */+#define MLD_CT_TESTING_SECRET(ptr, len) \+ do \+ { \+ } while (0)+#define MLD_CT_TESTING_DECLASSIFY(ptr, len) \+ do \+ { \+ } while (0)+#endif /* !(MLD_CONFIG_CT_TESTING_ENABLED && !__ASSEMBLER__) */++#if defined(__GNUC__) || defined(__clang__)+#define MLD_MUST_CHECK_RETURN_VALUE __attribute__((warn_unused_result))+#else+#define MLD_MUST_CHECK_RETURN_VALUE+#endif++/* The x86_64 assembly backend uses the SysV calling convention. On Windows,+ * where the Microsoft x64 calling convention is the default, it can still be+ * used with compilers that allow choosing the calling convention per+ * function: GCC and Clang support __attribute__((sysv_abi)), which makes+ * calls to the annotated function follow the SysV calling convention.+ *+ * MLD_SYSV_ABI_SUPPORTED signals that the toolchain can call SysV assembly+ * routines; the x86_64 assembly backend is only enabled if it is defined.+ * MLD_SYSV_ABI is the attribute carried by declarations of x86_64 assembly+ * routines. Both macros can be set externally for toolchains offering an+ * equivalent mechanism that is not recognized here. */+#if defined(MLD_SYS_X86_64) && !defined(MLD_SYSV_ABI_SUPPORTED)+#if !defined(MLD_SYS_WINDOWS) || defined(__GNUC__) || defined(__clang__)+#define MLD_SYSV_ABI_SUPPORTED+#endif+#endif++#if !defined(MLD_SYSV_ABI)+#if defined(MLD_SYS_WINDOWS) && defined(MLD_SYSV_ABI_SUPPORTED)+#define MLD_SYSV_ABI __attribute__((sysv_abi))+#else+#define MLD_SYSV_ABI+#endif+#endif /* !MLD_SYSV_ABI */++#if !defined(__ASSEMBLER__)+/* System capability enumeration */+typedef enum+{+ /* x86_64 */+ MLD_SYS_CAP_X86_64_AVX2,+ /* AArch64 */+ MLD_SYS_CAP_AARCH64_NEON,+ MLD_SYS_CAP_AARCH64_SHA3,+ /* Armv8.1-M */+ MLD_SYS_CAP_ARMV81M_MVE+} mld_sys_cap;++#if !defined(MLD_CONFIG_CUSTOM_CAPABILITY_FUNC)+#include "cbmc.h"++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_sys_check_capability(mld_sys_cap cap)+__contract__(+ ensures(return_value == 0 || return_value == 1)+)+{+ /* By default, we rely on compile-time feature detection/specification:+ * If a feature is enabled at compile-time, we assume it is supported by+ * the host that the resulting library/binary will be built on.+ * If this assumption is not true, you MUST overwrite this function.+ * See the documentation of MLD_CONFIG_CUSTOM_CAPABILITY_FUNC in+ * mldsa_native_config.h for more information. */+ (void)cap;+ return 1;+}+#endif /* !MLD_CONFIG_CUSTOM_CAPABILITY_FUNC */+#endif /* !__ASSEMBLER__ */++#endif /* !MLD_SYS_H */
+ cbits/mldsa/src/zetas.inc view
@@ -0,0 +1,55 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */+++/*+ * Table of zeta values used in the reference NTT and inverse NTT.+ * See autogen for details.+ */+static const int32_t mld_zetas[MLDSA_N] = {+ 0, 25847, -2608894, -518909, 237124, -777960, -876248,+ 466468, 1826347, 2353451, -359251, -2091905, 3119733, -2884855,+ 3111497, 2680103, 2725464, 1024112, -1079900, 3585928, -549488,+ -1119584, 2619752, -2108549, -2118186, -3859737, -1399561, -3277672,+ 1757237, -19422, 4010497, 280005, 2706023, 95776, 3077325,+ 3530437, -1661693, -3592148, -2537516, 3915439, -3861115, -3043716,+ 3574422, -2867647, 3539968, -300467, 2348700, -539299, -1699267,+ -1643818, 3505694, -3821735, 3507263, -2140649, -1600420, 3699596,+ 811944, 531354, 954230, 3881043, 3900724, -2556880, 2071892,+ -2797779, -3930395, -1528703, -3677745, -3041255, -1452451, 3475950,+ 2176455, -1585221, -1257611, 1939314, -4083598, -1000202, -3190144,+ -3157330, -3632928, 126922, 3412210, -983419, 2147896, 2715295,+ -2967645, -3693493, -411027, -2477047, -671102, -1228525, -22981,+ -1308169, -381987, 1349076, 1852771, -1430430, -3343383, 264944,+ 508951, 3097992, 44288, -1100098, 904516, 3958618, -3724342,+ -8578, 1653064, -3249728, 2389356, -210977, 759969, -1316856,+ 189548, -3553272, 3159746, -1851402, -2409325, -177440, 1315589,+ 1341330, 1285669, -1584928, -812732, -1439742, -3019102, -3881060,+ -3628969, 3839961, 2091667, 3407706, 2316500, 3817976, -3342478,+ 2244091, -2446433, -3562462, 266997, 2434439, -1235728, 3513181,+ -3520352, -3759364, -1197226, -3193378, 900702, 1859098, 909542,+ 819034, 495491, -1613174, -43260, -522500, -655327, -3122442,+ 2031748, 3207046, -3556995, -525098, -768622, -3595838, 342297,+ 286988, -2437823, 4108315, 3437287, -3342277, 1735879, 203044,+ 2842341, 2691481, -2590150, 1265009, 4055324, 1247620, 2486353,+ 1595974, -3767016, 1250494, 2635921, -3548272, -2994039, 1869119,+ 1903435, -1050970, -1333058, 1237275, -3318210, -1430225, -451100,+ 1312455, 3306115, -1962642, -1279661, 1917081, -2546312, -1374803,+ 1500165, 777191, 2235880, 3406031, -542412, -2831860, -1671176,+ -1846953, -2584293, -3724270, 594136, -3776993, -2013608, 2432395,+ 2454455, -164721, 1957272, 3369112, 185531, -1207385, -3183426,+ 162844, 1616392, 3014001, 810149, 1652634, -3694233, -1799107,+ -3038916, 3523897, 3866901, 269760, 2213111, -975884, 1717735,+ 472078, -426683, 1723600, -1803090, 1910376, -1667432, -1104333,+ -260646, -3833893, -2939036, -2235985, -420899, -2286327, 183443,+ -976891, 1612842, -3545687, -554416, 3919660, -48306, -1362209,+ 3937738, 1400424, -846154, 1976782,+};
+ cbits/mlkem/COMMIT view
@@ -0,0 +1,2 @@+d1b2fe782888bdb761a50336012923180be7f502+v2.0.0
+ cbits/mlkem/LICENSE view
@@ -0,0 +1,312 @@+mlkem-native is a fork of the public domain Kyber reference+implementation, available on https://github.com/pq-crystals/kyber.++All source code in mlkem/* and dev/* is licensed under your choice+of the Apache-2.0 license OR the ISC license OR the MIT license.+These licenses are reproduced at the bottom of this file.+The copyright holders are indicated at the top of each file.++Files outside the library itself may carry different terms. Every file+states its own SPDX-License-Identifier, which determines the terms that+apply to it. In particular:++The code in test/notrandombytes/*, and its copies in+examples/*/test_only_rng/* and scripts/notrandombytes, is derived from+https://cr.yp.to/papers.html#surf and licensed under+LicenseRef-PD-hp OR CC0-1.0 OR 0BSD OR MIT-0 OR MIT.+It is only used for testing purposes.++The benchmarking code in test/hal/* carries the+MIT license. It is only used for testing purposes.++The tiny_sha3 code in+examples/bring_your_own_fips202/custom_fips202/tiny_sha3/* and+examples/custom_backend/mlkem_native/src/fips202/native/custom/src/*+carries the MIT license. It is only used to demonstrate custom+FIPS-202 implementations.++The proofs in proofs/* are in part derived from Amazon Web Services+verification infrastructure. Individual files carry, among others,+Apache-2.0 OR ISC OR MIT-0, MIT-0, and MIT-0 AND Apache-2.0. None of the+proofs are part of the library.++Documentation is licensed under CC-BY-4.0.++```+Copyright (c) The mlkem-native project authors+Copyright (c) 2020 Dougall Johnson+Copyright (c) 2022 Arm Limited+SPDX-License-Identifier: MIT++Permission is hereby granted, free of charge, to any person obtaining a copy+of this software and associated documentation files (the "Software"), to deal+in the Software without restriction, including without limitation the rights+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell+copies of the Software, and to permit persons to whom the Software is+furnished to do so, subject to the following conditions:++The above copyright notice and this permission notice shall be included in+all copies or substantial portions of the Software.++THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE+SOFTWARE.+```++The tiny_sha3 implementation in examples/custom_backend/mlkem_native/src/fips202/native/custom/src/+carries the MIT license. It is only used for testing purposes.++```+SPDX-License-Identifier: MIT++19-Nov-11 Markku-Juhani O. Saarinen <mjos@iki.fi>+```++ISC license+-----------++Copyright <YEAR> <OWNER>++Permission to use, copy, modify, and/or distribute this software for any purpose+with or without fee is hereby granted, provided that the above copyright notice+and this permission notice appear in all copies.++THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH+REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY AND+FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY SPECIAL, DIRECT,+INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS+OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER+TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF+THIS SOFTWARE.+++MIT license+-----------++Copyright <YEAR> <COPYRIGHT HOLDER>++Permission is hereby granted, free of charge, to any person obtaining a copy of+this software and associated documentation files (the “Software”), to deal in+the Software without restriction, including without limitation the rights to+use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of+the Software, and to permit persons to whom the Software is furnished to do so,+subject to the following conditions:++The above copyright notice and this permission notice shall be included in all+copies or substantial portions of the Software.++THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS+FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR+COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER+IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN+CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.++Apache-2.0 license+------------------++ Apache License+ Version 2.0, January 2004+ http://www.apache.org/licenses/++ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION++ 1. Definitions.++ "License" shall mean the terms and conditions for use, reproduction,+ and distribution as defined by Sections 1 through 9 of this document.++ "Licensor" shall mean the copyright owner or entity authorized by+ the copyright owner that is granting the License.++ "Legal Entity" shall mean the union of the acting entity and all+ other entities that control, are controlled by, or are under common+ control with that entity. For the purposes of this definition,+ "control" means (i) the power, direct or indirect, to cause the+ direction or management of such entity, whether by contract or+ otherwise, or (ii) ownership of fifty percent (50%) or more of the+ outstanding shares, or (iii) beneficial ownership of such entity.++ "You" (or "Your") shall mean an individual or Legal Entity+ exercising permissions granted by this License.++ "Source" form shall mean the preferred form for making modifications,+ including but not limited to software source code, documentation+ source, and configuration files.++ "Object" form shall mean any form resulting from mechanical+ transformation or translation of a Source form, including but+ not limited to compiled object code, generated documentation,+ and conversions to other media types.++ "Work" shall mean the work of authorship, whether in Source or+ Object form, made available under the License, as indicated by a+ copyright notice that is included in or attached to the work+ (an example is provided in the Appendix below).++ "Derivative Works" shall mean any work, whether in Source or Object+ form, that is based on (or derived from) the Work and for which the+ editorial revisions, annotations, elaborations, or other modifications+ represent, as a whole, an original work of authorship. For the purposes+ of this License, Derivative Works shall not include works that remain+ separable from, or merely link (or bind by name) to the interfaces of,+ the Work and Derivative Works thereof.++ "Contribution" shall mean any work of authorship, including+ the original version of the Work and any modifications or additions+ to that Work or Derivative Works thereof, that is intentionally+ submitted to Licensor for inclusion in the Work by the copyright owner+ or by an individual or Legal Entity authorized to submit on behalf of+ the copyright owner. For the purposes of this definition, "submitted"+ means any form of electronic, verbal, or written communication sent+ to the Licensor or its representatives, including but not limited to+ communication on electronic mailing lists, source code control systems,+ and issue tracking systems that are managed by, or on behalf of, the+ Licensor for the purpose of discussing and improving the Work, but+ excluding communication that is conspicuously marked or otherwise+ designated in writing by the copyright owner as "Not a Contribution."++ "Contributor" shall mean Licensor and any individual or Legal Entity+ on behalf of whom a Contribution has been received by Licensor and+ subsequently incorporated within the Work.++ 2. Grant of Copyright License. Subject to the terms and conditions of+ this License, each Contributor hereby grants to You a perpetual,+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable+ copyright license to reproduce, prepare Derivative Works of,+ publicly display, publicly perform, sublicense, and distribute the+ Work and such Derivative Works in Source or Object form.++ 3. Grant of Patent License. Subject to the terms and conditions of+ this License, each Contributor hereby grants to You a perpetual,+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable+ (except as stated in this section) patent license to make, have made,+ use, offer to sell, sell, import, and otherwise transfer the Work,+ where such license applies only to those patent claims licensable+ by such Contributor that are necessarily infringed by their+ Contribution(s) alone or by combination of their Contribution(s)+ with the Work to which such Contribution(s) was submitted. If You+ institute patent litigation against any entity (including a+ cross-claim or counterclaim in a lawsuit) alleging that the Work+ or a Contribution incorporated within the Work constitutes direct+ or contributory patent infringement, then any patent licenses+ granted to You under this License for that Work shall terminate+ as of the date such litigation is filed.++ 4. Redistribution. You may reproduce and distribute copies of the+ Work or Derivative Works thereof in any medium, with or without+ modifications, and in Source or Object form, provided that You+ meet the following conditions:++ (a) You must give any other recipients of the Work or+ Derivative Works a copy of this License; and++ (b) You must cause any modified files to carry prominent notices+ stating that You changed the files; and++ (c) You must retain, in the Source form of any Derivative Works+ that You distribute, all copyright, patent, trademark, and+ attribution notices from the Source form of the Work,+ excluding those notices that do not pertain to any part of+ the Derivative Works; and++ (d) If the Work includes a "NOTICE" text file as part of its+ distribution, then any Derivative Works that You distribute must+ include a readable copy of the attribution notices contained+ within such NOTICE file, excluding those notices that do not+ pertain to any part of the Derivative Works, in at least one+ of the following places: within a NOTICE text file distributed+ as part of the Derivative Works; within the Source form or+ documentation, if provided along with the Derivative Works; or,+ within a display generated by the Derivative Works, if and+ wherever such third-party notices normally appear. The contents+ of the NOTICE file are for informational purposes only and+ do not modify the License. You may add Your own attribution+ notices within Derivative Works that You distribute, alongside+ or as an addendum to the NOTICE text from the Work, provided+ that such additional attribution notices cannot be construed+ as modifying the License.++ You may add Your own copyright statement to Your modifications and+ may provide additional or different license terms and conditions+ for use, reproduction, or distribution of Your modifications, or+ for any such Derivative Works as a whole, provided Your use,+ reproduction, and distribution of the Work otherwise complies with+ the conditions stated in this License.++ 5. Submission of Contributions. Unless You explicitly state otherwise,+ any Contribution intentionally submitted for inclusion in the Work+ by You to the Licensor shall be under the terms and conditions of+ this License, without any additional terms or conditions.+ Notwithstanding the above, nothing herein shall supersede or modify+ the terms of any separate license agreement you may have executed+ with Licensor regarding such Contributions.++ 6. Trademarks. This License does not grant permission to use the trade+ names, trademarks, service marks, or product names of the Licensor,+ except as required for reasonable and customary use in describing the+ origin of the Work and reproducing the content of the NOTICE file.++ 7. Disclaimer of Warranty. Unless required by applicable law or+ agreed to in writing, Licensor provides the Work (and each+ Contributor provides its Contributions) on an "AS IS" BASIS,+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or+ implied, including, without limitation, any warranties or conditions+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A+ PARTICULAR PURPOSE. You are solely responsible for determining the+ appropriateness of using or redistributing the Work and assume any+ risks associated with Your exercise of permissions under this License.++ 8. Limitation of Liability. In no event and under no legal theory,+ whether in tort (including negligence), contract, or otherwise,+ unless required by applicable law (such as deliberate and grossly+ negligent acts) or agreed to in writing, shall any Contributor be+ liable to You for damages, including any direct, indirect, special,+ incidental, or consequential damages of any character arising as a+ result of this License or out of the use or inability to use the+ Work (including but not limited to damages for loss of goodwill,+ work stoppage, computer failure or malfunction, or any and all+ other commercial damages or losses), even if such Contributor+ has been advised of the possibility of such damages.++ 9. Accepting Warranty or Additional Liability. While redistributing+ the Work or Derivative Works thereof, You may choose to offer,+ and charge a fee for, acceptance of support, warranty, indemnity,+ or other liability obligations and/or rights consistent with this+ License. However, in accepting such obligations, You may act only+ on Your own behalf and on Your sole responsibility, not on behalf+ of any other Contributor, and only if You agree to indemnify,+ defend, and hold each Contributor harmless for any liability+ incurred by, or claims asserted against, such Contributor by reason+ of your accepting any such warranty or additional liability.++ END OF TERMS AND CONDITIONS++ APPENDIX: How to apply the Apache License to your work.++ To apply the Apache License to your work, attach the following+ boilerplate notice, with the fields enclosed by brackets "[]"+ replaced with your own identifying information. (Don't include+ the brackets!) The text should be enclosed in the appropriate+ comment syntax for the file format. We also recommend that a+ file or class name and description of purpose be included on the+ same "printed page" as the copyright notice for easier+ identification within third-party archives.++ Copyright [yyyy] [name of copyright owner]++ Licensed under the Apache License, Version 2.0 (the "License");+ you may not use this file except in compliance with the License.+ You may obtain a copy of the License at++ http://www.apache.org/licenses/LICENSE-2.0++ Unless required by applicable law or agreed to in writing, software+ distributed under the License is distributed on an "AS IS" BASIS,+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.+ See the License for the specific language governing permissions and+ limitations under the License.
+ cbits/mlkem/README.md view
@@ -0,0 +1,59 @@+# mlkem-native++ML-KEM (FIPS 203) from the PQ Code Package, vendored here and built for all+three parameter sets.++## Where it comes from++<https://github.com/pq-code-package/mlkem-native>. `COMMIT` holds the+revision this tree is at; `import.sh` puts it there and is how the tree is+refreshed. Nothing here is edited by hand -- every choice crypton makes is+made in `crypton_mlkem.h`, `crypton_mlkem.c` and `crypton.cabal`, so that a+re-import is a straight overwrite.++## Licence++`Apache-2.0 OR ISC OR MIT`, the same three-way form as `cbits/s2n` and by+some of the same authors. crypton takes it under ISC, which is already in+the package's `license:` field. `LICENSE` is the upstream file and is+listed in `license-files:`.++## What is taken and what is not++The whole of `mlkem/`, less the backends for architectures crypton does not+build for: `src/native/riscv64`, `src/native/ppc64le` and the 32-bit+`src/fips202/native/armv81m`. Every reference to those is behind an+`MLK_SYS_` guard that cannot be true on the architectures crypton does+build for, so dropping them changes no build and keeps a few dozen files of+unreachable assembly out of the release tarball. To take one back, delete+its line from `import.sh` and add its directory to `extra-source-files:`.++## How it is built++Upstream builds for one parameter set at a time. `crypton_mlkem.c`+includes the amalgamation once per set -- the level-independent half kept by+exactly one of them -- which is how a single crypton offers ML-KEM-512, 768+and 1024. `crypton_mlkem_asm.S` does the same for the assembly, which is+level-independent and so is included once.++This is the shape `cbits/aes/armv8.c` already uses for the three AES key+sizes, and it has the same hazard: **cabal does not know that the wrapper+depends on the tree it includes.** After changing anything under+`cbits/mlkem`, touch `crypton_mlkem.c`, or the build keeps the object it+already has and the change is not tested.++The hand-written backends are selected by `CRYPTON_MLKEM_NATIVE_BACKEND`,+which `crypton.cabal` defines on x86-64 and AArch64 other than Windows --+the same exclusion, and for the same reasons, as `cbits/s2n`. Everywhere+else the portable C is built, which is the same code and passes the same+tests.++There is no randomised API: `MLK_CONFIG_NO_RANDOMIZED_API` is set, no+`randombytes()` is needed, and randomness is drawn in Haskell through+`MonadRandom` as it is for every other key crypton generates.++The symbols are `crypton_mlkem512_*`, `crypton_mlkem768_*` and+`crypton_mlkem1024_*` rather than upstream's defaults, so that an+application linking another copy of mlkem-native -- through some other+library, or its own -- does not present the linker with two sets of+functions answering to one set of names.
+ cbits/mlkem/crypton_mlkem.c view
@@ -0,0 +1,31 @@+/*+ * All three ML-KEM parameter sets in one translation unit.+ *+ * mlkem-native is built for one parameter set at a time; a build wanting+ * several includes the amalgamation once per set, with the level-independent+ * half kept by exactly one of them. This is the same shape as+ * cbits/aes/armv8.c, which includes cbits/aes/armv8_impl.c three times for+ * the three AES key sizes.+ *+ * NOTE, as there: cabal does not know that this file depends on the tree it+ * includes. After changing anything under cbits/mlkem, touch this file, or+ * the build will quietly keep the object it already has.+ */+#include "crypton_mlkem.h"++#define MLK_CONFIG_MULTILEVEL_WITH_SHARED 1+#define MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+#define MLK_CONFIG_PARAMETER_SET 512+#include "mlkem_native.c"+#undef MLK_CONFIG_MULTILEVEL_WITH_SHARED+#undef MLK_CONFIG_PARAMETER_SET++#define MLK_CONFIG_MULTILEVEL_NO_SHARED+#define MLK_CONFIG_PARAMETER_SET 768+#include "mlkem_native.c"+#undef MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+#undef MLK_CONFIG_PARAMETER_SET++#define MLK_CONFIG_PARAMETER_SET 1024+#include "mlkem_native.c"+#undef MLK_CONFIG_PARAMETER_SET
+ cbits/mlkem/crypton_mlkem.h view
@@ -0,0 +1,41 @@+/*+ * What crypton asks of the vendored mlkem-native, in one place. The tree+ * under cbits/mlkem is upstream's and is overwritten by import.sh, so every+ * choice crypton makes is made here instead of by editing it.+ *+ * Included first by both crypton_mlkem.c and crypton_mlkem_asm.S, so it must+ * hold nothing but preprocessor directives.+ */+#ifndef CRYPTON_MLKEM_H+#define CRYPTON_MLKEM_H++/*+ * The symbols are crypton's own, not the default PQCP_MLKEM_NATIVE_*. An+ * application is free to link another copy of mlkem-native -- through some+ * other library, or its own -- and two copies answering to one set of names+ * is a problem the linker resolves silently and in nobody's favour. With+ * MLK_CONFIG_MULTILEVEL_BUILD the level is appended, so the entry points+ * are crypton_mlkem512_*, crypton_mlkem768_* and crypton_mlkem1024_*.+ */+#define MLK_CONFIG_NAMESPACE_PREFIX crypton_mlkem+#define MLK_CONFIG_MULTILEVEL_BUILD++/*+ * No randomised API, so no randombytes() to provide. Randomness is drawn+ * in Haskell through MonadRandom, the way every other key in crypton is+ * generated, and the deterministic entry points are what the FFI calls.+ * That also keeps the C free of any opinion about where entropy comes from.+ */+#define MLK_CONFIG_NO_RANDOMIZED_API++/*+ * The hand-written backends, where crypton.cabal says the architecture has+ * them. Without this the portable C is built, which is correct everywhere+ * and is what every other architecture gets.+ */+#ifdef CRYPTON_MLKEM_NATIVE_BACKEND+#define MLK_CONFIG_USE_NATIVE_BACKEND_ARITH+#define MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202+#endif++#endif /* CRYPTON_MLKEM_H */
+ cbits/mlkem/crypton_mlkem_asm.S view
@@ -0,0 +1,14 @@+/*+ * The assembly half, which is level-independent: it is included once, with+ * the shared directives kept, and covers all three parameter sets. See the+ * comment at the top of mlkem_native_asm.S.+ *+ * Built only where crypton.cabal defines CRYPTON_MLKEM_NATIVE_BACKEND; on+ * every other architecture this file is not in asm-sources at all.+ */+#include "crypton_mlkem.h"++#define MLK_CONFIG_MULTILEVEL_WITH_SHARED 1+#define MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+#define MLK_CONFIG_PARAMETER_SET 768+#include "mlkem_native_asm.S"
+ cbits/mlkem/import.sh view
@@ -0,0 +1,42 @@+#!/bin/sh+# Re-import the vendored parts of the PQ Code Package's mlkem-native.+#+# The files are kept unmodified. Everything crypton decides -- which+# parameter sets exist, what the symbols are called, that there is no+# randomised API -- is decided in cbits/mlkem/crypton_mlkem.c and in+# crypton.cabal, not by editing anything here. Run this from cbits/mlkem:+#+# ./import.sh [tag-or-commit]+#+# and commit the result together with the COMMIT line it writes, so that the+# tree always says which upstream revision it holds.+set -eu++REPO=https://github.com/pq-code-package/mlkem-native+REV=${1:-v2.0.0}+HERE=$(cd "$(dirname "$0")" && pwd)+TMP=$(mktemp -d)+trap 'rm -rf "$TMP"' EXIT++git clone -q "$REPO" "$TMP/u"+git -C "$TMP/u" checkout -q "$REV"++rm -rf "$HERE/src"+cp "$TMP/u/mlkem/mlkem_native.c" "$HERE/"+cp "$TMP/u/mlkem/mlkem_native.h" "$HERE/"+cp "$TMP/u/mlkem/mlkem_native_asm.S" "$HERE/"+cp "$TMP/u/mlkem/mlkem_native_config.h" "$HERE/"+cp -R "$TMP/u/mlkem/src" "$HERE/src"+cp "$TMP/u/LICENSE" "$HERE/LICENSE"++# The backends for architectures crypton does not build for. Every+# reference to them is behind an MLK_SYS_ guard that cannot be true on the+# two it does, so dropping them changes no build and keeps a few dozen files+# of unreachable assembly out of the release tarball. If crypton ever+# wants one of them, delete its line here rather than patching anything.+rm -rf "$HERE/src/native/riscv64" "$HERE/src/native/ppc64le"+rm -rf "$HERE/src/fips202/native/armv81m"++git -C "$TMP/u" rev-parse HEAD > "$HERE/COMMIT"+git -C "$TMP/u" describe --tags --exact-match 2>/dev/null >> "$HERE/COMMIT" || true+echo "imported $(head -1 "$HERE/COMMIT")"
+ cbits/mlkem/mlkem_native.c view
@@ -0,0 +1,692 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++/******************************************************************************+ *+ * Single compilation unit (SCU) for fixed-level build of mlkem-native+ *+ * This compilation unit bundles together all source files for a build+ * of mlkem-native for a fixed security level (MLKEM-512/768/1024).+ *+ * # API+ *+ * The API exposed by this file is described in mlkem_native.h.+ *+ * # Multi-level build+ *+ * If you want an SCU build of mlkem-native with support for multiple security+ * levels, you need to include this file multiple times, and set+ * MLK_CONFIG_MULTILEVEL_WITH_SHARED and MLK_CONFIG_MULTILEVEL_NO_SHARED+ * appropriately. This is exemplified in examples/monolithic_build_multilevel+ * and examples/monolithic_build_multilevel_native.+ *+ * # Configuration+ *+ * The following options from the mlkem-native configuration are relevant:+ *+ * - MLK_CONFIG_FIPS202_CUSTOM_HEADER+ * Set this option if you use a custom FIPS202 implementation.+ *+ * - MLK_CONFIG_USE_NATIVE_BACKEND_ARITH+ * Set this option if you want to include the native arithmetic backends+ * in your build.+ *+ * - MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202+ * Set this option if you want to include the native FIPS202 backends+ * in your build.+ *+ * - MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+ * Set this option if you want to keep the directives defined in+ * level-independent headers. This is needed for a multi-level build.+ */++/* If parts of the mlkem-native source tree are not used,+ * consider reducing this header via `unifdef`.+ *+ * Example:+ * ```bash+ * unifdef -UMLK_CONFIG_USE_NATIVE_BACKEND_ARITH mlkem_native.c+ * ```+ */++#include "src/common.h"++#include "src/compress.c"+#include "src/debug.c"+#include "src/indcpa.c"+#include "src/kem.c"+#include "src/poly.c"+#include "src/poly_k.c"+#include "src/sampling.c"+#include "src/verify.c"++#if !defined(MLK_CONFIG_FIPS202_CUSTOM_HEADER)+#include "src/fips202/fips202.c"+#include "src/fips202/fips202x4.c"+#include "src/fips202/keccakf1600.c"+#endif++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH)+#if defined(MLK_SYS_AARCH64)+#include "src/native/aarch64/src/aarch64_zetas.c"+#include "src/native/aarch64/src/rej_uniform_table.c"+#endif+#if defined(MLK_SYS_X86_64)+#include "src/native/x86_64/src/compress_consts.c"+#include "src/native/x86_64/src/consts.c"+#include "src/native/x86_64/src/rej_uniform_table.c"+#endif+#if defined(MLK_SYS_RISCV64)+#include "src/native/riscv64/src/rv64v_debug.c"+#include "src/native/riscv64/src/rv64v_poly.c"+#endif+#if defined(MLK_SYS_PPC64LE)+#include "src/native/ppc64le/src/consts.c"+#endif+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_ARITH */++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202)+#if defined(MLK_SYS_AARCH64)+#include "src/fips202/native/aarch64/src/keccakf1600_round_constants.c"+#endif+#if defined(MLK_SYS_X86_64)+#include "src/fips202/native/x86_64/src/keccakf1600_constants.c"+#endif+#if defined(MLK_SYS_ARMV81M_MVE)+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_mve.c"+#include "src/fips202/native/armv81m/src/keccakf1600_round_constants.c"+#endif+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 */++/* Macro #undef's+ *+ * The following undefines macros from headers+ * included by the source files imported above.+ *+ * This is to allow building and linking multiple builds+ * of mlkem-native for varying parameter sets through concatenation+ * of this file, as if the files had been compiled separately.+ * If this is not relevant to you, you may remove the following.+ */++/*+ * Undefine macros from MLK_CONFIG_PARAMETER_SET-specific files+ */+/* mlkem/mlkem_native.h */+#undef MLKEM1024_BYTES+#undef MLKEM1024_CIPHERTEXTBYTES+#undef MLKEM1024_PUBLICKEYBYTES+#undef MLKEM1024_SECRETKEYBYTES+#undef MLKEM1024_SYMBYTES+#undef MLKEM512_BYTES+#undef MLKEM512_CIPHERTEXTBYTES+#undef MLKEM512_PUBLICKEYBYTES+#undef MLKEM512_SECRETKEYBYTES+#undef MLKEM512_SYMBYTES+#undef MLKEM768_BYTES+#undef MLKEM768_CIPHERTEXTBYTES+#undef MLKEM768_PUBLICKEYBYTES+#undef MLKEM768_SECRETKEYBYTES+#undef MLKEM768_SYMBYTES+#undef MLKEM_BYTES+#undef MLKEM_CIPHERTEXTBYTES+#undef MLKEM_CIPHERTEXTBYTES_+#undef MLKEM_PUBLICKEYBYTES+#undef MLKEM_PUBLICKEYBYTES_+#undef MLKEM_SECRETKEYBYTES+#undef MLKEM_SECRETKEYBYTES_+#undef MLKEM_SYMBYTES+#undef MLK_API_CONCAT+#undef MLK_API_CONCAT_+#undef MLK_API_CONCAT_UNDERSCORE+#undef MLK_API_MUST_CHECK_RETURN_VALUE+#undef MLK_API_NAMESPACE+#undef MLK_API_NAMESPACE_PREFIX+#undef MLK_API_QUALIFIER+#undef MLK_ERR_FAIL+#undef MLK_ERR_INVALID_PK+#undef MLK_ERR_INVALID_SK+#undef MLK_ERR_OUT_OF_MEMORY+#undef MLK_ERR_PCT_FAIL+#undef MLK_ERR_RNG_FAIL+#undef MLK_H+#undef MLK_MAX3_+#undef MLK_TOTAL_ALLOC_1024+#undef MLK_TOTAL_ALLOC_1024_DECAPS+#undef MLK_TOTAL_ALLOC_1024_ENCAPS+#undef MLK_TOTAL_ALLOC_1024_KEYPAIR+#undef MLK_TOTAL_ALLOC_1024_KEYPAIR_NO_PCT+#undef MLK_TOTAL_ALLOC_1024_KEYPAIR_PCT+#undef MLK_TOTAL_ALLOC_512+#undef MLK_TOTAL_ALLOC_512_DECAPS+#undef MLK_TOTAL_ALLOC_512_ENCAPS+#undef MLK_TOTAL_ALLOC_512_KEYPAIR+#undef MLK_TOTAL_ALLOC_512_KEYPAIR_NO_PCT+#undef MLK_TOTAL_ALLOC_512_KEYPAIR_PCT+#undef MLK_TOTAL_ALLOC_768+#undef MLK_TOTAL_ALLOC_768_DECAPS+#undef MLK_TOTAL_ALLOC_768_ENCAPS+#undef MLK_TOTAL_ALLOC_768_KEYPAIR+#undef MLK_TOTAL_ALLOC_768_KEYPAIR_NO_PCT+#undef MLK_TOTAL_ALLOC_768_KEYPAIR_PCT+/* mlkem/src/common.h */+#undef MLK_ADD_PARAM_SET+#undef MLK_ALLOC+#undef MLK_APPLY+#undef MLK_ASM_FN_SIZE+#undef MLK_ASM_FN_SYMBOL+#undef MLK_ASM_NAMESPACE+#undef MLK_BUILD_INTERNAL+#undef MLK_COMMON_H+#undef MLK_CONCAT+#undef MLK_CONCAT_+#undef MLK_EMPTY_CU+#undef MLK_ERR_FAIL+#undef MLK_ERR_INVALID_PK+#undef MLK_ERR_INVALID_SK+#undef MLK_ERR_OUT_OF_MEMORY+#undef MLK_ERR_PCT_FAIL+#undef MLK_ERR_RNG_FAIL+#undef MLK_EXTERNAL_API+#undef MLK_FIPS202X4_HEADER_FILE+#undef MLK_FIPS202_HEADER_FILE+#undef MLK_FREE+#undef MLK_INTERNAL_API+#undef MLK_INTERNAL_DATA_DECLARATION+#undef MLK_INTERNAL_DATA_DEFINITION+#undef MLK_NAMESPACE+#undef MLK_NAMESPACE_K+#undef MLK_NAMESPACE_PREFIX+#undef MLK_NAMESPACE_PREFIX_K+#undef mlk_memcpy+#undef mlk_memset+/* mlkem/src/indcpa.h */+#undef MLK_INDCPA_H+#undef mlk_gen_matrix+#undef mlk_indcpa_dec+#undef mlk_indcpa_enc+#undef mlk_indcpa_keypair_derand+/* mlkem/src/kem.h */+#undef MLK_KEM_H+#undef mlk_kem_check_pk+#undef mlk_kem_check_sk+#undef mlk_kem_dec+#undef mlk_kem_enc+#undef mlk_kem_enc_derand+#undef mlk_kem_keypair+#undef mlk_kem_keypair_derand+/* mlkem/src/params.h */+#undef MLKEM_DU+#undef MLKEM_DV+#undef MLKEM_ETA1+#undef MLKEM_ETA2+#undef MLKEM_INDCCA_CIPHERTEXTBYTES+#undef MLKEM_INDCCA_PUBLICKEYBYTES+#undef MLKEM_INDCCA_SECRETKEYBYTES+#undef MLKEM_INDCPA_BYTES+#undef MLKEM_INDCPA_MSGBYTES+#undef MLKEM_INDCPA_PUBLICKEYBYTES+#undef MLKEM_INDCPA_SECRETKEYBYTES+#undef MLKEM_K+#undef MLKEM_N+#undef MLKEM_POLYBYTES+#undef MLKEM_POLYCOMPRESSEDBYTES_D10+#undef MLKEM_POLYCOMPRESSEDBYTES_D11+#undef MLKEM_POLYCOMPRESSEDBYTES_D4+#undef MLKEM_POLYCOMPRESSEDBYTES_D5+#undef MLKEM_POLYCOMPRESSEDBYTES_DU+#undef MLKEM_POLYCOMPRESSEDBYTES_DV+#undef MLKEM_POLYVECBYTES+#undef MLKEM_POLYVECCOMPRESSEDBYTES_DU+#undef MLKEM_Q+#undef MLKEM_Q_HALF+#undef MLKEM_SSBYTES+#undef MLKEM_SYMBYTES+#undef MLKEM_UINT12_LIMIT+#undef MLK_PARAMS_H+/* mlkem/src/poly_k.h */+#undef MLK_POLY_K_H+#undef mlk_poly_compress_du+#undef mlk_poly_compress_dv+#undef mlk_poly_decompress_du+#undef mlk_poly_decompress_dv+#undef mlk_poly_getnoise_eta1122_4x+#undef mlk_poly_getnoise_eta1_4x+#undef mlk_poly_getnoise_eta2+#undef mlk_poly_getnoise_eta2_4x+#undef mlk_polymat+#undef mlk_polyvec+#undef mlk_polyvec_add+#undef mlk_polyvec_basemul_acc_montgomery_cached+#undef mlk_polyvec_compress_du+#undef mlk_polyvec_decompress_du+#undef mlk_polyvec_frombytes+#undef mlk_polyvec_invntt_tomont+#undef mlk_polyvec_mulcache+#undef mlk_polyvec_mulcache_compute+#undef mlk_polyvec_ntt+#undef mlk_polyvec_reduce+#undef mlk_polyvec_tobytes+#undef mlk_polyvec_tomont++#if !defined(MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS)+/*+ * Undefine macros from MLK_CONFIG_PARAMETER_SET-generic files+ */+/* mlkem/src/compress.h */+#undef MLK_COMPRESS_H+#undef mlk_poly_compress_d10+#undef mlk_poly_compress_d11+#undef mlk_poly_compress_d4+#undef mlk_poly_compress_d5+#undef mlk_poly_decompress_d10+#undef mlk_poly_decompress_d11+#undef mlk_poly_decompress_d4+#undef mlk_poly_decompress_d5+#undef mlk_poly_frombytes+#undef mlk_poly_frommsg+#undef mlk_poly_tobytes+#undef mlk_poly_tomsg+/* mlkem/src/context.h */+#undef MLK_CONTEXT_H+#undef MLK_CONTEXT_PARAMETERS_0+#undef MLK_CONTEXT_PARAMETERS_1+#undef MLK_CONTEXT_PARAMETERS_2+#undef MLK_CONTEXT_PARAMETERS_3+#undef MLK_CONTEXT_PARAMETERS_4+#undef MLK_CONTEXT_UNUSED+/* mlkem/src/debug.h */+#undef MLK_DEBUG_H+#undef mlk_assert+#undef mlk_assert_abs_bound+#undef mlk_assert_abs_bound_2d+#undef mlk_assert_bound+#undef mlk_assert_bound_2d+#undef mlk_debug_check_assert+#undef mlk_debug_check_bounds+/* mlkem/src/poly.h */+#undef MLK_INVNTT_BOUND+#undef MLK_NTT_BOUND+#undef MLK_POLY_H+#undef mlk_poly_add+#undef mlk_poly_invntt_tomont+#undef mlk_poly_mulcache_compute+#undef mlk_poly_ntt+#undef mlk_poly_reduce+#undef mlk_poly_sub+#undef mlk_poly_tomont+/* mlkem/src/randombytes.h */+#undef MLK_RANDOMBYTES_H+/* mlkem/src/sampling.h */+#undef MLK_SAMPLING_H+#undef mlk_poly_cbd2+#undef mlk_poly_cbd3+#undef mlk_poly_rej_uniform+#undef mlk_poly_rej_uniform_x4+/* mlkem/src/symmetric.h */+#undef MLK_SYMMETRIC_H+#undef MLK_XOF_RATE+#undef mlk_hash_g+#undef mlk_hash_h+#undef mlk_hash_j+#undef mlk_prf_eta+#undef mlk_prf_eta1+#undef mlk_prf_eta1_x4+#undef mlk_prf_eta2+#undef mlk_xof_absorb+#undef mlk_xof_ctx+#undef mlk_xof_init+#undef mlk_xof_release+#undef mlk_xof_squeezeblocks+#undef mlk_xof_x4_absorb+#undef mlk_xof_x4_ctx+#undef mlk_xof_x4_init+#undef mlk_xof_x4_release+#undef mlk_xof_x4_squeezeblocks+/* mlkem/src/sys.h */+#undef MLK_ALIGN+#undef MLK_ALIGN_UP+#undef MLK_ALWAYS_INLINE+#undef MLK_CET_ENDBR+#undef MLK_CT_TESTING_DECLASSIFY+#undef MLK_CT_TESTING_SECRET+#undef MLK_DEFAULT_ALIGN+#undef MLK_HAVE_INLINE_ASM+#undef MLK_INLINE+#undef MLK_MUST_CHECK_RETURN_VALUE+#undef MLK_NOINLINE+#undef MLK_RESTRICT+#undef MLK_STATIC_TESTABLE+#undef MLK_SYSV_ABI+#undef MLK_SYSV_ABI_SUPPORTED+#undef MLK_SYS_AARCH64+#undef MLK_SYS_AARCH64_EB+#undef MLK_SYS_AARCH64_NEON+#undef MLK_SYS_APPLE+#undef MLK_SYS_ARMV81M_MVE+#undef MLK_SYS_BIG_ENDIAN+#undef MLK_SYS_H+#undef MLK_SYS_LINUX+#undef MLK_SYS_LITTLE_ENDIAN+#undef MLK_SYS_PPC64LE+#undef MLK_SYS_RISCV32+#undef MLK_SYS_RISCV64+#undef MLK_SYS_RISCV64_RVV+#undef MLK_SYS_WINDOWS+#undef MLK_SYS_X86_64+#undef MLK_SYS_X86_64_AVX2+/* mlkem/src/verify.h */+#undef MLK_USE_ASM_VALUE_BARRIER+#undef MLK_VERIFY_H+#undef mlk_ct_opt_blocker_u64+/* mlkem/src/cbmc.h */+#undef MLK_CBMC_H+#undef __contract__+#undef __loop__++#if !defined(MLK_CONFIG_FIPS202_CUSTOM_HEADER)+/*+ * Undefine macros from FIPS-202 files+ */+/* mlkem/src/fips202/fips202.h */+#undef FIPS202_X4_DEFAULT_IMPLEMENTATION+#undef MLK_FIPS202_FIPS202_H+#undef SHA3_256_HASHBYTES+#undef SHA3_256_RATE+#undef SHA3_384_RATE+#undef SHA3_512_HASHBYTES+#undef SHA3_512_RATE+#undef SHAKE128_RATE+#undef SHAKE256_RATE+#undef mlk_sha3_256+#undef mlk_sha3_512+#undef mlk_shake128_absorb_once+#undef mlk_shake128_init+#undef mlk_shake128_release+#undef mlk_shake128_squeezeblocks+#undef mlk_shake256+/* mlkem/src/fips202/fips202x4.h */+#undef MLK_FIPS202_FIPS202X4_H+#undef mlk_shake128x4_absorb_once+#undef mlk_shake128x4_init+#undef mlk_shake128x4_release+#undef mlk_shake128x4_squeezeblocks+#undef mlk_shake256x4+/* mlkem/src/fips202/keccakf1600.h */+#undef MLK_FIPS202_KECCAKF1600_H+#undef MLK_KECCAK_LANES+#undef MLK_KECCAK_WAY+#undef mlk_keccakf1600_extract_bytes+#undef mlk_keccakf1600_permute+#undef mlk_keccakf1600_xor_bytes+#undef mlk_keccakf1600x4_extract_bytes+#undef mlk_keccakf1600x4_permute+#undef mlk_keccakf1600x4_xor_bytes+#endif /* !MLK_CONFIG_FIPS202_CUSTOM_HEADER */++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202)+/* mlkem/src/fips202/native/api.h */+#undef MLK_FIPS202_NATIVE_API_H+#undef MLK_NATIVE_FUNC_FALLBACK+#undef MLK_NATIVE_FUNC_SUCCESS+/* mlkem/src/fips202/native/auto.h */+#undef MLK_FIPS202_NATIVE_AUTO_H+#if defined(MLK_SYS_AARCH64)+/*+ * Undefine macros from native code (FIPS202, AArch64)+ */+/* mlkem/src/fips202/native/aarch64/auto.h */+#undef MLK_FIPS202_NATIVE_AARCH64_AUTO_H+/* mlkem/src/fips202/native/aarch64/src/fips202_native_aarch64.h */+#undef MLK_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H+#undef mlk_keccak_f1600_x1_scalar_aarch64_asm+#undef mlk_keccak_f1600_x1_v84a_aarch64_asm+#undef mlk_keccak_f1600_x2_v84a_aarch64_asm+#undef mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm+#undef mlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm+#undef mlk_keccakf1600_round_constants+/* mlkem/src/fips202/native/aarch64/x1_scalar.h */+#undef MLK_FIPS202_AARCH64_NEED_X1_SCALAR+#undef MLK_FIPS202_NATIVE_AARCH64_X1_SCALAR_H+#undef MLK_USE_NATIVE_FIPS202_X1+/* mlkem/src/fips202/native/aarch64/x1_v84a.h */+#undef MLK_FIPS202_AARCH64_NEED_X1_V84A+#undef MLK_FIPS202_NATIVE_AARCH64_X1_V84A_H+#undef MLK_USE_NATIVE_FIPS202_X1+/* mlkem/src/fips202/native/aarch64/x2_v84a.h */+#undef MLK_FIPS202_AARCH64_NEED_X2_V84A+#undef MLK_FIPS202_NATIVE_AARCH64_X2_V84A_H+#undef MLK_USE_NATIVE_FIPS202_X4+/* mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h */+#undef MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID+#undef MLK_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H+#undef MLK_USE_NATIVE_FIPS202_X4+/* mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h */+#undef MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID+#undef MLK_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H+#undef MLK_USE_NATIVE_FIPS202_X4+#endif /* MLK_SYS_AARCH64 */+#if defined(MLK_SYS_X86_64)+/*+ * Undefine macros from native code (FIPS202, x86_64)+ */+/* mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h */+#undef MLK_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H+#undef MLK_FIPS202_X86_64_NEED_X4_AVX2+#undef MLK_USE_NATIVE_FIPS202_X4+/* mlkem/src/fips202/native/x86_64/src/fips202_native_x86_64.h */+#undef MLK_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H+#undef mlk_keccak_f1600_x4_avx2_asm+#undef mlk_keccak_rho56+#undef mlk_keccak_rho8+#undef mlk_keccakf1600_round_constants+#endif /* MLK_SYS_X86_64 */+#if defined(MLK_SYS_ARMV81M_MVE)+/*+ * Undefine macros from native code (FIPS202, Armv8.1-M)+ */+/* mlkem/src/fips202/native/armv81m/mve.h */+#undef MLK_FIPS202_ARMV81M_NEED_X4+#undef MLK_FIPS202_NATIVE_ARMV81M+#undef MLK_FIPS202_NATIVE_ARMV81M_MVE_H+#undef MLK_USE_NATIVE_FIPS202_X4+#undef MLK_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES+#undef MLK_USE_NATIVE_FIPS202_X4_XOR_BYTES+#undef mlk_keccak_f1600_x4_native_impl+/* mlkem/src/fips202/native/armv81m/src/fips202_native_armv81m.h */+#undef MLK_FIPS202_NATIVE_ARMV81M_SRC_FIPS202_NATIVE_ARMV81M_H+#undef mlk_keccak_f1600_x4_mve_asm+#undef mlk_keccak_f1600_x4_state_extract_bytes_asm+#undef mlk_keccak_f1600_x4_state_xor_bytes_asm+#undef mlk_keccakf1600_round_constants+#endif /* MLK_SYS_ARMV81M_MVE */+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH)+/* mlkem/src/native/api.h */+#undef MLK_INVNTT_BOUND+#undef MLK_NATIVE_API_H+#undef MLK_NATIVE_FUNC_FALLBACK+#undef MLK_NATIVE_FUNC_SUCCESS+#undef MLK_NTT_BOUND+/* mlkem/src/native/meta.h */+#undef MLK_NATIVE_META_H+#if defined(MLK_SYS_AARCH64)+/*+ * Undefine macros from native code (Arith, AArch64)+ */+/* mlkem/src/native/aarch64/meta.h */+#undef MLK_ARITH_BACKEND_AARCH64+#undef MLK_NATIVE_AARCH64_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#undef MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOBYTES+#undef MLK_USE_NATIVE_POLY_TOMONT+#undef MLK_USE_NATIVE_REJ_UNIFORM+/* mlkem/src/native/aarch64/src/arith_native_aarch64.h */+#undef MLK_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H+#undef mlk_aarch64_invntt_zetas_layer12345+#undef mlk_aarch64_invntt_zetas_layer67+#undef mlk_aarch64_ntt_zetas_layer12345+#undef mlk_aarch64_ntt_zetas_layer67+#undef mlk_aarch64_zetas_mulcache_native+#undef mlk_aarch64_zetas_mulcache_twisted_native+#undef mlk_intt_aarch64_asm+#undef mlk_ntt_aarch64_asm+#undef mlk_poly_mulcache_compute_aarch64_asm+#undef mlk_poly_reduce_aarch64_asm+#undef mlk_poly_tobytes_aarch64_asm+#undef mlk_poly_tomont_aarch64_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm+#undef mlk_rej_uniform_aarch64_asm+#undef mlk_rej_uniform_table+#endif /* MLK_SYS_AARCH64 */+#if defined(MLK_SYS_X86_64)+/*+ * Undefine macros from native code (Arith, X86_64)+ */+/* mlkem/src/native/x86_64/meta.h */+#undef MLK_ARITH_BACKEND_X86_64_DEFAULT+#undef MLK_NATIVE_X86_64_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_NTT_CUSTOM_ORDER+#undef MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#undef MLK_USE_NATIVE_POLY_COMPRESS_D10+#undef MLK_USE_NATIVE_POLY_COMPRESS_D11+#undef MLK_USE_NATIVE_POLY_COMPRESS_D4+#undef MLK_USE_NATIVE_POLY_COMPRESS_D5+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D10+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D11+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D4+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D5+#undef MLK_USE_NATIVE_POLY_FROMBYTES+#undef MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOBYTES+#undef MLK_USE_NATIVE_POLY_TOMONT+#undef MLK_USE_NATIVE_REJ_UNIFORM+/* mlkem/src/native/x86_64/src/arith_native_x86_64.h */+#undef MLK_AVX2_REJ_UNIFORM_BUFLEN+#undef MLK_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H+#undef mlk_invntt_avx2_asm+#undef mlk_ntt_avx2_asm+#undef mlk_nttfrombytes_avx2_asm+#undef mlk_ntttobytes_avx2_asm+#undef mlk_nttunpack_avx2_asm+#undef mlk_poly_compress_d10_avx2_asm+#undef mlk_poly_compress_d11_avx2_asm+#undef mlk_poly_compress_d4_avx2_asm+#undef mlk_poly_compress_d5_avx2_asm+#undef mlk_poly_decompress_d10_avx2_asm+#undef mlk_poly_decompress_d11_avx2_asm+#undef mlk_poly_decompress_d4_avx2_asm+#undef mlk_poly_decompress_d5_avx2_asm+#undef mlk_poly_mulcache_compute_avx2_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm+#undef mlk_reduce_avx2_asm+#undef mlk_rej_uniform_avx2_asm+#undef mlk_rej_uniform_table+#undef mlk_tomont_avx2_asm+/* mlkem/src/native/x86_64/src/compress_consts.h */+#undef MLK_NATIVE_X86_64_SRC_COMPRESS_CONSTS_H+#undef mlk_compress_d10_data+#undef mlk_compress_d11_data+#undef mlk_compress_d4_data+#undef mlk_compress_d5_data+#undef mlk_decompress_d10_data+#undef mlk_decompress_d11_data+#undef mlk_decompress_d4_data+#undef mlk_decompress_d5_data+/* mlkem/src/native/x86_64/src/consts.h */+#undef MLK_AVX2_BACKEND_DATA_OFFSET_MULCACHE_TWIDDLES+#undef MLK_AVX2_BACKEND_DATA_OFFSET_REVIDXB+#undef MLK_AVX2_BACKEND_DATA_OFFSET_REVIDXD+#undef MLK_AVX2_BACKEND_DATA_OFFSET_ZETAS_EXP+#undef MLK_NATIVE_X86_64_SRC_CONSTS_H+#undef mlk_qdata+#endif /* MLK_SYS_X86_64 */+#if defined(MLK_SYS_RISCV64)+/*+ * Undefine macros from native code (Arith, RISC-V 64)+ */+/* mlkem/src/native/riscv64/meta.h */+#undef MLK_ARITH_BACKEND_RISCV64+#undef MLK_NATIVE_RISCV64_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#undef MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOMONT+#undef MLK_USE_NATIVE_REJ_UNIFORM+/* mlkem/src/native/riscv64/src/arith_native_riscv64.h */+#undef MLK_NATIVE_RISCV64_SRC_ARITH_NATIVE_RISCV64_H+#undef mlk_rv64v_poly_add+#undef mlk_rv64v_poly_basemul_mont_add_k2+#undef mlk_rv64v_poly_basemul_mont_add_k3+#undef mlk_rv64v_poly_basemul_mont_add_k4+#undef mlk_rv64v_poly_invntt_tomont+#undef mlk_rv64v_poly_ntt+#undef mlk_rv64v_poly_reduce+#undef mlk_rv64v_poly_sub+#undef mlk_rv64v_poly_tomont+#undef mlk_rv64v_rej_uniform+/* mlkem/src/native/riscv64/src/rv64v_debug.h */+#undef MLK_NATIVE_RISCV64_SRC_RV64V_DEBUG_H+#undef mlk_assert_abs_bound_int16m1+#undef mlk_assert_abs_bound_int16m2+#undef mlk_assert_bound_int16m1+#undef mlk_assert_bound_int16m2+#undef mlk_debug_check_bounds_int16m1+#undef mlk_debug_check_bounds_int16m2+#endif /* MLK_SYS_RISCV64 */+#if defined(MLK_SYS_PPC64LE)+/*+ * Undefine macros from native code (Arith, PPC64LE)+ */+/* mlkem/src/native/ppc64le/meta.h */+#undef MLK_ARITH_BACKEND_NAME+#undef MLK_ARITH_BACKEND_PPC64LE_DEFAULT+#undef MLK_NATIVE_PPC64LE_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOMONT+/* mlkem/src/native/ppc64le/src/arith_native_ppc64le.h */+#undef MLK_NATIVE_PPC64LE_SRC_ARITH_NATIVE_PPC64LE_H+#undef mlk_intt_ppc_asm+#undef mlk_ntt_ppc_asm+#undef mlk_poly_tomont_ppc_asm+#undef mlk_reduce_ppc_asm+/* mlkem/src/native/ppc64le/src/consts.h */+#undef MLK_NATIVE_PPC64LE_SRC_CONSTS_H+#undef MLK_PPC_C20159_OFFSET+#undef MLK_PPC_NQ_OFFSET+#undef MLK_PPC_N_INV_OFFSET+#undef MLK_PPC_N_INV_TW_OFFSET+#undef MLK_PPC_Q_OFFSET+#undef MLK_PPC_TOMONT_OFFSET+#undef MLK_PPC_TOMONT_TW_OFFSET+#undef MLK_PPC_ZETA_INTT_OFFSET+#undef MLK_PPC_ZETA_INTT_TW_OFFSET+#undef MLK_PPC_ZETA_NTT_OFFSET+#undef MLK_PPC_ZETA_NTT_TW_OFFSET+#undef mlk_ppc_qdata+#endif /* MLK_SYS_PPC64LE */+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_ARITH */+#endif /* !MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS */
+ cbits/mlkem/mlkem_native.h view
@@ -0,0 +1,464 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS140_3_IG]+ * Implementation Guidance for FIPS 140-3 and the Cryptographic Module+ * Validation Program+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/projects/cryptographic-module-validation-program/fips-140-3-ig-announcements+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#ifndef MLK_H+#define MLK_H++/*+ * Public API for mlkem-native.+ *+ * This header defines the public API of a single build of mlkem-native.+ *+ * Make sure the configuration file is in the include path+ * (this is "mlkem_native_config.h" by default, or MLK_CONFIG_FILE if defined).+ *+ * # API conventions+ *+ * Conventions shared by all functions below (return values, pointer validity,+ * output buffers on error) are documented in API-CONVENTIONS.md.+ *+ * # Multi-level builds+ *+ * This header specifies a build of mlkem-native for a fixed security level.+ * If you need multiple security levels, leave the security level unspecified+ * in the configuration file and include this header multiple times, setting+ * MLK_CONFIG_PARAMETER_SET accordingly for each, and #undef'ing the MLK_H+ * guard to allow multiple inclusions.+ *+ * In this case, the configuration file must also set+ * MLK_CONFIG_MULTILEVEL_BUILD. Without it, the parameter set is not appended+ * to the namespace prefix and all inclusions declare the same symbol names.+ */++/******************************* Key sizes ************************************/++/* Sizes of cryptographic material, per parameter set */+/* See mlkem/src/params.h for the arithmetic expressions giving rise to these */+/* check-magic: off */+#define MLKEM512_SECRETKEYBYTES 1632+#define MLKEM512_PUBLICKEYBYTES 800+#define MLKEM512_CIPHERTEXTBYTES 768++#define MLKEM768_SECRETKEYBYTES 2400+#define MLKEM768_PUBLICKEYBYTES 1184+#define MLKEM768_CIPHERTEXTBYTES 1088++#define MLKEM1024_SECRETKEYBYTES 3168+#define MLKEM1024_PUBLICKEYBYTES 1568+#define MLKEM1024_CIPHERTEXTBYTES 1568+/* check-magic: on */++/* Size of randomness coins in bytes (level-independent) */+#define MLKEM_SYMBYTES 32+#define MLKEM512_SYMBYTES MLKEM_SYMBYTES+#define MLKEM768_SYMBYTES MLKEM_SYMBYTES+#define MLKEM1024_SYMBYTES MLKEM_SYMBYTES+/* Size of shared secret in bytes (level-independent) */+#define MLKEM_BYTES 32+#define MLKEM512_BYTES MLKEM_BYTES+#define MLKEM768_BYTES MLKEM_BYTES+#define MLKEM1024_BYTES MLKEM_BYTES++/* Sizes of cryptographic material, as a function of LVL=512,768,1024 */+#define MLKEM_SECRETKEYBYTES_(LVL) MLKEM##LVL##_SECRETKEYBYTES+#define MLKEM_PUBLICKEYBYTES_(LVL) MLKEM##LVL##_PUBLICKEYBYTES+#define MLKEM_CIPHERTEXTBYTES_(LVL) MLKEM##LVL##_CIPHERTEXTBYTES+#define MLKEM_SECRETKEYBYTES(LVL) MLKEM_SECRETKEYBYTES_(LVL)+#define MLKEM_PUBLICKEYBYTES(LVL) MLKEM_PUBLICKEYBYTES_(LVL)+#define MLKEM_CIPHERTEXTBYTES(LVL) MLKEM_CIPHERTEXTBYTES_(LVL)++/****************************** Error codes ***********************************/++/* Generic failure condition. Currently not returned by any function;+ * reserved for failures that no more specific code covers. */+#define MLK_ERR_FAIL (-1)+/* An allocation failed. This can only happen if MLK_CONFIG_CUSTOM_ALLOC_FREE+ * is defined and the provided MLK_CUSTOM_ALLOC can fail. */+#define MLK_ERR_OUT_OF_MEMORY (-2)+/* An RNG failure occurred. Might be due to insufficient entropy or+ * system misconfiguration. */+#define MLK_ERR_RNG_FAIL (-3)+/* Public key validation failed: the @[FIPS203, Section 7.2, 'modulus check']+ * found a coefficient outside [0,q-1]. Returned by check_pk and by the+ * encapsulation API. */+#define MLK_ERR_INVALID_PK (-4)+/* Secret key validation failed: the @[FIPS203, Section 7.3, 'hash check']+ * found the embedded public key hash inconsistent. Returned by check_sk and+ * by the decapsulation API. */+#define MLK_ERR_INVALID_SK (-5)+/* The 'Pairwise Consistency Test' @[FIPS140_3_IG, p.87] and+ * @[FIPS203, Section 7.1, Pairwise Consistency] failed. Only possible when+ * MLK_CONFIG_KEYGEN_PCT is enabled; signals that the freshly generated key+ * pair failed its encaps/decaps self-test. */+#define MLK_ERR_PCT_FAIL (-6)++/********************* Namespacing and Qualifiers *****************************/++#define MLK_API_CONCAT_(x, y) x##y+#define MLK_API_CONCAT(x, y) MLK_API_CONCAT_(x, y)+#define MLK_API_CONCAT_UNDERSCORE(x, y) MLK_API_CONCAT(MLK_API_CONCAT(x, _), y)++/* You need to make sure the config file is in the include path. */+#if defined(MLK_CONFIG_FILE)+#include MLK_CONFIG_FILE+#else+#include "mlkem_native_config.h"+#endif++/* Namespace prefix for the public API symbols. For multi-level builds, the+ * parameter set is appended to disambiguate the security levels. */+#if defined(MLK_CONFIG_MULTILEVEL_BUILD)+#define MLK_API_NAMESPACE_PREFIX \+ MLK_API_CONCAT(MLK_CONFIG_NAMESPACE_PREFIX, MLK_CONFIG_PARAMETER_SET)+#else+#define MLK_API_NAMESPACE_PREFIX MLK_CONFIG_NAMESPACE_PREFIX+#endif++#define MLK_API_NAMESPACE(sym) \+ MLK_API_CONCAT_UNDERSCORE(MLK_API_NAMESPACE_PREFIX, sym)++#if defined(__GNUC__) || defined(__clang__)+#define MLK_API_MUST_CHECK_RETURN_VALUE __attribute__((warn_unused_result))+#else+#define MLK_API_MUST_CHECK_RETURN_VALUE+#endif++#if defined(MLK_CONFIG_EXTERNAL_API_QUALIFIER)+#define MLK_API_QUALIFIER MLK_CONFIG_EXTERNAL_API_QUALIFIER+#else+#define MLK_API_QUALIFIER+#endif++/****************************** Function API **********************************/++#if !defined(MLK_CONFIG_CONSTANTS_ONLY)++#include <stdint.h>++#ifdef __cplusplus+extern "C"+{+#endif++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+/**+ * Generate a public/private keypair for the ML-KEM key encapsulation mechanism.+ *+ * @spec{Implements @[FIPS203, Algorithm 16, ML-KEM.KeyGen_Internal].}+ *+ * @param[out] pk Output public key, an array of+ * MLKEM{512,768,1024}_PUBLICKEYBYTES bytes.+ * @param[out] sk Output private key, an array of+ * MLKEM{512,768,1024}_SECRETKEYBYTES bytes.+ * @param[in] coins Input randomness, an array of 2*MLKEM_SYMBYTES uniformly+ * random bytes.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_PCT_FAIL MLK_CONFIG_KEYGEN_PCT enabled and PCT failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ * @retval MLK_ERR_RNG_FAIL MLK_CONFIG_KEYGEN_PCT enabled and random+ * number generation failed within the PCT.+ */+MLK_API_QUALIFIER+MLK_API_MUST_CHECK_RETURN_VALUE+int MLK_API_NAMESPACE(keypair_derand)(+ uint8_t pk[MLKEM_PUBLICKEYBYTES(MLK_CONFIG_PARAMETER_SET)],+ uint8_t sk[MLKEM_SECRETKEYBYTES(MLK_CONFIG_PARAMETER_SET)],+ const uint8_t coins[2 * MLKEM_SYMBYTES]+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+ ,+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+++#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+/**+ * Generate a public/private keypair for the ML-KEM key encapsulation mechanism.+ *+ * @spec{Implements @[FIPS203, Algorithm 19, ML-KEM.KeyGen].}+ *+ * @param[out] pk Output public key, an array of+ * MLKEM{512,768,1024}_PUBLICKEYBYTES bytes.+ * @param[out] sk Output private key, an array of+ * MLKEM{512,768,1024}_SECRETKEYBYTES bytes.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_PCT_FAIL MLK_CONFIG_KEYGEN_PCT enabled and PCT failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ * @retval MLK_ERR_RNG_FAIL Random number generation failed.+ */+MLK_API_QUALIFIER+MLK_API_MUST_CHECK_RETURN_VALUE+int MLK_API_NAMESPACE(keypair)(+ uint8_t pk[MLKEM_PUBLICKEYBYTES(MLK_CONFIG_PARAMETER_SET)],+ uint8_t sk[MLKEM_SECRETKEYBYTES(MLK_CONFIG_PARAMETER_SET)]+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+ ,+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLK_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API)+/**+ * Generate ciphertext and shared secret for a given public key.+ *+ * @spec{Implements @[FIPS203, Algorithm 17, ML-KEM.Encaps_Internal].}+ *+ * @param[out] ct Output ciphertext, an array of+ * MLKEM{512,768,1024}_CIPHERTEXTBYTES bytes.+ * @param[out] ss Output shared secret, an array of MLKEM_BYTES bytes.+ * @param[in] pk Input public key, an array of+ * MLKEM{512,768,1024}_PUBLICKEYBYTES bytes.+ * @param[in] coins Input randomness, an array of MLKEM_SYMBYTES bytes.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_PK The 'modulus check' @[FIPS203, Section 7.2]+ * for the public key failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+MLK_API_QUALIFIER+MLK_API_MUST_CHECK_RETURN_VALUE+int MLK_API_NAMESPACE(enc_derand)(+ uint8_t ct[MLKEM_CIPHERTEXTBYTES(MLK_CONFIG_PARAMETER_SET)],+ uint8_t ss[MLKEM_BYTES],+ const uint8_t pk[MLKEM_PUBLICKEYBYTES(MLK_CONFIG_PARAMETER_SET)],+ const uint8_t coins[MLKEM_SYMBYTES]+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+ ,+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+/**+ * Generate ciphertext and shared secret for a given public key.+ *+ * @spec{Implements @[FIPS203, Algorithm 20, ML-KEM.Encaps].}+ *+ * @param[out] ct Output ciphertext, an array of+ * MLKEM{512,768,1024}_CIPHERTEXTBYTES bytes.+ * @param[out] ss Output shared secret, an array of MLKEM_BYTES bytes.+ * @param[in] pk Input public key, an array of+ * MLKEM{512,768,1024}_PUBLICKEYBYTES bytes.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_PK The 'modulus check' @[FIPS203, Section 7.2]+ * for the public key failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ * @retval MLK_ERR_RNG_FAIL Random number generation failed.+ */+MLK_API_QUALIFIER+MLK_API_MUST_CHECK_RETURN_VALUE+int MLK_API_NAMESPACE(enc)(+ uint8_t ct[MLKEM_CIPHERTEXTBYTES(MLK_CONFIG_PARAMETER_SET)],+ uint8_t ss[MLKEM_BYTES],+ const uint8_t pk[MLKEM_PUBLICKEYBYTES(MLK_CONFIG_PARAMETER_SET)]+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+ ,+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLK_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLK_CONFIG_NO_ENCAPS_API */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/**+ * Generate shared secret for a given ciphertext and private key.+ *+ * @spec{Implements @[FIPS203, Algorithm 21, ML-KEM.Decaps].}+ *+ * @param[out] ss Output shared secret, an array of MLKEM_BYTES bytes.+ * @param[in] ct Input ciphertext, an array of+ * MLKEM{512,768,1024}_CIPHERTEXTBYTES bytes.+ * @param[in] sk Input private key, an array of+ * MLKEM{512,768,1024}_SECRETKEYBYTES bytes.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_SK The 'hash check' @[FIPS203, Section 7.3]+ * for the secret key failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+MLK_API_QUALIFIER+MLK_API_MUST_CHECK_RETURN_VALUE+int MLK_API_NAMESPACE(dec)(+ uint8_t ss[MLKEM_BYTES],+ const uint8_t ct[MLKEM_CIPHERTEXTBYTES(MLK_CONFIG_PARAMETER_SET)],+ const uint8_t sk[MLKEM_SECRETKEYBYTES(MLK_CONFIG_PARAMETER_SET)]+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+ ,+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLK_CONFIG_NO_DECAPS_API */+++/**+ * Implements modulus check mandated by FIPS 203, i.e., ensures that+ * coefficients are in [0,q-1].+ *+ * @spec{Implements @[FIPS203, Section 7.2, 'modulus check'].}+ *+ * @param[in] pk Input public key, an array of+ * MLKEM{512,768,1024}_PUBLICKEYBYTES bytes.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_PK Modulus check failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+#if !defined(MLK_CONFIG_NO_ENCAPS_API)+MLK_API_QUALIFIER+MLK_API_MUST_CHECK_RETURN_VALUE+int MLK_API_NAMESPACE(check_pk)(+ const uint8_t pk[MLKEM_PUBLICKEYBYTES(MLK_CONFIG_PARAMETER_SET)]+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+ ,+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLK_CONFIG_NO_ENCAPS_API */++/**+ * Implements public key hash check mandated by FIPS 203, i.e., ensures that+ * sk[768𝑘+32 ∶ 768𝑘+64] = H(pk) = H(sk[384𝑘 : 768𝑘+32]).+ *+ * @spec{Implements @[FIPS203, Section 7.3, 'hash check'].}+ *+ * @param[in] sk Input private key, an array of+ * MLKEM{512,768,1024}_SECRETKEYBYTES bytes.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_SK Public key hash check failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_API_QUALIFIER+MLK_API_MUST_CHECK_RETURN_VALUE+int MLK_API_NAMESPACE(check_sk)(+ const uint8_t sk[MLKEM_SECRETKEYBYTES(MLK_CONFIG_PARAMETER_SET)]+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+ ,+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#ifdef __cplusplus+}+#endif++#undef MLK_API_NAMESPACE_PREFIX++#endif /* !MLK_CONFIG_CONSTANTS_ONLY */+++/***************************** Memory Usage **********************************/++/*+ * By default mlkem-native performs all memory allocations on the stack.+ * Alternatively, mlkem-native supports custom allocation of large structures+ * through the `MLK_CONFIG_CUSTOM_ALLOC_FREE` configuration option.+ * See mlkem_native_config.h for details.+ *+ * `MLK_TOTAL_ALLOC_{512,768,1024}_{KEYPAIR,ENCAPS,DECAPS}` indicates the+ * maximum (accumulative) allocation via MLK_ALLOC for each parameter set and+ * operation. Note that some stack allocation remains even when using custom+ * allocators, so these values are lower than total stack usage with the default+ * stack-only allocation.+ *+ * These constants may be used to implement custom allocations using a+ * fixed-sized buffer and a simple allocator (e.g., bump allocator).+ */+/* check-magic: off */+#define MLK_TOTAL_ALLOC_512_KEYPAIR_NO_PCT 5824+#define MLK_TOTAL_ALLOC_512_KEYPAIR_PCT 10048+#define MLK_TOTAL_ALLOC_512_ENCAPS 8384+#define MLK_TOTAL_ALLOC_512_DECAPS 9152+#define MLK_TOTAL_ALLOC_768_KEYPAIR_NO_PCT 10176+#define MLK_TOTAL_ALLOC_768_KEYPAIR_PCT 15552+#define MLK_TOTAL_ALLOC_768_ENCAPS 13248+#define MLK_TOTAL_ALLOC_768_DECAPS 14336+#define MLK_TOTAL_ALLOC_1024_KEYPAIR_NO_PCT 15552+#define MLK_TOTAL_ALLOC_1024_KEYPAIR_PCT 22400+#define MLK_TOTAL_ALLOC_1024_ENCAPS 19136+#define MLK_TOTAL_ALLOC_1024_DECAPS 20704+/* check-magic: on */++/*+ * MLK_TOTAL_ALLOC_*_KEYPAIR adapts based on MLK_CONFIG_KEYGEN_PCT.+ */+#if defined(MLK_CONFIG_KEYGEN_PCT)+#define MLK_TOTAL_ALLOC_512_KEYPAIR MLK_TOTAL_ALLOC_512_KEYPAIR_PCT+#define MLK_TOTAL_ALLOC_768_KEYPAIR MLK_TOTAL_ALLOC_768_KEYPAIR_PCT+#define MLK_TOTAL_ALLOC_1024_KEYPAIR MLK_TOTAL_ALLOC_1024_KEYPAIR_PCT+#else+#define MLK_TOTAL_ALLOC_512_KEYPAIR MLK_TOTAL_ALLOC_512_KEYPAIR_NO_PCT+#define MLK_TOTAL_ALLOC_768_KEYPAIR MLK_TOTAL_ALLOC_768_KEYPAIR_NO_PCT+#define MLK_TOTAL_ALLOC_1024_KEYPAIR MLK_TOTAL_ALLOC_1024_KEYPAIR_NO_PCT+#endif++#define MLK_MAX3_(a, b, c) \+ ((a) > (b) ? ((a) > (c) ? (a) : (c)) : ((b) > (c) ? (b) : (c)))++/*+ * `MLK_TOTAL_ALLOC_{512,768,1024}` is the maximum across all operations for+ * each parameter set.+ */+#define MLK_TOTAL_ALLOC_512 \+ MLK_MAX3_(MLK_TOTAL_ALLOC_512_KEYPAIR, MLK_TOTAL_ALLOC_512_ENCAPS, \+ MLK_TOTAL_ALLOC_512_DECAPS)+#define MLK_TOTAL_ALLOC_768 \+ MLK_MAX3_(MLK_TOTAL_ALLOC_768_KEYPAIR, MLK_TOTAL_ALLOC_768_ENCAPS, \+ MLK_TOTAL_ALLOC_768_DECAPS)+#define MLK_TOTAL_ALLOC_1024 \+ MLK_MAX3_(MLK_TOTAL_ALLOC_1024_KEYPAIR, MLK_TOTAL_ALLOC_1024_ENCAPS, \+ MLK_TOTAL_ALLOC_1024_DECAPS)++#endif /* !MLK_H */
+ cbits/mlkem/mlkem_native_asm.S view
@@ -0,0 +1,716 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++/******************************************************************************+ *+ * Single assembly unit for fixed-level build of mlkem-native+ *+ * This assembly unit bundles together all assembly files for a build+ * of mlkem-native for a fixed security level (MLKEM-512/768/1024).+ *+ * # Multi-level build+ *+ * If you want an SCU build of mlkem-native with support for multiple security+ * levels, you should include this file once with+ * MLK_CONFIG_MULTILEVEL_WITH_SHARED set.+ *+ * (You could also follow the same pattern as for mlkem_native.c+ * and include it for every level, setting MLK_CONFIG_MULTILEVEL_NO_SHARED+ * for all but one. For builds with MLK_CONFIG_MULTILEVEL_NO_SHARED, this+ * file will then be ignored.)+ *+ * # Configuration+ *+ * The following options from the mlkem-native configuration are relevant:+ *+ * - MLK_CONFIG_FIPS202_CUSTOM_HEADER+ * Set this option if you use a custom FIPS202 implementation.+ *+ * - MLK_CONFIG_USE_NATIVE_BACKEND_ARITH+ * Set this option if you want to include the native arithmetic backends+ * in your build.+ *+ * - MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202+ * Set this option if you want to include the native FIPS202 backends+ * in your build.+ *+ * - MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+ * Set this option if you want to keep the directives defined in+ * level-independent headers. This is needed for a multi-level build.+ */++/* If parts of the mlkem-native source tree are not used,+ * consider reducing this header via `unifdef`.+ *+ * Example:+ * ```bash+ * unifdef -UMLK_CONFIG_USE_NATIVE_BACKEND_ARITH mlkem_native_asm.S+ * ```+ */++#include "src/common.h"++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH)+#if defined(MLK_SYS_AARCH64)+#include "src/native/aarch64/src/mlkem_intt_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_ntt_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_poly_mulcache_compute_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_poly_reduce_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_poly_tobytes_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_poly_tomont_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_rej_uniform_aarch64_asm.S"+#endif /* MLK_SYS_AARCH64 */+#if defined(MLK_SYS_X86_64)+#include "src/native/x86_64/src/mlkem_intt_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_ntt_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_nttfrombytes_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_ntttobytes_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_nttunpack_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_compress_d10_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_compress_d11_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_compress_d4_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_compress_d5_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_decompress_d10_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_decompress_d11_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_decompress_d4_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_decompress_d5_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_mulcache_compute_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_reduce_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_rej_uniform_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_tomont_avx2_asm.S"+#endif /* MLK_SYS_X86_64 */+#if defined(MLK_SYS_RISCV64)+#endif+#if defined(MLK_SYS_PPC64LE)+#include "src/native/ppc64le/src/mlkem_intt_ppc_asm.S"+#include "src/native/ppc64le/src/mlkem_ntt_ppc_asm.S"+#include "src/native/ppc64le/src/mlkem_poly_tomont_ppc_asm.S"+#include "src/native/ppc64le/src/mlkem_reduce_ppc_asm.S"+#endif /* MLK_SYS_PPC64LE */+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_ARITH */++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202)+#if defined(MLK_SYS_AARCH64)+#include "src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S"+#endif /* MLK_SYS_AARCH64 */+#if defined(MLK_SYS_X86_64)+#include "src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S"+#endif+#if defined(MLK_SYS_ARMV81M_MVE)+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S"+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_state_extract_bytes_mve.S"+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_state_xor_bytes_mve.S"+#endif+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+++/* Macro #undef's+ *+ * The following undefines macros from headers+ * included by the source files imported above.+ *+ * This is to allow building and linking multiple builds+ * of mlkem-native for varying parameter sets through concatenation+ * of this file, as if the files had been compiled separately.+ * If this is not relevant to you, you may remove the following.+ *+ * NOTE: This is not needed for the assembly SCU since, at present,+ * there is no need to include it multiple times.+ * We keep it for uniformity with mlkem_native.c only.+ *+ * NOTE: To avoid having to distinguish between which headers are included+ * from the assembly files, we #undef the same set of directives+ * as in mlkem_native.c+ */++/*+ * Undefine macros from MLK_CONFIG_PARAMETER_SET-specific files+ */+/* mlkem/mlkem_native.h */+#undef MLKEM1024_BYTES+#undef MLKEM1024_CIPHERTEXTBYTES+#undef MLKEM1024_PUBLICKEYBYTES+#undef MLKEM1024_SECRETKEYBYTES+#undef MLKEM1024_SYMBYTES+#undef MLKEM512_BYTES+#undef MLKEM512_CIPHERTEXTBYTES+#undef MLKEM512_PUBLICKEYBYTES+#undef MLKEM512_SECRETKEYBYTES+#undef MLKEM512_SYMBYTES+#undef MLKEM768_BYTES+#undef MLKEM768_CIPHERTEXTBYTES+#undef MLKEM768_PUBLICKEYBYTES+#undef MLKEM768_SECRETKEYBYTES+#undef MLKEM768_SYMBYTES+#undef MLKEM_BYTES+#undef MLKEM_CIPHERTEXTBYTES+#undef MLKEM_CIPHERTEXTBYTES_+#undef MLKEM_PUBLICKEYBYTES+#undef MLKEM_PUBLICKEYBYTES_+#undef MLKEM_SECRETKEYBYTES+#undef MLKEM_SECRETKEYBYTES_+#undef MLKEM_SYMBYTES+#undef MLK_API_CONCAT+#undef MLK_API_CONCAT_+#undef MLK_API_CONCAT_UNDERSCORE+#undef MLK_API_MUST_CHECK_RETURN_VALUE+#undef MLK_API_NAMESPACE+#undef MLK_API_NAMESPACE_PREFIX+#undef MLK_API_QUALIFIER+#undef MLK_ERR_FAIL+#undef MLK_ERR_INVALID_PK+#undef MLK_ERR_INVALID_SK+#undef MLK_ERR_OUT_OF_MEMORY+#undef MLK_ERR_PCT_FAIL+#undef MLK_ERR_RNG_FAIL+#undef MLK_H+#undef MLK_MAX3_+#undef MLK_TOTAL_ALLOC_1024+#undef MLK_TOTAL_ALLOC_1024_DECAPS+#undef MLK_TOTAL_ALLOC_1024_ENCAPS+#undef MLK_TOTAL_ALLOC_1024_KEYPAIR+#undef MLK_TOTAL_ALLOC_1024_KEYPAIR_NO_PCT+#undef MLK_TOTAL_ALLOC_1024_KEYPAIR_PCT+#undef MLK_TOTAL_ALLOC_512+#undef MLK_TOTAL_ALLOC_512_DECAPS+#undef MLK_TOTAL_ALLOC_512_ENCAPS+#undef MLK_TOTAL_ALLOC_512_KEYPAIR+#undef MLK_TOTAL_ALLOC_512_KEYPAIR_NO_PCT+#undef MLK_TOTAL_ALLOC_512_KEYPAIR_PCT+#undef MLK_TOTAL_ALLOC_768+#undef MLK_TOTAL_ALLOC_768_DECAPS+#undef MLK_TOTAL_ALLOC_768_ENCAPS+#undef MLK_TOTAL_ALLOC_768_KEYPAIR+#undef MLK_TOTAL_ALLOC_768_KEYPAIR_NO_PCT+#undef MLK_TOTAL_ALLOC_768_KEYPAIR_PCT+/* mlkem/src/common.h */+#undef MLK_ADD_PARAM_SET+#undef MLK_ALLOC+#undef MLK_APPLY+#undef MLK_ASM_FN_SIZE+#undef MLK_ASM_FN_SYMBOL+#undef MLK_ASM_NAMESPACE+#undef MLK_BUILD_INTERNAL+#undef MLK_COMMON_H+#undef MLK_CONCAT+#undef MLK_CONCAT_+#undef MLK_EMPTY_CU+#undef MLK_ERR_FAIL+#undef MLK_ERR_INVALID_PK+#undef MLK_ERR_INVALID_SK+#undef MLK_ERR_OUT_OF_MEMORY+#undef MLK_ERR_PCT_FAIL+#undef MLK_ERR_RNG_FAIL+#undef MLK_EXTERNAL_API+#undef MLK_FIPS202X4_HEADER_FILE+#undef MLK_FIPS202_HEADER_FILE+#undef MLK_FREE+#undef MLK_INTERNAL_API+#undef MLK_INTERNAL_DATA_DECLARATION+#undef MLK_INTERNAL_DATA_DEFINITION+#undef MLK_NAMESPACE+#undef MLK_NAMESPACE_K+#undef MLK_NAMESPACE_PREFIX+#undef MLK_NAMESPACE_PREFIX_K+#undef mlk_memcpy+#undef mlk_memset+/* mlkem/src/indcpa.h */+#undef MLK_INDCPA_H+#undef mlk_gen_matrix+#undef mlk_indcpa_dec+#undef mlk_indcpa_enc+#undef mlk_indcpa_keypair_derand+/* mlkem/src/kem.h */+#undef MLK_KEM_H+#undef mlk_kem_check_pk+#undef mlk_kem_check_sk+#undef mlk_kem_dec+#undef mlk_kem_enc+#undef mlk_kem_enc_derand+#undef mlk_kem_keypair+#undef mlk_kem_keypair_derand+/* mlkem/src/params.h */+#undef MLKEM_DU+#undef MLKEM_DV+#undef MLKEM_ETA1+#undef MLKEM_ETA2+#undef MLKEM_INDCCA_CIPHERTEXTBYTES+#undef MLKEM_INDCCA_PUBLICKEYBYTES+#undef MLKEM_INDCCA_SECRETKEYBYTES+#undef MLKEM_INDCPA_BYTES+#undef MLKEM_INDCPA_MSGBYTES+#undef MLKEM_INDCPA_PUBLICKEYBYTES+#undef MLKEM_INDCPA_SECRETKEYBYTES+#undef MLKEM_K+#undef MLKEM_N+#undef MLKEM_POLYBYTES+#undef MLKEM_POLYCOMPRESSEDBYTES_D10+#undef MLKEM_POLYCOMPRESSEDBYTES_D11+#undef MLKEM_POLYCOMPRESSEDBYTES_D4+#undef MLKEM_POLYCOMPRESSEDBYTES_D5+#undef MLKEM_POLYCOMPRESSEDBYTES_DU+#undef MLKEM_POLYCOMPRESSEDBYTES_DV+#undef MLKEM_POLYVECBYTES+#undef MLKEM_POLYVECCOMPRESSEDBYTES_DU+#undef MLKEM_Q+#undef MLKEM_Q_HALF+#undef MLKEM_SSBYTES+#undef MLKEM_SYMBYTES+#undef MLKEM_UINT12_LIMIT+#undef MLK_PARAMS_H+/* mlkem/src/poly_k.h */+#undef MLK_POLY_K_H+#undef mlk_poly_compress_du+#undef mlk_poly_compress_dv+#undef mlk_poly_decompress_du+#undef mlk_poly_decompress_dv+#undef mlk_poly_getnoise_eta1122_4x+#undef mlk_poly_getnoise_eta1_4x+#undef mlk_poly_getnoise_eta2+#undef mlk_poly_getnoise_eta2_4x+#undef mlk_polymat+#undef mlk_polyvec+#undef mlk_polyvec_add+#undef mlk_polyvec_basemul_acc_montgomery_cached+#undef mlk_polyvec_compress_du+#undef mlk_polyvec_decompress_du+#undef mlk_polyvec_frombytes+#undef mlk_polyvec_invntt_tomont+#undef mlk_polyvec_mulcache+#undef mlk_polyvec_mulcache_compute+#undef mlk_polyvec_ntt+#undef mlk_polyvec_reduce+#undef mlk_polyvec_tobytes+#undef mlk_polyvec_tomont++#if !defined(MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS)+/*+ * Undefine macros from MLK_CONFIG_PARAMETER_SET-generic files+ */+/* mlkem/src/compress.h */+#undef MLK_COMPRESS_H+#undef mlk_poly_compress_d10+#undef mlk_poly_compress_d11+#undef mlk_poly_compress_d4+#undef mlk_poly_compress_d5+#undef mlk_poly_decompress_d10+#undef mlk_poly_decompress_d11+#undef mlk_poly_decompress_d4+#undef mlk_poly_decompress_d5+#undef mlk_poly_frombytes+#undef mlk_poly_frommsg+#undef mlk_poly_tobytes+#undef mlk_poly_tomsg+/* mlkem/src/context.h */+#undef MLK_CONTEXT_H+#undef MLK_CONTEXT_PARAMETERS_0+#undef MLK_CONTEXT_PARAMETERS_1+#undef MLK_CONTEXT_PARAMETERS_2+#undef MLK_CONTEXT_PARAMETERS_3+#undef MLK_CONTEXT_PARAMETERS_4+#undef MLK_CONTEXT_UNUSED+/* mlkem/src/debug.h */+#undef MLK_DEBUG_H+#undef mlk_assert+#undef mlk_assert_abs_bound+#undef mlk_assert_abs_bound_2d+#undef mlk_assert_bound+#undef mlk_assert_bound_2d+#undef mlk_debug_check_assert+#undef mlk_debug_check_bounds+/* mlkem/src/poly.h */+#undef MLK_INVNTT_BOUND+#undef MLK_NTT_BOUND+#undef MLK_POLY_H+#undef mlk_poly_add+#undef mlk_poly_invntt_tomont+#undef mlk_poly_mulcache_compute+#undef mlk_poly_ntt+#undef mlk_poly_reduce+#undef mlk_poly_sub+#undef mlk_poly_tomont+/* mlkem/src/randombytes.h */+#undef MLK_RANDOMBYTES_H+/* mlkem/src/sampling.h */+#undef MLK_SAMPLING_H+#undef mlk_poly_cbd2+#undef mlk_poly_cbd3+#undef mlk_poly_rej_uniform+#undef mlk_poly_rej_uniform_x4+/* mlkem/src/symmetric.h */+#undef MLK_SYMMETRIC_H+#undef MLK_XOF_RATE+#undef mlk_hash_g+#undef mlk_hash_h+#undef mlk_hash_j+#undef mlk_prf_eta+#undef mlk_prf_eta1+#undef mlk_prf_eta1_x4+#undef mlk_prf_eta2+#undef mlk_xof_absorb+#undef mlk_xof_ctx+#undef mlk_xof_init+#undef mlk_xof_release+#undef mlk_xof_squeezeblocks+#undef mlk_xof_x4_absorb+#undef mlk_xof_x4_ctx+#undef mlk_xof_x4_init+#undef mlk_xof_x4_release+#undef mlk_xof_x4_squeezeblocks+/* mlkem/src/sys.h */+#undef MLK_ALIGN+#undef MLK_ALIGN_UP+#undef MLK_ALWAYS_INLINE+#undef MLK_CET_ENDBR+#undef MLK_CT_TESTING_DECLASSIFY+#undef MLK_CT_TESTING_SECRET+#undef MLK_DEFAULT_ALIGN+#undef MLK_HAVE_INLINE_ASM+#undef MLK_INLINE+#undef MLK_MUST_CHECK_RETURN_VALUE+#undef MLK_NOINLINE+#undef MLK_RESTRICT+#undef MLK_STATIC_TESTABLE+#undef MLK_SYSV_ABI+#undef MLK_SYSV_ABI_SUPPORTED+#undef MLK_SYS_AARCH64+#undef MLK_SYS_AARCH64_EB+#undef MLK_SYS_AARCH64_NEON+#undef MLK_SYS_APPLE+#undef MLK_SYS_ARMV81M_MVE+#undef MLK_SYS_BIG_ENDIAN+#undef MLK_SYS_H+#undef MLK_SYS_LINUX+#undef MLK_SYS_LITTLE_ENDIAN+#undef MLK_SYS_PPC64LE+#undef MLK_SYS_RISCV32+#undef MLK_SYS_RISCV64+#undef MLK_SYS_RISCV64_RVV+#undef MLK_SYS_WINDOWS+#undef MLK_SYS_X86_64+#undef MLK_SYS_X86_64_AVX2+/* mlkem/src/verify.h */+#undef MLK_USE_ASM_VALUE_BARRIER+#undef MLK_VERIFY_H+#undef mlk_ct_opt_blocker_u64+/* mlkem/src/cbmc.h */+#undef MLK_CBMC_H+#undef __contract__+#undef __loop__++#if !defined(MLK_CONFIG_FIPS202_CUSTOM_HEADER)+/*+ * Undefine macros from FIPS-202 files+ */+/* mlkem/src/fips202/fips202.h */+#undef FIPS202_X4_DEFAULT_IMPLEMENTATION+#undef MLK_FIPS202_FIPS202_H+#undef SHA3_256_HASHBYTES+#undef SHA3_256_RATE+#undef SHA3_384_RATE+#undef SHA3_512_HASHBYTES+#undef SHA3_512_RATE+#undef SHAKE128_RATE+#undef SHAKE256_RATE+#undef mlk_sha3_256+#undef mlk_sha3_512+#undef mlk_shake128_absorb_once+#undef mlk_shake128_init+#undef mlk_shake128_release+#undef mlk_shake128_squeezeblocks+#undef mlk_shake256+/* mlkem/src/fips202/fips202x4.h */+#undef MLK_FIPS202_FIPS202X4_H+#undef mlk_shake128x4_absorb_once+#undef mlk_shake128x4_init+#undef mlk_shake128x4_release+#undef mlk_shake128x4_squeezeblocks+#undef mlk_shake256x4+/* mlkem/src/fips202/keccakf1600.h */+#undef MLK_FIPS202_KECCAKF1600_H+#undef MLK_KECCAK_LANES+#undef MLK_KECCAK_WAY+#undef mlk_keccakf1600_extract_bytes+#undef mlk_keccakf1600_permute+#undef mlk_keccakf1600_xor_bytes+#undef mlk_keccakf1600x4_extract_bytes+#undef mlk_keccakf1600x4_permute+#undef mlk_keccakf1600x4_xor_bytes+#endif /* !MLK_CONFIG_FIPS202_CUSTOM_HEADER */++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202)+/* mlkem/src/fips202/native/api.h */+#undef MLK_FIPS202_NATIVE_API_H+#undef MLK_NATIVE_FUNC_FALLBACK+#undef MLK_NATIVE_FUNC_SUCCESS+/* mlkem/src/fips202/native/auto.h */+#undef MLK_FIPS202_NATIVE_AUTO_H+#if defined(MLK_SYS_AARCH64)+/*+ * Undefine macros from native code (FIPS202, AArch64)+ */+/* mlkem/src/fips202/native/aarch64/auto.h */+#undef MLK_FIPS202_NATIVE_AARCH64_AUTO_H+/* mlkem/src/fips202/native/aarch64/src/fips202_native_aarch64.h */+#undef MLK_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H+#undef mlk_keccak_f1600_x1_scalar_aarch64_asm+#undef mlk_keccak_f1600_x1_v84a_aarch64_asm+#undef mlk_keccak_f1600_x2_v84a_aarch64_asm+#undef mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm+#undef mlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm+#undef mlk_keccakf1600_round_constants+/* mlkem/src/fips202/native/aarch64/x1_scalar.h */+#undef MLK_FIPS202_AARCH64_NEED_X1_SCALAR+#undef MLK_FIPS202_NATIVE_AARCH64_X1_SCALAR_H+#undef MLK_USE_NATIVE_FIPS202_X1+/* mlkem/src/fips202/native/aarch64/x1_v84a.h */+#undef MLK_FIPS202_AARCH64_NEED_X1_V84A+#undef MLK_FIPS202_NATIVE_AARCH64_X1_V84A_H+#undef MLK_USE_NATIVE_FIPS202_X1+/* mlkem/src/fips202/native/aarch64/x2_v84a.h */+#undef MLK_FIPS202_AARCH64_NEED_X2_V84A+#undef MLK_FIPS202_NATIVE_AARCH64_X2_V84A_H+#undef MLK_USE_NATIVE_FIPS202_X4+/* mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h */+#undef MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID+#undef MLK_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H+#undef MLK_USE_NATIVE_FIPS202_X4+/* mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h */+#undef MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID+#undef MLK_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H+#undef MLK_USE_NATIVE_FIPS202_X4+#endif /* MLK_SYS_AARCH64 */+#if defined(MLK_SYS_X86_64)+/*+ * Undefine macros from native code (FIPS202, x86_64)+ */+/* mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h */+#undef MLK_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H+#undef MLK_FIPS202_X86_64_NEED_X4_AVX2+#undef MLK_USE_NATIVE_FIPS202_X4+/* mlkem/src/fips202/native/x86_64/src/fips202_native_x86_64.h */+#undef MLK_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H+#undef mlk_keccak_f1600_x4_avx2_asm+#undef mlk_keccak_rho56+#undef mlk_keccak_rho8+#undef mlk_keccakf1600_round_constants+#endif /* MLK_SYS_X86_64 */+#if defined(MLK_SYS_ARMV81M_MVE)+/*+ * Undefine macros from native code (FIPS202, Armv8.1-M)+ */+/* mlkem/src/fips202/native/armv81m/mve.h */+#undef MLK_FIPS202_ARMV81M_NEED_X4+#undef MLK_FIPS202_NATIVE_ARMV81M+#undef MLK_FIPS202_NATIVE_ARMV81M_MVE_H+#undef MLK_USE_NATIVE_FIPS202_X4+#undef MLK_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES+#undef MLK_USE_NATIVE_FIPS202_X4_XOR_BYTES+#undef mlk_keccak_f1600_x4_native_impl+/* mlkem/src/fips202/native/armv81m/src/fips202_native_armv81m.h */+#undef MLK_FIPS202_NATIVE_ARMV81M_SRC_FIPS202_NATIVE_ARMV81M_H+#undef mlk_keccak_f1600_x4_mve_asm+#undef mlk_keccak_f1600_x4_state_extract_bytes_asm+#undef mlk_keccak_f1600_x4_state_xor_bytes_asm+#undef mlk_keccakf1600_round_constants+#endif /* MLK_SYS_ARMV81M_MVE */+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH)+/* mlkem/src/native/api.h */+#undef MLK_INVNTT_BOUND+#undef MLK_NATIVE_API_H+#undef MLK_NATIVE_FUNC_FALLBACK+#undef MLK_NATIVE_FUNC_SUCCESS+#undef MLK_NTT_BOUND+/* mlkem/src/native/meta.h */+#undef MLK_NATIVE_META_H+#if defined(MLK_SYS_AARCH64)+/*+ * Undefine macros from native code (Arith, AArch64)+ */+/* mlkem/src/native/aarch64/meta.h */+#undef MLK_ARITH_BACKEND_AARCH64+#undef MLK_NATIVE_AARCH64_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#undef MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOBYTES+#undef MLK_USE_NATIVE_POLY_TOMONT+#undef MLK_USE_NATIVE_REJ_UNIFORM+/* mlkem/src/native/aarch64/src/arith_native_aarch64.h */+#undef MLK_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H+#undef mlk_aarch64_invntt_zetas_layer12345+#undef mlk_aarch64_invntt_zetas_layer67+#undef mlk_aarch64_ntt_zetas_layer12345+#undef mlk_aarch64_ntt_zetas_layer67+#undef mlk_aarch64_zetas_mulcache_native+#undef mlk_aarch64_zetas_mulcache_twisted_native+#undef mlk_intt_aarch64_asm+#undef mlk_ntt_aarch64_asm+#undef mlk_poly_mulcache_compute_aarch64_asm+#undef mlk_poly_reduce_aarch64_asm+#undef mlk_poly_tobytes_aarch64_asm+#undef mlk_poly_tomont_aarch64_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm+#undef mlk_rej_uniform_aarch64_asm+#undef mlk_rej_uniform_table+#endif /* MLK_SYS_AARCH64 */+#if defined(MLK_SYS_X86_64)+/*+ * Undefine macros from native code (Arith, X86_64)+ */+/* mlkem/src/native/x86_64/meta.h */+#undef MLK_ARITH_BACKEND_X86_64_DEFAULT+#undef MLK_NATIVE_X86_64_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_NTT_CUSTOM_ORDER+#undef MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#undef MLK_USE_NATIVE_POLY_COMPRESS_D10+#undef MLK_USE_NATIVE_POLY_COMPRESS_D11+#undef MLK_USE_NATIVE_POLY_COMPRESS_D4+#undef MLK_USE_NATIVE_POLY_COMPRESS_D5+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D10+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D11+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D4+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D5+#undef MLK_USE_NATIVE_POLY_FROMBYTES+#undef MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOBYTES+#undef MLK_USE_NATIVE_POLY_TOMONT+#undef MLK_USE_NATIVE_REJ_UNIFORM+/* mlkem/src/native/x86_64/src/arith_native_x86_64.h */+#undef MLK_AVX2_REJ_UNIFORM_BUFLEN+#undef MLK_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H+#undef mlk_invntt_avx2_asm+#undef mlk_ntt_avx2_asm+#undef mlk_nttfrombytes_avx2_asm+#undef mlk_ntttobytes_avx2_asm+#undef mlk_nttunpack_avx2_asm+#undef mlk_poly_compress_d10_avx2_asm+#undef mlk_poly_compress_d11_avx2_asm+#undef mlk_poly_compress_d4_avx2_asm+#undef mlk_poly_compress_d5_avx2_asm+#undef mlk_poly_decompress_d10_avx2_asm+#undef mlk_poly_decompress_d11_avx2_asm+#undef mlk_poly_decompress_d4_avx2_asm+#undef mlk_poly_decompress_d5_avx2_asm+#undef mlk_poly_mulcache_compute_avx2_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm+#undef mlk_reduce_avx2_asm+#undef mlk_rej_uniform_avx2_asm+#undef mlk_rej_uniform_table+#undef mlk_tomont_avx2_asm+/* mlkem/src/native/x86_64/src/compress_consts.h */+#undef MLK_NATIVE_X86_64_SRC_COMPRESS_CONSTS_H+#undef mlk_compress_d10_data+#undef mlk_compress_d11_data+#undef mlk_compress_d4_data+#undef mlk_compress_d5_data+#undef mlk_decompress_d10_data+#undef mlk_decompress_d11_data+#undef mlk_decompress_d4_data+#undef mlk_decompress_d5_data+/* mlkem/src/native/x86_64/src/consts.h */+#undef MLK_AVX2_BACKEND_DATA_OFFSET_MULCACHE_TWIDDLES+#undef MLK_AVX2_BACKEND_DATA_OFFSET_REVIDXB+#undef MLK_AVX2_BACKEND_DATA_OFFSET_REVIDXD+#undef MLK_AVX2_BACKEND_DATA_OFFSET_ZETAS_EXP+#undef MLK_NATIVE_X86_64_SRC_CONSTS_H+#undef mlk_qdata+#endif /* MLK_SYS_X86_64 */+#if defined(MLK_SYS_RISCV64)+/*+ * Undefine macros from native code (Arith, RISC-V 64)+ */+/* mlkem/src/native/riscv64/meta.h */+#undef MLK_ARITH_BACKEND_RISCV64+#undef MLK_NATIVE_RISCV64_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#undef MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOMONT+#undef MLK_USE_NATIVE_REJ_UNIFORM+/* mlkem/src/native/riscv64/src/arith_native_riscv64.h */+#undef MLK_NATIVE_RISCV64_SRC_ARITH_NATIVE_RISCV64_H+#undef mlk_rv64v_poly_add+#undef mlk_rv64v_poly_basemul_mont_add_k2+#undef mlk_rv64v_poly_basemul_mont_add_k3+#undef mlk_rv64v_poly_basemul_mont_add_k4+#undef mlk_rv64v_poly_invntt_tomont+#undef mlk_rv64v_poly_ntt+#undef mlk_rv64v_poly_reduce+#undef mlk_rv64v_poly_sub+#undef mlk_rv64v_poly_tomont+#undef mlk_rv64v_rej_uniform+/* mlkem/src/native/riscv64/src/rv64v_debug.h */+#undef MLK_NATIVE_RISCV64_SRC_RV64V_DEBUG_H+#undef mlk_assert_abs_bound_int16m1+#undef mlk_assert_abs_bound_int16m2+#undef mlk_assert_bound_int16m1+#undef mlk_assert_bound_int16m2+#undef mlk_debug_check_bounds_int16m1+#undef mlk_debug_check_bounds_int16m2+#endif /* MLK_SYS_RISCV64 */+#if defined(MLK_SYS_PPC64LE)+/*+ * Undefine macros from native code (Arith, PPC64LE)+ */+/* mlkem/src/native/ppc64le/meta.h */+#undef MLK_ARITH_BACKEND_NAME+#undef MLK_ARITH_BACKEND_PPC64LE_DEFAULT+#undef MLK_NATIVE_PPC64LE_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOMONT+/* mlkem/src/native/ppc64le/src/arith_native_ppc64le.h */+#undef MLK_NATIVE_PPC64LE_SRC_ARITH_NATIVE_PPC64LE_H+#undef mlk_intt_ppc_asm+#undef mlk_ntt_ppc_asm+#undef mlk_poly_tomont_ppc_asm+#undef mlk_reduce_ppc_asm+/* mlkem/src/native/ppc64le/src/consts.h */+#undef MLK_NATIVE_PPC64LE_SRC_CONSTS_H+#undef MLK_PPC_C20159_OFFSET+#undef MLK_PPC_NQ_OFFSET+#undef MLK_PPC_N_INV_OFFSET+#undef MLK_PPC_N_INV_TW_OFFSET+#undef MLK_PPC_Q_OFFSET+#undef MLK_PPC_TOMONT_OFFSET+#undef MLK_PPC_TOMONT_TW_OFFSET+#undef MLK_PPC_ZETA_INTT_OFFSET+#undef MLK_PPC_ZETA_INTT_TW_OFFSET+#undef MLK_PPC_ZETA_NTT_OFFSET+#undef MLK_PPC_ZETA_NTT_TW_OFFSET+#undef mlk_ppc_qdata+#endif /* MLK_SYS_PPC64LE */+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_ARITH */+#endif /* !MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS */
+ cbits/mlkem/mlkem_native_config.h view
@@ -0,0 +1,683 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS140_3_IG]+ * Implementation Guidance for FIPS 140-3 and the Cryptographic Module+ * Validation Program+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/projects/cryptographic-module-validation-program/fips-140-3-ig-announcements+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#ifndef MLK_CONFIG_H+#define MLK_CONFIG_H++/**+ * Specifies the parameter set for ML-KEM:+ * - MLK_CONFIG_PARAMETER_SET=512 corresponds to ML-KEM-512+ * - MLK_CONFIG_PARAMETER_SET=768 corresponds to ML-KEM-768+ * - MLK_CONFIG_PARAMETER_SET=1024 corresponds to ML-KEM-1024+ *+ * If you want to support multiple parameter sets, build the library multiple+ * times and set MLK_CONFIG_MULTILEVEL_BUILD. See MLK_CONFIG_MULTILEVEL_BUILD+ * for how to do this while minimizing code duplication.+ *+ * This can also be set using CFLAGS.+ */+#ifndef MLK_CONFIG_PARAMETER_SET+#define MLK_CONFIG_PARAMETER_SET \+ 768 /* Change this for different security strengths */+#endif++/**+ * MLK_CONFIG_FILE+ *+ * If defined, this is a header that will be included instead of the default+ * configuration file mlkem/mlkem_native_config.h.+ *+ * When you need to build mlkem-native in multiple configurations, using+ * varying MLK_CONFIG_FILE can be more convenient than configuring everything+ * through CFLAGS.+ *+ * To use, MLK_CONFIG_FILE _must_ be defined prior to the inclusion of any+ * mlkem-native headers. For example, it can be set by passing+ * `-DMLK_CONFIG_FILE="..."` on the command line.+ */+/* #define MLK_CONFIG_FILE "mlkem_native_config.h" */++/**+ * The prefix to use to namespace global symbols from mlkem/.+ *+ * In a multi-level build, level-dependent symbols will additionally be+ * prefixed with the parameter set (512/768/1024).+ *+ * This can also be set using CFLAGS.+ */+#if !defined(MLK_CONFIG_NAMESPACE_PREFIX)+#define MLK_CONFIG_NAMESPACE_PREFIX MLK_DEFAULT_NAMESPACE_PREFIX+#endif++/**+ * MLK_CONFIG_MULTILEVEL_BUILD+ *+ * Set this if the build is part of a multi-level build supporting multiple+ * parameter sets.+ *+ * If you need only a single parameter set, keep this unset.+ *+ * To build mlkem-native with support for all parameter sets, build it three+ * times -- once per parameter set -- and set the option+ * MLK_CONFIG_MULTILEVEL_WITH_SHARED for exactly one of them, and+ * MLK_CONFIG_MULTILEVEL_NO_SHARED for the others.+ * MLK_CONFIG_MULTILEVEL_BUILD should be set for all of them.+ *+ * See examples/multilevel_build for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLK_CONFIG_MULTILEVEL_BUILD */++/**+ * MLK_CONFIG_EXTERNAL_API_QUALIFIER+ *+ * If set, this option provides an additional function qualifier to be added+ * to declarations of mlkem-native's public API.+ *+ * The primary use case for this option are single-CU builds where the public+ * API exposed by mlkem-native is wrapped by another API in the consuming+ * application. In this case, even mlkem-native's public API can be marked+ * `static`.+ */+/* #define MLK_CONFIG_EXTERNAL_API_QUALIFIER */++/**+ * MLK_CONFIG_NO_KEYPAIR_API+ *+ * By default, mlkem-native includes support for generating key pairs.+ * If you don't need this, set MLK_CONFIG_NO_KEYPAIR_API to exclude+ * keypair and keypair_derand, and all internal+ * APIs only needed by those functions.+ */+/* #define MLK_CONFIG_NO_KEYPAIR_API */++/**+ * MLK_CONFIG_NO_ENCAPS_API+ *+ * By default, mlkem-native includes support for encapsulation. If you+ * don't need this, set MLK_CONFIG_NO_ENCAPS_API to exclude+ * enc, enc_derand, check_pk, and+ * all internal APIs only needed by those functions.+ *+ * @note Setting this option is incompatible with MLK_CONFIG_KEYGEN_PCT+ * as the current PCT implementation requires enc().+ */+/* #define MLK_CONFIG_NO_ENCAPS_API */++/**+ * MLK_CONFIG_NO_DECAPS_API+ *+ * By default, mlkem-native includes support for decapsulation. If you+ * don't need this, set MLK_CONFIG_NO_DECAPS_API to exclude+ * dec, check_sk, and all internal APIs only+ * needed by those functions.+ *+ * @note Setting this option is incompatible with MLK_CONFIG_KEYGEN_PCT+ * as the current PCT implementation requires dec().+ */+/* #define MLK_CONFIG_NO_DECAPS_API */++/**+ * MLK_CONFIG_NO_RANDOMIZED_API+ *+ * If this option is set, mlkem-native will be built without the randomized+ * API functions (keypair and enc). This allows users+ * to build mlkem-native without providing a randombytes() implementation+ * if they only need the deterministic API (keypair_derand,+ * enc_derand, dec).+ *+ * @note This option is incompatible with MLK_CONFIG_KEYGEN_PCT as the+ * current PCT implementation requires enc().+ */+/* #define MLK_CONFIG_NO_RANDOMIZED_API */++/**+ * MLK_CONFIG_CONSTANTS_ONLY+ *+ * If you only need the size constants (MLKEM_PUBLICKEYBYTES, etc.) but no+ * function declarations, set MLK_CONFIG_CONSTANTS_ONLY.+ *+ * This only affects the public header mlkem_native.h, not the+ * implementation.+ */+/* #define MLK_CONFIG_CONSTANTS_ONLY */++/******************************************************************************+ *+ * Build-only configuration options+ *+ * The remaining configurations are build-options only.+ * They do not affect the API described in mlkem_native.h.+ *+ *****************************************************************************/++#if defined(MLK_BUILD_INTERNAL)+/**+ * MLK_CONFIG_MULTILEVEL_WITH_SHARED+ *+ * This is for multi-level builds of mlkem-native only. If you need only a+ * single parameter set, keep this unset.+ *+ * If this is set, all MLK_CONFIG_PARAMETER_SET-independent code will be+ * included in the build, including code needed only for other parameter+ * sets.+ *+ * Example: mlk_poly_cbd3 is only needed for MLK_CONFIG_PARAMETER_SET == 512.+ * Yet, if this option is set for a build with+ * MLK_CONFIG_PARAMETER_SET == 768/1024, it would be included.+ *+ * To build mlkem-native with support for all parameter sets, build it three+ * times -- once per parameter set -- and set the option+ * MLK_CONFIG_MULTILEVEL_WITH_SHARED for exactly one of them, and+ * MLK_CONFIG_MULTILEVEL_NO_SHARED for the others.+ * MLK_CONFIG_MULTILEVEL_BUILD should be set for all of them.+ *+ * See examples/multilevel_build for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLK_CONFIG_MULTILEVEL_WITH_SHARED */++/**+ * MLK_CONFIG_MULTILEVEL_NO_SHARED+ *+ * This is for multi-level builds of mlkem-native only. If you need only a+ * single parameter set, keep this unset.+ *+ * If this is set, no MLK_CONFIG_PARAMETER_SET-independent code will be+ * included in the build.+ *+ * To build mlkem-native with support for all parameter sets, build it three+ * times -- once per parameter set -- and set the option+ * MLK_CONFIG_MULTILEVEL_WITH_SHARED for exactly one of them, and+ * MLK_CONFIG_MULTILEVEL_NO_SHARED for the others.+ * MLK_CONFIG_MULTILEVEL_BUILD should be set for all of them.+ *+ * See examples/multilevel_build for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLK_CONFIG_MULTILEVEL_NO_SHARED */++/**+ * MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+ *+ * This is only relevant for single compilation unit (SCU) builds of+ * mlkem-native. In this case, it determines whether directives defined in+ * parameter-set-independent headers should be #undef'ined or not at the+ * end of the SCU file. This is needed in multilevel builds.+ *+ * See examples/multilevel_build_native for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS */++/**+ * MLK_CONFIG_USE_NATIVE_BACKEND_ARITH+ *+ * Determines whether a native arithmetic backend should be used.+ *+ * The arithmetic backend covers performance-critical functions such as the+ * number-theoretic transform (NTT).+ *+ * If this option is unset, the C backend will be used.+ *+ * If this option is set, the arithmetic backend to be used is determined+ * by MLK_CONFIG_ARITH_BACKEND_FILE: if the latter is unset, the default+ * backend for the target architecture will be used. If set, it must be the+ * name of a backend metadata file.+ *+ * This can also be set using CFLAGS.+ */+#if !defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH)+/* #define MLK_CONFIG_USE_NATIVE_BACKEND_ARITH */+#endif++/**+ * MLK_CONFIG_ARITH_BACKEND_FILE+ *+ * The arithmetic backend to use.+ *+ * If MLK_CONFIG_USE_NATIVE_BACKEND_ARITH is unset, this option is ignored.+ *+ * If MLK_CONFIG_USE_NATIVE_BACKEND_ARITH is set, this option must either+ * be undefined or the filename of an arithmetic backend. If unset, the+ * default backend will be used.+ *+ * This can be set using CFLAGS.+ */+#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH) && \+ !defined(MLK_CONFIG_ARITH_BACKEND_FILE)+#define MLK_CONFIG_ARITH_BACKEND_FILE "native/meta.h"+#endif++/**+ * MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202+ *+ * Determines whether a native FIPS202 backend should be used.+ *+ * The FIPS202 backend covers 1x/2x/4x-fold Keccak-f1600, which is the+ * performance bottleneck of SHA3 and SHAKE.+ *+ * If this option is unset, the C backend will be used.+ *+ * If this option is set, the FIPS202 backend to be used is determined by+ * MLK_CONFIG_FIPS202_BACKEND_FILE: if the latter is unset, the default+ * backend for the target architecture will be used. If set, it must be+ * the name of a backend metadata file.+ *+ * This can also be set using CFLAGS.+ */+#if !defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202)+/* #define MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+#endif++/**+ * MLK_CONFIG_FIPS202_BACKEND_FILE+ *+ * The FIPS-202 backend to use.+ *+ * If MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 is set, this option must either+ * be undefined or the filename of a FIPS202 backend. If unset, the default+ * backend will be used.+ *+ * This can be set using CFLAGS.+ */+#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202) && \+ !defined(MLK_CONFIG_FIPS202_BACKEND_FILE)+#define MLK_CONFIG_FIPS202_BACKEND_FILE "fips202/native/auto.h"+#endif++/**+ * MLK_CONFIG_FIPS202_CUSTOM_HEADER+ *+ * Custom header to use for FIPS-202.+ *+ * This should only be set if you intend to use a custom FIPS-202+ * implementation, different from the one shipped with mlkem-native.+ *+ * If set, it must be the name of a file serving as the replacement for+ * mlkem/src/fips202/fips202.h, and exposing the same API (see FIPS202.md).+ */+/* #define MLK_CONFIG_FIPS202_CUSTOM_HEADER "SOME_FILE.h" */++/**+ * MLK_CONFIG_FIPS202X4_CUSTOM_HEADER+ *+ * Custom header to use for FIPS-202-X4.+ *+ * This should only be set if you intend to use a custom FIPS-202+ * implementation, different from the one shipped with mlkem-native.+ *+ * If set, it must be the name of a file serving as the replacement for+ * mlkem/src/fips202/fips202x4.h, and exposing the same API (see FIPS202.md).+ */+/* #define MLK_CONFIG_FIPS202X4_CUSTOM_HEADER "SOME_FILE.h" */++/**+ * MLK_CONFIG_CUSTOM_ZEROIZE+ *+ * In compliance with @[FIPS203, Section 3.3], mlkem-native zeroizes+ * intermediate buffers before returning from function calls. By default,+ * those buffers are allocated from the stack; if MLK_CONFIG_CUSTOM_ALLOC_FREE+ * is set, they are (mostly -- few exceptions remain at present) allocated from+ * the configured custom allocator.+ *+ * mlkem-native also zeroizes caller-owned output buffers as needed to uphold+ * the API convention that outputs be either unmodified or zeroized upon+ * failure.+ *+ * Set this option and define `mlk_zeroize` if you want to use a custom+ * method to zeroize intermediate and output buffers.+ *+ * The default implementation uses SecureZeroMemory on Windows and a+ * memset + compiler barrier otherwise. If neither of those is available on+ * the target platform, compilation will fail, and you will need to use+ * MLK_CONFIG_CUSTOM_ZEROIZE to provide a custom implementation of+ * `mlk_zeroize()`.+ *+ * @warning+ * The zeroization conducted by mlkem-native reduces the likelihood of data+ * leaking on the stack or custom allocators, but it does not eliminate it.+ * For example, the C standard makes no guarantee about where a compiler+ * allocates local structures and whether/where it makes copies of them.+ * Also, in addition to entire structures, there may also be potentially+ * exploitable leakage of individual values on the stack. If you need+ * bullet-proof zeroization of the stack, you need to consider additional+ * measures instead of what this feature provides. In this case, you can+ * set mlk_zeroize to a no-op. Note that in this case you are also responsible+ * for zeroizing output buffers upon failure.+ */+/* #define MLK_CONFIG_CUSTOM_ZEROIZE+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/sys.h"+ static MLK_INLINE void mlk_zeroize(void *ptr, size_t len)+ {+ ... your implementation ...+ }+ #endif+*/++/**+ * MLK_CONFIG_CUSTOM_RANDOMBYTES+ *+ * mlkem-native does not provide a secure randombytes implementation. Such+ * an implementation has to be provided by the consumer.+ *+ * If this option is not set, mlkem-native expects a function+ * int randombytes(uint8_t *out, size_t outlen). It is expected to return+ * zero on success, and non-zero on failure. In case of failure, the+ * top-level APIs will return an MLK_ERR_RNG_FAIL error code.+ *+ * Set this option and define `mlk_randombytes` (with the same signature+ * and behaviour) if you want to use a custom method to sample randombytes+ * with a different name or signature.+ */+/* #define MLK_CONFIG_CUSTOM_RANDOMBYTES+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/sys.h"+ static MLK_INLINE int mlk_randombytes(uint8_t *ptr, size_t len)+ {+ ... your implementation ...+ return 0;+ }+ #endif+*/++/**+ * MLK_CONFIG_CUSTOM_CAPABILITY_FUNC+ *+ * mlkem-native backends may rely on specific hardware features. Those+ * backends will only be included in an mlkem-native build if support for+ * the respective features is enabled at compile-time. However, when+ * building for a heterogeneous set of CPUs to run the resulting+ * binary/library on, feature detection at _runtime_ is needed to decide+ * whether a backend can be used or not.+ *+ * Set this option and define `mlk_sys_check_capability` if you want to+ * use a custom method to dispatch between implementations.+ *+ * If this option is not set, mlkem-native uses compile-time feature+ * detection only to decide which backend to use.+ *+ * If you compile mlkem-native on a system with different capabilities+ * than the system that the resulting binary/library will be run on, you+ * must use this option.+ */+/* #define MLK_CONFIG_CUSTOM_CAPABILITY_FUNC+ static MLK_INLINE int mlk_sys_check_capability(mlk_sys_cap cap)+ __contract__(+ ensures(return_value == 0 || return_value == 1)+ )+ {+ ... your implementation ...+ }+*/++/**+ * MLK_CONFIG_CUSTOM_ALLOC_FREE [EXPERIMENTAL]+ *+ * Set this option and define `MLK_CUSTOM_ALLOC` and `MLK_CUSTOM_FREE` if+ * you want to use custom allocation for large local structures or buffers.+ *+ * By default, all buffers/structures are allocated on the stack. If this+ * option is set, most of them will be allocated via MLK_CUSTOM_ALLOC.+ *+ * Parameters to MLK_CUSTOM_ALLOC:+ * - T* v: Target pointer to declare.+ * - T: Type of structure to be allocated.+ * - N: Number of elements to be allocated.+ *+ * Parameters to MLK_CUSTOM_FREE:+ * - T* v: Target pointer to free. May be NULL.+ * - T: Type of structure to be freed.+ * - N: Number of elements to be freed.+ *+ * @warning This option is experimental. Its scope, configuration and+ * function/macro signatures may change at any time. We expect a+ * stable API in a future version.+ *+ * @note Even if this option is set, some allocations further down the call+ * stack will still be made from the stack, consuming up to 3KB of+ * stack space. Those will likely be added to the scope of this+ * option in the future.+ *+ * @note MLK_CUSTOM_ALLOC need not guarantee a successful allocation nor+ * include error handling. Upon failure, the target pointer should+ * simply be set to NULL. The calling code will handle this case and+ * invoke MLK_CUSTOM_FREE.+ */+/* #define MLK_CONFIG_CUSTOM_ALLOC_FREE+ #if !defined(__ASSEMBLER__)+ #include <stdlib.h>+ #define MLK_CUSTOM_ALLOC(v, T, N) \+ T* (v) = (T *)aligned_alloc(MLK_DEFAULT_ALIGN, \+ MLK_ALIGN_UP(sizeof(T) * (N)))+ #define MLK_CUSTOM_FREE(v, T, N) free(v)+ #endif+*/++/**+ * MLK_CONFIG_CUSTOM_MEMCPY+ *+ * Set this option and define `mlk_memcpy` if you want to use a custom+ * method to copy memory instead of the standard library memcpy function.+ *+ * The custom implementation must have the same signature and behavior as+ * the standard memcpy function:+ * void *mlk_memcpy(void *dest, const void *src, size_t n)+ */+/* #define MLK_CONFIG_CUSTOM_MEMCPY+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/sys.h"+ static MLK_INLINE void *mlk_memcpy(void *dest, const void *src, size_t n)+ {+ ... your implementation ...+ }+ #endif+*/++/**+ * MLK_CONFIG_CUSTOM_MEMSET+ *+ * Set this option and define `mlk_memset` if you want to use a custom+ * method to set memory instead of the standard library memset function.+ *+ * The custom implementation must have the same signature and behavior as+ * the standard memset function:+ * void *mlk_memset(void *s, int c, size_t n)+ */+/* #define MLK_CONFIG_CUSTOM_MEMSET+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/sys.h"+ static MLK_INLINE void *mlk_memset(void *s, int c, size_t n)+ {+ ... your implementation ...+ }+ #endif+*/++/**+ * MLK_CONFIG_INTERNAL_API_QUALIFIER+ *+ * If set, this option provides an additional qualifier to be added to+ * declarations of internal API functions and data.+ *+ * The primary use case for this option are single-CU builds, in which case+ * this option can be set to `static`.+ */+/* #define MLK_CONFIG_INTERNAL_API_QUALIFIER */++/**+ * MLK_CONFIG_CT_TESTING_ENABLED+ *+ * If set, mlkem-native annotates data as secret/public using valgrind's+ * annotations VALGRIND_MAKE_MEM_UNDEFINED and VALGRIND_MAKE_MEM_DEFINED,+ * enabling various checks for secret-dependent control flow or+ * variable-time execution (depending on the exact version of valgrind+ * installed).+ */+/* #define MLK_CONFIG_CT_TESTING_ENABLED */++/**+ * MLK_CONFIG_NO_ASM+ *+ * If this option is set, mlkem-native will be built without use of native+ * code or inline assembly.+ *+ * By default, inline assembly is used to implement value barriers. Without+ * inline assembly, mlkem-native will use a global volatile 'opt blocker'+ * instead; see verify.h.+ *+ * Inline assembly is also used to implement a secure zeroization function+ * on non-Windows platforms. If this option is set and the target platform+ * is not Windows, you MUST set MLK_CONFIG_CUSTOM_ZEROIZE and provide a+ * custom zeroization function.+ *+ * If this option is set, MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 and+ * MLK_CONFIG_USE_NATIVE_BACKEND_ARITH will be ignored, and no native+ * backends will be used.+ */+/* #define MLK_CONFIG_NO_ASM */++/**+ * MLK_CONFIG_NO_ASM_VALUE_BARRIER+ *+ * If this option is set, mlkem-native will be built without use of native+ * code or inline assembly for value barriers.+ *+ * By default, inline assembly (if available) is used to implement value+ * barriers. Without inline assembly, mlkem-native will use a global+ * volatile 'opt blocker' instead; see verify.h.+ */+/* #define MLK_CONFIG_NO_ASM_VALUE_BARRIER */++/**+ * MLK_CONFIG_KEYGEN_PCT+ *+ * Compliance with @[FIPS140_3_IG, p.87] requires a Pairwise Consistency+ * Test (PCT) to be carried out on a freshly generated keypair before it+ * can be exported.+ *+ * Set this option if such a check should be implemented. In this case,+ * keypair_derand and keypair will return+ * MLK_ERR_PCT_FAIL if the PCT failed.+ *+ * @note This feature will drastically lower the performance of key+ * generation.+ */+/* #define MLK_CONFIG_KEYGEN_PCT */++/**+ * MLK_CONFIG_KEYGEN_PCT_BREAKAGE_TEST+ *+ * If this option is set, the user must provide a runtime function+ * `static inline int mlk_break_pct() { ... }` to indicate whether the PCT+ * should be made to fail.+ *+ * This option only has an effect if MLK_CONFIG_KEYGEN_PCT is set.+ */+/* #define MLK_CONFIG_KEYGEN_PCT_BREAKAGE_TEST+ #if !defined(__ASSEMBLER__)+ #include "src/sys.h"+ static MLK_INLINE int mlk_break_pct(void)+ {+ ... return 0/1 depending on whether PCT should be broken ...+ }+ #endif+*/++/**+ * MLK_CONFIG_SERIAL_FIPS202_ONLY+ *+ * Set this to use a FIPS202 implementation with global state that supports+ * only one active Keccak computation at a time (e.g. some hardware+ * accelerators).+ *+ * If this option is set, batched Keccak operations are disabled for+ * rejection sampling during matrix generation. Instead, matrix entries+ * will be generated one at a time.+ *+ * This allows offloading Keccak computations to a hardware accelerator+ * that holds only a single Keccak state locally, rather than requiring+ * support for batched (4x) Keccak states.+ *+ * @note Depending on the target CPU, disabling batched Keccak may reduce+ * performance when using software FIPS202 implementations. Only+ * enable this when you have to.+ */+/* #define MLK_CONFIG_SERIAL_FIPS202_ONLY */++/**+ * MLK_CONFIG_CONTEXT_PARAMETER+ *+ * Set this to add a context parameter that is provided to public API+ * functions and is then available in custom callbacks.+ *+ * The type of the context parameter is configured via+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ */+/* #define MLK_CONFIG_CONTEXT_PARAMETER */++/**+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE+ *+ * Set this to define the type for the context parameter used by+ * MLK_CONFIG_CONTEXT_PARAMETER.+ *+ * This is only relevant if MLK_CONFIG_CONTEXT_PARAMETER is set.+ */+/* #define MLK_CONFIG_CONTEXT_PARAMETER_TYPE void* */++/************************* Config internals ********************************/++#endif /* MLK_BUILD_INTERNAL */++/* Default namespace+ *+ * Don't change this. If you need a different namespace, re-define+ * MLK_CONFIG_NAMESPACE_PREFIX above instead, and remove the following.+ *+ * The default MLKEM namespace is+ *+ * PQCP_MLKEM_NATIVE_MLKEM<LEVEL>_+ *+ * e.g., PQCP_MLKEM_NATIVE_MLKEM512_+ */++#if defined(MLK_CONFIG_MULTILEVEL_BUILD)+/* In a multi-level build the parameter set is appended by the namespacing+ * machinery, so the default prefix must not embed it. */+#define MLK_DEFAULT_NAMESPACE_PREFIX PQCP_MLKEM_NATIVE_MLKEM+#elif MLK_CONFIG_PARAMETER_SET == 512+#define MLK_DEFAULT_NAMESPACE_PREFIX PQCP_MLKEM_NATIVE_MLKEM512+#elif MLK_CONFIG_PARAMETER_SET == 768+#define MLK_DEFAULT_NAMESPACE_PREFIX PQCP_MLKEM_NATIVE_MLKEM768+#elif MLK_CONFIG_PARAMETER_SET == 1024+#define MLK_DEFAULT_NAMESPACE_PREFIX PQCP_MLKEM_NATIVE_MLKEM1024+#endif++#endif /* !MLK_CONFIG_H */
+ cbits/mlkem/src/cbmc.h view
@@ -0,0 +1,222 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_CBMC_H+#define MLK_CBMC_H+/***************************************************+ * Basic replacements for __CPROVER_XXX contracts+ ***************************************************/+/*+ * The `__contract__` / `__loop__` annotation macros use a+ * leading-double-underscore spelling in line with other CBMC macros.+ * clang-tidy flags these as reserved identifiers; we suppress the diagnostic+ * at each definition site (NOLINT) rather than disabling the check globally,+ * so it stays active for the rest of the tree.+ */+#ifndef CBMC++/* clang-format off */+#define __contract__(x) /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+#define __loop__(x) /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+/* clang-format on */++#else /* !CBMC */+++/* clang-format off */+#define __contract__(x) x /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+#define __loop__(x) x /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+/* clang-format on */++/* https://diffblue.github.io/cbmc/contracts-assigns.html */+#define assigns(...) __CPROVER_assigns(__VA_ARGS__)++/* https://diffblue.github.io/cbmc/contracts-requires-ensures.html */+#define requires(...) __CPROVER_requires(__VA_ARGS__)+#define ensures(...) __CPROVER_ensures(__VA_ARGS__)+/* https://diffblue.github.io/cbmc/contracts-loops.html */+#define invariant(...) __CPROVER_loop_invariant(__VA_ARGS__)+#define decreases(...) __CPROVER_decreases(__VA_ARGS__)+/* cassert to avoid confusion with in-built assert */+#define cassert(x) __CPROVER_assert(x, "cbmc assertion failed")+#define assume(...) __CPROVER_assume(__VA_ARGS__)++/***************************************************+ * Macros for "expression" forms that may appear+ * _inside_ top-level contracts.+ ***************************************************/++/*+ * function return value - useful inside ensures+ * https://diffblue.github.io/cbmc/contracts-functions.html+ */+#define return_value (__CPROVER_return_value)++/*+ * assigns l-value targets+ * https://diffblue.github.io/cbmc/contracts-assigns.html+ */+#define object_whole(...) __CPROVER_object_whole(__VA_ARGS__)+#define memory_slice(...) __CPROVER_object_upto(__VA_ARGS__)++/*+ * Pointer-related predicates+ * https://diffblue.github.io/cbmc/contracts-memory-predicates.html+ */+#define memory_no_alias(...) __CPROVER_is_fresh(__VA_ARGS__)+#define readable(...) __CPROVER_r_ok(__VA_ARGS__)+#define writeable(...) __CPROVER_w_ok(__VA_ARGS__)++/* Maximum supported buffer size+ *+ * Larger buffers may be supported, but due to internal modeling constraints+ * in CBMC, the proofs of memory- and type-safety won't be able to run.+ *+ * If you find yourself in need for a buffer size larger than this,+ * please contact the maintainers, so we can prioritize work to relax+ * this somewhat artificial bound.+ */+#define MLK_MAX_BUFFER_SIZE (SIZE_MAX >> 12)++/*+ * History variables+ * https://diffblue.github.io/cbmc/contracts-history-variables.html+ */+#define old(...) __CPROVER_old(__VA_ARGS__)+#define loop_entry(...) __CPROVER_loop_entry(__VA_ARGS__)++/*+ * Quantifiers+ * Note that the range on qvar is _exclusive_ between qvar_lb .. qvar_ub+ * https://diffblue.github.io/cbmc/contracts-quantifiers.html+ *+ * The quantified variable is declared as uint32_t, so these macros+ * quantify only over indices in [0, UINT32_MAX). Bounds larger than+ * UINT32_MAX (4 GiB) are NOT supported: the explicit (uint32_t) casts+ * on the bounds will trigger CBMC's conversion check if a wider bound+ * (e.g. a size_t > UINT32_MAX) is passed.+ *+ * Quantifying over size_t (64-bit) was found to blow up SMT proof+ * times, so we deliberately keep the index width at 32 bits. Callers+ * dealing with size_t-typed buffers must add an explicit+ * requires(len <= UINT32_MAX)+ * precondition.+ */++/*+ * Prevent clang-format from corrupting CBMC's special ==> operator+ */+/* clang-format off */+#define forall(qvar, qvar_lb, qvar_ub, predicate) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> (predicate) \+ }++#define exists(qvar, qvar_lb, qvar_ub, predicate) \+ __CPROVER_exists \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) && (predicate) \+ }+/* clang-format on */++/***************************************************+ * Convenience macros for common contract patterns+ ***************************************************/++/*+ * Boolean-value predidate that asserts that "all values of array_var are in+ * range value_lb (inclusive) .. value_ub (exclusive)"+ * Example:+ * array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q)+ * expands to+ * __CPROVER_forall { int k; (0 <= k && k <= MLKEM_N-1) ==> (+ * 0 <= a->coeffs[k]) && a->coeffs[k] < MLKEM_Q)) }+ */++/*+ * Prevent clang-format from corrupting CBMC's special ==> operator+ */+/* clang-format off */+#define CBMC_CONCAT_(left, right) left##right+#define CBMC_CONCAT(left, right) CBMC_CONCAT_(left, right)++#define array_bound_core(qvar, qvar_lb, qvar_ub, array_var, \+ value_lb, value_ub) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ (((int)(value_lb) <= ((array_var)[(qvar)])) && \+ (((array_var)[(qvar)]) < (int)(value_ub))) \+ }++#define array_bound(array_var, qvar_lb, qvar_ub, value_lb, value_ub) \+ array_bound_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), (qvar_lb), \+ (qvar_ub), (array_var), (value_lb), (value_ub))++#define array_unchanged_core(qvar, qvar_lb, qvar_ub, array_var) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ ((array_var)[(qvar)]) == (old(* (int16_t (*)[(qvar_ub)])(array_var)))[(qvar)] \+ }++#define array_unchanged(array_var, N) \+ array_unchanged_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), 0, (N), (array_var))++#define array_unchanged_u64_core(qvar, qvar_lb, qvar_ub, array_var) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ ((array_var)[(qvar)]) == (old(* (uint64_t (*)[(qvar_ub)])(array_var)))[(qvar)] \+ }++#define array_unchanged_u64(array_var, N) \+ array_unchanged_u64_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), 0, (N), (array_var))++#define array_unchanged_u8_core(qvar, array_var, N) \+ forall(qvar, 0, (N), \+ ((array_var)[(qvar)]) == (old(* (uint8_t (*)[(N)])(array_var)))[(qvar)])++#define array_unchanged_u8(array_var, N) \+ array_unchanged_u8_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), (array_var), (N))++#define array_zeroized_u8_core(qvar, array_var, N) \+ forall(qvar, 0, (N), ((array_var)[(qvar)]) == 0)++#define array_zeroized_u8(array_var, N) \+ array_zeroized_u8_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), (array_var), (N))+/* clang-format on */++/*+ * Output-buffer discipline on failure, as documented in API-CONVENTIONS.md:+ * when a function fails, each caller-owned output buffer is left either+ * fully unchanged or fully zeroized -- never holding partially computed or+ * stale data that could be mistaken for a valid result.+ *+ * Note the disjunction is over the buffer as a whole: it is not enough for+ * each byte to be individually either unchanged or zero.+ */+#define array_unchanged_or_zeroized_u8(array_var, N) \+ (array_unchanged_u8((array_var), (N)) || array_zeroized_u8((array_var), (N)))++/* Wrapper around array_bound operating on absolute values.+ *+ * The absolute value bound `k` is exclusive.+ *+ * Note that since the lower bound in array_bound is inclusive, we have to+ * raise it by 1 here.+ */+#define array_abs_bound(arr, lb, ub, k) \+ array_bound((arr), (lb), (ub), -((int)(k)) + 1, (k))++#endif /* CBMC */++#endif /* !MLK_CBMC_H */
+ cbits/mlkem/src/common.h view
@@ -0,0 +1,296 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS140_3_IG]+ * Implementation Guidance for FIPS 140-3 and the Cryptographic Module+ * Validation Program+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/projects/cryptographic-module-validation-program/fips-140-3-ig-announcements+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#ifndef MLK_COMMON_H+#define MLK_COMMON_H++#ifndef __ASSEMBLER__+#include <stdint.h>+#endif++#define MLK_BUILD_INTERNAL++#if defined(MLK_CONFIG_FILE)+#include MLK_CONFIG_FILE+#else+#include "mlkem_native_config.h"+#endif++#include "params.h"+#include "sys.h"++/* Internal and public API have external linkage by default, but+ * this can be overwritten by the user, e.g. for single-CU builds. */+#if !defined(MLK_CONFIG_INTERNAL_API_QUALIFIER)+#define MLK_INTERNAL_API+#define MLK_INTERNAL_DATA_DECLARATION extern+#define MLK_INTERNAL_DATA_DEFINITION+#else+#define MLK_INTERNAL_API MLK_CONFIG_INTERNAL_API_QUALIFIER+#define MLK_INTERNAL_DATA_DECLARATION MLK_CONFIG_INTERNAL_API_QUALIFIER+#define MLK_INTERNAL_DATA_DEFINITION MLK_CONFIG_INTERNAL_API_QUALIFIER+#endif++#if !defined(MLK_CONFIG_EXTERNAL_API_QUALIFIER)+#define MLK_EXTERNAL_API+#else+#define MLK_EXTERNAL_API MLK_CONFIG_EXTERNAL_API_QUALIFIER+#endif++#define MLK_CONCAT_(x1, x2) x1##x2+#define MLK_CONCAT(x1, x2) MLK_CONCAT_(x1, x2)++#if (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || \+ defined(MLK_CONFIG_MULTILEVEL_NO_SHARED))+#define MLK_ADD_PARAM_SET(s) MLK_CONCAT(s, MLK_CONFIG_PARAMETER_SET)+#else+#define MLK_ADD_PARAM_SET(s) s+#endif++#define MLK_NAMESPACE_PREFIX MLK_CONCAT(MLK_CONFIG_NAMESPACE_PREFIX, _)+#define MLK_NAMESPACE_PREFIX_K \+ MLK_CONCAT(MLK_ADD_PARAM_SET(MLK_CONFIG_NAMESPACE_PREFIX), _)++/* Functions are prefixed by MLK_CONFIG_NAMESPACE_PREFIX.+ *+ * If multiple parameter sets are used, functions depending on the parameter+ * set are additionally prefixed with 512/768/1024. See mlkem_native_config.h.+ *+ * Example: If MLK_CONFIG_NAMESPACE_PREFIX is mlkem, then+ * MLK_NAMESPACE_K(enc) becomes mlkem512_enc/mlkem768_enc/mlkem1024_enc.+ */+#define MLK_NAMESPACE(s) MLK_CONCAT(MLK_NAMESPACE_PREFIX, s)+#define MLK_NAMESPACE_K(s) MLK_CONCAT(MLK_NAMESPACE_PREFIX_K, s)++/* On Apple platforms, we need to emit leading underscore+ * in front of assembly symbols. We thus introduce a separate+ * namespace wrapper for ASM symbols. */+#if !defined(__APPLE__)+#define MLK_ASM_NAMESPACE(sym) MLK_NAMESPACE(sym)+#else+#define MLK_ASM_NAMESPACE(sym) MLK_CONCAT(_, MLK_NAMESPACE(sym))+#endif++/*+ * On X86_64 if control-flow protections (CET) are enabled (through+ * -fcf-protection=), we add an endbr64 instruction at every global function+ * label. See sys.h for more details+ */+#if defined(MLK_SYS_X86_64)+#define MLK_ASM_FN_SYMBOL(sym) MLK_ASM_NAMESPACE(sym) : MLK_CET_ENDBR+#elif defined(MLK_SYS_ARMV81M_MVE)+/* clang-format off */+#define MLK_ASM_FN_SYMBOL(sym) \+ .type MLK_ASM_NAMESPACE(sym), %function; \+ MLK_ASM_NAMESPACE(sym) :+/* clang-format on */+#else /* !MLK_SYS_X86_64 && MLK_SYS_ARMV81M_MVE */+#define MLK_ASM_FN_SYMBOL(sym) MLK_ASM_NAMESPACE(sym) :+#endif /* !MLK_SYS_X86_64 && !MLK_SYS_ARMV81M_MVE */++/*+ * Output the size of an assembly function.+ */+#if defined(__ELF__)+#define MLK_ASM_FN_SIZE(sym) \+ .size MLK_ASM_NAMESPACE(sym), .- MLK_ASM_NAMESPACE(sym)+#else+#define MLK_ASM_FN_SIZE(sym)+#endif++/* We aim to simplify the user's life by supporting builds where+ * all source files are included, even those that are not needed.+ * Those files are appropriately guarded and will be empty when unneeded.+ * The following is to avoid compilers complaining about this. */+#define MLK_EMPTY_CU(s) extern int MLK_NAMESPACE_K(empty_cu_##s);++/* MLK_CONFIG_NO_ASM takes precedence over MLK_USE_NATIVE_XXX */+#if defined(MLK_CONFIG_NO_ASM)+#undef MLK_CONFIG_USE_NATIVE_BACKEND_ARITH+#undef MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202+#endif++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH) && \+ !defined(MLK_CONFIG_ARITH_BACKEND_FILE)+#error Bad configuration: MLK_CONFIG_USE_NATIVE_BACKEND_ARITH is set, but MLK_CONFIG_ARITH_BACKEND_FILE is not.+#endif++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202) && \+ !defined(MLK_CONFIG_FIPS202_BACKEND_FILE)+#error Bad configuration: MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 is set, but MLK_CONFIG_FIPS202_BACKEND_FILE is not.+#endif++#if defined(MLK_CONFIG_NO_RANDOMIZED_API) && defined(MLK_CONFIG_KEYGEN_PCT)+#error Bad configuration: MLK_CONFIG_NO_RANDOMIZED_API is incompatible with MLK_CONFIG_KEYGEN_PCT as the current PCT implementation requires enc()+#endif++#if defined(MLK_CONFIG_NO_ENCAPS_API) && defined(MLK_CONFIG_KEYGEN_PCT)+#error Bad configuration: MLK_CONFIG_NO_ENCAPS_API is incompatible with MLK_CONFIG_KEYGEN_PCT as the current PCT implementation requires enc()+#endif++#if defined(MLK_CONFIG_NO_DECAPS_API) && defined(MLK_CONFIG_KEYGEN_PCT)+#error Bad configuration: MLK_CONFIG_NO_DECAPS_API is incompatible with MLK_CONFIG_KEYGEN_PCT as the current PCT implementation requires dec()+#endif++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH)+#include MLK_CONFIG_ARITH_BACKEND_FILE+/* Include to enforce consistency of API and implementation,+ * and conduct sanity checks on the backend.+ *+ * Keep this _after_ the inclusion of the backend; otherwise,+ * the sanity checks won't have an effect. */+#if defined(MLK_CHECK_APIS) && !defined(__ASSEMBLER__)+#include "native/api.h"+#endif+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_ARITH */++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202)+#include MLK_CONFIG_FIPS202_BACKEND_FILE+/* Include to enforce consistency of API and implementation,+ * and conduct sanity checks on the backend.+ *+ * Keep this _after_ the inclusion of the backend; otherwise,+ * the sanity checks won't have an effect. */+#if defined(MLK_CHECK_APIS) && !defined(__ASSEMBLER__)+#include "fips202/native/api.h"+#endif+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 */++#if !defined(MLK_CONFIG_FIPS202_CUSTOM_HEADER)+#define MLK_FIPS202_HEADER_FILE "fips202/fips202.h"+#else+#define MLK_FIPS202_HEADER_FILE MLK_CONFIG_FIPS202_CUSTOM_HEADER+#endif++#if !defined(MLK_CONFIG_FIPS202X4_CUSTOM_HEADER)+#define MLK_FIPS202X4_HEADER_FILE "fips202/fips202x4.h"+#else+#define MLK_FIPS202X4_HEADER_FILE MLK_CONFIG_FIPS202X4_CUSTOM_HEADER+#endif++/* Standard library function replacements */+#if !defined(__ASSEMBLER__)+#if !defined(MLK_CONFIG_CUSTOM_MEMCPY)+#include <string.h>+#define mlk_memcpy memcpy+#endif++#if !defined(MLK_CONFIG_CUSTOM_MEMSET)+#include <string.h>+#define mlk_memset memset+#endif+++/* Allocation macros for large local structures+ *+ * MLK_ALLOC(v, T, N) declares T *v and attempts to point it to an T[N]+ * MLK_FREE(v, T, N) zeroizes and frees the allocation+ *+ * Default implementation uses stack allocation.+ * Can be overridden by setting the config option MLK_CONFIG_CUSTOM_ALLOC_FREE+ * and defining MLK_CUSTOM_ALLOC and MLK_CUSTOM_FREE.+ */+#if defined(MLK_CONFIG_CUSTOM_ALLOC_FREE) != \+ (defined(MLK_CUSTOM_ALLOC) && defined(MLK_CUSTOM_FREE))+#error Bad configuration: MLK_CONFIG_CUSTOM_ALLOC_FREE must be set together with MLK_CUSTOM_ALLOC and MLK_CUSTOM_FREE+#endif++/* Context-parameter machinery (MLK_CONTEXT_PARAMETERS_n and related config+ * checks). Kept in a separate, level-generic header for readability; included+ * here so it is available to the allocation macros below and to all consumers+ * of common.h. */+#include "context.h"++#if !defined(MLK_CONFIG_CUSTOM_ALLOC_FREE)+/* Default: stack allocation */++/* This is a declaration macro, not an expression macro: T is a type and v is+ * a declarator, neither of which can be wrapped in parentheses. The+ * bugprone-macro-parentheses diagnostic is therefore a false positive here. */+#define MLK_ALLOC(v, T, N, context) \+ MLK_ALIGN T mlk_alloc_##v[N]; \+ T *v = mlk_alloc_##v /* NOLINT(bugprone-macro-parentheses) */++/* The MLK_FREE macro body references mlk_zeroize(), which is declared in+ * verify.h. We deliberately do NOT include verify.h here: doing so would+ * create a circular dependency (verify.h includes common.h), and common.h+ * itself never calls mlk_zeroize() -- only the macro expansion does. Each+ * translation unit that uses MLK_FREE therefore includes verify.h directly. */+#define MLK_FREE(v, T, N, context) \+ do \+ { \+ MLK_CONTEXT_UNUSED(context); \+ mlk_zeroize(mlk_alloc_##v, sizeof(mlk_alloc_##v)); \+ (v) = NULL; \+ } while (0)++#else /* !MLK_CONFIG_CUSTOM_ALLOC_FREE */++/* Custom allocation */++/*+ * The indirection here is necessary to use MLK_CONTEXT_PARAMETERS_3 here.+ */+#define MLK_APPLY(f, args) f args++#define MLK_ALLOC(v, T, N, context) \+ MLK_APPLY(MLK_CUSTOM_ALLOC, MLK_CONTEXT_PARAMETERS_3(v, T, N, context))++#define MLK_FREE(v, T, N, context) \+ do \+ { \+ if (v != NULL) \+ { \+ mlk_zeroize(v, sizeof(T) * (N)); \+ MLK_APPLY(MLK_CUSTOM_FREE, MLK_CONTEXT_PARAMETERS_3(v, T, N, context)); \+ v = NULL; \+ } \+ } while (0)++#endif /* MLK_CONFIG_CUSTOM_ALLOC_FREE */++/****************************** Error codes ***********************************/++/* Generic failure condition. Currently not returned by any function;+ * reserved for failures that no more specific code covers. */+#define MLK_ERR_FAIL (-1)+/* An allocation failed. This can only happen if MLK_CONFIG_CUSTOM_ALLOC_FREE+ * is defined and the provided MLK_CUSTOM_ALLOC can fail. */+#define MLK_ERR_OUT_OF_MEMORY (-2)+/* An RNG failure occurred. Might be due to insufficient entropy or+ * system misconfiguration. */+#define MLK_ERR_RNG_FAIL (-3)+/* Public key validation failed: the @[FIPS203, Section 7.2, 'modulus check']+ * found a coefficient outside [0,q-1]. Returned by check_pk and by the+ * encapsulation API. */+#define MLK_ERR_INVALID_PK (-4)+/* Secret key validation failed: the @[FIPS203, Section 7.3, 'hash check']+ * found the embedded public key hash inconsistent. Returned by check_sk and+ * by the decapsulation API. */+#define MLK_ERR_INVALID_SK (-5)+/* The 'Pairwise Consistency Test' @[FIPS140_3_IG, p.87] and+ * @[FIPS203, Section 7.1, Pairwise Consistency] failed. Only possible when+ * MLK_CONFIG_KEYGEN_PCT is enabled; signals that the freshly generated key+ * pair failed its encaps/decaps self-test. */+#define MLK_ERR_PCT_FAIL (-6)++#endif /* !__ASSEMBLER__ */++#endif /* !MLK_COMMON_H */
+ cbits/mlkem/src/compress.c view
@@ -0,0 +1,763 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ *+ * - [REF]+ * CRYSTALS-Kyber C reference implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/ref+ */++#include "common.h"+#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)+++#include "cbmc.h"+#include "compress.h"+#include "debug.h"+#include "verify.h"++#if (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API)) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2 || \+ MLKEM_K == 3)+/* Reference: `poly_compress()` in the reference implementation @[REF],+ * for ML-KEM-{512,768}.+ * - In contrast to the reference implementation, we assume+ * unsigned canonical coefficients here.+ * The reference implementation works with coefficients+ * in the range [-(MLKEM_Q-1), MLKEM_Q-1]. */+MLK_STATIC_TESTABLE void mlk_poly_compress_d4_c(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D4], const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+)+{+ unsigned i;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);++ for (i = 0; i < MLKEM_N / 8; i++)+ __loop__(invariant(i <= MLKEM_N / 8)+ decreases(MLKEM_N / 8 - i))+ {+ unsigned j;+ uint8_t t[8] = {0};+ for (j = 0; j < 8; j++)+ __loop__(+ invariant(i <= MLKEM_N / 8 && j <= 8)+ invariant(array_bound(t, 0, j, 0, 16))+ decreases(8 - j))+ {+ t[j] = mlk_scalar_compress_d4(a->coeffs[8 * i + j]);+ }++ /* All t[i] are 4-bit wide, so the truncations don't alter the value. */+ r[i * 4] = (uint8_t)(t[0] | (t[1] << 4));+ r[i * 4 + 1] = (uint8_t)(t[2] | (t[3] << 4));+ r[i * 4 + 2] = (uint8_t)(t[4] | (t[5] << 4));+ r[i * 4 + 3] = (uint8_t)(t[6] | (t[7] << 4));+ }+}++MLK_INTERNAL_API+void mlk_poly_compress_d4(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D4],+ const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+)+{+#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D4)+ int ret;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);+ ret = mlk_poly_compress_d4_native(r, a->coeffs);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D4 */++ mlk_poly_compress_d4_c(r, a);+}++/* Reference: Embedded into `polyvec_compress()` in the+ * reference implementation, for ML-KEM-{512,768}.+ * - In contrast to the reference implementation, we assume+ * unsigned canonical coefficients here.+ * The reference implementation works with coefficients+ * in the range [-(MLKEM_Q-1), MLKEM_Q-1]. */+MLK_STATIC_TESTABLE void mlk_poly_compress_d10_c(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D10], const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+)+{+ unsigned j;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);+ for (j = 0; j < MLKEM_N / 4; j++)+ __loop__(invariant(j <= MLKEM_N / 4)+ decreases(MLKEM_N / 4 - j))+ {+ unsigned k;+ uint16_t t[4];+ for (k = 0; k < 4; k++)+ __loop__(+ invariant(k <= 4)+ invariant(forall(r, 0, k, t[r] < (1u << 10)))+ decreases(4 - k))+ {+ t[k] = mlk_scalar_compress_d10(a->coeffs[4 * j + k]);+ }++ /*+ * Make all implicit truncation explicit. No data is being+ * truncated for the LHS's since each t[i] is 10-bit in size.+ */+ r[5 * j + 0] = (uint8_t)((t[0] >> 0) & 0xFF);+ r[5 * j + 1] = (uint8_t)((t[0] >> 8) | ((t[1] << 2) & 0xFF));+ r[5 * j + 2] = (uint8_t)((t[1] >> 6) | ((t[2] << 4) & 0xFF));+ r[5 * j + 3] = (uint8_t)((t[2] >> 4) | ((t[3] << 6) & 0xFF));+ r[5 * j + 4] = (uint8_t)(t[3] >> 2);+ }+}++MLK_INTERNAL_API+void mlk_poly_compress_d10(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D10],+ const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+)+{+#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D10)+ int ret;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);+ ret = mlk_poly_compress_d10_native(r, a->coeffs);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D10 */++ mlk_poly_compress_d10_c(r, a);+}++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `poly_decompress()` in the reference implementation @[REF],+ * for ML-KEM-{512,768}. */+MLK_STATIC_TESTABLE void mlk_poly_decompress_d4_c(+ mlk_poly *r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D4])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D4))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+ unsigned i;+ for (i = 0; i < MLKEM_N / 2; i++)+ __loop__(+ invariant(i <= MLKEM_N / 2)+ invariant(array_bound(r->coeffs, 0, 2 * i, 0, MLKEM_Q))+ decreases(MLKEM_N / 2 - i))+ {+ r->coeffs[2 * i + 0] = mlk_scalar_decompress_d4((a[i] >> 0) & 0xF);+ r->coeffs[2 * i + 1] = mlk_scalar_decompress_d4((a[i] >> 4) & 0xF);+ }++ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+}++MLK_INTERNAL_API+void mlk_poly_decompress_d4(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D4])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D4))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D4)+ int ret;+ ret = mlk_poly_decompress_d4_native(r->coeffs, a);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D4 */++ mlk_poly_decompress_d4_c(r, a);+}++/* Reference: Embedded into `polyvec_decompress()` in the+ * reference implementation, for ML-KEM-{512,768}. */+MLK_STATIC_TESTABLE void mlk_poly_decompress_d10_c(+ mlk_poly *r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D10])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D10))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+ unsigned j;+ for (j = 0; j < MLKEM_N / 4; j++)+ __loop__(+ invariant(j <= MLKEM_N / 4)+ invariant(array_bound(r->coeffs, 0, 4 * j, 0, MLKEM_Q))+ decreases(MLKEM_N / 4 - j))+ {+ unsigned k;+ uint16_t t[4];+ uint8_t const *base = &a[5 * j];++ t[0] = 0x3FF & ((base[0] >> 0) | ((uint16_t)base[1] << 8));+ t[1] = 0x3FF & ((base[1] >> 2) | ((uint16_t)base[2] << 6));+ t[2] = 0x3FF & ((base[2] >> 4) | ((uint16_t)base[3] << 4));+ t[3] = 0x3FF & ((base[3] >> 6) | ((uint16_t)base[4] << 2));++ for (k = 0; k < 4; k++)+ __loop__(+ invariant(k <= 4)+ invariant(array_bound(r->coeffs, 0, 4 * j + k, 0, MLKEM_Q))+ decreases(4 - k))+ {+ r->coeffs[4 * j + k] = mlk_scalar_decompress_d10(t[k]);+ }+ }++ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+}++MLK_INTERNAL_API+void mlk_poly_decompress_d10(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D10])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D10))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D10)+ int ret;+ ret = mlk_poly_decompress_d10_native(r->coeffs, a);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D10 */++ mlk_poly_decompress_d10_c(r, a);+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3) \+ */++#if (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API)) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4)+/* Reference: `poly_compress()` in the reference implementation @[REF],+ * for ML-KEM-1024.+ * - In contrast to the reference implementation, we assume+ * unsigned canonical coefficients here.+ * The reference implementation works with coefficients+ * in the range [-(MLKEM_Q-1), MLKEM_Q-1]. */+MLK_STATIC_TESTABLE void mlk_poly_compress_d5_c(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D5], const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+)+{+ unsigned i;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);++ for (i = 0; i < MLKEM_N / 8; i++)+ __loop__(invariant(i <= MLKEM_N / 8)+ decreases(MLKEM_N / 8 - i))+ {+ unsigned j;+ uint8_t t[8] = {0};+ for (j = 0; j < 8; j++)+ __loop__(+ invariant(i <= MLKEM_N / 8 && j <= 8)+ invariant(array_bound(t, 0, j, 0, 32))+ decreases(8 - j))+ {+ t[j] = mlk_scalar_compress_d5(a->coeffs[8 * i + j]);+ }++ r[i * 5] = (uint8_t)(0xFF & ((t[0] >> 0) | (t[1] << 5)));+ r[i * 5 + 1] = (uint8_t)(0xFF & ((t[1] >> 3) | (t[2] << 2) | (t[3] << 7)));+ r[i * 5 + 2] = (uint8_t)(0xFF & ((t[3] >> 1) | (t[4] << 4)));+ r[i * 5 + 3] = (uint8_t)(0xFF & ((t[4] >> 4) | (t[5] << 1) | (t[6] << 6)));+ r[i * 5 + 4] = (uint8_t)(0xFF & ((t[6] >> 2) | (t[7] << 3)));+ }+}++MLK_INTERNAL_API+void mlk_poly_compress_d5(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D5],+ const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+)+{+#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D5)+ int ret;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);+ ret = mlk_poly_compress_d5_native(r, a->coeffs);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D5 */++ mlk_poly_compress_d5_c(r, a);+}++/* Reference: Embedded into `polyvec_compress()` in the+ * reference implementation, for ML-KEM-1024.+ * - In contrast to the reference implementation, we assume+ * unsigned canonical coefficients here.+ * The reference implementation works with coefficients+ * in the range [-(MLKEM_Q-1), MLKEM_Q-1]. */+MLK_STATIC_TESTABLE void mlk_poly_compress_d11_c(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D11], const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+)+{+ unsigned j;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);++ for (j = 0; j < MLKEM_N / 8; j++)+ __loop__(invariant(j <= MLKEM_N / 8)+ decreases(MLKEM_N / 8 - j))+ {+ unsigned k;+ uint16_t t[8];+ for (k = 0; k < 8; k++)+ __loop__(+ invariant(k <= 8)+ invariant(forall(r, 0, k, t[r] < (1u << 11)))+ decreases(8 - k))+ {+ t[k] = mlk_scalar_compress_d11(a->coeffs[8 * j + k]);+ }++ /*+ * Make all implicit truncation explicit. No data is being+ * truncated for the LHS's since each t[i] is 11-bit in size.+ */+ r[11 * j + 0] = (uint8_t)((t[0] >> 0) & 0xFF);+ r[11 * j + 1] = (uint8_t)((t[0] >> 8) | ((t[1] << 3) & 0xFF));+ r[11 * j + 2] = (uint8_t)((t[1] >> 5) | ((t[2] << 6) & 0xFF));+ r[11 * j + 3] = (uint8_t)((t[2] >> 2) & 0xFF);+ r[11 * j + 4] = (uint8_t)((t[2] >> 10) | ((t[3] << 1) & 0xFF));+ r[11 * j + 5] = (uint8_t)((t[3] >> 7) | ((t[4] << 4) & 0xFF));+ r[11 * j + 6] = (uint8_t)((t[4] >> 4) | ((t[5] << 7) & 0xFF));+ r[11 * j + 7] = (uint8_t)((t[5] >> 1) & 0xFF);+ r[11 * j + 8] = (uint8_t)((t[5] >> 9) | ((t[6] << 2) & 0xFF));+ r[11 * j + 9] = (uint8_t)((t[6] >> 6) | ((t[7] << 5) & 0xFF));+ r[11 * j + 10] = (uint8_t)(t[7] >> 3);+ }+}++MLK_INTERNAL_API+void mlk_poly_compress_d11(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D11],+ const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+)+{+#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D11)+ int ret;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);+ ret = mlk_poly_compress_d11_native(r, a->coeffs);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D11 */++ mlk_poly_compress_d11_c(r, a);+}++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `poly_decompress()` in the reference implementation @[REF],+ * for ML-KEM-1024. */+MLK_STATIC_TESTABLE void mlk_poly_decompress_d5_c(+ mlk_poly *r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D5])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D5))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+ unsigned i;+ for (i = 0; i < MLKEM_N / 8; i++)+ __loop__(+ invariant(i <= MLKEM_N / 8)+ invariant(array_bound(r->coeffs, 0, 8 * i, 0, MLKEM_Q))+ decreases(MLKEM_N / 8 - i))+ {+ unsigned j;+ uint8_t t[8];+ const unsigned offset = i * 5;+ /*+ * Explicitly truncate to avoid warning about+ * implicit truncation in CBMC and unwind loop for ease+ * of proof.+ */++ /*+ * Decompress 5 8-bit bytes (so 40 bits) into+ * 8 5-bit values stored in t[]+ */+ t[0] = 0x1F & (a[offset + 0] >> 0);+ t[1] = 0x1F & ((a[offset + 0] >> 5) | (a[offset + 1] << 3));+ t[2] = 0x1F & (a[offset + 1] >> 2);+ t[3] = 0x1F & ((a[offset + 1] >> 7) | (a[offset + 2] << 1));+ t[4] = 0x1F & ((a[offset + 2] >> 4) | (a[offset + 3] << 4));+ t[5] = 0x1F & (a[offset + 3] >> 1);+ t[6] = 0x1F & ((a[offset + 3] >> 6) | (a[offset + 4] << 2));+ t[7] = 0x1F & (a[offset + 4] >> 3);++ /* and copy to the correct slice in r[] */+ for (j = 0; j < 8; j++)+ __loop__(+ invariant(j <= 8 && i <= MLKEM_N / 8)+ invariant(array_bound(r->coeffs, 0, 8 * i + j, 0, MLKEM_Q))+ decreases(8 - j))+ {+ r->coeffs[8 * i + j] = mlk_scalar_decompress_d5(t[j]);+ }+ }++ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+}++MLK_INTERNAL_API+void mlk_poly_decompress_d5(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D5])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D5))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D5)+ int ret;+ ret = mlk_poly_decompress_d5_native(r->coeffs, a);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D5 */++ mlk_poly_decompress_d5_c(r, a);+}++/* Reference: Embedded into `polyvec_decompress()` in the+ * reference implementation, for ML-KEM-1024. */+MLK_STATIC_TESTABLE void mlk_poly_decompress_d11_c(+ mlk_poly *r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D11])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D11))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+ unsigned j;+ for (j = 0; j < MLKEM_N / 8; j++)+ __loop__(+ invariant(j <= MLKEM_N / 8)+ invariant(array_bound(r->coeffs, 0, 8 * j, 0, MLKEM_Q))+ decreases(MLKEM_N / 8 - j))+ {+ unsigned k;+ uint16_t t[8];+ uint8_t const *base = &a[11 * j];+ t[0] = 0x7FF & ((base[0] >> 0) | ((uint16_t)base[1] << 8));+ t[1] = 0x7FF & ((base[1] >> 3) | ((uint16_t)base[2] << 5));+ t[2] = 0x7FF & ((base[2] >> 6) | ((uint16_t)base[3] << 2) |+ ((uint16_t)base[4] << 10));+ t[3] = 0x7FF & ((base[4] >> 1) | ((uint16_t)base[5] << 7));+ t[4] = 0x7FF & ((base[5] >> 4) | ((uint16_t)base[6] << 4));+ t[5] = 0x7FF & ((base[6] >> 7) | ((uint16_t)base[7] << 1) |+ ((uint16_t)base[8] << 9));+ t[6] = 0x7FF & ((base[8] >> 2) | ((uint16_t)base[9] << 6));+ t[7] = 0x7FF & ((base[9] >> 5) | ((uint16_t)base[10] << 3));++ for (k = 0; k < 8; k++)+ __loop__(+ invariant(k <= 8)+ invariant(array_bound(r->coeffs, 0, 8 * j + k, 0, MLKEM_Q))+ decreases(8 - k))+ {+ r->coeffs[8 * j + k] = mlk_scalar_decompress_d11(t[k]);+ }+ }++ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+}++MLK_INTERNAL_API+void mlk_poly_decompress_d11(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D11])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D11))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D11)+ int ret;+ ret = mlk_poly_decompress_d11_native(r->coeffs, a);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D11 */++ mlk_poly_decompress_d11_c(r, a);+}++#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4) */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API) || !defined(MLK_CONFIG_NO_ENCAPS_API)+/* Reference: `poly_tobytes()` in the reference implementation @[REF].+ * - In contrast to the reference implementation, we assume+ * unsigned canonical coefficients here.+ * The reference implementation works with coefficients+ * in the range [-(MLKEM_Q-1), MLKEM_Q-1]. */+MLK_STATIC_TESTABLE void mlk_poly_tobytes_c(uint8_t r[MLKEM_POLYBYTES],+ const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYBYTES))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYBYTES))+)+{+ unsigned i;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);++ for (i = 0; i < MLKEM_N / 2; i++)+ __loop__(invariant(i <= MLKEM_N / 2)+ decreases(MLKEM_N / 2 - i))+ {+ /* The conversion to uint16_t is safe since we assume that+ * the coefficients of `a` are non-negative. */+ const uint16_t t0 = (uint16_t)a->coeffs[2 * i];+ const uint16_t t1 = (uint16_t)a->coeffs[2 * i + 1];+ /*+ * t0 and t1 are both < MLKEM_Q, so contain at most 12 bits each of+ * significant data, so these can be packed into 24 bits or exactly+ * 3 bytes, as follows.+ */++ /* Least significant bits 0 - 7 of t0. */+ r[3 * i + 0] = (uint8_t)(t0 & 0xFF);++ /*+ * Most significant bits 8 - 11 of t0 become the least significant+ * nibble of the second byte. The least significant 4 bits+ * of t1 become the upper nibble of the second byte.+ *+ * The conversion to uint8_t does not alter the value.+ */+ r[3 * i + 1] = (uint8_t)((t0 >> 8) | ((t1 << 4) & 0xF0));++ /* Bits 4 - 11 of t1 become the third byte. The conversion to uint8_t+ * does not alter the value because t1 is 12-bit wide. */+ r[3 * i + 2] = (uint8_t)(t1 >> 4);+ }+}++MLK_INTERNAL_API+void mlk_poly_tobytes(uint8_t r[MLKEM_POLYBYTES], const mlk_poly *a)+{+#if defined(MLK_USE_NATIVE_POLY_TOBYTES)+ int ret;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);+ ret = mlk_poly_tobytes_native(r, a->coeffs);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_TOBYTES */++ mlk_poly_tobytes_c(r, a);+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `poly_frombytes()` in the reference implementation @[REF]. */+MLK_STATIC_TESTABLE void mlk_poly_frombytes_c(mlk_poly *r,+ const uint8_t a[MLKEM_POLYBYTES])+__contract__(+ requires(memory_no_alias(a, MLKEM_POLYBYTES))+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT))+)+{+ unsigned i;+ for (i = 0; i < MLKEM_N / 2; i++)+ __loop__(+ invariant(i <= MLKEM_N / 2)+ invariant(array_bound(r->coeffs, 0, 2 * i, 0, MLKEM_UINT12_LIMIT))+ decreases(MLKEM_N / 2 - i))+ {+ const uint8_t t0 = a[3 * i + 0];+ const uint8_t t1 = a[3 * i + 1];+ const uint8_t t2 = a[3 * i + 2];+ /* Safety:+ * - The explicit cast to uint16_t ensures that << 8 does+ * not signed-overflow even on a 16-bit system.+ * - The cast to int16_t is safe due to the explicit 0xFFF truncation.+ */+ r->coeffs[2 * i + 0] = (int16_t)(t0 | (((uint16_t)t1 << 8) & 0xFFF));+ r->coeffs[2 * i + 1] = (int16_t)((t1 >> 4) | (t2 << 4));+ }++ /* Note that the coefficients are not canonical */+ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_UINT12_LIMIT);+}++MLK_INTERNAL_API+void mlk_poly_frombytes(mlk_poly *r, const uint8_t a[MLKEM_POLYBYTES])+{+#if defined(MLK_USE_NATIVE_POLY_FROMBYTES)+ int ret;+ ret = mlk_poly_frombytes_native(r->coeffs, a);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_FROMBYTES */++ mlk_poly_frombytes_c(r, a);+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `poly_frommsg()` in the reference implementation @[REF].+ * - We use a value barrier around the bit-selection mask to+ * reduce the risk of compiler-introduced branches.+ * The reference implementation contains the expression+ * `(msg[i] >> j) & 1` which the compiler can reason must+ * be either 0 or 1. */+MLK_INTERNAL_API+void mlk_poly_frommsg(mlk_poly *r, const uint8_t msg[MLKEM_INDCPA_MSGBYTES])+{+ unsigned i;+#if (MLKEM_INDCPA_MSGBYTES != MLKEM_N / 8)+#error "MLKEM_INDCPA_MSGBYTES must be equal to MLKEM_N/8 bytes!"+#endif++ for (i = 0; i < MLKEM_N / 8; i++)+ __loop__(+ invariant(i <= MLKEM_N / 8)+ invariant(array_bound(r->coeffs, 0, 8 * i, 0, MLKEM_Q))+ decreases(MLKEM_N / 8 - i))+ {+ unsigned j;+ for (j = 0; j < 8; j++)+ __loop__(+ invariant(i < MLKEM_N / 8 && j <= 8)+ invariant(array_bound(r->coeffs, 0, 8 * i + j, 0, MLKEM_Q))+ decreases(8 - j))+ {+ /* mlk_ct_sel_int16(MLKEM_Q_HALF, 0, b) is `Decompress_1(b != 0)`+ * as per @[FIPS203, Eq (4.8)]. */++ /* Prevent the compiler from recognizing this as a bit selection */+ uint8_t mask = mlk_value_barrier_u8((uint8_t)(1u << j));+ r->coeffs[8 * i + j] = mlk_ct_sel_int16(MLKEM_Q_HALF, 0, msg[i] & mask);+ }+ }+ mlk_assert_abs_bound(r, MLKEM_N, MLKEM_Q);+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `poly_tomsg()` in the reference implementation @[REF].+ * - In contrast to the reference implementation, we assume+ * unsigned canonical coefficients here.+ * The reference implementation works with coefficients+ * in the range [-(MLKEM_Q-1), MLKEM_Q-1].+ */+MLK_INTERNAL_API+void mlk_poly_tomsg(uint8_t msg[MLKEM_INDCPA_MSGBYTES], const mlk_poly *r)+{+ unsigned i;+ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);++ for (i = 0; i < MLKEM_N / 8; i++)+ __loop__(invariant(i <= MLKEM_N / 8)+ decreases(MLKEM_N / 8 - i))+ {+ unsigned j;+ msg[i] = 0;+ for (j = 0; j < 8; j++)+ __loop__(+ invariant(i <= MLKEM_N / 8 && j <= 8)+ decreases(8 - j))+ {+ uint32_t t = mlk_scalar_compress_d1(r->coeffs[8 * i + j]);+ msg[i] |= (uint8_t)(t << j);+ }+ }+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#else /* !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(compress)++#endif /* MLK_CONFIG_MULTILEVEL_NO_SHARED */
+ cbits/mlkem/src/compress.h view
@@ -0,0 +1,613 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ *+ * - [REF]+ * CRYSTALS-Kyber C reference implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/ref+ */++#ifndef MLK_COMPRESS_H+#define MLK_COMPRESS_H+++#include "cbmc.h"+#include "common.h"+#include "debug.h"+#include "poly.h"+#include "verify.h"++/*+ * The multiplication in this routine will exceed UINT32_MAX+ * and wrap around for large values of u. This is expected and required.+ */+#ifdef CBMC+#pragma CPROVER check push+#pragma CPROVER check disable "unsigned-overflow"+#endif++/**+ * Compute round(u * 2 / MLKEM_Q).+ *+ * @spec{Compress_1 from @[FIPS203, Eq (4.7)].}+ *+ * @reference{Part of poly_tomsg() in the reference implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo MLKEM_Q to be compressed.+ *+ * @return Compressed value.+ */+static MLK_INLINE uint8_t mlk_scalar_compress_d1(int16_t u)+__contract__(+ requires(0 <= u && u <= MLKEM_Q - 1)+ ensures(return_value < 2)+ ensures(return_value == (((uint32_t)u * 2 + MLKEM_Q / 2) / MLKEM_Q) % 2) )+{+ /* Compute as follows:+ * ```+ * round(u * 2 / MLKEM_Q)+ * = round(u * 2 * (2^31 / MLKEM_Q) / 2^31)+ * ~= round(u * 2 * round(2^31 / MLKEM_Q) / 2^31)+ * ```+ */+ /* check-magic: 1290168 == 2*round(2^31 / MLKEM_Q) */+ uint32_t d0 = (uint32_t)u * 1290168;+ /* Unsigned shifting by 31 positions leaves only the top bit. */+ return (uint8_t)((d0 + ((uint32_t)1u << 30)) >> 31);+}+#ifdef CBMC+#pragma CPROVER check pop+#endif++/*+ * The multiplication in this routine will exceed UINT32_MAX+ * and wrap around for large values of u. This is expected and required.+ */+#ifdef CBMC+#pragma CPROVER check push+#pragma CPROVER check disable "unsigned-overflow"+#endif++/**+ * Compute round(u * 16 / MLKEM_Q) % 16.+ *+ * @spec{Compress_4 from @[FIPS203, Eq (4.7)].}+ *+ * @reference{Embedded into `poly_compress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo MLKEM_Q to be compressed.+ *+ * @return Compressed value.+ */+static MLK_INLINE uint8_t mlk_scalar_compress_d4(int16_t u)+__contract__(+ requires(0 <= u && u <= MLKEM_Q - 1)+ ensures(return_value < 16)+ ensures(return_value == (((uint32_t)u * 16 + MLKEM_Q / 2) / MLKEM_Q) % 16))+{+ /* Compute as follows:+ * ```+ * round(u * 16 / MLKEM_Q)+ * = round(u * 16 * (2^28 / MLKEM_Q) / 2^28)+ * ~= round(u * 16 * round(2^28 / MLKEM_Q) / 2^28)+ * ```+ */+ /* check-magic: 1290160 == 16 * round(2^28 / MLKEM_Q) */+ uint32_t d0 = (uint32_t)u * 1290160;+ /* The return value is < 16, so not altered by the conversion to uint8_t. */+ return (uint8_t)((d0 + ((uint32_t)1u << 27)) >> 28); /* round(d0/2^28) */+}+#ifdef CBMC+#pragma CPROVER check pop+#endif++/**+ * Compute round(u * MLKEM_Q / 16).+ *+ * @spec{Decompress_4 from @[FIPS203, Eq (4.8)].}+ *+ * @reference{Embedded into `poly_decompress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo 16 to be decompressed.+ *+ * @return Decompressed value.+ */+static MLK_INLINE int16_t mlk_scalar_decompress_d4(uint8_t u)+__contract__(+ requires(0 <= u && u < 16)+ ensures(return_value <= (MLKEM_Q - 1))+)+{+ /* The return value is in 0..MLKEM_Q-1, hence not altered by the+ * conversion to int16_t. */+ return (int16_t)((((uint32_t)u * MLKEM_Q) + 8) >> 4);+}++/*+ * The multiplication in this routine will exceed UINT32_MAX+ * and wrap around for large values of u. This is expected and required.+ */+#ifdef CBMC+#pragma CPROVER check push+#pragma CPROVER check disable "unsigned-overflow"+#endif++/**+ * Compute round(u * 32 / MLKEM_Q) % 32.+ *+ * @spec{Compress_5 from @[FIPS203, Eq (4.7)].}+ *+ * @reference{Embedded into `poly_compress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo MLKEM_Q to be compressed.+ *+ * @return Compressed value.+ */+static MLK_INLINE uint8_t mlk_scalar_compress_d5(int16_t u)+__contract__(+ requires(0 <= u && u <= MLKEM_Q - 1)+ ensures(return_value < 32)+ ensures(return_value == (((uint32_t)u * 32 + MLKEM_Q / 2) / MLKEM_Q) % 32) )+{+ /* Compute as follows:+ * ```+ * round(u * 32 / MLKEM_Q)+ * = round(u * 32 * (2^27 / MLKEM_Q) / 2^27)+ * ~= round(u * 32 * round(2^27 / MLKEM_Q) / 2^27)+ * ```+ */+ /* check-magic: 1290176 == 2^5 * round(2^27 / MLKEM_Q) */+ uint32_t d0 = (uint32_t)u * 1290176;+ /* The return value is < 32, so not altered by the conversion to uint8_t. */+ return (uint8_t)((d0 + ((uint32_t)1u << 26)) >> 27); /* round(d0/2^27) */+}+#ifdef CBMC+#pragma CPROVER check pop+#endif++/**+ * Compute round(u * MLKEM_Q / 32).+ *+ * @spec{Decompress_5 from @[FIPS203, Eq (4.8)].}+ *+ * @reference{Embedded into `poly_decompress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo 32 to be decompressed.+ *+ * @return Decompressed value.+ */+static MLK_INLINE int16_t mlk_scalar_decompress_d5(uint8_t u)+__contract__(+ requires(0 <= u && u < 32)+ ensures(0 <= return_value && return_value <= MLKEM_Q - 1)+)+{+ /* The return value is in 0..MLKEM_Q-1, hence not altered by the+ * conversion to int16_t. */+ return (int16_t)((((uint32_t)u * MLKEM_Q) + 16) >> 5);+}++/*+ * The multiplication in this routine will exceed UINT32_MAX+ * and wrap around for large values of u. This is expected and required.+ */+#ifdef CBMC+#pragma CPROVER check push+#pragma CPROVER check disable "unsigned-overflow"+#endif++/**+ * Compute round(u * 2**10 / MLKEM_Q) % 2**10.+ *+ * @spec{Compress_10 from @[FIPS203, Eq (4.7)].}+ *+ * @reference{Embedded into `polyvec_compress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo MLKEM_Q to be compressed.+ *+ * @return Compressed value.+ */+static MLK_INLINE uint16_t mlk_scalar_compress_d10(int16_t u)+__contract__(+ requires(0 <= u && u <= MLKEM_Q - 1)+ ensures(return_value < (1u << 10))+ ensures(return_value == (((uint32_t)u * (1u << 10) + MLKEM_Q / 2) / MLKEM_Q) % (1 << 10)))+{+ /* Compute as follows:+ * ```+ * round(u * 1024 / MLKEM_Q)+ * = round(u * 1024 * (2^33 / MLKEM_Q) / 2^33)+ * ~= round(u * 1024 * round(2^33 / MLKEM_Q) / 2^33)+ * ```+ */+ /* check-magic: 2642263040 == 2^10 * round(2^33 / MLKEM_Q) */+ uint64_t d0 = (uint64_t)u * 2642263040;+ d0 = (d0 + ((uint64_t)1u << 32)) >> 33; /* round(d0/2^33) */+ return (d0 & 0x3FF);+}+#ifdef CBMC+#pragma CPROVER check pop+#endif++/**+ * Compute round(u * MLKEM_Q / 1024).+ *+ * @spec{Decompress_10 from @[FIPS203, Eq (4.8)].}+ *+ * @reference{Embedded into `polyvec_decompress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo 1024 to be decompressed.+ *+ * @return Decompressed value.+ */+static MLK_INLINE int16_t mlk_scalar_decompress_d10(uint16_t u)+__contract__(+ requires(0 <= u && u < 1024)+ ensures(0 <= return_value && return_value <= (MLKEM_Q - 1))+)+{+ /* The return value is in 0..MLKEM_Q-1, hence not altered by the+ * conversion to int16_t. */+ return (int16_t)((((uint32_t)u * MLKEM_Q) + 512) >> 10);+}++/*+ * The multiplication in this routine will exceed UINT32_MAX+ * and wrap around for large values of u. This is expected and required.+ */+#ifdef CBMC+#pragma CPROVER check push+#pragma CPROVER check disable "unsigned-overflow"+#endif++/**+ * Compute round(u * 2**11 / MLKEM_Q) % 2**11.+ *+ * @spec{Compress_11 from @[FIPS203, Eq (4.7)].}+ *+ * @reference{Embedded into `polyvec_compress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo MLKEM_Q to be compressed.+ *+ * @return Compressed value.+ */+static MLK_INLINE uint16_t mlk_scalar_compress_d11(int16_t u)+__contract__(+ requires(0 <= u && u <= MLKEM_Q - 1)+ ensures(return_value < (1u << 11))+ ensures(return_value == (((uint32_t)u * (1u << 11) + MLKEM_Q / 2) / MLKEM_Q) % (1 << 11)))+{+ /* Compute as follows:+ * ```+ * round(u * 2048 / MLKEM_Q)+ * = round(u * 2048 * (2^33 / MLKEM_Q) / 2^33)+ * ~= round(u * 2048 * round(2^33 / MLKEM_Q) / 2^33)+ * ```+ */+ /* check-magic: 5284526080 == 2^11 * round(2^33 / MLKEM_Q) */+ uint64_t d0 = (uint64_t)u * 5284526080;+ d0 = (d0 + ((uint64_t)1u << 32)) >> 33; /* round(d0/2^33) */+ return (d0 & 0x7FF);+}+#ifdef CBMC+#pragma CPROVER check pop+#endif++/**+ * Compute round(u * MLKEM_Q / 2048).+ *+ * @spec{Decompress_11 from @[FIPS203, Eq (4.8)].}+ *+ * @reference{Embedded into `polyvec_decompress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo 2048 to be decompressed.+ *+ * @return Decompressed value.+ */+static MLK_INLINE int16_t mlk_scalar_decompress_d11(uint16_t u)+__contract__(+ requires(0 <= u && u < 2048)+ ensures(0 <= return_value && return_value <= (MLKEM_Q - 1))+)+{+ /* The return value is in 0..MLKEM_Q-1, hence not altered by the+ * conversion to int16_t. */+ return (int16_t)((((uint32_t)u * MLKEM_Q) + 1024) >> 11);+}++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || (MLKEM_K == 2 || MLKEM_K == 3)+#define mlk_poly_compress_d4 MLK_NAMESPACE(poly_compress_d4)+/**+ * Compression (4 bits) and subsequent serialization of a polynomial.+ *+ * @spec{Implements `ByteEncode_4 (Compress_4 (a))`: ByteEncode_d+ * @[FIPS203, Algorithm 5], Compress_d @[FIPS203, Eq (4.7)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `ByteEncode_{d_v} (Compress_{d_v} (v))` appears in @[FIPS203, Algorithm+ * 14 (K-PKE.Encrypt), L23], where `d_v=4` for ML-KEM-{512,768} @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D4+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ */+MLK_INTERNAL_API+void mlk_poly_compress_d4(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D4],+ const mlk_poly *a);++#define mlk_poly_compress_d10 MLK_NAMESPACE(poly_compress_d10)+/**+ * Compression (10 bits) and subsequent serialization of a polynomial.+ *+ * @spec{Implements `ByteEncode_10 (Compress_10 (a))`: ByteEncode_d+ * @[FIPS203, Algorithm 5], Compress_d @[FIPS203, Eq (4.7)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `ByteEncode_{d_u} (Compress_{d_u} (u))` appears in @[FIPS203, Algorithm+ * 14 (K-PKE.Encrypt), L22], where `d_u=10` for ML-KEM-{512,768} @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D10+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ */+MLK_INTERNAL_API+void mlk_poly_compress_d10(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D10],+ const mlk_poly *a);++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_decompress_d4 MLK_NAMESPACE(poly_decompress_d4)+/**+ * De-serialization and subsequent decompression (4 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d4.+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @spec{Implements `Decompress_4 (ByteDecode_4 (a))`: ByteDecode_d+ * @[FIPS203, Algorithm 6], Decompress_d @[FIPS203, Eq (4.8)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `Decompress_{d_v} (ByteDecode_{d_v} (v))` appears in @[FIPS203, Algorithm+ * 15 (K-PKE.Decrypt), L4], where `d_v=4` for ML-KEM-{512,768} @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D4+ * bytes).+ */+MLK_INTERNAL_API+void mlk_poly_decompress_d4(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D4]);++#define mlk_poly_decompress_d10 MLK_NAMESPACE(poly_decompress_d10)+/**+ * De-serialization and subsequent decompression (10 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d10.+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @spec{Implements `Decompress_10 (ByteDecode_10 (a))`: ByteDecode_d+ * @[FIPS203, Algorithm 6], Decompress_d @[FIPS203, Eq (4.8)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `Decompress_{d_u} (ByteDecode_{d_u} (u))` appears in @[FIPS203, Algorithm+ * 15 (K-PKE.Decrypt), L3], where `d_u=10` for ML-KEM-{512,768} @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D10+ * bytes).+ */+MLK_INTERNAL_API+void mlk_poly_decompress_d10(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D10]);+#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4+#define mlk_poly_compress_d5 MLK_NAMESPACE(poly_compress_d5)+/**+ * Compression (5 bits) and subsequent serialization of a polynomial.+ *+ * @spec{Implements `ByteEncode_5 (Compress_5 (a))`: ByteEncode_d+ * @[FIPS203, Algorithm 5], Compress_d @[FIPS203, Eq (4.7)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `ByteEncode_{d_v} (Compress_{d_v} (v))` appears in @[FIPS203, Algorithm+ * 14 (K-PKE.Encrypt), L23], where `d_v=5` for ML-KEM-1024 @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D5+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ */+MLK_INTERNAL_API+void mlk_poly_compress_d5(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D5],+ const mlk_poly *a);++#define mlk_poly_compress_d11 MLK_NAMESPACE(poly_compress_d11)+/**+ * Compression (11 bits) and subsequent serialization of a polynomial.+ *+ * @spec{`ByteEncode_11 (Compress_11 (a))`: ByteEncode_d @[FIPS203,+ * Algorithm 5], Compress_d @[FIPS203, Eq (4.7)], extended to vectors as+ * per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `ByteEncode_{d_u} (Compress_{d_u} (u))` appears in @[FIPS203, Algorithm+ * 14 (K-PKE.Encrypt), L22], where `d_u=11` for ML-KEM-1024 @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D11+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ */+MLK_INTERNAL_API+void mlk_poly_compress_d11(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D11],+ const mlk_poly *a);++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_decompress_d5 MLK_NAMESPACE(poly_decompress_d5)+/**+ * De-serialization and subsequent decompression (5 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d5.+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @spec{Implements `Decompress_5 (ByteDecode_5 (a))`: ByteDecode_d+ * @[FIPS203, Algorithm 6], Decompress_d @[FIPS203, Eq (4.8)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `Decompress_{d_v} (ByteDecode_{d_v} (v))` appears in @[FIPS203, Algorithm+ * 15 (K-PKE.Decrypt), L4], where `d_v=5` for ML-KEM-1024 @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D5+ * bytes).+ */+MLK_INTERNAL_API+void mlk_poly_decompress_d5(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D5]);++#define mlk_poly_decompress_d11 MLK_NAMESPACE(poly_decompress_d11)+/**+ * De-serialization and subsequent decompression (11 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d11.+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @spec{Implements `Decompress_11 (ByteDecode_11 (a))`: ByteDecode_d+ * @[FIPS203, Algorithm 6], Decompress_d @[FIPS203, Eq (4.8)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `Decompress_{d_u} (ByteDecode_{d_u} (u))` appears in @[FIPS203, Algorithm+ * 15 (K-PKE.Decrypt), L3], where `d_u=11` for ML-KEM-1024 @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D11+ * bytes).+ */+MLK_INTERNAL_API+void mlk_poly_decompress_d11(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D11]);+#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API) || !defined(MLK_CONFIG_NO_ENCAPS_API)+#define mlk_poly_tobytes MLK_NAMESPACE(poly_tobytes)+/**+ * Serialization of a polynomial. Signed coefficients are converted to+ * unsigned form before serialization.+ *+ * @spec{Implements ByteEncode_12 @[FIPS203, Algorithm 5]. Extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].}+ *+ * @param[out] r Output byte array (of MLKEM_POLYBYTES bytes).+ * @param[in] a Input polynomial, with each coefficient in the range+ * [0,1,..,MLKEM_Q-1].+ */+MLK_INTERNAL_API+void mlk_poly_tobytes(uint8_t r[MLKEM_POLYBYTES], const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYBYTES))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYBYTES))+);+#endif /* !MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API */+++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_frombytes MLK_NAMESPACE(poly_frombytes)+/**+ * De-serialization of a polynomial.+ *+ * @spec{Implements ByteDecode_12 @[FIPS203, Algorithm 6]. Extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].}+ *+ * @param[out] r Output polynomial, with each coefficient unsigned and in+ * the range 0..4095.+ * @param[in] a Input byte array (of MLKEM_POLYBYTES bytes).+ */+MLK_INTERNAL_API+void mlk_poly_frombytes(mlk_poly *r, const uint8_t a[MLKEM_POLYBYTES])+__contract__(+ requires(memory_no_alias(a, MLKEM_POLYBYTES))+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT))+);+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */+++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_frommsg MLK_NAMESPACE(poly_frommsg)+/**+ * Convert a 32-byte message to a polynomial.+ *+ * @spec{Implements `Decompress_1 (ByteDecode_1 (a))`: ByteDecode_d+ * @[FIPS203, Algorithm 6], Decompress_d @[FIPS203, Eq (4.8)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `Decompress_1 (ByteDecode_1 (w))` appears in @[FIPS203, Algorithm 15+ * (K-PKE.Encrypt), L20].}+ *+ * @param[out] r Output polynomial.+ * @param[in] msg Input message.+ */+MLK_INTERNAL_API+void mlk_poly_frommsg(mlk_poly *r, const uint8_t msg[MLKEM_INDCPA_MSGBYTES])+__contract__(+ requires(memory_no_alias(msg, MLKEM_INDCPA_MSGBYTES))+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+);+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_tomsg MLK_NAMESPACE(poly_tomsg)+/**+ * Convert a polynomial to a 32-byte message.+ *+ * @spec{Implements `ByteEncode_1 (Compress_1 (a))`: ByteEncode_d+ * @[FIPS203, Algorithm 5], Compress_d @[FIPS203, Eq (4.7)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `ByteEncode_1 (Compress_1 (w))` appears in @[FIPS203, Algorithm 14+ * (K-PKE.Decrypt), L7].}+ *+ * @param[out] msg Output message.+ * @param[in] r Input polynomial. Coefficients must be unsigned canonical.+ */+MLK_INTERNAL_API+void mlk_poly_tomsg(uint8_t msg[MLKEM_INDCPA_MSGBYTES], const mlk_poly *r)+__contract__(+ requires(memory_no_alias(msg, MLKEM_INDCPA_MSGBYTES))+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(msg, MLKEM_INDCPA_MSGBYTES))+);+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#endif /* !MLK_COMPRESS_H */
+ cbits/mlkem/src/context.h view
@@ -0,0 +1,51 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_CONTEXT_H+#define MLK_CONTEXT_H++/* This header is included by common.h once the configuration has been pulled+ * in; it is not meant to be included directly. */++/*+ * If the integration wants to provide a context parameter for use in+ * platform-specific hooks, then it should define this parameter.+ *+ * The MLK_CONTEXT_PARAMETERS_n macros are intended to be used with macros+ * defining the function names and expand to either pass or discard the context+ * argument as required by the current build. If there is no context parameter+ * requested then these are removed from the prototypes and from all calls.+ */+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+#define MLK_CONTEXT_PARAMETERS_0(context) (context)+#define MLK_CONTEXT_PARAMETERS_1(arg0, context) (arg0, context)+#define MLK_CONTEXT_PARAMETERS_2(arg0, arg1, context) (arg0, arg1, context)+#define MLK_CONTEXT_PARAMETERS_3(arg0, arg1, arg2, context) \+ (arg0, arg1, arg2, context)+#define MLK_CONTEXT_PARAMETERS_4(arg0, arg1, arg2, arg3, context) \+ (arg0, arg1, arg2, arg3, context)+#else /* MLK_CONFIG_CONTEXT_PARAMETER */+#define MLK_CONTEXT_PARAMETERS_0(context) ()+#define MLK_CONTEXT_PARAMETERS_1(arg0, context) (arg0)+#define MLK_CONTEXT_PARAMETERS_2(arg0, arg1, context) (arg0, arg1)+#define MLK_CONTEXT_PARAMETERS_3(arg0, arg1, arg2, context) (arg0, arg1, arg2)+#define MLK_CONTEXT_PARAMETERS_4(arg0, arg1, arg2, arg3, context) \+ (arg0, arg1, arg2, arg3)+#endif /* !MLK_CONFIG_CONTEXT_PARAMETER */++/* Consume a context parameter carried only for the integration's benefit,+ * avoiding -Wunused-parameter; expands to nothing when no context is+ * configured. */+#if defined(MLK_CONFIG_CONTEXT_PARAMETER)+#define MLK_CONTEXT_UNUSED(context) ((void)(context))+#else+#define MLK_CONTEXT_UNUSED(context) ((void)0)+#endif++#if defined(MLK_CONFIG_CONTEXT_PARAMETER_TYPE) != \+ defined(MLK_CONFIG_CONTEXT_PARAMETER)+#error MLK_CONFIG_CONTEXT_PARAMETER_TYPE must be defined if and only if MLK_CONFIG_CONTEXT_PARAMETER is defined+#endif++#endif /* !MLK_CONTEXT_H */
+ cbits/mlkem/src/debug.c view
@@ -0,0 +1,64 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* NOTE: You can remove this file unless you compile with MLKEM_DEBUG. */++#include "common.h"++#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && defined(MLKEM_DEBUG)+++#include <stdio.h>+#include <stdlib.h>+#include "debug.h"++#define MLK_DEBUG_ERROR_HEADER "[ERROR:%s:%04d] "++void mlk_debug_check_assert(const char *file, int line, const int val)+{+ if (val == 0)+ {+ fprintf(stderr, MLK_DEBUG_ERROR_HEADER "Assertion failed (value %d)\n",+ file, line, val);+ exit(1);+ }+}++void mlk_debug_check_bounds(const char *file, int line, const int16_t *ptr,+ unsigned len, int lower_bound_exclusive,+ int upper_bound_exclusive)+{+ int err = 0;+ unsigned i;+ for (i = 0; i < len; i++)+ {+ int16_t val = ptr[i];+ if (!(val > lower_bound_exclusive && val < upper_bound_exclusive))+ {+ fprintf(+ stderr,+ MLK_DEBUG_ERROR_HEADER+ "Bounds assertion failed: Index %u, value %d out of bounds (%d,%d)\n",+ file, line, i, (int)val, lower_bound_exclusive,+ upper_bound_exclusive);+ err = 1;+ }+ }++ if (err == 1)+ {+ exit(1);+ }+}++#else /* !MLK_CONFIG_MULTILEVEL_NO_SHARED && MLKEM_DEBUG */++MLK_EMPTY_CU(debug)++#endif /* !(!MLK_CONFIG_MULTILEVEL_NO_SHARED && MLKEM_DEBUG) */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef MLK_DEBUG_ERROR_HEADER
+ cbits/mlkem/src/debug.h view
@@ -0,0 +1,121 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_DEBUG_H+#define MLK_DEBUG_H+#include "common.h"++#if defined(MLKEM_DEBUG)++/**+ * Check debug assertion.+ *+ * Prints an error message to stderr and calls exit(1) on failure.+ *+ * @param[in] file Filename.+ * @param line Line number.+ * @param val Value asserted to be non-zero.+ */+#define mlk_debug_check_assert MLK_NAMESPACE(mlkem_debug_assert)+void mlk_debug_check_assert(const char *file, int line, const int val);++/**+ * Check whether values in an array of int16_t are within specified bounds.+ *+ * Prints an error message to stderr and calls exit(1) on failure.+ *+ * @param[in] file Filename.+ * @param line Line number.+ * @param[in] ptr Base of array to be checked.+ * @param len Number of int16_t in @p ptr.+ * @param lower_bound_exclusive Exclusive lower bound.+ * @param upper_bound_exclusive Exclusive upper bound.+ */+#define mlk_debug_check_bounds MLK_NAMESPACE(mlkem_debug_check_bounds)+void mlk_debug_check_bounds(const char *file, int line, const int16_t *ptr,+ unsigned len, int lower_bound_exclusive,+ int upper_bound_exclusive);++/* Check assertion, calling exit() upon failure+ *+ * val: Value that's asserted to be non-zero+ */+#define mlk_assert(val) mlk_debug_check_assert(__FILE__, __LINE__, (val))++/* Check bounds in array of int16_t's+ * ptr: Base of int16_t array; will be explicitly cast to int16_t*,+ * so you may pass a byte-compatible type such as mlk_poly or mlk_polyvec.+ * len: Number of int16_t in array+ * value_lb: Inclusive lower value bound+ * value_ub: Exclusive upper value bound */+#define mlk_assert_bound(ptr, len, value_lb, value_ub) \+ mlk_debug_check_bounds(__FILE__, __LINE__, (const int16_t *)(ptr), (len), \+ (value_lb) - 1, (value_ub))++/* Check absolute bounds in array of int16_t's+ * ptr: Base of array, expression of type int16_t*+ * len: Number of int16_t in array+ * value_abs_bd: Exclusive absolute upper bound */+#define mlk_assert_abs_bound(ptr, len, value_abs_bd) \+ mlk_assert_bound((ptr), (len), (-(value_abs_bd) + 1), (value_abs_bd))++/* Version of bounds assertions for 2-dimensional arrays */+#define mlk_assert_bound_2d(ptr, len0, len1, value_lb, value_ub) \+ mlk_assert_bound((ptr), ((len0) * (len1)), (value_lb), (value_ub))++#define mlk_assert_abs_bound_2d(ptr, len0, len1, value_abs_bd) \+ mlk_assert_abs_bound((ptr), ((len0) * (len1)), (value_abs_bd))++/* When running CBMC, convert debug assertions into proof obligations */+#elif defined(CBMC)+#include "cbmc.h"++#define mlk_assert(val) cassert(val)++#define mlk_assert_bound(ptr, len, value_lb, value_ub) \+ cassert(array_bound(((int16_t *)(ptr)), 0, (len), (value_lb), (value_ub)))++#define mlk_assert_abs_bound(ptr, len, value_abs_bd) \+ cassert(array_abs_bound(((int16_t *)(ptr)), 0, (len), (value_abs_bd)))++/* Because of https://github.com/diffblue/cbmc/issues/8570, we can't+ * just use a single flattened array_bound(...) here. */+#define mlk_assert_bound_2d(ptr, M, N, value_lb, value_ub) \+ cassert(forall(kN, 0, (M), \+ array_bound(&((int16_t (*)[(N)])(ptr))[kN][0], 0, (N), \+ (value_lb), (value_ub))))++#define mlk_assert_abs_bound_2d(ptr, M, N, value_abs_bd) \+ cassert(forall(kN, 0, (M), \+ array_abs_bound(&((int16_t (*)[(N)])(ptr))[kN][0], 0, (N), \+ (value_abs_bd))))++#else /* !MLKEM_DEBUG && CBMC */++#define mlk_assert(val) \+ do \+ { \+ } while (0)+#define mlk_assert_bound(ptr, len, value_lb, value_ub) \+ do \+ { \+ } while (0)+#define mlk_assert_abs_bound(ptr, len, value_abs_bd) \+ do \+ { \+ } while (0)++#define mlk_assert_bound_2d(ptr, len0, len1, value_lb, value_ub) \+ do \+ { \+ } while (0)++#define mlk_assert_abs_bound_2d(ptr, len0, len1, value_abs_bd) \+ do \+ { \+ } while (0)+++#endif /* !MLKEM_DEBUG && !CBMC */+#endif /* !MLK_DEBUG_H */
+ cbits/mlkem/src/fips202/fips202.c view
@@ -0,0 +1,249 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ *+ * - [mupq]+ * Common files for pqm4, pqm3, pqriscv+ * Kannwischer, Petri, Rijneveld, Schwabe, Stoffelen+ * https://github.com/mupq/mupq+ *+ * - [supercop]+ * SUPERCOP benchmarking framework+ * Daniel J. Bernstein+ * http://bench.cr.yp.to/supercop.html+ *+ * - [tweetfips]+ * 'tweetfips202' FIPS202 implementation+ * Van Assche, Bernstein, Schwabe+ * https://keccak.team/2015/tweetfips202.html+ */++/* Based on the CC0 implementation from @[mupq] and the public domain+ * implementation @[supercop, crypto_hash/keccakc512/simple/]+ * by Ronny Van Keer, and the public domain @[tweetfips] implementation. */++#include "../common.h"+#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)+++#include "../verify.h"+#include "fips202.h"+#include "keccakf1600.h"++/**+ * Absorb step of Keccak; non-incremental, starts by zeroeing the state.+ *+ * @warning Must only be called once.+ *+ * @param[out] s Pointer to (uninitialized) output Keccak state.+ * @param r Rate in bytes (e.g., 168 for SHAKE128).+ * @param[in] m Input to be absorbed into @p s.+ * @param mlen Length of input in bytes.+ * @param p Domain-separation byte for different Keccak-derived+ * functions.+ */+static void mlk_keccak_absorb_once(uint64_t *s, unsigned r, const uint8_t *m,+ size_t mlen, uint8_t p)+__contract__(+ requires(mlen <= MLK_MAX_BUFFER_SIZE)+ requires(r > 0)+ requires(r <= sizeof(uint64_t) * MLK_KECCAK_LANES)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLK_KECCAK_LANES))+ requires(memory_no_alias(m, mlen))+ assigns(memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES)))+{+ /* Initialize state */+ size_t i;+ for (i = 0; i < 25; ++i)+ __loop__(invariant(i <= 25)+ decreases(25 - i))+ {+ s[i] = 0;+ }++ while (mlen >= r)+ __loop__(+ assigns(mlen, m, memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES))+ invariant(mlen <= loop_entry(mlen))+ invariant(m == loop_entry(m) + (loop_entry(mlen) - mlen))+ decreases(mlen))+ {+ mlk_keccakf1600_xor_bytes(s, m, 0, r);+ mlk_keccakf1600_permute(s);+ mlen -= r;+ m += r;+ }++ /* At this point, mlen < r, so the truncations to unsigned are safe below. */++ if (mlen > 0)+ {+ mlk_keccakf1600_xor_bytes(s, m, 0, (unsigned int)mlen);+ }++ if (mlen == r - 1)+ {+ p |= 128;+ mlk_keccakf1600_xor_bytes(s, &p, (unsigned int)mlen, 1);+ }+ else+ {+ mlk_keccakf1600_xor_bytes(s, &p, (unsigned int)mlen, 1);+ p = 128;+ mlk_keccakf1600_xor_bytes(s, &p, r - 1, 1);+ }+}++/**+ * Block-level Keccak squeeze.+ *+ * @param[out] h Output bytes.+ * @param nblocks Number of blocks to be squeezed.+ * @param[in,out] s Input/output state.+ * @param r Rate in bytes (e.g., 168 for SHAKE128).+ */+static void mlk_keccak_squeezeblocks(uint8_t *h, size_t nblocks, uint64_t *s,+ unsigned r)+__contract__(+ requires(r <= sizeof(uint64_t) * MLK_KECCAK_LANES)+ requires(nblocks <= 8 /* somewhat arbitrary bound */)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLK_KECCAK_LANES))+ requires(memory_no_alias(h, nblocks * r))+ assigns(memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES))+ assigns(memory_slice(h, nblocks * r)))+{+ while (nblocks > 0)+ __loop__(+ assigns(h, nblocks,+ memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES),+ memory_slice(h, nblocks * r))+ invariant(nblocks <= loop_entry(nblocks) &&+ h == loop_entry(h) + r * (loop_entry(nblocks) - nblocks))+ decreases(nblocks))+ {+ mlk_keccakf1600_permute(s);+ mlk_keccakf1600_extract_bytes(s, h, 0, r);+ h += r;+ nblocks--;+ }+}++/**+ * Keccak squeeze; can be called on byte-level.+ *+ * @warning Must only be called once.+ *+ * @param[out] h Output bytes.+ * @param outlen Number of bytes to be squeezed.+ * @param[in,out] s Keccak state.+ * @param r Rate in bytes (e.g., 168 for SHAKE128).+ */+static void mlk_keccak_squeeze_once(uint8_t *h, size_t outlen, uint64_t *s,+ unsigned r)+__contract__(+ requires(outlen <= MLK_MAX_BUFFER_SIZE)+ requires(r > 0)+ requires(r <= sizeof(uint64_t) * MLK_KECCAK_LANES)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLK_KECCAK_LANES))+ requires(memory_no_alias(h, outlen))+ assigns(memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES))+ assigns(memory_slice(h, outlen)))+{+ size_t len;+ while (outlen > 0)+ __loop__(+ assigns(len, h, outlen,+ memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES),+ memory_slice(h, outlen))+ invariant(outlen <= loop_entry(outlen) &&+ h == loop_entry(h) + (loop_entry(outlen) - outlen))+ decreases(outlen))+ {+ mlk_keccakf1600_permute(s);++ if (outlen < r)+ {+ len = outlen;+ }+ else+ {+ len = r;+ }+ mlk_keccakf1600_extract_bytes(s, h, 0, (unsigned int)len);+ h += len;+ outlen -= len;+ }+}++void mlk_shake128_absorb_once(mlk_shake128ctx *state, const uint8_t *input,+ size_t inlen)+{+ mlk_keccak_absorb_once(state->ctx, SHAKE128_RATE, input, inlen, 0x1F);+}++void mlk_shake128_squeezeblocks(uint8_t *output, size_t nblocks,+ mlk_shake128ctx *state)+{+ mlk_keccak_squeezeblocks(output, nblocks, state->ctx, SHAKE128_RATE);+}++void mlk_shake128_init(mlk_shake128ctx *state) { (void)state; }+void mlk_shake128_release(mlk_shake128ctx *state)+{+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(state, sizeof(mlk_shake128ctx));+}++typedef mlk_shake128ctx mlk_shake256ctx;+void mlk_shake256(uint8_t *output, size_t outlen, const uint8_t *input,+ size_t inlen)+{+ mlk_shake256ctx state;+ /* Absorb input */+ mlk_keccak_absorb_once(state.ctx, SHAKE256_RATE, input, inlen, 0x1F);+ /* Squeeze output */+ mlk_keccak_squeeze_once(output, outlen, state.ctx, SHAKE256_RATE);+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(&state, sizeof(state));+}++void mlk_sha3_256(uint8_t *output, const uint8_t *input, size_t inlen)+{+ uint64_t ctx[25];+ /* Absorb input */+ mlk_keccak_absorb_once(ctx, SHA3_256_RATE, input, inlen, 0x06);+ /* Squeeze output */+ mlk_keccak_squeeze_once(output, 32, ctx, SHA3_256_RATE);+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(ctx, sizeof(ctx));+}++void mlk_sha3_512(uint8_t *output, const uint8_t *input, size_t inlen)+{+ uint64_t ctx[25];+ /* Absorb input */+ mlk_keccak_absorb_once(ctx, SHA3_512_RATE, input, inlen, 0x06);+ /* Squeeze output */+ mlk_keccak_squeeze_once(output, 64, ctx, SHA3_512_RATE);+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(ctx, sizeof(ctx));+}++#else /* !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(fips202)++#endif /* MLK_CONFIG_MULTILEVEL_NO_SHARED */
+ cbits/mlkem/src/fips202/fips202.h view
@@ -0,0 +1,144 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_FIPS202_FIPS202_H+#define MLK_FIPS202_FIPS202_H++#include "../cbmc.h"+#include "../common.h"++#define SHAKE128_RATE 168+#define SHAKE256_RATE 136+#define SHA3_256_RATE 136+#define SHA3_384_RATE 104+#define SHA3_512_RATE 72++/** Context for the non-incremental SHAKE128 API. */+typedef struct+{+ uint64_t ctx[25]; /**< Keccak state. */+} MLK_ALIGN mlk_shake128ctx;++#define mlk_shake128_absorb_once MLK_NAMESPACE(shake128_absorb_once)+/**+ * One-shot absorb step of the SHAKE128 XOF.+ *+ * For call-sites (in mlkem-native):+ * - This function MUST ONLY be called straight after mlk_shake128_init().+ * - This function MUST ONLY be called once.+ *+ * Consequently, for providers of custom FIPS202 code to be used with+ * mlkem-native:+ * - You may assume that the input context is freshly initialized via+ * mlk_shake128_init().+ * - You may assume that this function is called exactly once.+ *+ * @param[in,out] state SHAKE128 context.+ * @param[in] input Input to be absorbed into the state.+ * @param inlen Length of input in bytes.+ */+void mlk_shake128_absorb_once(mlk_shake128ctx *state, const uint8_t *input,+ size_t inlen)+__contract__(+ requires(inlen <= MLK_MAX_BUFFER_SIZE)+ requires(memory_no_alias(state, sizeof(mlk_shake128ctx)))+ requires(memory_no_alias(input, inlen))+ assigns(memory_slice(state, sizeof(mlk_shake128ctx)))+);++#define mlk_shake128_squeezeblocks MLK_NAMESPACE(shake128_squeezeblocks)+/**+ * Squeeze step of SHAKE128 XOF. Squeezes full blocks of SHAKE128_RATE bytes+ * each. Modifies the state. Can be called multiple times to keep squeezing,+ * i.e., is incremental.+ *+ * @param[out] output Output blocks.+ * @param nblocks Number of blocks to be squeezed (written to output).+ * @param[in,out] state Keccak state.+ */+void mlk_shake128_squeezeblocks(uint8_t *output, size_t nblocks,+ mlk_shake128ctx *state)+__contract__(+ requires(nblocks <= 8 /* somewhat arbitrary bound */)+ requires(memory_no_alias(state, sizeof(mlk_shake128ctx)))+ requires(memory_no_alias(output, nblocks * SHAKE128_RATE))+ assigns(memory_slice(output, nblocks * SHAKE128_RATE), memory_slice(state, sizeof(mlk_shake128ctx)))+);++#define mlk_shake128_init MLK_NAMESPACE(shake128_init)+void mlk_shake128_init(mlk_shake128ctx *state);++#define mlk_shake128_release MLK_NAMESPACE(shake128_release)+void mlk_shake128_release(mlk_shake128ctx *state);++/* One-stop SHAKE256 call. Aliasing between input and+ * output is not permitted */+#define mlk_shake256 MLK_NAMESPACE(shake256)+/**+ * SHAKE256 XOF with non-incremental API.+ *+ * @param[out] output Output buffer.+ * @param outlen Requested output length in bytes.+ * @param[in] input Input buffer.+ * @param inlen Length of input in bytes.+ */+void mlk_shake256(uint8_t *output, size_t outlen, const uint8_t *input,+ size_t inlen)+__contract__(+ requires(inlen <= MLK_MAX_BUFFER_SIZE)+ requires(outlen <= MLK_MAX_BUFFER_SIZE)+ requires(memory_no_alias(input, inlen))+ requires(memory_no_alias(output, outlen))+ assigns(memory_slice(output, outlen))+);++/* One-stop SHA3_256 call. Aliasing between input and+ * output is not permitted */+#define SHA3_256_HASHBYTES 32+#define mlk_sha3_256 MLK_NAMESPACE(sha3_256)+/**+ * SHA3-256 with non-incremental API.+ *+ * @param[out] output Output buffer.+ * @param[in] input Input buffer.+ * @param inlen Length of input in bytes.+ */+void mlk_sha3_256(uint8_t *output, const uint8_t *input, size_t inlen)+__contract__(+ requires(inlen <= MLK_MAX_BUFFER_SIZE)+ requires(memory_no_alias(input, inlen))+ requires(memory_no_alias(output, SHA3_256_HASHBYTES))+ assigns(memory_slice(output, SHA3_256_HASHBYTES))+);++/* One-stop SHA3_512 call. Aliasing between input and+ * output is not permitted */+#define SHA3_512_HASHBYTES 64+#define mlk_sha3_512 MLK_NAMESPACE(sha3_512)+/**+ * SHA3-512 with non-incremental API.+ *+ * @param[out] output Output buffer.+ * @param[in] input Input buffer.+ * @param inlen Length of input in bytes.+ */+void mlk_sha3_512(uint8_t *output, const uint8_t *input, size_t inlen)+__contract__(+ requires(inlen <= MLK_MAX_BUFFER_SIZE)+ requires(memory_no_alias(input, inlen))+ requires(memory_no_alias(output, SHA3_512_HASHBYTES))+ assigns(memory_slice(output, SHA3_512_HASHBYTES))+);++#if !defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202) || \+ !defined(MLK_USE_NATIVE_FIPS202_X4)+/* If you provide your own FIPS-202 implementation where the x4-+ * Keccak-f1600-x4 implementation falls back to 4-fold Keccak-f1600,+ * set this to gain a small speedup. */+#define FIPS202_X4_DEFAULT_IMPLEMENTATION+#endif /* !MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 || !MLK_USE_NATIVE_FIPS202_X4 \+ */+++#endif /* !MLK_FIPS202_FIPS202_H */
+ cbits/mlkem/src/fips202/fips202x4.c view
@@ -0,0 +1,207 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#include "../common.h"+#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#include "../verify.h"+#include "fips202.h"+#include "fips202x4.h"+#include "keccakf1600.h"++typedef mlk_shake128x4ctx mlk_shake256x4_ctx;++static void mlk_keccak_absorb_once_x4(uint64_t *s, unsigned r,+ const uint8_t *in0, const uint8_t *in1,+ const uint8_t *in2, const uint8_t *in3,+ size_t inlen, uint8_t p)+__contract__(+ requires(inlen <= MLK_MAX_BUFFER_SIZE)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ requires(r > 0)+ requires(r <= sizeof(uint64_t) * MLK_KECCAK_LANES)+ requires(memory_no_alias(in0, inlen))+ requires(memory_no_alias(in1, inlen))+ requires(memory_no_alias(in2, inlen))+ requires(memory_no_alias(in3, inlen))+ assigns(memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY)))+{+ while (inlen >= r)+ __loop__(+ assigns(inlen, in0, in1, in2, in3, memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ invariant(inlen <= loop_entry(inlen))+ invariant(in0 == loop_entry(in0) + (loop_entry(inlen) - inlen))+ invariant(in1 == loop_entry(in1) + (loop_entry(inlen) - inlen))+ invariant(in2 == loop_entry(in2) + (loop_entry(inlen) - inlen))+ invariant(in3 == loop_entry(in3) + (loop_entry(inlen) - inlen))+ decreases(inlen))+ {+ mlk_keccakf1600x4_xor_bytes(s, in0, in1, in2, in3, 0, r);+ mlk_keccakf1600x4_permute(s);++ in0 += r;+ in1 += r;+ in2 += r;+ in3 += r;+ inlen -= r;+ }++ /* At this point, inlen < r, so the truncations to unsigned are safe below. */++ if (inlen > 0)+ {+ mlk_keccakf1600x4_xor_bytes(s, in0, in1, in2, in3, 0, (unsigned int)inlen);+ }++ if (inlen == r - 1)+ {+ p |= 128;+ mlk_keccakf1600x4_xor_bytes(s, &p, &p, &p, &p, (unsigned int)inlen, 1);+ }+ else+ {+ mlk_keccakf1600x4_xor_bytes(s, &p, &p, &p, &p, (unsigned int)inlen, 1);+ p = 128;+ mlk_keccakf1600x4_xor_bytes(s, &p, &p, &p, &p, r - 1, 1);+ }+}++static void mlk_keccak_squeezeblocks_x4(uint8_t *out0, uint8_t *out1,+ uint8_t *out2, uint8_t *out3,+ size_t nblocks, uint64_t *s, unsigned r)+__contract__(+ requires(r <= sizeof(uint64_t) * MLK_KECCAK_LANES)+ requires(r == SHAKE128_RATE || r == SHAKE256_RATE)+ requires(nblocks <= (MLK_MAX_BUFFER_SIZE / SHAKE256_RATE))+ requires(memory_no_alias(s, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ requires(memory_no_alias(out0, nblocks * r))+ requires(memory_no_alias(out1, nblocks * r))+ requires(memory_no_alias(out2, nblocks * r))+ requires(memory_no_alias(out3, nblocks * r))+ assigns(memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ assigns(memory_slice(out0, nblocks * r))+ assigns(memory_slice(out1, nblocks * r))+ assigns(memory_slice(out2, nblocks * r))+ assigns(memory_slice(out3, nblocks * r)))+{+ size_t current_offset = 0;+ while (nblocks > 0)+ __loop__(+ assigns(nblocks, current_offset,+ memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY),+ memory_slice(out0, nblocks * r),+ memory_slice(out1, nblocks * r),+ memory_slice(out2, nblocks * r),+ memory_slice(out3, nblocks * r))+ invariant(nblocks <= loop_entry(nblocks))+ invariant(current_offset == (loop_entry(nblocks) - nblocks) * r)+ decreases(nblocks))+ {+ mlk_keccakf1600x4_permute(s);+ mlk_keccakf1600x4_extract_bytes(+ s, &out0[current_offset], &out1[current_offset], &out2[current_offset],+ &out3[current_offset], 0, r);+ current_offset += r;+ nblocks--;+ }+}++void mlk_shake128x4_absorb_once(mlk_shake128x4ctx *state, const uint8_t *in0,+ const uint8_t *in1, const uint8_t *in2,+ const uint8_t *in3, size_t inlen)+{+ mlk_memset(state, 0, sizeof(mlk_shake128x4ctx));+ mlk_keccak_absorb_once_x4(state->ctx, SHAKE128_RATE, in0, in1, in2, in3,+ inlen, 0x1F);+}++void mlk_shake128x4_squeezeblocks(uint8_t *out0, uint8_t *out1, uint8_t *out2,+ uint8_t *out3, size_t nblocks,+ mlk_shake128x4ctx *state)+{+ mlk_keccak_squeezeblocks_x4(out0, out1, out2, out3, nblocks, state->ctx,+ SHAKE128_RATE);+}++void mlk_shake128x4_init(mlk_shake128x4ctx *state) { (void)state; }+void mlk_shake128x4_release(mlk_shake128x4ctx *state)+{+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(state, sizeof(mlk_shake128x4ctx));+}++static void mlk_shake256x4_absorb_once(mlk_shake256x4_ctx *state,+ const uint8_t *in0, const uint8_t *in1,+ const uint8_t *in2, const uint8_t *in3,+ size_t inlen)+{+ mlk_memset(state, 0, sizeof(mlk_shake128x4ctx));+ mlk_keccak_absorb_once_x4(state->ctx, SHAKE256_RATE, in0, in1, in2, in3,+ inlen, 0x1F);+}++static void mlk_shake256x4_squeezeblocks(uint8_t *out0, uint8_t *out1,+ uint8_t *out2, uint8_t *out3,+ size_t nblocks,+ mlk_shake256x4_ctx *state)+{+ mlk_keccak_squeezeblocks_x4(out0, out1, out2, out3, nblocks, state->ctx,+ SHAKE256_RATE);+}++void mlk_shake256x4(uint8_t *out0, uint8_t *out1, uint8_t *out2, uint8_t *out3,+ size_t outlen, const uint8_t *in0, const uint8_t *in1,+ const uint8_t *in2, const uint8_t *in3, size_t inlen)+{+ mlk_shake256x4_ctx statex;+ size_t nblocks = outlen / SHAKE256_RATE;+ uint8_t tmp0[SHAKE256_RATE];+ uint8_t tmp1[SHAKE256_RATE];+ uint8_t tmp2[SHAKE256_RATE];+ uint8_t tmp3[SHAKE256_RATE];++ mlk_shake256x4_absorb_once(&statex, in0, in1, in2, in3, inlen);+ mlk_shake256x4_squeezeblocks(out0, out1, out2, out3, nblocks, &statex);++ out0 += nblocks * SHAKE256_RATE;+ out1 += nblocks * SHAKE256_RATE;+ out2 += nblocks * SHAKE256_RATE;+ out3 += nblocks * SHAKE256_RATE;++ outlen -= nblocks * SHAKE256_RATE;++ if (outlen)+ {+ mlk_shake256x4_squeezeblocks(tmp0, tmp1, tmp2, tmp3, 1, &statex);+ mlk_memcpy(out0, tmp0, outlen);+ mlk_memcpy(out1, tmp1, outlen);+ mlk_memcpy(out2, tmp2, outlen);+ mlk_memcpy(out3, tmp3, outlen);+ }++ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(&statex, sizeof(statex));+ mlk_zeroize(tmp0, sizeof(tmp0));+ mlk_zeroize(tmp1, sizeof(tmp1));+ mlk_zeroize(tmp2, sizeof(tmp2));+ mlk_zeroize(tmp3, sizeof(tmp3));+}++#else /* !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(fips202x4)++#endif /* MLK_CONFIG_MULTILEVEL_NO_SHARED */
+ cbits/mlkem/src/fips202/fips202x4.h view
@@ -0,0 +1,81 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_FIPS202_FIPS202X4_H+#define MLK_FIPS202_FIPS202X4_H+++#include "../cbmc.h"+#include "../common.h"++#include "fips202.h"+#include "keccakf1600.h"++/** Context for the non-incremental 4-way SHAKE128 API. */+typedef struct+{+ uint64_t ctx[MLK_KECCAK_LANES *+ MLK_KECCAK_WAY]; /**< 4-way Keccak state, stored sequentially. */+} MLK_ALIGN mlk_shake128x4ctx;++#define mlk_shake128x4_absorb_once MLK_NAMESPACE(shake128x4_absorb_once)+void mlk_shake128x4_absorb_once(mlk_shake128x4ctx *state, const uint8_t *in0,+ const uint8_t *in1, const uint8_t *in2,+ const uint8_t *in3, size_t inlen)+__contract__(+ requires(inlen <= MLK_MAX_BUFFER_SIZE)+ requires(memory_no_alias(state, sizeof(mlk_shake128x4ctx)))+ requires(memory_no_alias(in0, inlen))+ requires(memory_no_alias(in1, inlen))+ requires(memory_no_alias(in2, inlen))+ requires(memory_no_alias(in3, inlen))+ assigns(memory_slice(state, sizeof(mlk_shake128x4ctx)))+);++#define mlk_shake128x4_squeezeblocks MLK_NAMESPACE(shake128x4_squeezeblocks)+void mlk_shake128x4_squeezeblocks(uint8_t *out0, uint8_t *out1, uint8_t *out2,+ uint8_t *out3, size_t nblocks,+ mlk_shake128x4ctx *state)+__contract__(+ requires(nblocks <= 8 /* somewhat arbitrary bound */)+ requires(memory_no_alias(state, sizeof(mlk_shake128x4ctx)))+ requires(memory_no_alias(out0, nblocks * SHAKE128_RATE))+ requires(memory_no_alias(out1, nblocks * SHAKE128_RATE))+ requires(memory_no_alias(out2, nblocks * SHAKE128_RATE))+ requires(memory_no_alias(out3, nblocks * SHAKE128_RATE))+ assigns(memory_slice(out0, nblocks * SHAKE128_RATE),+ memory_slice(out1, nblocks * SHAKE128_RATE),+ memory_slice(out2, nblocks * SHAKE128_RATE),+ memory_slice(out3, nblocks * SHAKE128_RATE),+ memory_slice(state, sizeof(mlk_shake128x4ctx)))+);++#define mlk_shake128x4_init MLK_NAMESPACE(shake128x4_init)+void mlk_shake128x4_init(mlk_shake128x4ctx *state);++#define mlk_shake128x4_release MLK_NAMESPACE(shake128x4_release)+void mlk_shake128x4_release(mlk_shake128x4ctx *state);++#define mlk_shake256x4 MLK_NAMESPACE(shake256x4)+void mlk_shake256x4(uint8_t *out0, uint8_t *out1, uint8_t *out2, uint8_t *out3,+ size_t outlen, const uint8_t *in0, const uint8_t *in1,+ const uint8_t *in2, const uint8_t *in3, size_t inlen)+__contract__(+ requires(inlen <= MLK_MAX_BUFFER_SIZE)+ requires(outlen <= MLK_MAX_BUFFER_SIZE)+ requires(memory_no_alias(in0, inlen))+ requires(memory_no_alias(in1, inlen))+ requires(memory_no_alias(in2, inlen))+ requires(memory_no_alias(in3, inlen))+ requires(memory_no_alias(out0, outlen))+ requires(memory_no_alias(out1, outlen))+ requires(memory_no_alias(out2, outlen))+ requires(memory_no_alias(out3, outlen))+ assigns(memory_slice(out0, outlen))+ assigns(memory_slice(out1, outlen))+ assigns(memory_slice(out2, outlen))+ assigns(memory_slice(out3, outlen))+);++#endif /* !MLK_FIPS202_FIPS202X4_H */
+ cbits/mlkem/src/fips202/keccakf1600.c view
@@ -0,0 +1,499 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [mupq]+ * Common files for pqm4, pqm3, pqriscv+ * Kannwischer, Petri, Rijneveld, Schwabe, Stoffelen+ * https://github.com/mupq/mupq+ *+ * - [supercop]+ * SUPERCOP benchmarking framework+ * Daniel J. Bernstein+ * http://bench.cr.yp.to/supercop.html+ *+ * - [tweetfips]+ * 'tweetfips202' FIPS202 implementation+ * Van Assche, Bernstein, Schwabe+ * https://keccak.team/2015/tweetfips202.html+ */++/* Based on the CC0 implementation from @[mupq] and the public domain+ * implementation @[supercop, crypto_hash/keccakc512/simple/]+ * by Ronny Van Keer, and the public domain @[tweetfips] implementation. */+++#include "keccakf1600.h"+#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#define MLK_KECCAK_NROUNDS 24+#define MLK_KECCAK_ROL(a, offset) (((a) << (offset)) ^ ((a) >> (64 - (offset))))++void mlk_keccakf1600_extract_bytes(uint64_t *state, unsigned char *data,+ unsigned offset, unsigned length)+{+ unsigned i;+#if defined(MLK_SYS_LITTLE_ENDIAN)+ uint8_t *state_ptr = (uint8_t *)state + offset;+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ data[i] = state_ptr[i];+ }+#else /* MLK_SYS_LITTLE_ENDIAN */+ /* Portable version */+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ data[i] = (state[(offset + i) >> 3] >> (8 * ((offset + i) & 0x07))) & 0xFF;+ }+#endif /* !MLK_SYS_LITTLE_ENDIAN */+}++void mlk_keccakf1600_xor_bytes(uint64_t *state, const unsigned char *data,+ unsigned offset, unsigned length)+{+ unsigned i;+#if defined(MLK_SYS_LITTLE_ENDIAN)+ uint8_t *state_ptr = (uint8_t *)state + offset;+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ state_ptr[i] ^= data[i];+ }+#else /* MLK_SYS_LITTLE_ENDIAN */+ /* Portable version */+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ state[(offset + i) >> 3] ^= (uint64_t)data[i]+ << (8 * ((offset + i) & 0x07));+ }+#endif /* !MLK_SYS_LITTLE_ENDIAN */+}++static void mlk_keccakf1600x4_extract_bytes_c(uint64_t *state,+ unsigned char *data0,+ unsigned char *data1,+ unsigned char *data2,+ unsigned char *data3,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLK_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLK_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ requires(memory_no_alias(data1, length))+ requires(memory_no_alias(data2, length))+ requires(memory_no_alias(data3, length))+ assigns(memory_slice(data0, length))+ assigns(memory_slice(data1, length))+ assigns(memory_slice(data2, length))+ assigns(memory_slice(data3, length))+)+{+ mlk_keccakf1600_extract_bytes(state + MLK_KECCAK_LANES * 0, data0, offset,+ length);+ mlk_keccakf1600_extract_bytes(state + MLK_KECCAK_LANES * 1, data1, offset,+ length);+ mlk_keccakf1600_extract_bytes(state + MLK_KECCAK_LANES * 2, data2, offset,+ length);+ mlk_keccakf1600_extract_bytes(state + MLK_KECCAK_LANES * 3, data3, offset,+ length);+}++void mlk_keccakf1600x4_extract_bytes(uint64_t *state, unsigned char *data0,+ unsigned char *data1, unsigned char *data2,+ unsigned char *data3, unsigned offset,+ unsigned length)+{+#if defined(MLK_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES)+ if (mlk_keccakf1600_extract_bytes_x4_native(state, data0, data1, data2, data3,+ offset, length) ==+ MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES */+ mlk_keccakf1600x4_extract_bytes_c(state, data0, data1, data2, data3, offset,+ length);+}++static void mlk_keccakf1600x4_xor_bytes_c(uint64_t *state,+ const unsigned char *data0,+ const unsigned char *data1,+ const unsigned char *data2,+ const unsigned char *data3,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLK_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLK_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ /* Case 1: all input buffers are distinct; Case 2: All input buffers are the same */+ requires((data0 == data1 &&+ data0 == data2 &&+ data0 == data3) ||+ (memory_no_alias(data1, length) &&+ memory_no_alias(data2, length) &&+ memory_no_alias(data3, length)))+ assigns(memory_slice(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+)+{+ mlk_keccakf1600_xor_bytes(state + MLK_KECCAK_LANES * 0, data0, offset,+ length);+ mlk_keccakf1600_xor_bytes(state + MLK_KECCAK_LANES * 1, data1, offset,+ length);+ mlk_keccakf1600_xor_bytes(state + MLK_KECCAK_LANES * 2, data2, offset,+ length);+ mlk_keccakf1600_xor_bytes(state + MLK_KECCAK_LANES * 3, data3, offset,+ length);+}++void mlk_keccakf1600x4_xor_bytes(uint64_t *state, const unsigned char *data0,+ const unsigned char *data1,+ const unsigned char *data2,+ const unsigned char *data3, unsigned offset,+ unsigned length)+{+#if defined(MLK_USE_NATIVE_FIPS202_X4_XOR_BYTES)+ if (mlk_keccakf1600_xor_bytes_x4_native(state, data0, data1, data2, data3,+ offset,+ length) == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_FIPS202_X4_XOR_BYTES */+ mlk_keccakf1600x4_xor_bytes_c(state, data0, data1, data2, data3, offset,+ length);+}++void mlk_keccakf1600x4_permute(uint64_t *state)+{+#if defined(MLK_USE_NATIVE_FIPS202_X4)+ if (mlk_keccak_f1600_x4_native(state) == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_FIPS202_X4 */+ mlk_keccakf1600_permute(state + MLK_KECCAK_LANES * 0);+ mlk_keccakf1600_permute(state + MLK_KECCAK_LANES * 1);+ mlk_keccakf1600_permute(state + MLK_KECCAK_LANES * 2);+ mlk_keccakf1600_permute(state + MLK_KECCAK_LANES * 3);+}++static const uint64_t mlk_KeccakF_RoundConstants[MLK_KECCAK_NROUNDS] = {+ (uint64_t)0x0000000000000001ULL, (uint64_t)0x0000000000008082ULL,+ (uint64_t)0x800000000000808aULL, (uint64_t)0x8000000080008000ULL,+ (uint64_t)0x000000000000808bULL, (uint64_t)0x0000000080000001ULL,+ (uint64_t)0x8000000080008081ULL, (uint64_t)0x8000000000008009ULL,+ (uint64_t)0x000000000000008aULL, (uint64_t)0x0000000000000088ULL,+ (uint64_t)0x0000000080008009ULL, (uint64_t)0x000000008000000aULL,+ (uint64_t)0x000000008000808bULL, (uint64_t)0x800000000000008bULL,+ (uint64_t)0x8000000000008089ULL, (uint64_t)0x8000000000008003ULL,+ (uint64_t)0x8000000000008002ULL, (uint64_t)0x8000000000000080ULL,+ (uint64_t)0x000000000000800aULL, (uint64_t)0x800000008000000aULL,+ (uint64_t)0x8000000080008081ULL, (uint64_t)0x8000000000008080ULL,+ (uint64_t)0x0000000080000001ULL, (uint64_t)0x8000000080008008ULL};++MLK_STATIC_TESTABLE+void mlk_keccakf1600_permute_c(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES))+ assigns(memory_slice(state, sizeof(uint64_t) * MLK_KECCAK_LANES))+)+{+ unsigned round;++ uint64_t Aba, Abe, Abi, Abo, Abu;+ uint64_t Aga, Age, Agi, Ago, Agu;+ uint64_t Aka, Ake, Aki, Ako, Aku;+ uint64_t Ama, Ame, Ami, Amo, Amu;+ uint64_t Asa, Ase, Asi, Aso, Asu;+ uint64_t BCa, BCe, BCi, BCo, BCu;+ uint64_t Da, De, Di, Do, Du;+ uint64_t Eba, Ebe, Ebi, Ebo, Ebu;+ uint64_t Ega, Ege, Egi, Ego, Egu;+ uint64_t Eka, Eke, Eki, Eko, Eku;+ uint64_t Ema, Eme, Emi, Emo, Emu;+ uint64_t Esa, Ese, Esi, Eso, Esu;++ /* copyFromState(A, state) */+ Aba = state[0];+ Abe = state[1];+ Abi = state[2];+ Abo = state[3];+ Abu = state[4];+ Aga = state[5];+ Age = state[6];+ Agi = state[7];+ Ago = state[8];+ Agu = state[9];+ Aka = state[10];+ Ake = state[11];+ Aki = state[12];+ Ako = state[13];+ Aku = state[14];+ Ama = state[15];+ Ame = state[16];+ Ami = state[17];+ Amo = state[18];+ Amu = state[19];+ Asa = state[20];+ Ase = state[21];+ Asi = state[22];+ Aso = state[23];+ Asu = state[24];++ for (round = 0; round < MLK_KECCAK_NROUNDS; round += 2)+ __loop__(invariant(round <= MLK_KECCAK_NROUNDS && round % 2 == 0)+ decreases(MLK_KECCAK_NROUNDS - round))+ {+ /* prepareTheta */+ BCa = Aba ^ Aga ^ Aka ^ Ama ^ Asa;+ BCe = Abe ^ Age ^ Ake ^ Ame ^ Ase;+ BCi = Abi ^ Agi ^ Aki ^ Ami ^ Asi;+ BCo = Abo ^ Ago ^ Ako ^ Amo ^ Aso;+ BCu = Abu ^ Agu ^ Aku ^ Amu ^ Asu;++ /* thetaRhoPiChiIotaPrepareTheta(round, A, E) */+ Da = BCu ^ MLK_KECCAK_ROL(BCe, 1);+ De = BCa ^ MLK_KECCAK_ROL(BCi, 1);+ Di = BCe ^ MLK_KECCAK_ROL(BCo, 1);+ Do = BCi ^ MLK_KECCAK_ROL(BCu, 1);+ Du = BCo ^ MLK_KECCAK_ROL(BCa, 1);++ Aba ^= Da;+ BCa = Aba;+ Age ^= De;+ BCe = MLK_KECCAK_ROL(Age, 44);+ Aki ^= Di;+ BCi = MLK_KECCAK_ROL(Aki, 43);+ Amo ^= Do;+ BCo = MLK_KECCAK_ROL(Amo, 21);+ Asu ^= Du;+ BCu = MLK_KECCAK_ROL(Asu, 14);+ Eba = BCa ^ ((~BCe) & BCi);+ Eba ^= (uint64_t)mlk_KeccakF_RoundConstants[round];+ Ebe = BCe ^ ((~BCi) & BCo);+ Ebi = BCi ^ ((~BCo) & BCu);+ Ebo = BCo ^ ((~BCu) & BCa);+ Ebu = BCu ^ ((~BCa) & BCe);++ Abo ^= Do;+ BCa = MLK_KECCAK_ROL(Abo, 28);+ Agu ^= Du;+ BCe = MLK_KECCAK_ROL(Agu, 20);+ Aka ^= Da;+ BCi = MLK_KECCAK_ROL(Aka, 3);+ Ame ^= De;+ BCo = MLK_KECCAK_ROL(Ame, 45);+ Asi ^= Di;+ BCu = MLK_KECCAK_ROL(Asi, 61);+ Ega = BCa ^ ((~BCe) & BCi);+ Ege = BCe ^ ((~BCi) & BCo);+ Egi = BCi ^ ((~BCo) & BCu);+ Ego = BCo ^ ((~BCu) & BCa);+ Egu = BCu ^ ((~BCa) & BCe);++ Abe ^= De;+ BCa = MLK_KECCAK_ROL(Abe, 1);+ Agi ^= Di;+ BCe = MLK_KECCAK_ROL(Agi, 6);+ Ako ^= Do;+ BCi = MLK_KECCAK_ROL(Ako, 25);+ Amu ^= Du;+ BCo = MLK_KECCAK_ROL(Amu, 8);+ Asa ^= Da;+ BCu = MLK_KECCAK_ROL(Asa, 18);+ Eka = BCa ^ ((~BCe) & BCi);+ Eke = BCe ^ ((~BCi) & BCo);+ Eki = BCi ^ ((~BCo) & BCu);+ Eko = BCo ^ ((~BCu) & BCa);+ Eku = BCu ^ ((~BCa) & BCe);++ Abu ^= Du;+ BCa = MLK_KECCAK_ROL(Abu, 27);+ Aga ^= Da;+ BCe = MLK_KECCAK_ROL(Aga, 36);+ Ake ^= De;+ BCi = MLK_KECCAK_ROL(Ake, 10);+ Ami ^= Di;+ BCo = MLK_KECCAK_ROL(Ami, 15);+ Aso ^= Do;+ BCu = MLK_KECCAK_ROL(Aso, 56);+ Ema = BCa ^ ((~BCe) & BCi);+ Eme = BCe ^ ((~BCi) & BCo);+ Emi = BCi ^ ((~BCo) & BCu);+ Emo = BCo ^ ((~BCu) & BCa);+ Emu = BCu ^ ((~BCa) & BCe);++ Abi ^= Di;+ BCa = MLK_KECCAK_ROL(Abi, 62);+ Ago ^= Do;+ BCe = MLK_KECCAK_ROL(Ago, 55);+ Aku ^= Du;+ BCi = MLK_KECCAK_ROL(Aku, 39);+ Ama ^= Da;+ BCo = MLK_KECCAK_ROL(Ama, 41);+ Ase ^= De;+ BCu = MLK_KECCAK_ROL(Ase, 2);+ Esa = BCa ^ ((~BCe) & BCi);+ Ese = BCe ^ ((~BCi) & BCo);+ Esi = BCi ^ ((~BCo) & BCu);+ Eso = BCo ^ ((~BCu) & BCa);+ Esu = BCu ^ ((~BCa) & BCe);++ /* prepareTheta */+ BCa = Eba ^ Ega ^ Eka ^ Ema ^ Esa;+ BCe = Ebe ^ Ege ^ Eke ^ Eme ^ Ese;+ BCi = Ebi ^ Egi ^ Eki ^ Emi ^ Esi;+ BCo = Ebo ^ Ego ^ Eko ^ Emo ^ Eso;+ BCu = Ebu ^ Egu ^ Eku ^ Emu ^ Esu;++ /* thetaRhoPiChiIotaPrepareTheta(round+1, E, A) */+ Da = BCu ^ MLK_KECCAK_ROL(BCe, 1);+ De = BCa ^ MLK_KECCAK_ROL(BCi, 1);+ Di = BCe ^ MLK_KECCAK_ROL(BCo, 1);+ Do = BCi ^ MLK_KECCAK_ROL(BCu, 1);+ Du = BCo ^ MLK_KECCAK_ROL(BCa, 1);++ Eba ^= Da;+ BCa = Eba;+ Ege ^= De;+ BCe = MLK_KECCAK_ROL(Ege, 44);+ Eki ^= Di;+ BCi = MLK_KECCAK_ROL(Eki, 43);+ Emo ^= Do;+ BCo = MLK_KECCAK_ROL(Emo, 21);+ Esu ^= Du;+ BCu = MLK_KECCAK_ROL(Esu, 14);+ Aba = BCa ^ ((~BCe) & BCi);+ Aba ^= (uint64_t)mlk_KeccakF_RoundConstants[round + 1];+ Abe = BCe ^ ((~BCi) & BCo);+ Abi = BCi ^ ((~BCo) & BCu);+ Abo = BCo ^ ((~BCu) & BCa);+ Abu = BCu ^ ((~BCa) & BCe);++ Ebo ^= Do;+ BCa = MLK_KECCAK_ROL(Ebo, 28);+ Egu ^= Du;+ BCe = MLK_KECCAK_ROL(Egu, 20);+ Eka ^= Da;+ BCi = MLK_KECCAK_ROL(Eka, 3);+ Eme ^= De;+ BCo = MLK_KECCAK_ROL(Eme, 45);+ Esi ^= Di;+ BCu = MLK_KECCAK_ROL(Esi, 61);+ Aga = BCa ^ ((~BCe) & BCi);+ Age = BCe ^ ((~BCi) & BCo);+ Agi = BCi ^ ((~BCo) & BCu);+ Ago = BCo ^ ((~BCu) & BCa);+ Agu = BCu ^ ((~BCa) & BCe);++ Ebe ^= De;+ BCa = MLK_KECCAK_ROL(Ebe, 1);+ Egi ^= Di;+ BCe = MLK_KECCAK_ROL(Egi, 6);+ Eko ^= Do;+ BCi = MLK_KECCAK_ROL(Eko, 25);+ Emu ^= Du;+ BCo = MLK_KECCAK_ROL(Emu, 8);+ Esa ^= Da;+ BCu = MLK_KECCAK_ROL(Esa, 18);+ Aka = BCa ^ ((~BCe) & BCi);+ Ake = BCe ^ ((~BCi) & BCo);+ Aki = BCi ^ ((~BCo) & BCu);+ Ako = BCo ^ ((~BCu) & BCa);+ Aku = BCu ^ ((~BCa) & BCe);++ Ebu ^= Du;+ BCa = MLK_KECCAK_ROL(Ebu, 27);+ Ega ^= Da;+ BCe = MLK_KECCAK_ROL(Ega, 36);+ Eke ^= De;+ BCi = MLK_KECCAK_ROL(Eke, 10);+ Emi ^= Di;+ BCo = MLK_KECCAK_ROL(Emi, 15);+ Eso ^= Do;+ BCu = MLK_KECCAK_ROL(Eso, 56);+ Ama = BCa ^ ((~BCe) & BCi);+ Ame = BCe ^ ((~BCi) & BCo);+ Ami = BCi ^ ((~BCo) & BCu);+ Amo = BCo ^ ((~BCu) & BCa);+ Amu = BCu ^ ((~BCa) & BCe);++ Ebi ^= Di;+ BCa = MLK_KECCAK_ROL(Ebi, 62);+ Ego ^= Do;+ BCe = MLK_KECCAK_ROL(Ego, 55);+ Eku ^= Du;+ BCi = MLK_KECCAK_ROL(Eku, 39);+ Ema ^= Da;+ BCo = MLK_KECCAK_ROL(Ema, 41);+ Ese ^= De;+ BCu = MLK_KECCAK_ROL(Ese, 2);+ Asa = BCa ^ ((~BCe) & BCi);+ Ase = BCe ^ ((~BCi) & BCo);+ Asi = BCi ^ ((~BCo) & BCu);+ Aso = BCo ^ ((~BCu) & BCa);+ Asu = BCu ^ ((~BCa) & BCe);+ }++ /* copyToState(state, A) */+ state[0] = Aba;+ state[1] = Abe;+ state[2] = Abi;+ state[3] = Abo;+ state[4] = Abu;+ state[5] = Aga;+ state[6] = Age;+ state[7] = Agi;+ state[8] = Ago;+ state[9] = Agu;+ state[10] = Aka;+ state[11] = Ake;+ state[12] = Aki;+ state[13] = Ako;+ state[14] = Aku;+ state[15] = Ama;+ state[16] = Ame;+ state[17] = Ami;+ state[18] = Amo;+ state[19] = Amu;+ state[20] = Asa;+ state[21] = Ase;+ state[22] = Asi;+ state[23] = Aso;+ state[24] = Asu;+}++void mlk_keccakf1600_permute(uint64_t *state)+{+#if defined(MLK_USE_NATIVE_FIPS202_X1)+ if (mlk_keccak_f1600_x1_native(state) == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_FIPS202_X1 */+ mlk_keccakf1600_permute_c(state);+}++#else /* !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(keccakf1600)++#endif /* MLK_CONFIG_MULTILEVEL_NO_SHARED */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef MLK_KECCAK_NROUNDS+#undef MLK_KECCAK_ROL
+ cbits/mlkem/src/fips202/keccakf1600.h view
@@ -0,0 +1,98 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_FIPS202_KECCAKF1600_H+#define MLK_FIPS202_KECCAKF1600_H+#include "../cbmc.h"+#include "../common.h"++#define MLK_KECCAK_LANES 25+#define MLK_KECCAK_WAY 4++/*+ * WARNING:+ * The contents of this structure, including the placement+ * and interleaving of Keccak lanes, are IMPLEMENTATION-DEFINED.+ * The struct is only exposed here to allow its construction on the stack.+ */++#define mlk_keccakf1600_extract_bytes MLK_NAMESPACE(keccakf1600_extract_bytes)+void mlk_keccakf1600_extract_bytes(uint64_t *state, unsigned char *data,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLK_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLK_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES))+ requires(memory_no_alias(data, length))+ assigns(memory_slice(data, length))+);++#define mlk_keccakf1600_xor_bytes MLK_NAMESPACE(keccakf1600_xor_bytes)+void mlk_keccakf1600_xor_bytes(uint64_t *state, const unsigned char *data,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLK_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLK_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES))+ requires(memory_no_alias(data, length))+ assigns(memory_slice(state, sizeof(uint64_t) * MLK_KECCAK_LANES))+);++#define mlk_keccakf1600x4_extract_bytes \+ MLK_NAMESPACE(keccakf1600x4_extract_bytes)+void mlk_keccakf1600x4_extract_bytes(uint64_t *state, unsigned char *data0,+ unsigned char *data1, unsigned char *data2,+ unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLK_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLK_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ requires(memory_no_alias(data1, length))+ requires(memory_no_alias(data2, length))+ requires(memory_no_alias(data3, length))+ assigns(memory_slice(data0, length))+ assigns(memory_slice(data1, length))+ assigns(memory_slice(data2, length))+ assigns(memory_slice(data3, length))+);++#define mlk_keccakf1600x4_xor_bytes MLK_NAMESPACE(keccakf1600x4_xor_bytes)+void mlk_keccakf1600x4_xor_bytes(uint64_t *state, const unsigned char *data0,+ const unsigned char *data1,+ const unsigned char *data2,+ const unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLK_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLK_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ /* Case 1: all input buffers are distinct; Case 2: All input buffers are the same */+ requires((data0 == data1 &&+ data0 == data2 &&+ data0 == data3) ||+ (memory_no_alias(data1, length) &&+ memory_no_alias(data2, length) &&+ memory_no_alias(data3, length)))+ assigns(memory_slice(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+);+++#define mlk_keccakf1600x4_permute MLK_NAMESPACE(keccakf1600x4_permute)+void mlk_keccakf1600x4_permute(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ assigns(memory_slice(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+);++#define mlk_keccakf1600_permute MLK_NAMESPACE(keccakf1600_permute)+void mlk_keccakf1600_permute(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES))+ assigns(memory_slice(state, sizeof(uint64_t) * MLK_KECCAK_LANES))+);++#endif /* !MLK_FIPS202_KECCAKF1600_H */
+ cbits/mlkem/src/fips202/native/aarch64/auto.h view
@@ -0,0 +1,78 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [HYBRID]+ * Hybrid scalar/vector implementations of Keccak and SPHINCS+ on AArch64+ * Becker, Kannwischer+ * https://eprint.iacr.org/2022/1243+ */++#ifndef MLK_FIPS202_NATIVE_AARCH64_AUTO_H+#define MLK_FIPS202_NATIVE_AARCH64_AUTO_H+/* Default FIPS202 assembly profile for AArch64 systems */++/*+ * Default logic to decide which implementation to use.+ *+ */++/*+ * Keccak-f1600+ *+ * - On Arm-based Apple CPUs, or if MLK_SYS_AARCH64_FAST_SHA3 is set,+ * we pick a pure Neon implementation.+ * - Otherwise, unless MLK_SYS_AARCH64_SLOW_BARREL_SHIFTER is set,+ * we use lazy-rotation scalar assembly from @[HYBRID].+ * - Otherwise, if MLK_SYS_AARCH64_SLOW_BARREL_SHIFTER is set, we+ * fall back to the standard C implementation.+ */+#if defined(__ARM_FEATURE_SHA3) && \+ (defined(__APPLE__) || defined(MLK_SYS_AARCH64_FAST_SHA3))+#include "x1_v84a.h"+#elif !defined(MLK_SYS_AARCH64_SLOW_BARREL_SHIFTER)+#include "x1_scalar.h"+#endif++/* Batched, SIMD-based Keccak-f1600 implementations. */+#if defined(MLK_SYS_AARCH64_NEON)++/*+ * Keccak-f1600x2/x4+ *+ * The optimal implementation is highly CPU-specific; see @[HYBRID].+ *+ * For now, if v8.4-A is not implemented, we fall back to Keccak-f1600.+ * If v8.4-A is implemented and we are on an Apple CPU or+ * MLK_SYS_AARCH64_FAST_SHA3 is set, we use a plain Neon-based+ * implementation.+ * Otherwise, if v8.4-A is implemented, we use a scalar/Neon/Neon hybrid.+ * The reason for this distinction is that Apple CPUs (and CPUs flagged with+ * MLK_SYS_AARCH64_FAST_SHA3) implement the SHA3 instructions on all SIMD+ * units, while Arm CPUs prior to Cortex-X4 don't, and ordinary Neon+ * instructions are still needed.+ */+#if defined(__ARM_FEATURE_SHA3)+/*+ * For Apple-M cores (and CPUs flagged with MLK_SYS_AARCH64_FAST_SHA3), we+ * use a plain implementation leveraging SHA3 instructions only.+ */+#if defined(__APPLE__) || defined(MLK_SYS_AARCH64_FAST_SHA3)+#include "x2_v84a.h"+#else+#include "x4_v8a_v84a_scalar.h"+#endif++#else /* __ARM_FEATURE_SHA3 */++#include "x4_v8a_scalar.h"++#endif /* !__ARM_FEATURE_SHA3 */++#endif /* MLK_SYS_AARCH64_NEON */++#endif /* !MLK_FIPS202_NATIVE_AARCH64_AUTO_H */
+ cbits/mlkem/src/fips202/native/aarch64/src/fips202_native_aarch64.h view
@@ -0,0 +1,80 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H+#define MLK_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H++#include "../../../../cbmc.h"+#include "../../../../common.h"++#define mlk_keccakf1600_round_constants \+ MLK_NAMESPACE(keccakf1600_round_constants)+MLK_INTERNAL_DATA_DECLARATION const uint64_t+ mlk_keccakf1600_round_constants[24];++#define mlk_keccak_f1600_x1_scalar_aarch64_asm \+ MLK_NAMESPACE(keccak_f1600_x1_scalar_aarch64_asm)+void mlk_keccak_f1600_x1_scalar_aarch64_asm(uint64_t state[25],+ const uint64_t rc[24])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/keccak_f1600_x1_scalar_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 1))+ requires(rc == mlk_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 1))+);++#define mlk_keccak_f1600_x1_v84a_aarch64_asm \+ MLK_NAMESPACE(keccak_f1600_x1_v84a_aarch64_asm)+void mlk_keccak_f1600_x1_v84a_aarch64_asm(uint64_t state[25],+ const uint64_t rc[24])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/keccak_f1600_x1_v84a_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 1))+ requires(rc == mlk_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 1))+);++#define mlk_keccak_f1600_x2_v84a_aarch64_asm \+ MLK_NAMESPACE(keccak_f1600_x2_v84a_aarch64_asm)+void mlk_keccak_f1600_x2_v84a_aarch64_asm(uint64_t state[50],+ const uint64_t rc[24])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/keccak_f1600_x2_v84a_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 2))+ requires(rc == mlk_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 2))+);++#define mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm \+ MLK_NAMESPACE(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)+void mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(uint64_t state[100],+ const uint64_t rc[24])+/* This must be kept in sync with the HOL-Light specification+ * in+ * proofs/hol_light/aarch64/proofs/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.ml+ */+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(rc == mlk_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+);++#define mlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm \+ MLK_NAMESPACE(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)+void mlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm(+ uint64_t state[100], const uint64_t rc[24])+/* This must be kept in sync with the HOL-Light specification+ * in+ * proofs/hol_light/aarch64/proofs/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.ml+ */+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(rc == mlk_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+);++#endif /* !MLK_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H */
+ cbits/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S view
@@ -0,0 +1,377 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++// Author: Hanno Becker <hanno.becker@arm.com>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>++/*yaml+ Name: keccak_f1600_x1_scalar_aarch64_asm+ Description: AArch64 scalar implementation of Keccak-f[1600] permutation for single state+ Signature: void mlk_keccak_f1600_x1_scalar_aarch64_asm(uint64_t state[25], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ x0:+ type: buffer+ size_bytes: 200+ permissions: read/write+ c_parameter: uint64_t state[25]+ description: Keccak state (25 x uint64_t)+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: uint64_t const *rc+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 128+ description: register preservation and temporary storage+*/++#include "../../../../common.h"+#if defined(MLK_FIPS202_AARCH64_NEED_X1_SCALAR) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(keccak_f1600_x1_scalar_aarch64_asm)+MLK_ASM_FN_SYMBOL(keccak_f1600_x1_scalar_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x80+ .cfi_adjust_cfa_offset 0x80+ stp x19, x20, [sp, #0x20]+ .cfi_rel_offset x19, 0x20+ .cfi_rel_offset x20, 0x28+ stp x21, x22, [sp, #0x30]+ .cfi_rel_offset x21, 0x30+ .cfi_rel_offset x22, 0x38+ stp x23, x24, [sp, #0x40]+ .cfi_rel_offset x23, 0x40+ .cfi_rel_offset x24, 0x48+ stp x25, x26, [sp, #0x50]+ .cfi_rel_offset x25, 0x50+ .cfi_rel_offset x26, 0x58+ stp x27, x28, [sp, #0x60]+ .cfi_rel_offset x27, 0x60+ .cfi_rel_offset x28, 0x68+ stp x29, x30, [sp, #0x70]+ .cfi_rel_offset x29, 0x70+ .cfi_rel_offset x30, 0x78++Lmlk_keccak_f1600_x1_scalar_initial:+ mov x26, x1+ str x1, [sp, #0x8]+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ str x0, [sp]+ eor x30, x24, x25+ eor x27, x9, x10+ eor x0, x30, x21+ eor x26, x27, x6+ eor x27, x26, x7+ eor x29, x0, x22+ eor x26, x29, x23+ eor x29, x4, x5+ eor x30, x29, x1+ eor x0, x27, x8+ eor x29, x30, x2+ eor x30, x19, x20+ eor x30, x30, x16+ eor x27, x26, x0, ror #63+ eor x4, x4, x27+ eor x30, x30, x17+ eor x30, x30, x28+ eor x29, x29, x3+ eor x0, x0, x30, ror #63+ eor x30, x30, x29, ror #63+ eor x22, x22, x30+ eor x23, x23, x30+ str x23, [sp, #0x18]+ eor x23, x14, x15+ eor x14, x14, x0+ eor x23, x23, x11+ eor x15, x15, x0+ eor x1, x1, x27+ eor x23, x23, x12+ eor x23, x23, x13+ eor x11, x11, x0+ eor x29, x29, x23, ror #63+ eor x23, x23, x26, ror #63+ eor x26, x13, x0+ eor x13, x28, x23+ eor x28, x24, x30+ eor x24, x16, x23+ eor x16, x21, x30+ eor x21, x25, x30+ eor x30, x19, x23+ eor x19, x20, x23+ eor x20, x17, x23+ eor x17, x12, x0+ eor x0, x2, x27+ eor x2, x6, x29+ eor x6, x8, x29+ bic x8, x28, x13, ror #47+ eor x12, x3, x27+ bic x3, x13, x17, ror #19+ eor x5, x5, x27+ ldr x27, [sp, #0x18]+ bic x25, x17, x2, ror #5+ eor x9, x9, x29+ eor x23, x25, x5, ror #52+ eor x3, x3, x2, ror #24+ eor x8, x8, x17, ror #2+ eor x17, x10, x29+ bic x25, x12, x22, ror #47+ eor x29, x7, x29+ bic x10, x4, x27, ror #2+ bic x7, x5, x28, ror #10+ eor x10, x10, x20, ror #50+ eor x13, x7, x13, ror #57+ bic x7, x2, x5, ror #47+ eor x2, x25, x24, ror #39+ bic x25, x20, x11, ror #57+ bic x5, x17, x4, ror #25+ eor x25, x25, x17, ror #53+ bic x17, x11, x17, ror #60+ eor x28, x7, x28, ror #57+ bic x7, x9, x12, ror #42+ eor x7, x7, x22, ror #25+ bic x22, x22, x24, ror #56+ bic x24, x24, x15, ror #31+ eor x22, x22, x15, ror #23+ bic x20, x27, x20, ror #48+ bic x15, x15, x9, ror #16+ eor x12, x15, x12, ror #58+ eor x15, x5, x27, ror #27+ eor x5, x20, x11, ror #41+ ldr x11, [sp, #0x8]+ eor x20, x17, x4, ror #21+ eor x17, x24, x9, ror #47+ mov x24, #0x1 // =1+ bic x9, x0, x16, ror #9+ str x24, [sp, #0x10]+ bic x24, x29, x1, ror #44+ bic x27, x1, x21, ror #50+ bic x4, x26, x29, ror #63+ eor x1, x1, x4, ror #21+ ldr x11, [x11]+ bic x4, x21, x30, ror #57+ eor x21, x24, x21, ror #30+ eor x24, x9, x19, ror #44+ bic x9, x14, x6, ror #5+ eor x9, x9, x0, ror #43+ bic x0, x6, x0, ror #38+ eor x1, x1, x11+ eor x11, x4, x26, ror #35+ eor x4, x0, x16, ror #47+ bic x0, x16, x19, ror #35+ eor x16, x27, x30, ror #43+ bic x27, x30, x26, ror #42+ bic x26, x19, x14, ror #41+ eor x19, x0, x14, ror #12+ eor x14, x26, x6, ror #46+ eor x6, x27, x29, ror #41++Lmlk_keccak_f1600_x1_scalar_loop:+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor x26, x8, x9, ror #57+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ eor x30, x23, x22, ror #50+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ eor x26, x30, x21, ror #26+ eor x26, x26, x25, ror #15+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ eor x16, x30, x16+ eor x28, x30, x28, ror #63+ str x28, [sp, #0x18]+ eor x29, x29, x17, ror #36+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ eor x27, x28, x27, ror #61+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ eor x20, x26, x3, ror #39+ eor x11, x0, x11, ror #50+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ eor x1, x30, x17, ror #36+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ eor x17, x27, x7, ror #19+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ eor x4, x26, x4, ror #54+ eor x0, x0, x12, ror #3+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ eor x26, x26, x5, ror #25+ eor x2, x7, x16, ror #39+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ eor x7, x7, x22, ror #25+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ eor x30, x27, x6, ror #43+ eor x22, x20, x15, ror #23+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ bic x5, x13, x17, ror #63+ eor x5, x21, x5, ror #21+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ bic x21, x21, x25, ror #50+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ eor x21, x17, x25, ror #30+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x10]+ eor x17, x10, x9, ror #47+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ bic x20, x4, x28, ror #2+ eor x10, x20, x1, ror #50+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ bic x4, x28, x1, ror #48+ bic x1, x1, x11, ror #57+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0x18]+ add x25, x25, #0x1+ str x25, [sp, #0x10]+ cmp x25, #0x17+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ eor x11, x19, x13, ror #35+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ bic x27, x24, x9, ror #47+ bic x19, x23, x3, ror #9+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic x29, x3, x29, ror #35+ eor x13, x13, x9, ror #57+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ bic x14, x14, x8, ror #5+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ b.le Lmlk_keccak_f1600_x1_scalar_loop+ ror x6, x6, #0x2b+ ror x11, x11, #0x32+ ror x21, x21, #0x14+ ror x2, x2, #0x3d+ ror x7, x7, #0x13+ ror x12, x12, #0x3+ ror x17, x17, #0x24+ ror x22, x22, #0x2c+ ror x3, x3, #0x27+ ror x8, x8, #0x38+ ror x13, x13, #0x2e+ ror x28, x28, #0x3f+ ror x23, x23, #0x3a+ ror x4, x4, #0x36+ ror x9, x9, #0x31+ ror x14, x14, #0x8+ ror x19, x19, #0x25+ ror x24, x24, #0x1c+ ror x5, x5, #0x19+ ror x10, x10, #0x17+ ror x15, x15, #0x3e+ ror x20, x20, #0x2+ ror x25, x25, #0x9+ ldr x0, [sp]+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ ldp x19, x20, [sp, #0x20]+ .cfi_restore x19+ .cfi_restore x20+ ldp x21, x22, [sp, #0x30]+ .cfi_restore x21+ .cfi_restore x22+ ldp x23, x24, [sp, #0x40]+ .cfi_restore x23+ .cfi_restore x24+ ldp x25, x26, [sp, #0x50]+ .cfi_restore x25+ .cfi_restore x26+ ldp x27, x28, [sp, #0x60]+ .cfi_restore x27+ .cfi_restore x28+ ldp x29, x30, [sp, #0x70]+ .cfi_restore x29+ .cfi_restore x30+ add sp, sp, #0x80+ .cfi_adjust_cfa_offset -0x80+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(keccak_f1600_x1_scalar_aarch64_asm)++#endif /* MLK_FIPS202_AARCH64_NEED_X1_SCALAR && \+ !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S view
@@ -0,0 +1,206 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [HYBRID]+ * Hybrid scalar/vector implementations of Keccak and SPHINCS+ on AArch64+ * Becker, Kannwischer+ * https://eprint.iacr.org/2022/1243+ */++/*yaml+ Name: keccak_f1600_x1_v84a_aarch64_asm+ Description: AArch64 ARMv8.4-A implementation of Keccak-f[1600] permutation for single state+ Signature: void mlk_keccak_f1600_x1_v84a_aarch64_asm(uint64_t state[25], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON, SHA3]+ x0:+ type: buffer+ size_bytes: 200+ permissions: read/write+ c_parameter: uint64_t state[25]+ description: Keccak state (25 x uint64_t)+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: const uint64_t rc[24]+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 64+ description: register preservation+*/++//+// Author: Hanno Becker <hanno.becker@arm.com>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>+//+// This implementation is essentially from the paper @[HYBRID].+// The only difference is interleaving/deinterleaving of Keccak state+// during load and store, so that the caller need not do this.+//++#include "../../../../common.h"+#if defined(MLK_FIPS202_AARCH64_NEED_X1_V84A) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#if defined(__ARM_FEATURE_SHA3)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(keccak_f1600_x1_v84a_aarch64_asm)+MLK_ASM_FN_SYMBOL(keccak_f1600_x1_v84a_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ ldp d0, d1, [x0]+ ldp d2, d3, [x0, #0x10]+ ldp d4, d5, [x0, #0x20]+ ldp d6, d7, [x0, #0x30]+ ldp d8, d9, [x0, #0x40]+ ldp d10, d11, [x0, #0x50]+ ldp d12, d13, [x0, #0x60]+ ldp d14, d15, [x0, #0x70]+ ldp d16, d17, [x0, #0x80]+ ldp d18, d19, [x0, #0x90]+ ldp d20, d21, [x0, #0xa0]+ ldp d22, d23, [x0, #0xb0]+ ldr d24, [x0, #0xc0]+ mov x2, #0x18 // =24++Lmlk_keccak_f1600_x1_v84a_loop:+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor3 v30.16b, v30.16b, v15.16b, v20.16b+ eor3 v29.16b, v29.16b, v16.16b, v21.16b+ eor3 v28.16b, v28.16b, v17.16b, v22.16b+ eor3 v27.16b, v27.16b, v18.16b, v23.16b+ eor3 v26.16b, v26.16b, v19.16b, v24.16b+ rax1 v25.2d, v30.2d, v28.2d+ rax1 v28.2d, v28.2d, v26.2d+ rax1 v26.2d, v26.2d, v29.2d+ rax1 v29.2d, v29.2d, v27.2d+ rax1 v27.2d, v27.2d, v30.2d+ eor v30.16b, v0.16b, v26.16b+ xar v0.2d, v2.2d, v29.2d, #0x2+ xar v2.2d, v12.2d, v29.2d, #0x15+ xar v12.2d, v13.2d, v28.2d, #0x27+ xar v13.2d, v19.2d, v27.2d, #0x38+ xar v19.2d, v23.2d, v28.2d, #0x8+ xar v23.2d, v15.2d, v26.2d, #0x17+ xar v15.2d, v1.2d, v25.2d, #0x3f+ xar v1.2d, v8.2d, v28.2d, #0x9+ xar v8.2d, v16.2d, v25.2d, #0x13+ xar v16.2d, v7.2d, v29.2d, #0x3a+ xar v7.2d, v10.2d, v26.2d, #0x3d+ xar v10.2d, v3.2d, v28.2d, #0x24+ xar v3.2d, v18.2d, v28.2d, #0x2b+ xar v18.2d, v17.2d, v29.2d, #0x31+ xar v17.2d, v11.2d, v25.2d, #0x36+ xar v11.2d, v9.2d, v27.2d, #0x2c+ xar v9.2d, v22.2d, v29.2d, #0x3+ xar v22.2d, v14.2d, v27.2d, #0x19+ xar v14.2d, v20.2d, v26.2d, #0x2e+ xar v20.2d, v4.2d, v27.2d, #0x25+ xar v4.2d, v24.2d, v27.2d, #0x32+ xar v24.2d, v21.2d, v25.2d, #0x3e+ xar v21.2d, v5.2d, v26.2d, #0x1c+ xar v27.2d, v6.2d, v25.2d, #0x14+ ld1r { v31.2d }, [x1], #8+ bcax v5.16b, v10.16b, v7.16b, v11.16b+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ bcax v7.16b, v7.16b, v9.16b, v8.16b+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ bcax v9.16b, v9.16b, v11.16b, v10.16b+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ bcax v11.16b, v16.16b, v13.16b, v12.16b+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ bcax v13.16b, v13.16b, v15.16b, v14.16b+ bcax v14.16b, v14.16b, v16.16b, v15.16b+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ bcax v16.16b, v21.16b, v18.16b, v17.16b+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bcax v18.16b, v18.16b, v20.16b, v19.16b+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ bcax v20.16b, v0.16b, v22.16b, v1.16b+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ bcax v22.16b, v22.16b, v24.16b, v23.16b+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ bcax v24.16b, v24.16b, v1.16b, v0.16b+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ bcax v1.16b, v27.16b, v3.16b, v2.16b+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bcax v3.16b, v3.16b, v30.16b, v4.16b+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ eor v0.16b, v0.16b, v31.16b+ sub x2, x2, #0x1+ cbnz x2, Lmlk_keccak_f1600_x1_v84a_loop+ stp d0, d1, [x0]+ stp d2, d3, [x0, #0x10]+ stp d4, d5, [x0, #0x20]+ stp d6, d7, [x0, #0x30]+ stp d8, d9, [x0, #0x40]+ stp d10, d11, [x0, #0x50]+ stp d12, d13, [x0, #0x60]+ stp d14, d15, [x0, #0x70]+ stp d16, d17, [x0, #0x80]+ stp d18, d19, [x0, #0x90]+ stp d20, d21, [x0, #0xa0]+ stp d22, d23, [x0, #0xb0]+ str d24, [x0, #0xc0]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(keccak_f1600_x1_v84a_aarch64_asm)++#endif /* __ARM_FEATURE_SHA3 */++#endif /* MLK_FIPS202_AARCH64_NEED_X1_V84A && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S view
@@ -0,0 +1,261 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [HYBRID]+ * Hybrid scalar/vector implementations of Keccak and SPHINCS+ on AArch64+ * Becker, Kannwischer+ * https://eprint.iacr.org/2022/1243+ */++/*yaml+ Name: keccak_f1600_x2_v84a_aarch64_asm+ Description: AArch64 ARMv8.4-A implementation of Keccak-f[1600] permutation for two sequential states+ Signature: void mlk_keccak_f1600_x2_v84a_aarch64_asm(uint64_t state[50], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON, SHA3]+ x0:+ type: buffer+ size_bytes: 400+ permissions: read/write+ c_parameter: uint64_t state[50]+ description: Two sequential Keccak states (state0[25], state1[25])+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: const uint64_t rc[24]+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 64+ description: register preservation+*/++//+// Author: Hanno Becker <hanno.becker@arm.com>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>+//+// This implementation is essentially from the paper @[HYBRID].+// The only difference is interleaving/deinterleaving of Keccak state+// during load and store, so that the caller need not do this.+//++#include "../../../../common.h"+#if defined(MLK_FIPS202_AARCH64_NEED_X2_V84A) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#if defined(__ARM_FEATURE_SHA3)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(keccak_f1600_x2_v84a_aarch64_asm)+MLK_ASM_FN_SYMBOL(keccak_f1600_x2_v84a_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ add x2, x0, #0xc8+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v0.2d, v25.2d, v27.2d+ trn2 v1.2d, v25.2d, v27.2d+ trn1 v2.2d, v26.2d, v28.2d+ trn2 v3.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v4.2d, v25.2d, v27.2d+ trn2 v5.2d, v25.2d, v27.2d+ trn1 v6.2d, v26.2d, v28.2d+ trn2 v7.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v8.2d, v25.2d, v27.2d+ trn2 v9.2d, v25.2d, v27.2d+ trn1 v10.2d, v26.2d, v28.2d+ trn2 v11.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v12.2d, v25.2d, v27.2d+ trn2 v13.2d, v25.2d, v27.2d+ trn1 v14.2d, v26.2d, v28.2d+ trn2 v15.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v16.2d, v25.2d, v27.2d+ trn2 v17.2d, v25.2d, v27.2d+ trn1 v18.2d, v26.2d, v28.2d+ trn2 v19.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v20.2d, v25.2d, v27.2d+ trn2 v21.2d, v25.2d, v27.2d+ trn1 v22.2d, v26.2d, v28.2d+ trn2 v23.2d, v26.2d, v28.2d+ ldr d25, [x0]+ ldr d27, [x2]+ trn1 v24.2d, v25.2d, v27.2d+ mov x2, #0x18 // =24++Lmlk_keccak_f1600_x2_v84a_loop:+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor3 v30.16b, v30.16b, v15.16b, v20.16b+ eor3 v29.16b, v29.16b, v16.16b, v21.16b+ eor3 v28.16b, v28.16b, v17.16b, v22.16b+ eor3 v27.16b, v27.16b, v18.16b, v23.16b+ eor3 v26.16b, v26.16b, v19.16b, v24.16b+ rax1 v25.2d, v30.2d, v28.2d+ rax1 v28.2d, v28.2d, v26.2d+ rax1 v26.2d, v26.2d, v29.2d+ rax1 v29.2d, v29.2d, v27.2d+ rax1 v27.2d, v27.2d, v30.2d+ eor v30.16b, v0.16b, v26.16b+ xar v0.2d, v2.2d, v29.2d, #0x2+ xar v2.2d, v12.2d, v29.2d, #0x15+ xar v12.2d, v13.2d, v28.2d, #0x27+ xar v13.2d, v19.2d, v27.2d, #0x38+ xar v19.2d, v23.2d, v28.2d, #0x8+ xar v23.2d, v15.2d, v26.2d, #0x17+ xar v15.2d, v1.2d, v25.2d, #0x3f+ xar v1.2d, v8.2d, v28.2d, #0x9+ xar v8.2d, v16.2d, v25.2d, #0x13+ xar v16.2d, v7.2d, v29.2d, #0x3a+ xar v7.2d, v10.2d, v26.2d, #0x3d+ xar v10.2d, v3.2d, v28.2d, #0x24+ xar v3.2d, v18.2d, v28.2d, #0x2b+ xar v18.2d, v17.2d, v29.2d, #0x31+ xar v17.2d, v11.2d, v25.2d, #0x36+ xar v11.2d, v9.2d, v27.2d, #0x2c+ xar v9.2d, v22.2d, v29.2d, #0x3+ xar v22.2d, v14.2d, v27.2d, #0x19+ xar v14.2d, v20.2d, v26.2d, #0x2e+ xar v20.2d, v4.2d, v27.2d, #0x25+ xar v4.2d, v24.2d, v27.2d, #0x32+ xar v24.2d, v21.2d, v25.2d, #0x3e+ xar v21.2d, v5.2d, v26.2d, #0x1c+ xar v27.2d, v6.2d, v25.2d, #0x14+ ld1r { v31.2d }, [x1], #8+ bcax v5.16b, v10.16b, v7.16b, v11.16b+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ bcax v7.16b, v7.16b, v9.16b, v8.16b+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ bcax v9.16b, v9.16b, v11.16b, v10.16b+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ bcax v11.16b, v16.16b, v13.16b, v12.16b+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ bcax v13.16b, v13.16b, v15.16b, v14.16b+ bcax v14.16b, v14.16b, v16.16b, v15.16b+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ bcax v16.16b, v21.16b, v18.16b, v17.16b+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bcax v18.16b, v18.16b, v20.16b, v19.16b+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ bcax v20.16b, v0.16b, v22.16b, v1.16b+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ bcax v22.16b, v22.16b, v24.16b, v23.16b+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ bcax v24.16b, v24.16b, v1.16b, v0.16b+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ bcax v1.16b, v27.16b, v3.16b, v2.16b+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bcax v3.16b, v3.16b, v30.16b, v4.16b+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ eor v0.16b, v0.16b, v31.16b+ sub x2, x2, #0x1+ cbnz x2, Lmlk_keccak_f1600_x2_v84a_loop+ sub x0, x0, #0xc0+ add x2, x0, #0xc8+ trn1 v25.2d, v0.2d, v1.2d+ trn1 v26.2d, v2.2d, v3.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v0.2d, v1.2d+ trn2 v28.2d, v2.2d, v3.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v4.2d, v5.2d+ trn1 v26.2d, v6.2d, v7.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v4.2d, v5.2d+ trn2 v28.2d, v6.2d, v7.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v8.2d, v9.2d+ trn1 v26.2d, v10.2d, v11.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v8.2d, v9.2d+ trn2 v28.2d, v10.2d, v11.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v12.2d, v13.2d+ trn1 v26.2d, v14.2d, v15.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v12.2d, v13.2d+ trn2 v28.2d, v14.2d, v15.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v16.2d, v17.2d+ trn1 v26.2d, v18.2d, v19.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v16.2d, v17.2d+ trn2 v28.2d, v18.2d, v19.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v20.2d, v21.2d+ trn1 v26.2d, v22.2d, v23.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v20.2d, v21.2d+ trn2 v28.2d, v22.2d, v23.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ str d24, [x0]+ trn2 v25.2d, v24.2d, v24.2d+ str d25, [x2]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(keccak_f1600_x2_v84a_aarch64_asm)++#endif /* __ARM_FEATURE_SHA3 */++#endif /* MLK_FIPS202_AARCH64_NEED_X2_V84A && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S view
@@ -0,0 +1,1079 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++// Author: Hanno Becker <hannobecker@posteo.de>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>++/*yaml+ Name: keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm+ Description: AArch64 hybrid scalar/vector implementation of Keccak-f[1600] permutation for four sequential states+ Signature: void mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(uint64_t state[100], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 800+ permissions: read/write+ c_parameter: uint64_t state[100]+ description: Four sequential Keccak states (state0[25], state1[25], state2[25], state3[25])+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: const uint64_t rc[24]+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 224+ description: register preservation and temporary storage+*/++#include "../../../../common.h"+#if defined(MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)+MLK_ASM_FN_SYMBOL(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0xe0+ .cfi_adjust_cfa_offset 0xe0+ stp x19, x20, [sp, #0x30]+ .cfi_rel_offset x19, 0x30+ .cfi_rel_offset x20, 0x38+ stp x21, x22, [sp, #0x40]+ .cfi_rel_offset x21, 0x40+ .cfi_rel_offset x22, 0x48+ stp x23, x24, [sp, #0x50]+ .cfi_rel_offset x23, 0x50+ .cfi_rel_offset x24, 0x58+ stp x25, x26, [sp, #0x60]+ .cfi_rel_offset x25, 0x60+ .cfi_rel_offset x26, 0x68+ stp x27, x28, [sp, #0x70]+ .cfi_rel_offset x27, 0x70+ .cfi_rel_offset x28, 0x78+ stp x29, x30, [sp, #0x80]+ .cfi_rel_offset x29, 0x80+ .cfi_rel_offset x30, 0x88+ stp d8, d9, [sp, #0x90]+ .cfi_rel_offset d8, 0x90+ .cfi_rel_offset d9, 0x98+ stp d10, d11, [sp, #0xa0]+ .cfi_rel_offset d10, 0xa0+ .cfi_rel_offset d11, 0xa8+ stp d12, d13, [sp, #0xb0]+ .cfi_rel_offset d12, 0xb0+ .cfi_rel_offset d13, 0xb8+ stp d14, d15, [sp, #0xc0]+ .cfi_rel_offset d14, 0xc0+ .cfi_rel_offset d15, 0xc8+ mov x29, x1+ mov x30, #0x0 // =0+ str x30, [sp, #0x20]+ str x29, [sp, #0x8]+ str x29, [sp, #0x10]+ str x0, [sp]+ add x4, x0, #0xc8+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v0.2d, v25.2d, v27.2d+ trn2 v1.2d, v25.2d, v27.2d+ trn1 v2.2d, v26.2d, v28.2d+ trn2 v3.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v4.2d, v25.2d, v27.2d+ trn2 v5.2d, v25.2d, v27.2d+ trn1 v6.2d, v26.2d, v28.2d+ trn2 v7.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v8.2d, v25.2d, v27.2d+ trn2 v9.2d, v25.2d, v27.2d+ trn1 v10.2d, v26.2d, v28.2d+ trn2 v11.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v12.2d, v25.2d, v27.2d+ trn2 v13.2d, v25.2d, v27.2d+ trn1 v14.2d, v26.2d, v28.2d+ trn2 v15.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v16.2d, v25.2d, v27.2d+ trn2 v17.2d, v25.2d, v27.2d+ trn1 v18.2d, v26.2d, v28.2d+ trn2 v19.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v20.2d, v25.2d, v27.2d+ trn2 v21.2d, v25.2d, v27.2d+ trn1 v22.2d, v26.2d, v28.2d+ trn2 v23.2d, v26.2d, v28.2d+ ldr d25, [x0]+ ldr d27, [x4]+ trn1 v24.2d, v25.2d, v27.2d+ sub x0, x0, #0xc0+ add x0, x0, #0x190+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x190++Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_initial:+ eor x30, x24, x25+ eor x27, x9, x10+ eor v30.16b, v0.16b, v5.16b+ eor v30.16b, v30.16b, v10.16b+ eor x0, x30, x21+ eor v30.16b, v30.16b, v15.16b+ eor x26, x27, x6+ eor x27, x26, x7+ eor v30.16b, v30.16b, v20.16b+ eor x29, x0, x22+ eor v29.16b, v1.16b, v6.16b+ eor x26, x29, x23+ eor v29.16b, v29.16b, v11.16b+ eor x29, x4, x5+ eor x30, x29, x1+ eor v29.16b, v29.16b, v16.16b+ eor x0, x27, x8+ eor v29.16b, v29.16b, v21.16b+ eor x29, x30, x2+ eor v28.16b, v2.16b, v7.16b+ eor x30, x19, x20+ eor x30, x30, x16+ eor v28.16b, v28.16b, v12.16b+ eor x27, x26, x0, ror #63+ eor v28.16b, v28.16b, v17.16b+ eor x4, x4, x27+ eor v28.16b, v28.16b, v22.16b+ eor x30, x30, x17+ eor x30, x30, x28+ eor v27.16b, v3.16b, v8.16b+ eor x29, x29, x3+ eor v27.16b, v27.16b, v13.16b+ eor x0, x0, x30, ror #63+ eor v27.16b, v27.16b, v18.16b+ eor x30, x30, x29, ror #63+ eor x22, x22, x30+ eor v27.16b, v27.16b, v23.16b+ eor x23, x23, x30+ eor v26.16b, v4.16b, v9.16b+ str x23, [sp, #0xd0]+ eor v26.16b, v26.16b, v14.16b+ eor x23, x14, x15+ eor x14, x14, x0+ eor v26.16b, v26.16b, v19.16b+ eor x23, x23, x11+ eor v26.16b, v26.16b, v24.16b+ eor x15, x15, x0+ eor x1, x1, x27+ add v31.2d, v28.2d, v28.2d+ eor x23, x23, x12+ sri v31.2d, v28.2d, #0x3f+ eor x23, x23, x13+ eor v25.16b, v31.16b, v30.16b+ eor x11, x11, x0+ eor x29, x29, x23, ror #63+ add v31.2d, v26.2d, v26.2d+ eor x23, x23, x26, ror #63+ sri v31.2d, v26.2d, #0x3f+ eor x26, x13, x0+ eor v28.16b, v31.16b, v28.16b+ eor x13, x28, x23+ eor x28, x24, x30+ add v31.2d, v29.2d, v29.2d+ eor x24, x16, x23+ sri v31.2d, v29.2d, #0x3f+ eor x16, x21, x30+ eor v26.16b, v31.16b, v26.16b+ eor x21, x25, x30+ eor x30, x19, x23+ add v31.2d, v27.2d, v27.2d+ eor x19, x20, x23+ sri v31.2d, v27.2d, #0x3f+ eor x20, x17, x23+ eor v29.16b, v31.16b, v29.16b+ eor x17, x12, x0+ eor x0, x2, x27+ add v31.2d, v30.2d, v30.2d+ eor x2, x6, x29+ sri v31.2d, v30.2d, #0x3f+ eor x6, x8, x29+ eor v27.16b, v31.16b, v27.16b+ bic x8, x28, x13, ror #47+ eor x12, x3, x27+ eor v30.16b, v0.16b, v26.16b+ bic x3, x13, x17, ror #19+ eor v31.16b, v2.16b, v29.16b+ eor x5, x5, x27+ ldr x27, [sp, #0xd0]+ shl v0.2d, v31.2d, #0x3e+ bic x25, x17, x2, ror #5+ sri v0.2d, v31.2d, #0x2+ eor x9, x9, x29+ eor v31.16b, v12.16b, v29.16b+ eor x23, x25, x5, ror #52+ eor x3, x3, x2, ror #24+ shl v2.2d, v31.2d, #0x2b+ eor x8, x8, x17, ror #2+ sri v2.2d, v31.2d, #0x15+ eor x17, x10, x29+ eor v31.16b, v13.16b, v28.16b+ bic x25, x12, x22, ror #47+ eor x29, x7, x29+ shl v12.2d, v31.2d, #0x19+ bic x10, x4, x27, ror #2+ sri v12.2d, v31.2d, #0x27+ bic x7, x5, x28, ror #10+ eor v31.16b, v19.16b, v27.16b+ eor x10, x10, x20, ror #50+ eor x13, x7, x13, ror #57+ shl v13.2d, v31.2d, #0x8+ bic x7, x2, x5, ror #47+ sri v13.2d, v31.2d, #0x38+ eor x2, x25, x24, ror #39+ eor v31.16b, v23.16b, v28.16b+ bic x25, x20, x11, ror #57+ bic x5, x17, x4, ror #25+ shl v19.2d, v31.2d, #0x38+ eor x25, x25, x17, ror #53+ sri v19.2d, v31.2d, #0x8+ bic x17, x11, x17, ror #60+ eor v31.16b, v15.16b, v26.16b+ eor x28, x7, x28, ror #57+ bic x7, x9, x12, ror #42+ shl v23.2d, v31.2d, #0x29+ eor x7, x7, x22, ror #25+ sri v23.2d, v31.2d, #0x17+ bic x22, x22, x24, ror #56+ bic x24, x24, x15, ror #31+ eor v31.16b, v1.16b, v25.16b+ eor x22, x22, x15, ror #23+ shl v15.2d, v31.2d, #0x1+ bic x20, x27, x20, ror #48+ sri v15.2d, v31.2d, #0x3f+ bic x15, x15, x9, ror #16+ eor x12, x15, x12, ror #58+ eor v31.16b, v8.16b, v28.16b+ eor x15, x5, x27, ror #27+ shl v1.2d, v31.2d, #0x37+ eor x5, x20, x11, ror #41+ sri v1.2d, v31.2d, #0x9+ ldr x11, [sp, #0x8]+ eor x20, x17, x4, ror #21+ eor v31.16b, v16.16b, v25.16b+ eor x17, x24, x9, ror #47+ shl v8.2d, v31.2d, #0x2d+ mov x24, #0x1 // =1+ sri v8.2d, v31.2d, #0x13+ bic x9, x0, x16, ror #9+ str x24, [sp, #0x18]+ eor v31.16b, v7.16b, v29.16b+ bic x24, x29, x1, ror #44+ shl v16.2d, v31.2d, #0x6+ bic x27, x1, x21, ror #50+ sri v16.2d, v31.2d, #0x3a+ bic x4, x26, x29, ror #63+ eor x1, x1, x4, ror #21+ eor v31.16b, v10.16b, v26.16b+ ldr x11, [x11]+ shl v7.2d, v31.2d, #0x3+ bic x4, x21, x30, ror #57+ sri v7.2d, v31.2d, #0x3d+ eor x21, x24, x21, ror #30+ eor x24, x9, x19, ror #44+ eor v31.16b, v3.16b, v28.16b+ bic x9, x14, x6, ror #5+ shl v10.2d, v31.2d, #0x1c+ eor x9, x9, x0, ror #43+ sri v10.2d, v31.2d, #0x24+ bic x0, x6, x0, ror #38+ eor x1, x1, x11+ eor v31.16b, v18.16b, v28.16b+ eor x11, x4, x26, ror #35+ shl v3.2d, v31.2d, #0x15+ eor x4, x0, x16, ror #47+ bic x0, x16, x19, ror #35+ sri v3.2d, v31.2d, #0x2b+ eor x16, x27, x30, ror #43+ eor v31.16b, v17.16b, v29.16b+ bic x27, x30, x26, ror #42+ shl v18.2d, v31.2d, #0xf+ bic x26, x19, x14, ror #41+ eor x19, x0, x14, ror #12+ sri v18.2d, v31.2d, #0x31+ eor x14, x26, x6, ror #46+ eor v31.16b, v11.16b, v25.16b+ eor x6, x27, x29, ror #41+ shl v17.2d, v31.2d, #0xa+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ sri v17.2d, v31.2d, #0x36+ eor x26, x8, x9, ror #57+ eor v31.16b, v9.16b, v27.16b+ eor x27, x0, x14, ror #10+ shl v11.2d, v31.2d, #0x14+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ sri v11.2d, v31.2d, #0x2c+ eor x30, x23, x22, ror #50+ eor v31.16b, v22.16b, v29.16b+ eor x0, x26, x10, ror #31+ shl v9.2d, v31.2d, #0x3d+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ sri v9.2d, v31.2d, #0x3+ eor x30, x30, x24, ror #34+ eor v31.16b, v14.16b, v27.16b+ eor x0, x0, x7, ror #27+ shl v22.2d, v31.2d, #0x27+ eor x26, x30, x21, ror #26+ eor x26, x26, x25, ror #15+ sri v22.2d, v31.2d, #0x19+ ror x30, x27, #0x3e+ eor v31.16b, v20.16b, v26.16b+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ shl v14.2d, v31.2d, #0x12+ eor x16, x30, x16+ sri v14.2d, v31.2d, #0x2e+ eor x28, x30, x28, ror #63+ eor v31.16b, v4.16b, v27.16b+ str x28, [sp, #0xd0]+ eor x29, x29, x17, ror #36+ shl v20.2d, v31.2d, #0x1b+ eor x28, x1, x2, ror #61+ sri v20.2d, v31.2d, #0x25+ eor x19, x30, x19, ror #37+ eor v31.16b, v24.16b, v27.16b+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ shl v4.2d, v31.2d, #0xe+ eor x26, x26, x0, ror #55+ sri v4.2d, v31.2d, #0x32+ eor x28, x28, x3, ror #39+ eor v31.16b, v21.16b, v25.16b+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ shl v24.2d, v31.2d, #0x2+ eor x0, x0, x29, ror #63+ sri v24.2d, v31.2d, #0x3e+ eor x27, x28, x27, ror #61+ eor v31.16b, v5.16b, v26.16b+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ shl v21.2d, v31.2d, #0x24+ eor x29, x30, x20, ror #2+ sri v21.2d, v31.2d, #0x1c+ eor x20, x26, x3, ror #39+ eor v31.16b, v6.16b, v25.16b+ eor x11, x0, x11, ror #50+ eor x25, x28, x25, ror #9+ shl v27.2d, v31.2d, #0x2c+ eor x3, x28, x21, ror #20+ sri v27.2d, v31.2d, #0x14+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ bic v31.16b, v7.16b, v11.16b+ eor x24, x28, x24, ror #28+ eor v5.16b, v31.16b, v10.16b+ eor x1, x30, x17, ror #36+ bic v31.16b, v8.16b, v7.16b+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ eor v6.16b, v31.16b, v11.16b+ eor x8, x27, x8, ror #56+ bic v31.16b, v9.16b, v8.16b+ eor x17, x27, x7, ror #19+ eor v7.16b, v31.16b, v7.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ bic v31.16b, v10.16b, v9.16b+ eor x4, x26, x4, ror #54+ eor v8.16b, v31.16b, v8.16b+ eor x0, x0, x12, ror #3+ bic v31.16b, v11.16b, v10.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ eor v9.16b, v31.16b, v9.16b+ eor x26, x26, x5, ror #25+ bic v31.16b, v12.16b, v16.16b+ eor x2, x7, x16, ror #39+ eor v10.16b, v31.16b, v15.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ bic v31.16b, v13.16b, v12.16b+ eor x7, x7, x22, ror #25+ eor v11.16b, v31.16b, v16.16b+ eor x12, x30, x20, ror #58+ bic v31.16b, v14.16b, v13.16b+ bic x20, x22, x16, ror #56+ eor x30, x27, x6, ror #43+ eor v12.16b, v31.16b, v12.16b+ eor x22, x20, x15, ror #23+ bic v31.16b, v15.16b, v14.16b+ bic x6, x19, x13, ror #42+ eor v13.16b, v31.16b, v13.16b+ eor x6, x6, x17, ror #41+ bic x5, x13, x17, ror #63+ bic v31.16b, v16.16b, v15.16b+ eor x5, x21, x5, ror #21+ eor v14.16b, v31.16b, v14.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ bic v31.16b, v17.16b, v21.16b+ bic x21, x21, x25, ror #50+ eor v15.16b, v31.16b, v20.16b+ bic x20, x27, x4, ror #25+ bic v31.16b, v18.16b, v17.16b+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ eor v16.16b, v31.16b, v21.16b+ eor x21, x17, x25, ror #30+ bic v31.16b, v19.16b, v18.16b+ bic x19, x25, x19, ror #57+ eor v17.16b, v31.16b, v17.16b+ ldr x25, [sp, #0x18]+ eor x17, x10, x9, ror #47+ bic v31.16b, v20.16b, v19.16b+ ldr x9, [sp, #0x8]+ eor v18.16b, v31.16b, v18.16b+ eor x15, x20, x28, ror #27+ bic v31.16b, v21.16b, v20.16b+ bic x20, x4, x28, ror #2+ eor x10, x20, x1, ror #50+ eor v19.16b, v31.16b, v19.16b+ bic x20, x11, x27, ror #60+ bic v31.16b, v22.16b, v1.16b+ eor x20, x20, x4, ror #21+ eor v20.16b, v31.16b, v0.16b+ bic x4, x28, x1, ror #48+ bic x1, x1, x11, ror #57+ bic v31.16b, v23.16b, v22.16b+ ldr x28, [x9, x25, lsl #3]+ eor v21.16b, v31.16b, v1.16b+ ldr x9, [sp, #0xd0]+ bic v31.16b, v24.16b, v23.16b+ add x25, x25, #0x1+ str x25, [sp, #0x18]+ eor v22.16b, v31.16b, v22.16b+ cmp x25, #0x17+ bic v31.16b, v0.16b, v24.16b+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ eor v23.16b, v31.16b, v23.16b+ eor x1, x5, x28+ bic v31.16b, v1.16b, v0.16b+ eor x5, x4, x11, ror #41+ eor v24.16b, v31.16b, v24.16b+ eor x11, x19, x13, ror #35+ bic x13, x26, x24, ror #10+ bic v31.16b, v2.16b, v27.16b+ eor x28, x27, x24, ror #57+ eor v0.16b, v31.16b, v30.16b+ bic x27, x24, x9, ror #47+ bic v31.16b, v3.16b, v2.16b+ bic x19, x23, x3, ror #9+ bic x4, x29, x14, ror #41+ eor v1.16b, v31.16b, v27.16b+ eor x24, x19, x29, ror #44+ bic v31.16b, v4.16b, v3.16b+ bic x29, x3, x29, ror #35+ eor v2.16b, v31.16b, v2.16b+ eor x13, x13, x9, ror #57+ eor x19, x29, x14, ror #12+ bic v31.16b, v30.16b, v4.16b+ bic x29, x9, x0, ror #19+ eor v3.16b, v31.16b, v3.16b+ bic x14, x14, x8, ror #5+ bic v31.16b, v27.16b, v30.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ eor v4.16b, v31.16b, v4.16b+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_loop:+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor v30.16b, v0.16b, v5.16b+ eor v30.16b, v30.16b, v10.16b+ eor x26, x8, x9, ror #57+ eor v30.16b, v30.16b, v15.16b+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ eor v30.16b, v30.16b, v20.16b+ eor x26, x26, x6, ror #51+ eor v29.16b, v1.16b, v6.16b+ eor x30, x23, x22, ror #50+ eor v29.16b, v29.16b, v11.16b+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ eor v29.16b, v29.16b, v16.16b+ eor x27, x27, x12, ror #5+ eor v29.16b, v29.16b, v21.16b+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ eor v28.16b, v2.16b, v7.16b+ eor x26, x30, x21, ror #26+ eor v28.16b, v28.16b, v12.16b+ eor x26, x26, x25, ror #15+ eor v28.16b, v28.16b, v17.16b+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ eor v28.16b, v28.16b, v22.16b+ ror x26, x26, #0x3a+ eor v27.16b, v3.16b, v8.16b+ eor x16, x30, x16+ eor v27.16b, v27.16b, v13.16b+ eor x28, x30, x28, ror #63+ str x28, [sp, #0xd0]+ eor v27.16b, v27.16b, v18.16b+ eor x29, x29, x17, ror #36+ eor v27.16b, v27.16b, v23.16b+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ eor v26.16b, v4.16b, v9.16b+ eor x29, x29, x20, ror #2+ eor v26.16b, v26.16b, v14.16b+ eor x28, x28, x4, ror #54+ eor v26.16b, v26.16b, v19.16b+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ eor v26.16b, v26.16b, v24.16b+ eor x28, x28, x5, ror #25+ add v31.2d, v28.2d, v28.2d+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ sri v31.2d, v28.2d, #0x3f+ eor x27, x28, x27, ror #61+ eor v25.16b, v31.16b, v30.16b+ eor x13, x0, x13, ror #46+ add v31.2d, v26.2d, v26.2d+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ sri v31.2d, v26.2d, #0x3f+ eor x20, x26, x3, ror #39+ eor v28.16b, v31.16b, v28.16b+ eor x11, x0, x11, ror #50+ add v31.2d, v29.2d, v29.2d+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ sri v31.2d, v29.2d, #0x3f+ eor x21, x26, x1+ eor v26.16b, v31.16b, v26.16b+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ add v31.2d, v27.2d, v27.2d+ eor x1, x30, x17, ror #36+ sri v31.2d, v27.2d, #0x3f+ eor x14, x0, x14, ror #8+ eor v29.16b, v31.16b, v29.16b+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ add v31.2d, v30.2d, v30.2d+ eor x17, x27, x7, ror #19+ sri v31.2d, v30.2d, #0x3f+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ eor v27.16b, v31.16b, v27.16b+ eor x4, x26, x4, ror #54+ eor v30.16b, v0.16b, v26.16b+ eor x0, x0, x12, ror #3+ eor v31.16b, v2.16b, v29.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ shl v0.2d, v31.2d, #0x3e+ eor x26, x26, x5, ror #25+ sri v0.2d, v31.2d, #0x2+ eor x2, x7, x16, ror #39+ eor v31.16b, v12.16b, v29.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ shl v2.2d, v31.2d, #0x2b+ eor x7, x7, x22, ror #25+ sri v2.2d, v31.2d, #0x15+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ eor v31.16b, v13.16b, v28.16b+ eor x30, x27, x6, ror #43+ shl v12.2d, v31.2d, #0x19+ eor x22, x20, x15, ror #23+ sri v12.2d, v31.2d, #0x27+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ eor v31.16b, v19.16b, v27.16b+ bic x5, x13, x17, ror #63+ shl v13.2d, v31.2d, #0x8+ eor x5, x21, x5, ror #21+ sri v13.2d, v31.2d, #0x38+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ eor v31.16b, v23.16b, v28.16b+ bic x21, x21, x25, ror #50+ shl v19.2d, v31.2d, #0x38+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ sri v19.2d, v31.2d, #0x8+ eor x16, x21, x19, ror #43+ eor v31.16b, v15.16b, v26.16b+ eor x21, x17, x25, ror #30+ shl v23.2d, v31.2d, #0x29+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x18]+ sri v23.2d, v31.2d, #0x17+ eor x17, x10, x9, ror #47+ eor v31.16b, v1.16b, v25.16b+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ shl v15.2d, v31.2d, #0x1+ bic x20, x4, x28, ror #2+ sri v15.2d, v31.2d, #0x3f+ eor x10, x20, x1, ror #50+ eor v31.16b, v8.16b, v28.16b+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ shl v1.2d, v31.2d, #0x37+ bic x4, x28, x1, ror #48+ sri v1.2d, v31.2d, #0x9+ bic x1, x1, x11, ror #57+ eor v31.16b, v16.16b, v25.16b+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0xd0]+ shl v8.2d, v31.2d, #0x2d+ add x25, x25, #0x1+ sri v8.2d, v31.2d, #0x13+ str x25, [sp, #0x18]+ cmp x25, #0x17+ eor v31.16b, v7.16b, v29.16b+ eor x25, x1, x27, ror #53+ shl v16.2d, v31.2d, #0x6+ bic x27, x30, x26, ror #47+ sri v16.2d, v31.2d, #0x3a+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ eor v31.16b, v10.16b, v26.16b+ eor x11, x19, x13, ror #35+ shl v7.2d, v31.2d, #0x3+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ sri v7.2d, v31.2d, #0x3d+ bic x27, x24, x9, ror #47+ eor v31.16b, v3.16b, v28.16b+ bic x19, x23, x3, ror #9+ shl v10.2d, v31.2d, #0x1c+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ sri v10.2d, v31.2d, #0x24+ bic x29, x3, x29, ror #35+ eor v31.16b, v18.16b, v28.16b+ eor x13, x13, x9, ror #57+ shl v3.2d, v31.2d, #0x15+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ sri v3.2d, v31.2d, #0x2b+ bic x14, x14, x8, ror #5+ eor v31.16b, v17.16b, v29.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ shl v18.2d, v31.2d, #0xf+ bic x23, x8, x23, ror #38+ sri v18.2d, v31.2d, #0x31+ eor x8, x27, x0, ror #2+ eor v31.16b, v11.16b, v25.16b+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ shl v17.2d, v31.2d, #0xa+ eor x23, x3, x26, ror #52+ sri v17.2d, v31.2d, #0x36+ eor x3, x29, x30, ror #24+ eor x0, x15, x11, ror #52+ eor v31.16b, v9.16b, v27.16b+ eor x0, x0, x13, ror #48+ shl v11.2d, v31.2d, #0x14+ eor x26, x8, x9, ror #57+ sri v11.2d, v31.2d, #0x2c+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ eor v31.16b, v22.16b, v29.16b+ eor x26, x26, x6, ror #51+ shl v9.2d, v31.2d, #0x3d+ eor x30, x23, x22, ror #50+ sri v9.2d, v31.2d, #0x3+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ eor v31.16b, v14.16b, v27.16b+ eor x27, x27, x12, ror #5+ shl v22.2d, v31.2d, #0x27+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ sri v22.2d, v31.2d, #0x19+ eor x26, x30, x21, ror #26+ eor v31.16b, v20.16b, v26.16b+ eor x26, x26, x25, ror #15+ shl v14.2d, v31.2d, #0x12+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ sri v14.2d, v31.2d, #0x2e+ ror x26, x26, #0x3a+ eor v31.16b, v4.16b, v27.16b+ eor x16, x30, x16+ shl v20.2d, v31.2d, #0x1b+ eor x28, x30, x28, ror #63+ str x28, [sp, #0xd0]+ sri v20.2d, v31.2d, #0x25+ eor x29, x29, x17, ror #36+ eor v31.16b, v24.16b, v27.16b+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ shl v4.2d, v31.2d, #0xe+ eor x29, x29, x20, ror #2+ sri v4.2d, v31.2d, #0x32+ eor x28, x28, x4, ror #54+ eor v31.16b, v21.16b, v25.16b+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ shl v24.2d, v31.2d, #0x2+ eor x28, x28, x5, ror #25+ sri v24.2d, v31.2d, #0x3e+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ eor v31.16b, v5.16b, v26.16b+ eor x27, x28, x27, ror #61+ shl v21.2d, v31.2d, #0x24+ eor x13, x0, x13, ror #46+ sri v21.2d, v31.2d, #0x1c+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ eor v31.16b, v6.16b, v25.16b+ eor x20, x26, x3, ror #39+ shl v27.2d, v31.2d, #0x2c+ eor x11, x0, x11, ror #50+ sri v27.2d, v31.2d, #0x14+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ bic v31.16b, v7.16b, v11.16b+ eor x21, x26, x1+ eor v5.16b, v31.16b, v10.16b+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ bic v31.16b, v8.16b, v7.16b+ eor x1, x30, x17, ror #36+ eor v6.16b, v31.16b, v11.16b+ eor x14, x0, x14, ror #8+ bic v31.16b, v9.16b, v8.16b+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ eor v7.16b, v31.16b, v7.16b+ eor x17, x27, x7, ror #19+ bic v31.16b, v10.16b, v9.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ eor v8.16b, v31.16b, v8.16b+ eor x4, x26, x4, ror #54+ bic v31.16b, v11.16b, v10.16b+ eor x0, x0, x12, ror #3+ eor v9.16b, v31.16b, v9.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ bic v31.16b, v12.16b, v16.16b+ eor x26, x26, x5, ror #25+ eor v10.16b, v31.16b, v15.16b+ eor x2, x7, x16, ror #39+ bic v31.16b, v13.16b, v12.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ eor v11.16b, v31.16b, v16.16b+ eor x7, x7, x22, ror #25+ bic v31.16b, v14.16b, v13.16b+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ eor v12.16b, v31.16b, v12.16b+ eor x30, x27, x6, ror #43+ bic v31.16b, v15.16b, v14.16b+ eor x22, x20, x15, ror #23+ eor v13.16b, v31.16b, v13.16b+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ bic v31.16b, v16.16b, v15.16b+ bic x5, x13, x17, ror #63+ eor v14.16b, v31.16b, v14.16b+ eor x5, x21, x5, ror #21+ bic v31.16b, v17.16b, v21.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ eor v15.16b, v31.16b, v20.16b+ bic x21, x21, x25, ror #50+ bic v31.16b, v18.16b, v17.16b+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ eor v16.16b, v31.16b, v21.16b+ eor x16, x21, x19, ror #43+ bic v31.16b, v19.16b, v18.16b+ eor x21, x17, x25, ror #30+ eor v17.16b, v31.16b, v17.16b+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x18]+ bic v31.16b, v20.16b, v19.16b+ eor x17, x10, x9, ror #47+ eor v18.16b, v31.16b, v18.16b+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ bic v31.16b, v21.16b, v20.16b+ bic x20, x4, x28, ror #2+ eor v19.16b, v31.16b, v19.16b+ eor x10, x20, x1, ror #50+ bic v31.16b, v22.16b, v1.16b+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ eor v20.16b, v31.16b, v0.16b+ bic x4, x28, x1, ror #48+ bic v31.16b, v23.16b, v22.16b+ bic x1, x1, x11, ror #57+ eor v21.16b, v31.16b, v1.16b+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0xd0]+ bic v31.16b, v24.16b, v23.16b+ add x25, x25, #0x1+ eor v22.16b, v31.16b, v22.16b+ str x25, [sp, #0x18]+ cmp x25, #0x17+ bic v31.16b, v0.16b, v24.16b+ eor x25, x1, x27, ror #53+ eor v23.16b, v31.16b, v23.16b+ bic x27, x30, x26, ror #47+ bic v31.16b, v1.16b, v0.16b+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ eor v24.16b, v31.16b, v24.16b+ eor x11, x19, x13, ror #35+ bic v31.16b, v2.16b, v27.16b+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ eor v0.16b, v31.16b, v30.16b+ bic x27, x24, x9, ror #47+ bic v31.16b, v3.16b, v2.16b+ bic x19, x23, x3, ror #9+ eor v1.16b, v31.16b, v27.16b+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic v31.16b, v4.16b, v3.16b+ bic x29, x3, x29, ror #35+ eor v2.16b, v31.16b, v2.16b+ eor x13, x13, x9, ror #57+ bic v31.16b, v30.16b, v4.16b+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ eor v3.16b, v31.16b, v3.16b+ bic x14, x14, x8, ror #5+ bic v31.16b, v27.16b, v30.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ eor v4.16b, v31.16b, v4.16b+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_loop_end:+ b.le Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_loop+ ror x2, x2, #0x3d+ ror x3, x3, #0x27+ ror x4, x4, #0x36+ ror x5, x5, #0x19+ ror x6, x6, #0x2b+ ror x7, x7, #0x13+ ror x8, x8, #0x38+ ror x9, x9, #0x31+ ror x10, x10, #0x17+ ror x11, x11, #0x32+ ror x12, x12, #0x3+ ror x13, x13, #0x2e+ ror x14, x14, #0x8+ ror x15, x15, #0x3e+ ror x17, x17, #0x24+ ror x28, x28, #0x3f+ ror x19, x19, #0x25+ ror x20, x20, #0x2+ ror x21, x21, #0x14+ ror x22, x22, #0x2c+ ror x23, x23, #0x3a+ ror x24, x24, #0x1c+ ror x25, x25, #0x9+ ldr x30, [sp, #0x20]+ cmp x30, #0x1+ b.eq Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_done+ mov x30, #0x1 // =1+ str x30, [sp, #0x20]+ ldr x0, [sp]+ add x0, x0, #0x190+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x190+ add x0, x0, #0x258+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x258+ b Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_initial++Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_done:+ ldr x0, [sp]+ add x0, x0, #0x258+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x258+ add x4, x0, #0xc8+ trn1 v25.2d, v0.2d, v1.2d+ trn1 v26.2d, v2.2d, v3.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v0.2d, v1.2d+ trn2 v28.2d, v2.2d, v3.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v4.2d, v5.2d+ trn1 v26.2d, v6.2d, v7.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v4.2d, v5.2d+ trn2 v28.2d, v6.2d, v7.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v8.2d, v9.2d+ trn1 v26.2d, v10.2d, v11.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v8.2d, v9.2d+ trn2 v28.2d, v10.2d, v11.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v12.2d, v13.2d+ trn1 v26.2d, v14.2d, v15.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v12.2d, v13.2d+ trn2 v28.2d, v14.2d, v15.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v16.2d, v17.2d+ trn1 v26.2d, v18.2d, v19.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v16.2d, v17.2d+ trn2 v28.2d, v18.2d, v19.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v20.2d, v21.2d+ trn1 v26.2d, v22.2d, v23.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v20.2d, v21.2d+ trn2 v28.2d, v22.2d, v23.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ str d24, [x0]+ trn2 v25.2d, v24.2d, v24.2d+ str d25, [x4]+ ldp d8, d9, [sp, #0x90]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0xa0]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0xb0]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0xc0]+ .cfi_restore d14+ .cfi_restore d15+ ldp x19, x20, [sp, #0x30]+ .cfi_restore x19+ .cfi_restore x20+ ldp x21, x22, [sp, #0x40]+ .cfi_restore x21+ .cfi_restore x22+ ldp x23, x24, [sp, #0x50]+ .cfi_restore x23+ .cfi_restore x24+ ldp x25, x26, [sp, #0x60]+ .cfi_restore x25+ .cfi_restore x26+ ldp x27, x28, [sp, #0x70]+ .cfi_restore x27+ .cfi_restore x28+ ldp x29, x30, [sp, #0x80]+ .cfi_restore x29+ .cfi_restore x30+ add sp, sp, #0xe0+ .cfi_adjust_cfa_offset -0xe0+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)++#endif /* MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID && \+ !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S view
@@ -0,0 +1,989 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++// Author: Hanno Becker <hannobecker@posteo.de>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>++/*yaml+ Name: keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm+ Description: AArch64 hybrid scalar/vector implementation of Keccak-f[1600] permutation for four sequential states with ARMv8.4-A optimizations+ Signature: void mlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm(uint64_t state[100], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON, SHA3]+ x0:+ type: buffer+ size_bytes: 800+ permissions: read/write+ c_parameter: uint64_t state[100]+ description: Four sequential Keccak states (state0[25], state1[25], state2[25], state3[25])+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: const uint64_t rc[24]+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 224+ description: register preservation and temporary storage+*/++#include "../../../../common.h"+#if defined(MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#if defined(__ARM_FEATURE_SHA3)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)+MLK_ASM_FN_SYMBOL(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0xe0+ .cfi_adjust_cfa_offset 0xe0+ stp x19, x20, [sp, #0x30]+ .cfi_rel_offset x19, 0x30+ .cfi_rel_offset x20, 0x38+ stp x21, x22, [sp, #0x40]+ .cfi_rel_offset x21, 0x40+ .cfi_rel_offset x22, 0x48+ stp x23, x24, [sp, #0x50]+ .cfi_rel_offset x23, 0x50+ .cfi_rel_offset x24, 0x58+ stp x25, x26, [sp, #0x60]+ .cfi_rel_offset x25, 0x60+ .cfi_rel_offset x26, 0x68+ stp x27, x28, [sp, #0x70]+ .cfi_rel_offset x27, 0x70+ .cfi_rel_offset x28, 0x78+ stp x29, x30, [sp, #0x80]+ .cfi_rel_offset x29, 0x80+ .cfi_rel_offset x30, 0x88+ stp d8, d9, [sp, #0x90]+ .cfi_rel_offset d8, 0x90+ .cfi_rel_offset d9, 0x98+ stp d10, d11, [sp, #0xa0]+ .cfi_rel_offset d10, 0xa0+ .cfi_rel_offset d11, 0xa8+ stp d12, d13, [sp, #0xb0]+ .cfi_rel_offset d12, 0xb0+ .cfi_rel_offset d13, 0xb8+ stp d14, d15, [sp, #0xc0]+ .cfi_rel_offset d14, 0xc0+ .cfi_rel_offset d15, 0xc8+ mov x29, x1+ mov x30, #0x0 // =0+ str x30, [sp, #0x20]+ str x29, [sp, #0x8]+ str x29, [sp, #0x10]+ str x0, [sp]+ add x4, x0, #0xc8+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v0.2d, v25.2d, v27.2d+ trn2 v1.2d, v25.2d, v27.2d+ trn1 v2.2d, v26.2d, v28.2d+ trn2 v3.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v4.2d, v25.2d, v27.2d+ trn2 v5.2d, v25.2d, v27.2d+ trn1 v6.2d, v26.2d, v28.2d+ trn2 v7.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v8.2d, v25.2d, v27.2d+ trn2 v9.2d, v25.2d, v27.2d+ trn1 v10.2d, v26.2d, v28.2d+ trn2 v11.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v12.2d, v25.2d, v27.2d+ trn2 v13.2d, v25.2d, v27.2d+ trn1 v14.2d, v26.2d, v28.2d+ trn2 v15.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v16.2d, v25.2d, v27.2d+ trn2 v17.2d, v25.2d, v27.2d+ trn1 v18.2d, v26.2d, v28.2d+ trn2 v19.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v20.2d, v25.2d, v27.2d+ trn2 v21.2d, v25.2d, v27.2d+ trn1 v22.2d, v26.2d, v28.2d+ trn2 v23.2d, v26.2d, v28.2d+ ldr d25, [x0]+ ldr d27, [x4]+ trn1 v24.2d, v25.2d, v27.2d+ sub x0, x0, #0xc0+ add x0, x0, #0x190+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x190++Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_initial:+ eor x30, x24, x25+ eor x27, x9, x10+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor v30.16b, v30.16b, v15.16b+ eor x0, x30, x21+ eor x26, x27, x6+ eor v30.16b, v30.16b, v20.16b+ eor x27, x26, x7+ eor x29, x0, x22+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor x26, x29, x23+ eor x29, x4, x5+ eor v29.16b, v29.16b, v16.16b+ eor x30, x29, x1+ eor x0, x27, x8+ eor v29.16b, v29.16b, v21.16b+ eor x29, x30, x2+ eor x30, x19, x20+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor x30, x30, x16+ eor x27, x26, x0, ror #63+ eor v28.16b, v28.16b, v17.16b+ eor x4, x4, x27+ eor x30, x30, x17+ eor v28.16b, v28.16b, v22.16b+ eor x30, x30, x28+ eor x29, x29, x3+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor x0, x0, x30, ror #63+ eor x30, x30, x29, ror #63+ eor v27.16b, v27.16b, v18.16b+ eor x22, x22, x30+ eor v27.16b, v27.16b, v23.16b+ eor x23, x23, x30+ str x23, [sp, #0xd0]+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor x23, x14, x15+ eor x14, x14, x0+ eor v26.16b, v26.16b, v19.16b+ eor x23, x23, x11+ eor x15, x15, x0+ eor v26.16b, v26.16b, v24.16b+ eor x1, x1, x27+ eor x23, x23, x12+ rax1 v25.2d, v30.2d, v28.2d+ eor x23, x23, x13+ eor x11, x11, x0+ add v31.2d, v26.2d, v26.2d+ eor x29, x29, x23, ror #63+ eor x23, x23, x26, ror #63+ sri v31.2d, v26.2d, #0x3f+ eor x26, x13, x0+ eor x13, x28, x23+ eor v28.16b, v31.16b, v28.16b+ eor x28, x24, x30+ eor x24, x16, x23+ rax1 v26.2d, v26.2d, v29.2d+ eor x16, x21, x30+ eor x21, x25, x30+ add v31.2d, v27.2d, v27.2d+ eor x30, x19, x23+ sri v31.2d, v27.2d, #0x3f+ eor x19, x20, x23+ eor x20, x17, x23+ eor v29.16b, v31.16b, v29.16b+ eor x17, x12, x0+ eor x0, x2, x27+ rax1 v27.2d, v27.2d, v30.2d+ eor x2, x6, x29+ eor x6, x8, x29+ eor v30.16b, v0.16b, v26.16b+ bic x8, x28, x13, ror #47+ eor x12, x3, x27+ eor v31.16b, v2.16b, v29.16b+ bic x3, x13, x17, ror #19+ eor x5, x5, x27+ shl v0.2d, v31.2d, #0x3e+ ldr x27, [sp, #0xd0]+ bic x25, x17, x2, ror #5+ sri v0.2d, v31.2d, #0x2+ eor x9, x9, x29+ eor x23, x25, x5, ror #52+ xar v2.2d, v12.2d, v29.2d, #0x15+ eor x3, x3, x2, ror #24+ eor x8, x8, x17, ror #2+ eor v31.16b, v13.16b, v28.16b+ eor x17, x10, x29+ bic x25, x12, x22, ror #47+ shl v12.2d, v31.2d, #0x19+ eor x29, x7, x29+ bic x10, x4, x27, ror #2+ sri v12.2d, v31.2d, #0x27+ bic x7, x5, x28, ror #10+ xar v13.2d, v19.2d, v27.2d, #0x38+ eor x10, x10, x20, ror #50+ eor x13, x7, x13, ror #57+ eor v31.16b, v23.16b, v28.16b+ bic x7, x2, x5, ror #47+ eor x2, x25, x24, ror #39+ shl v19.2d, v31.2d, #0x38+ bic x25, x20, x11, ror #57+ bic x5, x17, x4, ror #25+ sri v19.2d, v31.2d, #0x8+ eor x25, x25, x17, ror #53+ bic x17, x11, x17, ror #60+ xar v23.2d, v15.2d, v26.2d, #0x17+ eor x28, x7, x28, ror #57+ bic x7, x9, x12, ror #42+ eor v31.16b, v1.16b, v25.16b+ eor x7, x7, x22, ror #25+ bic x22, x22, x24, ror #56+ shl v15.2d, v31.2d, #0x1+ bic x24, x24, x15, ror #31+ eor x22, x22, x15, ror #23+ sri v15.2d, v31.2d, #0x3f+ bic x20, x27, x20, ror #48+ bic x15, x15, x9, ror #16+ xar v1.2d, v8.2d, v28.2d, #0x9+ eor x12, x15, x12, ror #58+ eor x15, x5, x27, ror #27+ eor v31.16b, v16.16b, v25.16b+ eor x5, x20, x11, ror #41+ shl v8.2d, v31.2d, #0x2d+ ldr x11, [sp, #0x8]+ eor x20, x17, x4, ror #21+ sri v8.2d, v31.2d, #0x13+ eor x17, x24, x9, ror #47+ mov x24, #0x1 // =1+ xar v16.2d, v7.2d, v29.2d, #0x3a+ bic x9, x0, x16, ror #9+ str x24, [sp, #0x18]+ eor v31.16b, v10.16b, v26.16b+ bic x24, x29, x1, ror #44+ bic x27, x1, x21, ror #50+ shl v7.2d, v31.2d, #0x3+ bic x4, x26, x29, ror #63+ eor x1, x1, x4, ror #21+ sri v7.2d, v31.2d, #0x3d+ ldr x11, [x11]+ bic x4, x21, x30, ror #57+ xar v10.2d, v3.2d, v28.2d, #0x24+ eor x21, x24, x21, ror #30+ eor x24, x9, x19, ror #44+ eor v31.16b, v18.16b, v28.16b+ bic x9, x14, x6, ror #5+ eor x9, x9, x0, ror #43+ shl v3.2d, v31.2d, #0x15+ bic x0, x6, x0, ror #38+ eor x1, x1, x11+ sri v3.2d, v31.2d, #0x2b+ eor x11, x4, x26, ror #35+ eor x4, x0, x16, ror #47+ xar v18.2d, v17.2d, v29.2d, #0x31+ bic x0, x16, x19, ror #35+ eor v31.16b, v11.16b, v25.16b+ eor x16, x27, x30, ror #43+ bic x27, x30, x26, ror #42+ shl v17.2d, v31.2d, #0xa+ bic x26, x19, x14, ror #41+ eor x19, x0, x14, ror #12+ sri v17.2d, v31.2d, #0x36+ eor x14, x26, x6, ror #46+ eor x6, x27, x29, ror #41+ xar v11.2d, v9.2d, v27.2d, #0x2c+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor v31.16b, v22.16b, v29.16b+ eor x26, x8, x9, ror #57+ eor x27, x0, x14, ror #10+ shl v9.2d, v31.2d, #0x3d+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ sri v9.2d, v31.2d, #0x3+ eor x30, x23, x22, ror #50+ eor x0, x26, x10, ror #31+ xar v22.2d, v14.2d, v27.2d, #0x19+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ eor v31.16b, v20.16b, v26.16b+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ shl v14.2d, v31.2d, #0x12+ eor x26, x30, x21, ror #26+ sri v14.2d, v31.2d, #0x2e+ eor x26, x26, x25, ror #15+ ror x30, x27, #0x3e+ xar v20.2d, v4.2d, v27.2d, #0x25+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ eor v31.16b, v24.16b, v27.16b+ eor x16, x30, x16+ eor x28, x30, x28, ror #63+ shl v4.2d, v31.2d, #0xe+ str x28, [sp, #0xd0]+ eor x29, x29, x17, ror #36+ sri v4.2d, v31.2d, #0x32+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ xar v24.2d, v21.2d, v25.2d, #0x3e+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ eor v31.16b, v5.16b, v26.16b+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ shl v21.2d, v31.2d, #0x24+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ sri v21.2d, v31.2d, #0x1c+ eor x0, x0, x29, ror #63+ eor x27, x28, x27, ror #61+ xar v27.2d, v6.2d, v25.2d, #0x14+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ bic v31.16b, v7.16b, v11.16b+ eor x29, x30, x20, ror #2+ eor v5.16b, v31.16b, v10.16b+ eor x20, x26, x3, ror #39+ eor x11, x0, x11, ror #50+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ bic v31.16b, v9.16b, v8.16b+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ eor v7.16b, v31.16b, v7.16b+ eor x24, x28, x24, ror #28+ eor x1, x30, x17, ror #36+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ bic v31.16b, v11.16b, v10.16b+ eor x8, x27, x8, ror #56+ eor x17, x27, x7, ror #19+ eor v9.16b, v31.16b, v9.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ eor x4, x26, x4, ror #54+ eor x0, x0, x12, ror #3+ bic v31.16b, v13.16b, v12.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ eor v11.16b, v31.16b, v16.16b+ eor x26, x26, x5, ror #25+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ eor x2, x7, x16, ror #39+ bic x7, x9, x20, ror #42+ bic v31.16b, v15.16b, v14.16b+ bic x30, x15, x9, ror #16+ eor x7, x7, x22, ror #25+ eor v13.16b, v31.16b, v13.16b+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ bic v31.16b, v16.16b, v15.16b+ eor x30, x27, x6, ror #43+ eor x22, x20, x15, ror #23+ eor v14.16b, v31.16b, v14.16b+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ bic x5, x13, x17, ror #63+ eor x5, x21, x5, ror #21+ bic v31.16b, v18.16b, v17.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ eor v16.16b, v31.16b, v21.16b+ bic x21, x21, x25, ror #50+ bic x20, x27, x4, ror #25+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ bic v31.16b, v20.16b, v19.16b+ eor x21, x17, x25, ror #30+ bic x19, x25, x19, ror #57+ eor v18.16b, v31.16b, v18.16b+ ldr x25, [sp, #0x18]+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ eor x17, x10, x9, ror #47+ ldr x9, [sp, #0x8]+ bic v31.16b, v22.16b, v1.16b+ eor x15, x20, x28, ror #27+ bic x20, x4, x28, ror #2+ eor v20.16b, v31.16b, v0.16b+ eor x10, x20, x1, ror #50+ bic x20, x11, x27, ror #60+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ eor x20, x20, x4, ror #21+ bic x4, x28, x1, ror #48+ bic v31.16b, v24.16b, v23.16b+ bic x1, x1, x11, ror #57+ ldr x28, [x9, x25, lsl #3]+ eor v22.16b, v31.16b, v22.16b+ ldr x9, [sp, #0xd0]+ add x25, x25, #0x1+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ str x25, [sp, #0x18]+ cmp x25, #0x17+ bic v31.16b, v1.16b, v0.16b+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ eor v24.16b, v31.16b, v24.16b+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ eor x11, x19, x13, ror #35+ bic v31.16b, v3.16b, v2.16b+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ eor v1.16b, v31.16b, v27.16b+ bic x27, x24, x9, ror #47+ bic x19, x23, x3, ror #9+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic v31.16b, v30.16b, v4.16b+ bic x29, x3, x29, ror #35+ eor x13, x13, x9, ror #57+ eor v3.16b, v31.16b, v3.16b+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ bic x14, x14, x8, ror #5+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop:+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor v30.16b, v30.16b, v15.16b+ eor x26, x8, x9, ror #57+ eor x27, x0, x14, ror #10+ eor v30.16b, v30.16b, v20.16b+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor x30, x23, x22, ror #50+ eor x0, x26, x10, ror #31+ eor v29.16b, v29.16b, v16.16b+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ eor v29.16b, v29.16b, v21.16b+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor x26, x30, x21, ror #26+ eor x26, x26, x25, ror #15+ eor v28.16b, v28.16b, v17.16b+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ eor v28.16b, v28.16b, v22.16b+ ror x26, x26, #0x3a+ eor x16, x30, x16+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor x28, x30, x28, ror #63+ str x28, [sp, #0xd0]+ eor v27.16b, v27.16b, v18.16b+ eor x29, x29, x17, ror #36+ eor x28, x1, x2, ror #61+ eor v27.16b, v27.16b, v23.16b+ eor x19, x30, x19, ror #37+ eor x29, x29, x20, ror #2+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor x28, x28, x4, ror #54+ eor x26, x26, x0, ror #55+ eor v26.16b, v26.16b, v19.16b+ eor x28, x28, x3, ror #39+ eor x28, x28, x5, ror #25+ eor v26.16b, v26.16b, v24.16b+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ rax1 v25.2d, v30.2d, v28.2d+ eor x27, x28, x27, ror #61+ eor x13, x0, x13, ror #46+ add v31.2d, v26.2d, v26.2d+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ sri v31.2d, v26.2d, #0x3f+ eor x20, x26, x3, ror #39+ eor x11, x0, x11, ror #50+ eor v28.16b, v31.16b, v28.16b+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ rax1 v26.2d, v26.2d, v29.2d+ eor x21, x26, x1+ add v31.2d, v27.2d, v27.2d+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ sri v31.2d, v27.2d, #0x3f+ eor x1, x30, x17, ror #36+ eor x14, x0, x14, ror #8+ eor v29.16b, v31.16b, v29.16b+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ rax1 v27.2d, v27.2d, v30.2d+ eor x17, x27, x7, ror #19+ eor x15, x0, x15, ror #62+ eor v30.16b, v0.16b, v26.16b+ bic x7, x20, x22, ror #47+ eor x4, x26, x4, ror #54+ eor v31.16b, v2.16b, v29.16b+ eor x0, x0, x12, ror #3+ eor x28, x28, x23, ror #58+ shl v0.2d, v31.2d, #0x3e+ eor x23, x26, x2, ror #61+ eor x26, x26, x5, ror #25+ sri v0.2d, v31.2d, #0x2+ eor x2, x7, x16, ror #39+ bic x7, x9, x20, ror #42+ xar v2.2d, v12.2d, v29.2d, #0x15+ bic x30, x15, x9, ror #16+ eor x7, x7, x22, ror #25+ eor v31.16b, v13.16b, v28.16b+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ shl v12.2d, v31.2d, #0x19+ eor x30, x27, x6, ror #43+ eor x22, x20, x15, ror #23+ sri v12.2d, v31.2d, #0x27+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ xar v13.2d, v19.2d, v27.2d, #0x38+ bic x5, x13, x17, ror #63+ eor x5, x21, x5, ror #21+ eor v31.16b, v23.16b, v28.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ shl v19.2d, v31.2d, #0x38+ bic x21, x21, x25, ror #50+ bic x20, x27, x4, ror #25+ sri v19.2d, v31.2d, #0x8+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ xar v23.2d, v15.2d, v26.2d, #0x17+ eor x21, x17, x25, ror #30+ bic x19, x25, x19, ror #57+ eor v31.16b, v1.16b, v25.16b+ ldr x25, [sp, #0x18]+ eor x17, x10, x9, ror #47+ shl v15.2d, v31.2d, #0x1+ ldr x9, [sp, #0x8]+ sri v15.2d, v31.2d, #0x3f+ eor x15, x20, x28, ror #27+ bic x20, x4, x28, ror #2+ xar v1.2d, v8.2d, v28.2d, #0x9+ eor x10, x20, x1, ror #50+ bic x20, x11, x27, ror #60+ eor v31.16b, v16.16b, v25.16b+ eor x20, x20, x4, ror #21+ bic x4, x28, x1, ror #48+ shl v8.2d, v31.2d, #0x2d+ bic x1, x1, x11, ror #57+ ldr x28, [x9, x25, lsl #3]+ sri v8.2d, v31.2d, #0x13+ ldr x9, [sp, #0xd0]+ add x25, x25, #0x1+ xar v16.2d, v7.2d, v29.2d, #0x3a+ str x25, [sp, #0x18]+ cmp x25, #0x17+ eor v31.16b, v10.16b, v26.16b+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ shl v7.2d, v31.2d, #0x3+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ sri v7.2d, v31.2d, #0x3d+ eor x11, x19, x13, ror #35+ bic x13, x26, x24, ror #10+ xar v10.2d, v3.2d, v28.2d, #0x24+ eor x28, x27, x24, ror #57+ bic x27, x24, x9, ror #47+ eor v31.16b, v18.16b, v28.16b+ bic x19, x23, x3, ror #9+ bic x4, x29, x14, ror #41+ shl v3.2d, v31.2d, #0x15+ eor x24, x19, x29, ror #44+ bic x29, x3, x29, ror #35+ sri v3.2d, v31.2d, #0x2b+ eor x13, x13, x9, ror #57+ eor x19, x29, x14, ror #12+ xar v18.2d, v17.2d, v29.2d, #0x31+ bic x29, x9, x0, ror #19+ bic x14, x14, x8, ror #5+ eor v31.16b, v11.16b, v25.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ shl v17.2d, v31.2d, #0xa+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ sri v17.2d, v31.2d, #0x36+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ xar v11.2d, v9.2d, v27.2d, #0x2c+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ eor v31.16b, v22.16b, v29.16b+ eor x0, x15, x11, ror #52+ shl v9.2d, v31.2d, #0x3d+ eor x0, x0, x13, ror #48+ eor x26, x8, x9, ror #57+ sri v9.2d, v31.2d, #0x3+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ xar v22.2d, v14.2d, v27.2d, #0x19+ eor x26, x26, x6, ror #51+ eor x30, x23, x22, ror #50+ eor v31.16b, v20.16b, v26.16b+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ shl v14.2d, v31.2d, #0x12+ eor x27, x27, x12, ror #5+ eor x30, x30, x24, ror #34+ sri v14.2d, v31.2d, #0x2e+ eor x0, x0, x7, ror #27+ eor x26, x30, x21, ror #26+ xar v20.2d, v4.2d, v27.2d, #0x25+ eor x26, x26, x25, ror #15+ ror x30, x27, #0x3e+ eor v31.16b, v24.16b, v27.16b+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ shl v4.2d, v31.2d, #0xe+ eor x16, x30, x16+ eor x28, x30, x28, ror #63+ sri v4.2d, v31.2d, #0x32+ str x28, [sp, #0xd0]+ eor x29, x29, x17, ror #36+ xar v24.2d, v21.2d, v25.2d, #0x3e+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ eor v31.16b, v5.16b, v26.16b+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ shl v21.2d, v31.2d, #0x24+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ sri v21.2d, v31.2d, #0x1c+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ xar v27.2d, v6.2d, v25.2d, #0x14+ eor x0, x0, x29, ror #63+ eor x27, x28, x27, ror #61+ bic v31.16b, v7.16b, v11.16b+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ eor v5.16b, v31.16b, v10.16b+ eor x29, x30, x20, ror #2+ eor x20, x26, x3, ror #39+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ eor x11, x0, x11, ror #50+ eor x25, x28, x25, ror #9+ bic v31.16b, v9.16b, v8.16b+ eor x3, x28, x21, ror #20+ eor v7.16b, v31.16b, v7.16b+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ eor x24, x28, x24, ror #28+ eor x1, x30, x17, ror #36+ bic v31.16b, v11.16b, v10.16b+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ eor v9.16b, v31.16b, v9.16b+ eor x8, x27, x8, ror #56+ eor x17, x27, x7, ror #19+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ bic v31.16b, v13.16b, v12.16b+ eor x4, x26, x4, ror #54+ eor x0, x0, x12, ror #3+ eor v11.16b, v31.16b, v16.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ eor x26, x26, x5, ror #25+ eor x2, x7, x16, ror #39+ bic v31.16b, v15.16b, v14.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ eor v13.16b, v31.16b, v13.16b+ eor x7, x7, x22, ror #25+ eor x12, x30, x20, ror #58+ bic v31.16b, v16.16b, v15.16b+ bic x20, x22, x16, ror #56+ eor x30, x27, x6, ror #43+ eor v14.16b, v31.16b, v14.16b+ eor x22, x20, x15, ror #23+ bic x6, x19, x13, ror #42+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ eor x6, x6, x17, ror #41+ bic x5, x13, x17, ror #63+ bic v31.16b, v18.16b, v17.16b+ eor x5, x21, x5, ror #21+ bic x17, x17, x21, ror #44+ eor v16.16b, v31.16b, v21.16b+ eor x27, x27, x10, ror #23+ bic x21, x21, x25, ror #50+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ bic v31.16b, v20.16b, v19.16b+ eor x16, x21, x19, ror #43+ eor x21, x17, x25, ror #30+ eor v18.16b, v31.16b, v18.16b+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x18]+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ eor x17, x10, x9, ror #47+ bic v31.16b, v22.16b, v1.16b+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ eor v20.16b, v31.16b, v0.16b+ bic x20, x4, x28, ror #2+ eor x10, x20, x1, ror #50+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ bic v31.16b, v24.16b, v23.16b+ bic x4, x28, x1, ror #48+ bic x1, x1, x11, ror #57+ eor v22.16b, v31.16b, v22.16b+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0xd0]+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ add x25, x25, #0x1+ str x25, [sp, #0x18]+ bic v31.16b, v1.16b, v0.16b+ cmp x25, #0x17+ eor x25, x1, x27, ror #53+ eor v24.16b, v31.16b, v24.16b+ bic x27, x30, x26, ror #47+ eor x1, x5, x28+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ eor x5, x4, x11, ror #41+ eor x11, x19, x13, ror #35+ bic v31.16b, v3.16b, v2.16b+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ eor v1.16b, v31.16b, v27.16b+ bic x27, x24, x9, ror #47+ bic x19, x23, x3, ror #9+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic v31.16b, v30.16b, v4.16b+ bic x29, x3, x29, ror #35+ eor x13, x13, x9, ror #57+ eor v3.16b, v31.16b, v3.16b+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ bic x14, x14, x8, ror #5+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop_end:+ b.le Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop+ ror x2, x2, #0x3d+ ror x3, x3, #0x27+ ror x4, x4, #0x36+ ror x5, x5, #0x19+ ror x6, x6, #0x2b+ ror x7, x7, #0x13+ ror x8, x8, #0x38+ ror x9, x9, #0x31+ ror x10, x10, #0x17+ ror x11, x11, #0x32+ ror x12, x12, #0x3+ ror x13, x13, #0x2e+ ror x14, x14, #0x8+ ror x15, x15, #0x3e+ ror x17, x17, #0x24+ ror x28, x28, #0x3f+ ror x19, x19, #0x25+ ror x20, x20, #0x2+ ror x21, x21, #0x14+ ror x22, x22, #0x2c+ ror x23, x23, #0x3a+ ror x24, x24, #0x1c+ ror x25, x25, #0x9+ ldr x30, [sp, #0x20]+ cmp x30, #0x1+ b.eq Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_done+ mov x30, #0x1 // =1+ str x30, [sp, #0x20]+ ldr x0, [sp]+ add x0, x0, #0x190+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x190+ add x0, x0, #0x258+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x258+ b Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_initial++Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_done:+ ldr x0, [sp]+ add x0, x0, #0x258+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x258+ add x4, x0, #0xc8+ trn1 v25.2d, v0.2d, v1.2d+ trn1 v26.2d, v2.2d, v3.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v0.2d, v1.2d+ trn2 v28.2d, v2.2d, v3.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v4.2d, v5.2d+ trn1 v26.2d, v6.2d, v7.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v4.2d, v5.2d+ trn2 v28.2d, v6.2d, v7.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v8.2d, v9.2d+ trn1 v26.2d, v10.2d, v11.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v8.2d, v9.2d+ trn2 v28.2d, v10.2d, v11.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v12.2d, v13.2d+ trn1 v26.2d, v14.2d, v15.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v12.2d, v13.2d+ trn2 v28.2d, v14.2d, v15.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v16.2d, v17.2d+ trn1 v26.2d, v18.2d, v19.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v16.2d, v17.2d+ trn2 v28.2d, v18.2d, v19.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v20.2d, v21.2d+ trn1 v26.2d, v22.2d, v23.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v20.2d, v21.2d+ trn2 v28.2d, v22.2d, v23.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ str d24, [x0]+ trn2 v25.2d, v24.2d, v24.2d+ str d25, [x4]+ ldp d8, d9, [sp, #0x90]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0xa0]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0xb0]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0xc0]+ .cfi_restore d14+ .cfi_restore d15+ ldp x19, x20, [sp, #0x30]+ .cfi_restore x19+ .cfi_restore x20+ ldp x21, x22, [sp, #0x40]+ .cfi_restore x21+ .cfi_restore x22+ ldp x23, x24, [sp, #0x50]+ .cfi_restore x23+ .cfi_restore x24+ ldp x25, x26, [sp, #0x60]+ .cfi_restore x25+ .cfi_restore x26+ ldp x27, x28, [sp, #0x70]+ .cfi_restore x27+ .cfi_restore x28+ ldp x29, x30, [sp, #0x80]+ .cfi_restore x29+ .cfi_restore x30+ add sp, sp, #0xe0+ .cfi_adjust_cfa_offset -0xe0+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)++#endif /* __ARM_FEATURE_SHA3 */++#endif /* MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID && \+ !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/fips202/native/aarch64/src/keccakf1600_round_constants.c view
@@ -0,0 +1,47 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#include "../../../../common.h"++#if (defined(MLK_FIPS202_AARCH64_NEED_X1_SCALAR) || \+ defined(MLK_FIPS202_AARCH64_NEED_X1_V84A) || \+ defined(MLK_FIPS202_AARCH64_NEED_X2_V84A) || \+ defined(MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID) || \+ defined(MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID)) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#include "fips202_native_aarch64.h"++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint64_t+ mlk_keccakf1600_round_constants[24] = {+ 0x0000000000000001, 0x0000000000008082, 0x800000000000808a,+ 0x8000000080008000, 0x000000000000808b, 0x0000000080000001,+ 0x8000000080008081, 0x8000000000008009, 0x000000000000008a,+ 0x0000000000000088, 0x0000000080008009, 0x000000008000000a,+ 0x000000008000808b, 0x800000000000008b, 0x8000000000008089,+ 0x8000000000008003, 0x8000000000008002, 0x8000000000000080,+ 0x000000000000800a, 0x800000008000000a, 0x8000000080008081,+ 0x8000000000008080, 0x0000000080000001, 0x8000000080008008,+};++#else /* (MLK_FIPS202_AARCH64_NEED_X1_SCALAR || \+ MLK_FIPS202_AARCH64_NEED_X1_V84A || MLK_FIPS202_AARCH64_NEED_X2_V84A \+ || MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID || \+ MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID) && \+ !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(fips202_aarch64_round_constants)++#endif /* !((MLK_FIPS202_AARCH64_NEED_X1_SCALAR || \+ MLK_FIPS202_AARCH64_NEED_X1_V84A || MLK_FIPS202_AARCH64_NEED_X2_V84A \+ || MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID || \+ MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID) && \+ !MLK_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mlkem/src/fips202/native/aarch64/x1_scalar.h view
@@ -0,0 +1,26 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_AARCH64_X1_SCALAR_H+#define MLK_FIPS202_NATIVE_AARCH64_X1_SCALAR_H++/* Part of backend API */+#define MLK_USE_NATIVE_FIPS202_X1+/* Guard for assembly file */+#define MLK_FIPS202_AARCH64_NEED_X1_SCALAR++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x1_native(uint64_t *state)+{+ mlk_keccak_f1600_x1_scalar_aarch64_asm(state,+ mlk_keccakf1600_round_constants);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLK_FIPS202_NATIVE_AARCH64_X1_SCALAR_H */
+ cbits/mlkem/src/fips202/native/aarch64/x1_v84a.h view
@@ -0,0 +1,35 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_AARCH64_X1_V84A_H+#define MLK_FIPS202_NATIVE_AARCH64_X1_V84A_H++#if !defined(__ARM_FEATURE_SHA3)+#error This backend can only be used if SHA3 extensions are available.+#endif++/* Part of backend API */+#define MLK_USE_NATIVE_FIPS202_X1+/* Guard for assembly file */+#define MLK_FIPS202_AARCH64_NEED_X1_V84A++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x1_native(uint64_t *state)+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON) ||+ !mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_SHA3))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_keccak_f1600_x1_v84a_aarch64_asm(state, mlk_keccakf1600_round_constants);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLK_FIPS202_NATIVE_AARCH64_X1_V84A_H */
+ cbits/mlkem/src/fips202/native/aarch64/x2_v84a.h view
@@ -0,0 +1,38 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_AARCH64_X2_V84A_H+#define MLK_FIPS202_NATIVE_AARCH64_X2_V84A_H++#if !defined(__ARM_FEATURE_SHA3)+#error This backend can only be used if SHA3 extensions are available.+#endif++/* Part of backend API */+#define MLK_USE_NATIVE_FIPS202_X4+/* Guard for assembly file */+#define MLK_FIPS202_AARCH64_NEED_X2_V84A++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON) ||+ !mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_SHA3))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_keccak_f1600_x2_v84a_aarch64_asm(state + 0 * 25,+ mlk_keccakf1600_round_constants);+ mlk_keccak_f1600_x2_v84a_aarch64_asm(state + 2 * 25,+ mlk_keccakf1600_round_constants);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLK_FIPS202_NATIVE_AARCH64_X2_V84A_H */
+ cbits/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h view
@@ -0,0 +1,31 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H+#define MLK_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H++/* Part of backend API */+#define MLK_USE_NATIVE_FIPS202_X4+/* Guard for assembly file */+#define MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(+ state, mlk_keccakf1600_round_constants);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLK_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H */
+ cbits/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h view
@@ -0,0 +1,36 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H+#define MLK_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H++#if !defined(__ARM_FEATURE_SHA3)+#error This backend can only be used if SHA3 extensions are available.+#endif++/* Part of backend API */+#define MLK_USE_NATIVE_FIPS202_X4+/* Guard for assembly file */+#define MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON) ||+ !mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_SHA3))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm(+ state, mlk_keccakf1600_round_constants);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLK_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H */
+ cbits/mlkem/src/fips202/native/api.h view
@@ -0,0 +1,117 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_API_H+#define MLK_FIPS202_NATIVE_API_H+/*+ * FIPS-202 native interface+ *+ * This header is primarily for documentation purposes.+ * It should not be included by backend implementations.+ */++#include "../../cbmc.h"++/* Backends must return MLK_NATIVE_FUNC_SUCCESS upon success. */+#define MLK_NATIVE_FUNC_SUCCESS (0)+/* Backends may return MLK_NATIVE_FUNC_FALLBACK to signal to the frontend that+ * the target/parameters are unsupported; typically, this would be because of+ * dependencies on CPU features not detected on the host CPU. In this case,+ * the frontend falls back to the default C implementation. */+#define MLK_NATIVE_FUNC_FALLBACK (-1)++/*+ * This is the C<->native interface allowing for the drop-in+ * of custom Keccak-F1600 implementations.+ *+ * A _backend_ is a specific implementation of parts of this interface.+ *+ * You can replace 1-fold or 4-fold batched Keccak-F1600.+ * To enable, set MLK_USE_NATIVE_FIPS202_X1 or MLK_USE_NATIVE_FIPS202_X4+ * in your backend, and define the inline wrappers mlk_keccak_f1600_x1_native()+ * and/or mlk_keccak_f1600_x4_native(), respectively, to forward to your+ * implementation.+ */++#if defined(MLK_USE_NATIVE_FIPS202_X1)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x1_native(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 1))+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 1))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged_u64(state, 25 * 1)));+#endif /* MLK_USE_NATIVE_FIPS202_X1 */+#if defined(MLK_USE_NATIVE_FIPS202_X4)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged_u64(state, 25 * 4)));+#endif /* MLK_USE_NATIVE_FIPS202_X4 */++/*+ * Native x4 XOR bytes and extract bytes interface.+ *+ * These functions allow backends to provide optimized implementations for+ * XORing input data into the state and extracting output data from the state.+ * This is particularly useful for backends that use a different internal state+ * representation (e.g., bit-interleaved), as conversion can happen during+ * XOR/extract rather than before/after each permutation.+ *+ * NOTE: We assume that the custom representation of the zero state is the+ * all-zero state.+ *+ * MLK_USE_NATIVE_FIPS202_X4_XOR_BYTES: Backend provides native XOR bytes+ * MLK_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES: Backend provides native extract+ * bytes+ */++#if defined(MLK_USE_NATIVE_FIPS202_X4_XOR_BYTES)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccakf1600_xor_bytes_x4_native(+ uint64_t *state, const unsigned char *data0, const unsigned char *data1,+ const unsigned char *data2, const unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= 25 * sizeof(uint64_t) &&+ 0 <= length && length <= 25 * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(memory_no_alias(data0, length))+ requires((data0 == data1 &&+ data0 == data2 &&+ data0 == data3) ||+ (memory_no_alias(data1, length) &&+ memory_no_alias(data2, length) &&+ memory_no_alias(data3, length)))+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged_u64(state, 25 * 4)));+#endif /* MLK_USE_NATIVE_FIPS202_X4_XOR_BYTES */++#if defined(MLK_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccakf1600_extract_bytes_x4_native(+ uint64_t *state, unsigned char *data0, unsigned char *data1,+ unsigned char *data2, unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= 25 * sizeof(uint64_t) &&+ 0 <= length && length <= 25 * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(memory_no_alias(data0, length))+ requires(memory_no_alias(data1, length))+ requires(memory_no_alias(data2, length))+ requires(memory_no_alias(data3, length))+ assigns(memory_slice(data0, length))+ assigns(memory_slice(data1, length))+ assigns(memory_slice(data2, length))+ assigns(memory_slice(data3, length))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS));+#endif /* MLK_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES */++#endif /* !MLK_FIPS202_NATIVE_API_H */
+ cbits/mlkem/src/fips202/native/auto.h view
@@ -0,0 +1,29 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_AUTO_H+#define MLK_FIPS202_NATIVE_AUTO_H++/*+ * Default FIPS202 backend+ */+#include "../../sys.h"++#if defined(MLK_SYS_AARCH64)+#include "aarch64/auto.h"+#endif++/* The x86_64 backend requires toolchain support for the SysV ABI */+#if defined(MLK_SYS_X86_64_AVX2) && defined(MLK_SYSV_ABI_SUPPORTED)+#include "x86_64/keccak_f1600_x4_avx2.h"+#endif++/* We do not yet include the FIPS202 backend for Armv8.1-M+MVE by default+ * as it is still experimental and undergoing review. */+/* #if defined(MLK_SYS_ARMV81M_MVE) */+/* #include "armv81m/mve.h" */+/* #endif */++#endif /* !MLK_FIPS202_NATIVE_AUTO_H */
+ cbits/mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h view
@@ -0,0 +1,33 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H+#define MLK_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H++#include "../../../common.h"++#define MLK_FIPS202_X86_64_NEED_X4_AVX2++/* Part of backend API */+#define MLK_USE_NATIVE_FIPS202_X4++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_x86_64.h"+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_keccak_f1600_x4_avx2_asm(state, mlk_keccakf1600_round_constants,+ mlk_keccak_rho8, mlk_keccak_rho56);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLK_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H */
+ cbits/mlkem/src/fips202/native/x86_64/src/fips202_native_x86_64.h view
@@ -0,0 +1,44 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H+#define MLK_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H++#include "../../../../cbmc.h"+#include "../../../../common.h"++/* TODO: Reconsider whether this check is needed -- x86_64 is always+ * little-endian, so the backend selection already implies this. */+#ifndef MLK_SYS_LITTLE_ENDIAN+#error Expecting a little-endian platform+#endif++#define mlk_keccakf1600_round_constants \+ MLK_NAMESPACE(keccakf1600_round_constants)+MLK_INTERNAL_DATA_DECLARATION const uint64_t+ mlk_keccakf1600_round_constants[24];++#define mlk_keccak_rho8 MLK_NAMESPACE(keccak_rho8)+MLK_INTERNAL_DATA_DECLARATION const uint64_t mlk_keccak_rho8[4];++#define mlk_keccak_rho56 MLK_NAMESPACE(keccak_rho56)+MLK_INTERNAL_DATA_DECLARATION const uint64_t mlk_keccak_rho56[4];++#define mlk_keccak_f1600_x4_avx2_asm MLK_NAMESPACE(keccak_f1600_x4_avx2_asm)+MLK_SYSV_ABI+void mlk_keccak_f1600_x4_avx2_asm(uint64_t states[100], const uint64_t rc[24],+ const uint64_t rho8[4],+ const uint64_t rho56[4])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/keccak_f1600_x4_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(states, sizeof(uint64_t) * 25 * 4))+ requires(rc == mlk_keccakf1600_round_constants)+ requires(rho8 == mlk_keccak_rho8)+ requires(rho56 == mlk_keccak_rho56)+ assigns(memory_slice(states, sizeof(uint64_t) * 25 * 4))+);++#endif /* !MLK_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H */
+ cbits/mlkem/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S view
@@ -0,0 +1,487 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#include "../../../../common.h"++#if defined(MLK_FIPS202_X86_64_NEED_X4_AVX2) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)+/*yaml+ Name: keccak_f1600_x4_avx2_asm+ Description: x86_64 AVX2 Keccak-f[1600] permutation for four sequential states+ Signature: void mlk_keccak_f1600_x4_avx2_asm(uint64_t states[100], const uint64_t rc[24], const uint64_t rho8[4], const uint64_t rho56[4])+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 800+ permissions: read/write+ c_parameter: uint64_t states[100]+ description: Four sequential Keccak states (4 x 25 x uint64_t)+ rsi:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: const uint64_t rc[24]+ description: Round constants (24 x uint64_t)+ rdx:+ type: buffer+ size_bytes: 32+ permissions: read-only+ c_parameter: const uint64_t rho8[4]+ description: Rotation constant rho8 (4 x uint64_t)+ rcx:+ type: buffer+ size_bytes: 32+ permissions: read-only+ c_parameter: const uint64_t rho56[4]+ description: Rotation constant rho56 (4 x uint64_t)+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/fips202/x86_64/src/keccak_f1600_x4_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(keccak_f1600_x4_avx2_asm)+MLK_ASM_FN_SYMBOL(keccak_f1600_x4_avx2_asm)++ .cfi_startproc+ movq %rsp, %r11+ .cfi_def_cfa_register %r11+ andq $-0x20, %rsp+ subq $0x300, %rsp # imm = 0x300+ vmovdqu (%rdi), %ymm0+ vmovdqu 0xc8(%rdi), %ymm3+ vmovdqu 0x190(%rdi), %ymm1+ vmovdqu 0x258(%rdi), %ymm4+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm7 # ymm7 = ymm2[0,1],ymm3[0,1]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vmovdqu 0x278(%rdi), %ymm4+ vmovdqu %ymm3, 0x40(%rsp)+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm3 # ymm3 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm7, (%rsp)+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu 0x20(%rdi), %ymm0+ vmovdqu 0x1b0(%rdi), %ymm1+ vmovdqu %ymm3, 0x60(%rsp)+ vmovdqu 0xe8(%rdi), %ymm3+ vmovdqu %ymm7, 0x20(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm7 # ymm7 = ymm2[0,1],ymm3[0,1]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vmovdqu 0x298(%rdi), %ymm4+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm14 # ymm14 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm7, 0x80(%rsp)+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu 0x40(%rdi), %ymm0+ vmovdqu 0x1d0(%rdi), %ymm1+ vmovdqu %ymm3, 0xc0(%rsp)+ vmovdqu 0x108(%rdi), %ymm3+ vmovdqu %ymm14, %ymm10+ vmovdqu %ymm7, 0xa0(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm11 # ymm11 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu %ymm3, 0x100(%rsp)+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm8 # ymm8 = ymm0[2,3],ymm1[2,3]+ vmovdqu 0x128(%rdi), %ymm3+ vmovdqu 0x60(%rdi), %ymm0+ vmovdqu 0x1f0(%rdi), %ymm1+ vmovdqu %ymm7, 0xe0(%rsp)+ vmovdqu %ymm11, %ymm14+ vmovdqu 0x2b8(%rdi), %ymm4+ vmovdqu 0x2f8(%rdi), %ymm5+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vmovdqu 0x2d8(%rdi), %ymm4+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm15 # ymm15 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm9 # ymm9 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm3, 0x140(%rsp)+ vmovdqu 0x80(%rdi), %ymm0+ vmovdqu 0x148(%rdi), %ymm3+ vmovdqu 0x210(%rdi), %ymm1+ vmovdqu %ymm7, 0x120(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm7 # ymm7 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm13 # ymm13 = ymm2[2,3],ymm3[2,3]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm3 # ymm3 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm7, 0x160(%rsp)+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu 0xa0(%rdi), %ymm0+ vmovdqu 0x230(%rdi), %ymm1+ vmovdqu %ymm3, 0x1a0(%rsp)+ vmovdqu 0x168(%rdi), %ymm3+ vpunpcklqdq %ymm5, %ymm1, %ymm4 # ymm4 = ymm1[0],ymm5[0],ymm1[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm5[1],ymm1[3],ymm5[3]+ vmovdqu %ymm7, 0x180(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm12 # ymm12 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm3 # ymm3 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm7 # ymm7 = ymm2[2,3],ymm4[2,3]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm4 # ymm4 = ymm0[2,3],ymm1[2,3]+ vmovq 0x250(%rdi), %xmm0+ vmovq 0xc0(%rdi), %xmm1+ vmovdqu %ymm12, 0x1c0(%rsp)+ vmovdqu %ymm4, 0x1e0(%rsp)+ vpinsrq $0x1, 0x318(%rdi), %xmm0, %xmm0+ vpinsrq $0x1, 0x188(%rdi), %xmm1, %xmm1+ vinserti128 $0x1, %xmm0, %ymm1, %ymm2+ movq $0x0, %r10++LLmlk_keccak_f1600_x4_avx2:+ vmovdqu 0xa0(%rsp), %ymm4+ vpxor 0x1c0(%rsp), %ymm9, %ymm0+ vmovdqu %ymm9, 0x200(%rsp)+ vmovdqu %ymm10, %ymm9+ vmovdqu 0xc0(%rsp), %ymm11+ vmovdqu 0x160(%rsp), %ymm12+ vmovdqu %ymm3, 0x240(%rsp)+ vpxor 0x100(%rsp), %ymm4, %ymm1+ vmovdqu 0x40(%rsp), %ymm10+ vmovdqu %ymm4, 0x220(%rsp)+ vpxor %ymm3, %ymm12, %ymm12+ vmovdqu 0x20(%rsp), %ymm6+ vmovdqu 0x140(%rsp), %ymm4+ vmovdqu %ymm14, 0x2a0(%rsp)+ vpxor %ymm1, %ymm0, %ymm0+ vpxor %ymm8, %ymm11, %ymm1+ vpxor 0x180(%rsp), %ymm7, %ymm11+ vmovdqu %ymm10, 0x280(%rsp)+ vpxor %ymm1, %ymm12, %ymm12+ vpxor %ymm15, %ymm9, %ymm1+ vmovdqu 0xe0(%rsp), %ymm3+ vmovdqu %ymm8, 0x260(%rsp)+ vpxor %ymm1, %ymm11, %ymm11+ vpxor 0x120(%rsp), %ymm14, %ymm1+ vpxor %ymm6, %ymm12, %ymm12+ vmovdqu 0x60(%rsp), %ymm8+ vpxor %ymm10, %ymm11, %ymm11+ vpxor 0x1e0(%rsp), %ymm13, %ymm10+ vpxor %ymm4, %ymm3, %ymm3+ vmovdqu %ymm4, 0x2c0(%rsp)+ vpsrlq $0x3f, %ymm12, %ymm4+ vpsrlq $0x3f, %ymm11, %ymm5+ vpxor (%rsp), %ymm0, %ymm0+ vpxor %ymm1, %ymm10, %ymm10+ vmovdqu 0x80(%rsp), %ymm1+ vpxor %ymm8, %ymm10, %ymm10+ vmovdqu %ymm1, %ymm14+ vpxor 0x1a0(%rsp), %ymm2, %ymm1+ vmovdqu %ymm14, 0x2e0(%rsp)+ vpxor %ymm3, %ymm1, %ymm1+ vpsllq $0x1, %ymm12, %ymm3+ vpor %ymm4, %ymm3, %ymm3+ vpsllq $0x1, %ymm11, %ymm4+ vpxor %ymm14, %ymm1, %ymm1+ vpor %ymm5, %ymm4, %ymm4+ vpsrlq $0x3f, %ymm10, %ymm14+ vpxor %ymm1, %ymm3, %ymm3+ vpsllq $0x1, %ymm10, %ymm5+ vpxor %ymm0, %ymm4, %ymm4+ vpor %ymm14, %ymm5, %ymm5+ vpxor %ymm6, %ymm4, %ymm6+ vpxor %ymm12, %ymm5, %ymm5+ vpsrlq $0x3f, %ymm1, %ymm12+ vpsllq $0x1, %ymm1, %ymm1+ vpxor %ymm7, %ymm5, %ymm7+ vpxor %ymm9, %ymm5, %ymm9+ vpor %ymm12, %ymm1, %ymm1+ vpxor (%rsp), %ymm3, %ymm12+ vpxor %ymm11, %ymm1, %ymm1+ vpsrlq $0x3f, %ymm0, %ymm11+ vpsllq $0x1, %ymm0, %ymm0+ vpxor %ymm13, %ymm1, %ymm13+ vpxor %ymm8, %ymm1, %ymm8+ vpor %ymm11, %ymm0, %ymm0+ vpxor %ymm10, %ymm0, %ymm0+ vpxor 0xc0(%rsp), %ymm4, %ymm10+ vpxor %ymm2, %ymm0, %ymm2+ vpsrlq $0x14, %ymm10, %ymm11+ vpsllq $0x2c, %ymm10, %ymm10+ vpor %ymm11, %ymm10, %ymm10+ vpxor %ymm15, %ymm5, %ymm11+ vpbroadcastq (%rsi), %ymm15+ vpsrlq $0x15, %ymm11, %ymm14+ vpsllq $0x2b, %ymm11, %ymm11+ vpor %ymm14, %ymm11, %ymm11+ vpandn %ymm11, %ymm10, %ymm14+ vpxor %ymm15, %ymm14, %ymm14+ vpxor %ymm12, %ymm14, %ymm15+ vpsrlq $0x2b, %ymm13, %ymm14+ vpsllq $0x15, %ymm13, %ymm13+ vmovdqu %ymm15, (%rsp)+ vpor %ymm14, %ymm13, %ymm13+ vpandn %ymm13, %ymm11, %ymm14+ vpxor %ymm10, %ymm14, %ymm15+ vpsrlq $0x32, %ymm2, %ymm14+ vpsllq $0xe, %ymm2, %ymm2+ vmovdqu %ymm15, 0x20(%rsp)+ vpor %ymm14, %ymm2, %ymm2+ vpandn %ymm2, %ymm13, %ymm14+ vpxor %ymm11, %ymm14, %ymm11+ vmovdqu %ymm11, 0x40(%rsp)+ vpandn %ymm12, %ymm2, %ymm11+ vpandn %ymm10, %ymm12, %ymm12+ vpxor %ymm13, %ymm11, %ymm11+ vmovdqu %ymm11, 0x60(%rsp)+ vpxor %ymm2, %ymm12, %ymm11+ vpsrlq $0x24, %ymm8, %ymm2+ vpsllq $0x1c, %ymm8, %ymm8+ vmovdqu %ymm11, 0x80(%rsp)+ vpor %ymm2, %ymm8, %ymm8+ vpxor 0xe0(%rsp), %ymm0, %ymm2+ vpsrlq $0x2c, %ymm2, %ymm10+ vpsllq $0x14, %ymm2, %ymm2+ vpor %ymm10, %ymm2, %ymm2+ vpxor 0x100(%rsp), %ymm3, %ymm10+ vpsrlq $0x3d, %ymm10, %ymm11+ vpsllq $0x3, %ymm10, %ymm10+ vpor %ymm11, %ymm10, %ymm10+ vpandn %ymm10, %ymm2, %ymm11+ vpxor %ymm8, %ymm11, %ymm11+ vmovdqu %ymm11, 0xa0(%rsp)+ vpxor 0x160(%rsp), %ymm4, %ymm11+ vpsrlq $0x13, %ymm11, %ymm12+ vpsllq $0x2d, %ymm11, %ymm11+ vpor %ymm12, %ymm11, %ymm11+ vpandn %ymm11, %ymm10, %ymm12+ vpxor %ymm2, %ymm12, %ymm12+ vmovdqu %ymm12, 0xc0(%rsp)+ vpsrlq $0x3, %ymm7, %ymm12+ vpsllq $0x3d, %ymm7, %ymm7+ vpor %ymm12, %ymm7, %ymm7+ vpandn %ymm7, %ymm11, %ymm12+ vpxor %ymm10, %ymm12, %ymm10+ vpandn %ymm8, %ymm7, %ymm12+ vpandn %ymm2, %ymm8, %ymm8+ vpsrlq $0x3f, %ymm6, %ymm2+ vpsllq $0x1, %ymm6, %ymm6+ vpxor %ymm11, %ymm12, %ymm14+ vpor %ymm2, %ymm6, %ymm6+ vpsrlq $0x3a, %ymm9, %ymm2+ vpxor %ymm7, %ymm8, %ymm12+ vpsllq $0x6, %ymm9, %ymm9+ vmovdqu %ymm12, 0xe0(%rsp)+ vpxor 0x1a0(%rsp), %ymm0, %ymm7+ vpor %ymm2, %ymm9, %ymm9+ vpxor 0x120(%rsp), %ymm1, %ymm2+ vpshufb (%rdx), %ymm7, %ymm7+ vpsrlq $0x27, %ymm2, %ymm11+ vpsllq $0x19, %ymm2, %ymm2+ vpor %ymm2, %ymm11, %ymm11+ vpandn %ymm11, %ymm9, %ymm2+ vpandn %ymm7, %ymm11, %ymm8+ vpxor %ymm6, %ymm2, %ymm12+ vpxor 0x1c0(%rsp), %ymm3, %ymm2+ vpxor %ymm9, %ymm8, %ymm8+ vmovdqu %ymm12, 0x100(%rsp)+ vpsrlq $0x2e, %ymm2, %ymm12+ vpsllq $0x12, %ymm2, %ymm2+ vpor %ymm2, %ymm12, %ymm2+ vpandn %ymm2, %ymm7, %ymm12+ vpxor %ymm11, %ymm12, %ymm15+ vpandn %ymm6, %ymm2, %ymm11+ vpandn %ymm9, %ymm6, %ymm6+ vpxor %ymm7, %ymm11, %ymm12+ vmovdqu %ymm12, 0x120(%rsp)+ vpxor %ymm2, %ymm6, %ymm12+ vpxor 0x2e0(%rsp), %ymm0, %ymm6+ vpxor 0x2c0(%rsp), %ymm0, %ymm0+ vmovdqu %ymm12, 0x140(%rsp)+ vpsrlq $0x25, %ymm6, %ymm2+ vpsllq $0x1b, %ymm6, %ymm6+ vpor %ymm6, %ymm2, %ymm2+ vpxor 0x220(%rsp), %ymm3, %ymm6+ vpxor 0x200(%rsp), %ymm3, %ymm3+ vpsrlq $0x1c, %ymm6, %ymm7+ vpsllq $0x24, %ymm6, %ymm6+ vpor %ymm6, %ymm7, %ymm7+ vpxor 0x260(%rsp), %ymm4, %ymm6+ vpxor 0x240(%rsp), %ymm4, %ymm4+ vpsrlq $0x36, %ymm6, %ymm12+ vpsllq $0xa, %ymm6, %ymm6+ vpor %ymm6, %ymm12, %ymm12+ vpxor 0x180(%rsp), %ymm5, %ymm6+ vpxor 0x280(%rsp), %ymm5, %ymm5+ vpandn %ymm12, %ymm7, %ymm9+ vpsrlq $0x31, %ymm6, %ymm11+ vpsllq $0xf, %ymm6, %ymm6+ vpxor %ymm2, %ymm9, %ymm9+ vpor %ymm6, %ymm11, %ymm11+ vpandn %ymm11, %ymm12, %ymm6+ vpxor %ymm7, %ymm6, %ymm6+ vmovdqu %ymm6, 0x160(%rsp)+ vpxor 0x1e0(%rsp), %ymm1, %ymm6+ vpxor 0x2a0(%rsp), %ymm1, %ymm1+ vpshufb (%rcx), %ymm6, %ymm6+ vpandn %ymm6, %ymm11, %ymm13+ vpxor %ymm12, %ymm13, %ymm13+ vmovdqu %ymm13, 0x180(%rsp)+ vpandn %ymm2, %ymm6, %ymm13+ vpandn %ymm7, %ymm2, %ymm2+ vpxor %ymm6, %ymm2, %ymm2+ vpsrlq $0x3e, %ymm4, %ymm6+ vpxor %ymm11, %ymm13, %ymm13+ vmovdqu %ymm2, 0x1a0(%rsp)+ vpsrlq $0x2, %ymm5, %ymm2+ vpsllq $0x3e, %ymm5, %ymm5+ vpor %ymm5, %ymm2, %ymm2+ vpsrlq $0x9, %ymm1, %ymm5+ vpsllq $0x37, %ymm1, %ymm1+ vpsllq $0x2, %ymm4, %ymm4+ vpor %ymm1, %ymm5, %ymm1+ vpsrlq $0x19, %ymm0, %ymm5+ vpor %ymm4, %ymm6, %ymm4+ vpsllq $0x27, %ymm0, %ymm0+ vpor %ymm0, %ymm5, %ymm5+ vpandn %ymm5, %ymm1, %ymm0+ vpxor %ymm2, %ymm0, %ymm0+ vmovdqu %ymm0, 0x1c0(%rsp)+ vpsrlq $0x17, %ymm3, %ymm0+ vpsllq $0x29, %ymm3, %ymm3+ vpor %ymm3, %ymm0, %ymm0+ vpandn %ymm4, %ymm0, %ymm7+ vpandn %ymm0, %ymm5, %ymm3+ vpxor %ymm5, %ymm7, %ymm7+ vpandn %ymm2, %ymm4, %ymm5+ vpandn %ymm1, %ymm2, %ymm2+ vpxor %ymm0, %ymm5, %ymm5+ vpxor %ymm1, %ymm3, %ymm3+ vpxor %ymm4, %ymm2, %ymm2+ vmovdqu %ymm5, 0x1e0(%rsp)+ addq $0x8, %rsi+ addq $0x1, %r10+ cmpq $0x18, %r10+ jne LLmlk_keccak_f1600_x4_avx2+ vmovdqu (%rsp), %ymm4+ vmovdqu 0x40(%rsp), %ymm5+ vmovdqu 0x20(%rsp), %ymm0+ vmovdqu 0x60(%rsp), %ymm1+ vmovdqu 0x1c0(%rsp), %ymm12+ vmovdqu %ymm2, 0x1c0(%rsp)+ vpunpcklqdq %ymm0, %ymm4, %ymm2 # ymm2 = ymm4[0],ymm0[0],ymm4[2],ymm0[2]+ vpunpckhqdq %ymm0, %ymm4, %ymm0 # ymm0 = ymm4[1],ymm0[1],ymm4[3],ymm0[3]+ vpunpcklqdq %ymm1, %ymm5, %ymm4 # ymm4 = ymm5[0],ymm1[0],ymm5[2],ymm1[2]+ vpunpckhqdq %ymm1, %ymm5, %ymm1 # ymm1 = ymm5[1],ymm1[1],ymm5[3],ymm1[3]+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vmovdqu 0x80(%rsp), %ymm4+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm5 # ymm5 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm0 # ymm0 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm6, (%rdi)+ vmovdqu %ymm5, 0xc8(%rdi)+ vmovdqu %ymm2, 0x190(%rdi)+ vmovdqu %ymm0, 0x258(%rdi)+ vmovdqu 0xa0(%rsp), %ymm0+ vpunpcklqdq %ymm0, %ymm4, %ymm2 # ymm2 = ymm4[0],ymm0[0],ymm4[2],ymm0[2]+ vpunpckhqdq %ymm0, %ymm4, %ymm1 # ymm1 = ymm4[1],ymm0[1],ymm4[3],ymm0[3]+ vmovdqu 0xc0(%rsp), %ymm0+ vpunpcklqdq %ymm10, %ymm0, %ymm4 # ymm4 = ymm0[0],ymm10[0],ymm0[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm10[1],ymm0[3],ymm10[3]+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x20, %ymm0, %ymm1, %ymm5 # ymm5 = ymm1[0,1],ymm0[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vmovdqu 0xe0(%rsp), %ymm4+ vperm2i128 $0x31, %ymm0, %ymm1, %ymm1 # ymm1 = ymm1[2,3],ymm0[2,3]+ vmovdqu 0x100(%rsp), %ymm0+ vmovdqu %ymm2, 0x1b0(%rdi)+ vmovdqu %ymm1, 0x278(%rdi)+ vpunpcklqdq %ymm4, %ymm14, %ymm2 # ymm2 = ymm14[0],ymm4[0],ymm14[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm14, %ymm1 # ymm1 = ymm14[1],ymm4[1],ymm14[3],ymm4[3]+ vpunpcklqdq %ymm8, %ymm0, %ymm4 # ymm4 = ymm0[0],ymm8[0],ymm0[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm8[1],ymm0[3],ymm8[3]+ vmovdqu %ymm6, 0x20(%rdi)+ vmovdqu %ymm5, 0xe8(%rdi)+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x20, %ymm0, %ymm1, %ymm5 # ymm5 = ymm1[0,1],ymm0[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vperm2i128 $0x31, %ymm0, %ymm1, %ymm1 # ymm1 = ymm1[2,3],ymm0[2,3]+ vmovdqu 0x120(%rsp), %ymm4+ vmovdqu 0x140(%rsp), %ymm0+ vmovdqu %ymm2, 0x1d0(%rdi)+ vmovdqu %ymm1, 0x298(%rdi)+ vpunpcklqdq %ymm4, %ymm15, %ymm2 # ymm2 = ymm15[0],ymm4[0],ymm15[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm15, %ymm1 # ymm1 = ymm15[1],ymm4[1],ymm15[3],ymm4[3]+ vpunpcklqdq %ymm9, %ymm0, %ymm4 # ymm4 = ymm0[0],ymm9[0],ymm0[2],ymm9[2]+ vmovdqu %ymm5, 0x108(%rdi)+ vpunpckhqdq %ymm9, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm9[1],ymm0[3],ymm9[3]+ vmovdqu %ymm6, 0x40(%rdi)+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vperm2i128 $0x20, %ymm0, %ymm1, %ymm5 # ymm5 = ymm1[0,1],ymm0[0,1]+ vmovdqu 0x160(%rsp), %ymm4+ vperm2i128 $0x31, %ymm0, %ymm1, %ymm1 # ymm1 = ymm1[2,3],ymm0[2,3]+ vmovdqu 0x180(%rsp), %ymm0+ vmovdqu %ymm5, 0x128(%rdi)+ vmovdqu 0x1a0(%rsp), %ymm5+ vmovdqu %ymm2, 0x1f0(%rdi)+ vpunpcklqdq %ymm0, %ymm4, %ymm2 # ymm2 = ymm4[0],ymm0[0],ymm4[2],ymm0[2]+ vpunpckhqdq %ymm0, %ymm4, %ymm0 # ymm0 = ymm4[1],ymm0[1],ymm4[3],ymm0[3]+ vpunpcklqdq %ymm5, %ymm13, %ymm4 # ymm4 = ymm13[0],ymm5[0],ymm13[2],ymm5[2]+ vmovdqu %ymm6, 0x60(%rdi)+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vmovdqu %ymm1, 0x2b8(%rdi)+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vpunpckhqdq %ymm5, %ymm13, %ymm1 # ymm1 = ymm13[1],ymm5[1],ymm13[3],ymm5[3]+ vmovdqu %ymm6, 0x80(%rdi)+ vmovdqu 0x1e0(%rsp), %ymm4+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm5 # ymm5 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm0 # ymm0 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm2, 0x210(%rdi)+ vpunpcklqdq %ymm3, %ymm12, %ymm2 # ymm2 = ymm12[0],ymm3[0],ymm12[2],ymm3[2]+ vmovdqu %ymm0, 0x2d8(%rdi)+ vpunpckhqdq %ymm3, %ymm12, %ymm0 # ymm0 = ymm12[1],ymm3[1],ymm12[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm7, %ymm3 # ymm3 = ymm7[0],ymm4[0],ymm7[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm7, %ymm1 # ymm1 = ymm7[1],ymm4[1],ymm7[3],ymm4[3]+ vmovdqu %ymm5, 0x148(%rdi)+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm5 # ymm5 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm3[2,3]+ vmovdqu 0x1c0(%rsp), %ymm3+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm4 # ymm4 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm0 # ymm0 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm5, 0xa0(%rdi)+ vextracti128 $0x1, %ymm3, %xmm15+ vmovdqu %ymm4, 0x168(%rdi)+ vmovdqu %ymm2, 0x230(%rdi)+ vmovdqu %ymm0, 0x2f8(%rdi)+ vmovq %xmm3, 0xc0(%rdi)+ vmovhpd %xmm3, 0x188(%rdi)+ vmovq %xmm15, 0x250(%rdi)+ vmovhpd %xmm15, 0x318(%rdi)+ movq %r11, %rsp+ .cfi_def_cfa_register %rsp+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(keccak_f1600_x4_avx2_asm)++#endif /* MLK_FIPS202_X86_64_NEED_X4_AVX2 && !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/fips202/native/x86_64/src/keccakf1600_constants.c view
@@ -0,0 +1,52 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#include "../../../../common.h"+#if defined(MLK_FIPS202_X86_64_NEED_X4_AVX2) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#include <stdint.h>++#include "fips202_native_x86_64.h"++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint64_t+ mlk_keccakf1600_round_constants[24] = {+ 0x0000000000000001, 0x0000000000008082, 0x800000000000808a,+ 0x8000000080008000, 0x000000000000808b, 0x0000000080000001,+ 0x8000000080008081, 0x8000000000008009, 0x000000000000008a,+ 0x0000000000000088, 0x0000000080008009, 0x000000008000000a,+ 0x000000008000808b, 0x800000000000008b, 0x8000000000008089,+ 0x8000000000008003, 0x8000000000008002, 0x8000000000000080,+ 0x000000000000800a, 0x800000008000000a, 0x8000000080008081,+ 0x8000000000008080, 0x0000000080000001, 0x8000000080008008,+};++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint64_t mlk_keccak_rho8[4] = {+ 0x0605040302010007,+ 0x0e0d0c0b0a09080f,+ 0x1615141312111017,+ 0x1e1d1c1b1a19181f,+};++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint64_t mlk_keccak_rho56[4] = {+ 0x0007060504030201,+ 0x080f0e0d0c0b0a09,+ 0x1017161514131211,+ 0x181f1e1d1c1b1a19,+};++#else /* MLK_FIPS202_X86_64_NEED_X4_AVX2 && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ */++MLK_EMPTY_CU(fips202_x86_64_constants)++#endif /* !(MLK_FIPS202_X86_64_NEED_X4_AVX2 && \+ !MLK_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mlkem/src/indcpa.c view
@@ -0,0 +1,675 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ *+ * - [REF]+ * CRYSTALS-Kyber C reference implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/ref+ */++#include "indcpa.h"++#include "debug.h"+#include "randombytes.h"+#include "sampling.h"+#include "symmetric.h"+#include "verify.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mlkem-native (e.g. with varying parameter sets)+ * within a single compilation unit. */+#define mlk_pack_pk MLK_ADD_PARAM_SET(mlk_pack_pk)+#define mlk_unpack_pk MLK_ADD_PARAM_SET(mlk_unpack_pk)+#define mlk_pack_sk MLK_ADD_PARAM_SET(mlk_pack_sk)+#define mlk_unpack_sk MLK_ADD_PARAM_SET(mlk_unpack_sk)+#define mlk_pack_ciphertext MLK_ADD_PARAM_SET(mlk_pack_ciphertext)+#define mlk_unpack_ciphertext MLK_ADD_PARAM_SET(mlk_unpack_ciphertext)+#define mlk_matvec_mul MLK_ADD_PARAM_SET(mlk_matvec_mul)+#define mlk_polyvec_permute_bitrev_to_custom \+ MLK_ADD_PARAM_SET(mlk_polyvec_permute_bitrev_to_custom)+#define mlk_polymat_permute_bitrev_to_custom \+ MLK_ADD_PARAM_SET(mlk_polymat_permute_bitrev_to_custom)+#define mlk_keypair_getnoise_eta1 MLK_ADD_PARAM_SET(mlk_keypair_getnoise_eta1)+#define mlk_enc_getnoise_eta1_eta2 MLK_ADD_PARAM_SET(mlk_enc_getnoise_eta1_eta2)+/* End of parameter set namespacing */++/**+ * Serialize the public key as the concatenation of the serialized vector of+ * polynomials pk and the public seed used to generate the matrix A.+ *+ * @spec{Implements @[FIPS203, Algorithm 13 (K-PKE.KeyGen), L19].}+ *+ * @param[out] r Output serialized public key.+ * @param[in] pk Input public-key polyvec. Must have coefficients within+ * [0,..,MLKEM_Q-1].+ * @param[in] seed Input public seed.+ */+#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+static void mlk_pack_pk(uint8_t r[MLKEM_INDCPA_PUBLICKEYBYTES],+ const mlk_polyvec *pk,+ const uint8_t seed[MLKEM_SYMBYTES])+{+ mlk_assert_bound_2d(pk->vec, MLKEM_K, MLKEM_N, 0, MLKEM_Q);+ mlk_polyvec_tobytes(r, pk);+ mlk_memcpy(r + MLKEM_POLYVECBYTES, seed, MLKEM_SYMBYTES);+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++/**+ * De-serialize public key from a byte array; approximate inverse of+ * mlk_pack_pk.+ *+ * @spec{Implements @[FIPS203, Algorithm 14 (K-PKE.Encrypt), L2-3].}+ *+ * @param[out] pk Output public-key polynomial vector. Coefficients+ * will be normalized to [0,1,..,MLKEM_Q-1].+ * @param[out] seed Output seed to generate matrix A.+ * @param[in] packedpk Input serialized public key.+ */+#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+static void mlk_unpack_pk(mlk_polyvec *pk, uint8_t seed[MLKEM_SYMBYTES],+ const uint8_t packedpk[MLKEM_INDCPA_PUBLICKEYBYTES])+{+ mlk_polyvec_frombytes(pk, packedpk);+ mlk_memcpy(seed, packedpk + MLKEM_POLYVECBYTES, MLKEM_SYMBYTES);++ /* NOTE: If a modulus check was conducted on the PK, we know at this+ * point that the coefficients of `pk` are unsigned canonical. The+ * specifications and proofs, however, do _not_ assume this, and instead+ * work with the easily provable bound by MLKEM_UINT12_LIMIT. */+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++/**+ * Serialize the secret key.+ *+ * @spec{Implements @[FIPS203, Algorithm 13 (K-PKE.KeyGen), L20].}+ *+ * @param[out] r Output serialized secret key.+ * @param[in] sk Input vector of polynomials (secret key).+ */+#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+static void mlk_pack_sk(uint8_t r[MLKEM_INDCPA_SECRETKEYBYTES],+ const mlk_polyvec *sk)+{+ mlk_assert_bound_2d(sk->vec, MLKEM_K, MLKEM_N, 0, MLKEM_Q);+ mlk_polyvec_tobytes(r, sk);+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++/**+ * De-serialize the secret key; inverse of mlk_pack_sk.+ *+ * @spec{Implements @[FIPS203, Algorithm 15 (K-PKE.Decrypt), L5].}+ *+ * @param[out] sk Output vector of polynomials (secret key).+ * @param[in] packedsk Input serialized secret key.+ */+#if !defined(MLK_CONFIG_NO_DECAPS_API)+static void mlk_unpack_sk(mlk_polyvec *sk,+ const uint8_t packedsk[MLKEM_INDCPA_SECRETKEYBYTES])+{+ mlk_polyvec_frombytes(sk, packedsk);+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++/**+ * Serialize the ciphertext as the concatenation of the compressed and+ * serialized vector of polynomials b and the compressed and serialized+ * polynomial v.+ *+ * @spec{Implements @[FIPS203, Algorithm 14 (K-PKE.Encrypt), L22-23].}+ *+ * @param[out] r Output serialized ciphertext.+ * @param[in] b Input vector of polynomials b.+ * @param[in] v Input polynomial v.+ */+#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+static void mlk_pack_ciphertext(uint8_t r[MLKEM_INDCPA_BYTES],+ const mlk_polyvec *b, mlk_poly *v)+{+ mlk_polyvec_compress_du(r, b);+ mlk_poly_compress_dv(r + MLKEM_POLYVECCOMPRESSEDBYTES_DU, v);+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++/**+ * De-serialize and decompress ciphertext from a byte array; approximate+ * inverse of mlk_pack_ciphertext.+ *+ * @spec{Implements @[FIPS203, Algorithm 15 (K-PKE.Decrypt), L1-4].}+ *+ * @param[out] b Output vector of polynomials b.+ * @param[out] v Output polynomial v.+ * @param[in] c Input serialized ciphertext.+ */+#if !defined(MLK_CONFIG_NO_DECAPS_API)+static void mlk_unpack_ciphertext(mlk_polyvec *b, mlk_poly *v,+ const uint8_t c[MLKEM_INDCPA_BYTES])+{+ mlk_polyvec_decompress_du(b, c);+ mlk_poly_decompress_dv(v, c + MLKEM_POLYVECCOMPRESSEDBYTES_DU);+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++/* Helper function to ensure that the polynomial entries in the output+ * of gen_matrix use the standard (bitreversed) ordering of coefficients.+ * No-op unless a native backend with a custom ordering is used.+ *+ * We don't inline this into gen_matrix to avoid having to split the CBMC+ * proof for gen_matrix based on MLK_USE_NATIVE_NTT_CUSTOM_ORDER. */+static void mlk_polyvec_permute_bitrev_to_custom(mlk_polyvec *v)+__contract__(+ /* We don't specify that this should be a permutation, but only+ * that it does not change the bound established at the end of mlk_gen_matrix. */+ requires(memory_no_alias(v, sizeof(mlk_polyvec)))+ requires(forall(x, 0, MLKEM_K,+ array_bound(v->vec[x].coeffs, 0, MLKEM_N, 0, MLKEM_Q)))+ assigns(memory_slice(v, sizeof(mlk_polyvec)))+ ensures(forall(x, 0, MLKEM_K,+ array_bound(v->vec[x].coeffs, 0, MLKEM_N, 0, MLKEM_Q))))+{+#if defined(MLK_USE_NATIVE_NTT_CUSTOM_ORDER)+ unsigned i;+ for (i = 0; i < MLKEM_K; i++)+ __loop__(+ assigns(i, memory_slice(v, sizeof(mlk_polyvec)))+ invariant(i <= MLKEM_K)+ invariant(forall(x, 0, MLKEM_K,+ array_bound(v->vec[x].coeffs, 0, MLKEM_N, 0, MLKEM_Q)))+ decreases(MLKEM_K - i))+ {+ mlk_poly_permute_bitrev_to_custom(v->vec[i].coeffs);+ }+#else /* MLK_USE_NATIVE_NTT_CUSTOM_ORDER */+ /* Nothing to do */+ (void)v;+#endif /* !MLK_USE_NATIVE_NTT_CUSTOM_ORDER */+}++static void mlk_polymat_permute_bitrev_to_custom(mlk_polymat *a)+__contract__(+ /* We don't specify that this should be a permutation, but only+ * that it does not change the bound established at the end of mlk_gen_matrix. */+ requires(memory_no_alias(a, sizeof(mlk_polymat)))+ requires(forall(x, 0, MLKEM_K, forall(y, 0, MLKEM_K,+ array_bound(a->vec[x].vec[y].coeffs, 0, MLKEM_N, 0, MLKEM_Q))))+ assigns(memory_slice(a, sizeof(mlk_polymat)))+ ensures(forall(x, 0, MLKEM_K, forall(y, 0, MLKEM_K,+ array_bound(a->vec[x].vec[y].coeffs, 0, MLKEM_N, 0, MLKEM_Q)))))+{+ unsigned i;+ for (i = 0; i < MLKEM_K; i++)+ __loop__(+ assigns(i, memory_slice(a, sizeof(mlk_polymat)))+ invariant(i <= MLKEM_K)+ invariant(forall(x, 0, MLKEM_K, forall(y, 0, MLKEM_K,+ array_bound(a->vec[x].vec[y].coeffs, 0, MLKEM_N, 0, MLKEM_Q))))+ decreases(MLKEM_K - i))+ {+ mlk_polyvec_permute_bitrev_to_custom(&a->vec[i]);+ }+}++/* Reference: `gen_matrix()` in the reference implementation @[REF].+ * - We use a special subroutine to generate 4 polynomials+ * at a time, to be able to leverage batched Keccak-f1600+ * implementations. The reference implementation generates+ * one matrix entry a time.+ *+ * Not static for benchmarking */+MLK_INTERNAL_API+void mlk_gen_matrix(mlk_polymat *a, const uint8_t seed[MLKEM_SYMBYTES],+ int transposed)+{+ unsigned i, j;+ MLK_ALIGN uint8_t seed_ext[4][MLK_ALIGN_UP(MLKEM_SYMBYTES + 2)];++ for (j = 0; j < 4; j++)+ {+ mlk_memcpy(seed_ext[j], seed, MLKEM_SYMBYTES);+ }++#if !defined(MLK_CONFIG_SERIAL_FIPS202_ONLY)+ /* Sample 4 matrix entries a time. */+ for (i = 0; i < (MLKEM_K * MLKEM_K / 4) * 4; i += 4)+ {+ for (j = 0; j < 4; j++)+ {+ uint8_t x, y;+ /* MLKEM_K <= 4, so the values fit in uint8_t. */+ x = (uint8_t)((i + j) / MLKEM_K);+ y = (uint8_t)((i + j) % MLKEM_K);+ if (transposed)+ {+ seed_ext[j][MLKEM_SYMBYTES + 0] = x;+ seed_ext[j][MLKEM_SYMBYTES + 1] = y;+ }+ else+ {+ seed_ext[j][MLKEM_SYMBYTES + 0] = y;+ seed_ext[j][MLKEM_SYMBYTES + 1] = x;+ }+ }++ mlk_poly_rej_uniform_x4(&a->vec[i / MLKEM_K].vec[i % MLKEM_K],+ &a->vec[(i + 1) / MLKEM_K].vec[(i + 1) % MLKEM_K],+ &a->vec[(i + 2) / MLKEM_K].vec[(i + 2) % MLKEM_K],+ &a->vec[(i + 3) / MLKEM_K].vec[(i + 3) % MLKEM_K],+ seed_ext);+ }+#else /* !MLK_CONFIG_SERIAL_FIPS202_ONLY */+ /* When using serial FIPS202, sample all entries individually. */+ i = 0;+#endif /* MLK_CONFIG_SERIAL_FIPS202_ONLY */++ /* For MLKEM_K == 3, sample the last entry individually.+ * When MLK_CONFIG_SERIAL_FIPS202_ONLY is set, sample all entries+ * individually. */+ for (; i < MLKEM_K * MLKEM_K; i++)+ {+ uint8_t x, y;+ /* MLKEM_K <= 4, so the values fit in uint8_t. */+ x = (uint8_t)(i / MLKEM_K);+ y = (uint8_t)(i % MLKEM_K);++ if (transposed)+ {+ seed_ext[0][MLKEM_SYMBYTES + 0] = x;+ seed_ext[0][MLKEM_SYMBYTES + 1] = y;+ }+ else+ {+ seed_ext[0][MLKEM_SYMBYTES + 0] = y;+ seed_ext[0][MLKEM_SYMBYTES + 1] = x;+ }++ mlk_poly_rej_uniform(&a->vec[i / MLKEM_K].vec[i % MLKEM_K], seed_ext[0]);+ }++ mlk_assert(i == MLKEM_K * MLKEM_K);++ /*+ * The public matrix is generated in NTT domain. If the native backend+ * uses a custom order in NTT domain, permute A accordingly.+ */+ mlk_polymat_permute_bitrev_to_custom(a);++ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(seed_ext, sizeof(seed_ext));+}++/**+ * Compute matrix-vector product in NTT domain, via Montgomery multiplication.+ *+ * @spec{Implements @[FIPS203, Section 2.4.7, Eq (2.12), (2.13)].}+ *+ * @param[out] out Output polynomial vector.+ * @param[in] a Input matrix. Must be in NTT domain and have coefficients+ * of absolute value < 4096.+ * @param[in] v Input polynomial vector. Must be in NTT domain.+ * @param[in] vc Mulcache for @p v, computed via+ * mlk_polyvec_mulcache_compute().+ */+static void mlk_matvec_mul(mlk_polyvec *out, const mlk_polymat *a,+ const mlk_polyvec *v, const mlk_polyvec_mulcache *vc)+__contract__(+ requires(memory_no_alias(out, sizeof(mlk_polyvec)))+ requires(memory_no_alias(a, sizeof(mlk_polymat)))+ requires(memory_no_alias(v, sizeof(mlk_polyvec)))+ requires(memory_no_alias(vc, sizeof(mlk_polyvec_mulcache)))+ requires(forall(k0, 0, MLKEM_K,+ forall(k1, 0, MLKEM_K,+ array_bound(a->vec[k0].vec[k1].coeffs, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT))))+ assigns(memory_slice(out, sizeof(mlk_polyvec))))+{+ unsigned i;+ for (i = 0; i < MLKEM_K; i++)+ __loop__(+ assigns(i, memory_slice(out, sizeof(mlk_polyvec)))+ invariant(i <= MLKEM_K)+ decreases(MLKEM_K - i))+ {+ mlk_polyvec_basemul_acc_montgomery_cached(&out->vec[i], &a->vec[i], v, vc);+ }+}++/**+ * Compute and fill the pv and e polyvec structures needed by+ * mlk_keypair_derand(). Uses x4-batched versions of `poly_getnoise` to+ * leverage batched Keccak-f1600.+ *+ * @spec{Implements @[FIPS203, Algorithm 13 (K-PKE.KeyGen)] steps 8-15.}+ *+ * @param[out] pv Output polynomial vector.+ * @param[out] e Output polynomial vector.+ * @param[in] seed Seed bytes for sampling.+ */+#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+static void mlk_keypair_getnoise_eta1(mlk_polyvec *pv, mlk_polyvec *e,+ const uint8_t seed[MLKEM_SYMBYTES])+__contract__(+ requires(memory_no_alias(pv, sizeof(mlk_polyvec)))+ requires(memory_no_alias(e, sizeof(mlk_polyvec)))+ requires(memory_no_alias(seed, MLKEM_SYMBYTES))+ assigns(memory_slice(pv, sizeof(mlk_polyvec)))+ assigns(memory_slice(e, sizeof(mlk_polyvec)))+ ensures(forall(k0, 0, MLKEM_K, array_abs_bound(pv->vec[k0].coeffs, 0, MLKEM_N, MLKEM_ETA1 + 1)))+ ensures(forall(k1, 0, MLKEM_K, array_abs_bound(e->vec[k1].coeffs, 0, MLKEM_N, MLKEM_ETA1 + 1)))+)+{+#if MLKEM_K == 2+ mlk_poly_getnoise_eta1_4x(&pv->vec[0], &pv->vec[1], /* Fill elements of pv */+ &e->vec[0], &e->vec[1], /* and two elements of e */+ seed, 0, 1, 2, 3);+#elif MLKEM_K == 3+ /*+ * Only the first three output buffers are needed, so we pass NULL as+ * the fourth parameter, and 0xFF as its dummy nonce.+ */+ mlk_poly_getnoise_eta1_4x(&pv->vec[0], &pv->vec[1], &pv->vec[2], NULL, seed,+ 0, 1, 2, 0xFF);+ /* Same here */+ mlk_poly_getnoise_eta1_4x(&e->vec[0], &e->vec[1], &e->vec[2], NULL, seed, 3,+ 4, 5, 0xFF);+#elif MLKEM_K == 4+ mlk_poly_getnoise_eta1_4x(&pv->vec[0], &pv->vec[1], &pv->vec[2], &pv->vec[3],+ seed, 0, 1, 2, 3);+ mlk_poly_getnoise_eta1_4x(&e->vec[0], &e->vec[1], &e->vec[2], &e->vec[3],+ seed, 4, 5, 6, 7);+#endif /* MLKEM_K == 4 */+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++/**+ * Compute and fill the sp, ep, and epp polynomial structures needed by+ * mlk_indcpa_enc(). Uses x4-batched versions of `poly_getnoise` to leverage+ * batched Keccak-f1600.+ *+ * @spec{Implements @[FIPS203, Algorithm 14 (K-PKE.Encrypt)] steps 9-16.}+ *+ * @param[out] sp Output polynomial vector.+ * @param[out] ep Output polynomial vector.+ * @param[out] epp Output polynomial.+ * @param[in] coins Seed bytes for sampling.+ */+#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+static void mlk_enc_getnoise_eta1_eta2(mlk_polyvec *sp, mlk_polyvec *ep,+ mlk_poly *epp,+ const uint8_t coins[MLKEM_SYMBYTES])+__contract__(+ requires(memory_no_alias(sp, sizeof(mlk_polyvec)))+ requires(memory_no_alias(ep, sizeof(mlk_polyvec)))+ requires(memory_no_alias(epp, sizeof(mlk_poly)))+ requires(memory_no_alias(coins, MLKEM_SYMBYTES))+ assigns(memory_slice(sp, sizeof(mlk_polyvec)))+ assigns(memory_slice(ep, sizeof(mlk_polyvec)))+ assigns(memory_slice(epp, sizeof(mlk_poly)))+ ensures(forall(k0, 0, MLKEM_K, array_abs_bound(sp->vec[k0].coeffs, 0, MLKEM_N, MLKEM_ETA1 + 1)))+ ensures(forall(k1, 0, MLKEM_K, array_abs_bound(ep->vec[k1].coeffs, 0, MLKEM_N, MLKEM_ETA2 + 1)))+ ensures(array_abs_bound(epp->coeffs, 0, MLKEM_N, MLKEM_ETA2 + 1))+)+{+#if MLKEM_K == 2+ mlk_poly_getnoise_eta1122_4x(&sp->vec[0], &sp->vec[1], &ep->vec[0],+ &ep->vec[1], coins, 0, 1, 2, 3);+ mlk_poly_getnoise_eta2(epp, coins, 4);+#elif MLKEM_K == 3+ /*+ * In this call, only the first three output buffers are needed,+ * so we pass NULL as the fourth parameter, and 0xFF as its dummy nonce.+ */+ mlk_poly_getnoise_eta1_4x(&sp->vec[0], &sp->vec[1], &sp->vec[2], NULL, coins,+ 0, 1, 2, 0xFF /* irrelevant */);+ /* The fourth output buffer in this call _is_ used. */+ mlk_poly_getnoise_eta2_4x(&ep->vec[0], &ep->vec[1], &ep->vec[2], epp, coins,+ 3, 4, 5, 6);+#elif MLKEM_K == 4+ mlk_poly_getnoise_eta1_4x(&sp->vec[0], &sp->vec[1], &sp->vec[2], &sp->vec[3],+ coins, 0, 1, 2, 3);+ mlk_poly_getnoise_eta2_4x(&ep->vec[0], &ep->vec[1], &ep->vec[2], &ep->vec[3],+ coins, 4, 5, 6, 7);+ mlk_poly_getnoise_eta2(epp, coins, 8);+#endif /* MLKEM_K == 4 */+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */+++/* Reference: `indcpa_keypair_derand()` in the reference implementation @[REF].+ * - We use a different implementation of `gen_matrix()` which+ * uses x4-batched Keccak-f1600 (see `mlk_gen_matrix()` above).+ * - We use a mulcache to speed up matrix-vector multiplication.+ * - We include buffer zeroization.+ */+#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+MLK_INTERNAL_API+int mlk_indcpa_keypair_derand(uint8_t pk[MLKEM_INDCPA_PUBLICKEYBYTES],+ uint8_t sk[MLKEM_INDCPA_SECRETKEYBYTES],+ const uint8_t coins[MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ const uint8_t *publicseed;+ const uint8_t *noiseseed;+ MLK_ALLOC(buf, uint8_t, 2 * MLKEM_SYMBYTES, context);+ MLK_ALLOC(coins_with_domain_separator, uint8_t, MLKEM_SYMBYTES + 1, context);+ MLK_ALLOC(a, mlk_polymat, 1, context);+ MLK_ALLOC(e, mlk_polyvec, 1, context);+ MLK_ALLOC(pkpv, mlk_polyvec, 1, context);+ MLK_ALLOC(skpv, mlk_polyvec, 1, context);+ MLK_ALLOC(skpv_cache, mlk_polyvec_mulcache, 1, context);++ if (buf == NULL || coins_with_domain_separator == NULL || a == NULL ||+ e == NULL || pkpv == NULL || skpv == NULL || skpv_cache == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ publicseed = buf;+ noiseseed = buf + MLKEM_SYMBYTES;++ /* Concatenate coins with MLKEM_K for domain separation of security levels */+ mlk_memcpy(coins_with_domain_separator, coins, MLKEM_SYMBYTES);+ coins_with_domain_separator[MLKEM_SYMBYTES] = MLKEM_K;++ mlk_hash_g(buf, coins_with_domain_separator, MLKEM_SYMBYTES + 1);++ /*+ * Declassify the public seed.+ * Required to use it in conditional-branches in rejection sampling.+ * This is needed because all output of randombytes is marked as secret+ * (=undefined)+ */+ MLK_CT_TESTING_DECLASSIFY(publicseed, MLKEM_SYMBYTES);++ mlk_gen_matrix(a, publicseed, 0 /* no transpose */);++ mlk_keypair_getnoise_eta1(skpv, e, noiseseed);++ mlk_polyvec_ntt(skpv);+ mlk_polyvec_ntt(e);++ mlk_polyvec_mulcache_compute(skpv_cache, skpv);+ mlk_matvec_mul(pkpv, a, skpv, skpv_cache);+ mlk_polyvec_tomont(pkpv);++ mlk_polyvec_add(pkpv, e);+ mlk_polyvec_reduce(pkpv);+ mlk_polyvec_reduce(skpv);++ mlk_pack_sk(sk, skpv);+ mlk_pack_pk(pk, pkpv, publicseed);++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(skpv_cache, mlk_polyvec_mulcache, 1, context);+ MLK_FREE(skpv, mlk_polyvec, 1, context);+ MLK_FREE(pkpv, mlk_polyvec, 1, context);+ MLK_FREE(e, mlk_polyvec, 1, context);+ MLK_FREE(a, mlk_polymat, 1, context);+ MLK_FREE(coins_with_domain_separator, uint8_t, MLKEM_SYMBYTES + 1, context);+ MLK_FREE(buf, uint8_t, 2 * MLKEM_SYMBYTES, context);+ return ret;+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++/* Reference: `indcpa_enc()` in the reference implementation @[REF].+ * - We use x4-batched versions of `poly_getnoise` to leverage+ * batched x4-batched Keccak-f1600.+ * - We use a different implementation of `gen_matrix()` which+ * uses x4-batched Keccak-f1600 (see `mlk_gen_matrix()` above).+ * - We use a mulcache to speed up matrix-vector multiplication.+ * - We include buffer zeroization.+ */+#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_INTERNAL_API+int mlk_indcpa_enc(uint8_t c[MLKEM_INDCPA_BYTES],+ const uint8_t m[MLKEM_INDCPA_MSGBYTES],+ const uint8_t pk[MLKEM_INDCPA_PUBLICKEYBYTES],+ const uint8_t coins[MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(seed, uint8_t, MLKEM_SYMBYTES, context);+ MLK_ALLOC(at, mlk_polymat, 1, context);+ MLK_ALLOC(sp, mlk_polyvec, 1, context);+ MLK_ALLOC(pkpv, mlk_polyvec, 1, context);+ MLK_ALLOC(ep, mlk_polyvec, 1, context);+ MLK_ALLOC(b, mlk_polyvec, 1, context);+ MLK_ALLOC(v, mlk_poly, 1, context);+ MLK_ALLOC(k, mlk_poly, 1, context);+ MLK_ALLOC(epp, mlk_poly, 1, context);+ MLK_ALLOC(sp_cache, mlk_polyvec_mulcache, 1, context);++ if (seed == NULL || at == NULL || sp == NULL || pkpv == NULL || ep == NULL ||+ b == NULL || v == NULL || k == NULL || epp == NULL || sp_cache == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ mlk_unpack_pk(pkpv, seed, pk);+ mlk_poly_frommsg(k, m);++ /*+ * Declassify the public seed.+ * Required to use it in conditional-branches in rejection sampling.+ * This is needed because in re-encryption the publicseed originated from sk+ * which is marked undefined.+ */+ MLK_CT_TESTING_DECLASSIFY(seed, MLKEM_SYMBYTES);++ mlk_gen_matrix(at, seed, 1 /* transpose */);++ mlk_enc_getnoise_eta1_eta2(sp, ep, epp, coins);++ mlk_polyvec_ntt(sp);++ mlk_polyvec_mulcache_compute(sp_cache, sp);+ mlk_matvec_mul(b, at, sp, sp_cache);+ mlk_polyvec_basemul_acc_montgomery_cached(v, pkpv, sp, sp_cache);++ mlk_polyvec_invntt_tomont(b);+ mlk_poly_invntt_tomont(v);++ mlk_polyvec_add(b, ep);+ mlk_poly_add(v, epp);+ mlk_poly_add(v, k);++ mlk_polyvec_reduce(b);+ mlk_poly_reduce(v);++ mlk_pack_ciphertext(c, b, v);++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(sp_cache, mlk_polyvec_mulcache, 1, context);+ MLK_FREE(epp, mlk_poly, 1, context);+ MLK_FREE(k, mlk_poly, 1, context);+ MLK_FREE(v, mlk_poly, 1, context);+ MLK_FREE(b, mlk_polyvec, 1, context);+ MLK_FREE(ep, mlk_polyvec, 1, context);+ MLK_FREE(pkpv, mlk_polyvec, 1, context);+ MLK_FREE(sp, mlk_polyvec, 1, context);+ MLK_FREE(at, mlk_polymat, 1, context);+ MLK_FREE(seed, uint8_t, MLKEM_SYMBYTES, context);+ return ret;+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++/* Reference: `indcpa_dec()` in the reference implementation @[REF].+ * - We use a mulcache for the scalar product.+ * - We include buffer zeroization. */+#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_INTERNAL_API+int mlk_indcpa_dec(uint8_t m[MLKEM_INDCPA_MSGBYTES],+ const uint8_t c[MLKEM_INDCPA_BYTES],+ const uint8_t sk[MLKEM_INDCPA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(b, mlk_polyvec, 1, context);+ MLK_ALLOC(skpv, mlk_polyvec, 1, context);+ MLK_ALLOC(v, mlk_poly, 1, context);+ MLK_ALLOC(sb, mlk_poly, 1, context);+ MLK_ALLOC(b_cache, mlk_polyvec_mulcache, 1, context);++ if (b == NULL || skpv == NULL || v == NULL || sb == NULL || b_cache == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ mlk_unpack_ciphertext(b, v, c);+ mlk_unpack_sk(skpv, sk);++ mlk_polyvec_ntt(b);+ mlk_polyvec_mulcache_compute(b_cache, b);+ mlk_polyvec_basemul_acc_montgomery_cached(sb, skpv, b, b_cache);+ mlk_poly_invntt_tomont(sb);++ mlk_poly_sub(v, sb);+ mlk_poly_reduce(v);++ mlk_poly_tomsg(m, v);++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(b_cache, mlk_polyvec_mulcache, 1, context);+ MLK_FREE(sb, mlk_poly, 1, context);+ MLK_FREE(v, mlk_poly, 1, context);+ MLK_FREE(skpv, mlk_polyvec, 1, context);+ MLK_FREE(b, mlk_polyvec, 1, context);+ return ret;+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef mlk_pack_pk+#undef mlk_unpack_pk+#undef mlk_pack_sk+#undef mlk_unpack_sk+#undef mlk_pack_ciphertext+#undef mlk_unpack_ciphertext+#undef mlk_matvec_mul+#undef mlk_polyvec_permute_bitrev_to_custom+#undef mlk_polymat_permute_bitrev_to_custom+#undef mlk_keypair_getnoise_eta1+#undef mlk_enc_getnoise_eta1_eta2
+ cbits/mlkem/src/indcpa.h view
@@ -0,0 +1,172 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#ifndef MLK_INDCPA_H+#define MLK_INDCPA_H++#include "cbmc.h"+#include "common.h"+#include "poly_k.h"++#define mlk_gen_matrix MLK_NAMESPACE_K(gen_matrix)+/**+ * Deterministically generate matrix A (or the transpose of A) from a seed.+ * Entries of the matrix are polynomials that look uniformly random.+ * Performs rejection sampling on the output of an XOF.+ *+ * @spec{Implements @[FIPS203, Algorithm 13 (K-PKE.KeyGen), L3-7] and+ * @[FIPS203, Algorithm 14 (K-PKE.Encrypt), L4-8]. The @p transposed+ * parameter only affects internal presentation.}+ *+ * @param[out] a Output matrix A.+ * @param[in] seed Input seed.+ * @param transposed Boolean deciding whether A or A^T is generated.+ */+MLK_INTERNAL_API+void mlk_gen_matrix(mlk_polymat *a, const uint8_t seed[MLKEM_SYMBYTES],+ int transposed)+__contract__(+ requires(memory_no_alias(a, sizeof(mlk_polymat)))+ requires(memory_no_alias(seed, MLKEM_SYMBYTES))+ requires(transposed == 0 || transposed == 1)+ assigns(memory_slice(a, sizeof(mlk_polymat)))+ ensures(forall(x, 0, MLKEM_K, forall(y, 0, MLKEM_K,+ array_bound(a->vec[x].vec[y].coeffs, 0, MLKEM_N, 0, MLKEM_Q))))+);++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+#define mlk_indcpa_keypair_derand \+ MLK_NAMESPACE_K(indcpa_keypair_derand) MLK_CONTEXT_PARAMETERS_3+/**+ * Generate public and private key for the CPA-secure public-key encryption+ * scheme underlying ML-KEM.+ *+ * @spec{Implements @[FIPS203, Algorithm 13 (K-PKE.KeyGen)].}+ *+ * @param[out] pk Output public key+ * (length MLKEM_INDCPA_PUBLICKEYBYTES bytes).+ * @param[out] sk Output private key+ * (length MLKEM_INDCPA_SECRETKEYBYTES bytes).+ * @param[in] coins Input randomness (length MLKEM_SYMBYTES bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+MLK_INTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_indcpa_keypair_derand(uint8_t pk[MLKEM_INDCPA_PUBLICKEYBYTES],+ uint8_t sk[MLKEM_INDCPA_SECRETKEYBYTES],+ const uint8_t coins[MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLKEM_INDCPA_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLKEM_INDCPA_SECRETKEYBYTES))+ requires(memory_no_alias(coins, MLKEM_SYMBYTES))+ assigns(memory_slice(pk, MLKEM_INDCPA_PUBLICKEYBYTES))+ assigns(memory_slice(sk, MLKEM_INDCPA_SECRETKEYBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_OUT_OF_MEMORY)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(pk, MLKEM_INDCPA_PUBLICKEYBYTES))+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sk, MLKEM_INDCPA_SECRETKEYBYTES))+);+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_indcpa_enc MLK_NAMESPACE_K(indcpa_enc) MLK_CONTEXT_PARAMETERS_4+/**+ * Encryption function of the CPA-secure public-key encryption scheme+ * underlying ML-KEM.+ *+ * @spec{Implements @[FIPS203, Algorithm 14 (K-PKE.Encrypt)].}+ *+ * @param[out] c Output ciphertext (length MLKEM_INDCPA_BYTES bytes).+ * @param[in] m Input message (length MLKEM_INDCPA_MSGBYTES bytes).+ * @param[in] pk Input public key+ * (length MLKEM_INDCPA_PUBLICKEYBYTES bytes).+ * @param[in] coins Input random coins used as seed (length MLKEM_SYMBYTES+ * bytes) to deterministically generate all randomness.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+MLK_INTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_indcpa_enc(uint8_t c[MLKEM_INDCPA_BYTES],+ const uint8_t m[MLKEM_INDCPA_MSGBYTES],+ const uint8_t pk[MLKEM_INDCPA_PUBLICKEYBYTES],+ const uint8_t coins[MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(c, MLKEM_INDCPA_BYTES))+ requires(memory_no_alias(m, MLKEM_INDCPA_MSGBYTES))+ requires(memory_no_alias(pk, MLKEM_INDCPA_PUBLICKEYBYTES))+ requires(memory_no_alias(coins, MLKEM_SYMBYTES))+ assigns(memory_slice(c, MLKEM_INDCPA_BYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_OUT_OF_MEMORY)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(c, MLKEM_INDCPA_BYTES))+);+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_indcpa_dec MLK_NAMESPACE_K(indcpa_dec) MLK_CONTEXT_PARAMETERS_3+/**+ * Decryption function of the CPA-secure public-key encryption scheme+ * underlying ML-KEM.+ *+ * @spec{Implements @[FIPS203, Algorithm 15 (K-PKE.Decrypt)].}+ *+ * @param[out] m Output decrypted message+ * (length MLKEM_INDCPA_MSGBYTES bytes).+ * @param[in] c Input ciphertext (length MLKEM_INDCPA_BYTES bytes).+ * @param[in] sk Input secret key+ * (length MLKEM_INDCPA_SECRETKEYBYTES bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+MLK_INTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_indcpa_dec(uint8_t m[MLKEM_INDCPA_MSGBYTES],+ const uint8_t c[MLKEM_INDCPA_BYTES],+ const uint8_t sk[MLKEM_INDCPA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(c, MLKEM_INDCPA_BYTES))+ requires(memory_no_alias(m, MLKEM_INDCPA_MSGBYTES))+ requires(memory_no_alias(sk, MLKEM_INDCPA_SECRETKEYBYTES))+ assigns(memory_slice(m, MLKEM_INDCPA_MSGBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_OUT_OF_MEMORY)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(m, MLKEM_INDCPA_MSGBYTES))+);+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#endif /* !MLK_INDCPA_H */
+ cbits/mlkem/src/kem.c view
@@ -0,0 +1,474 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS140_3_IG]+ * Implementation Guidance for FIPS 140-3 and the Cryptographic Module+ * Validation Program+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/projects/cryptographic-module-validation-program/fips-140-3-ig-announcements+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ *+ * - [REF]+ * CRYSTALS-Kyber C reference implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/ref+ */++#include "kem.h"++#include "indcpa.h"+#include "randombytes.h"+#include "symmetric.h"+#include "verify.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mlkem-native (e.g. with varying security levels)+ * within a single compilation unit. */+#define mlk_check_pct MLK_ADD_PARAM_SET(mlk_check_pct) MLK_CONTEXT_PARAMETERS_2+/* End of parameter set namespacing */++#if !defined(MLK_CONFIG_NO_ENCAPS_API)+/* Reference: Not implemented in the reference implementation @[REF]. */+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_check_pk(const uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(p, mlk_polyvec, 1, context);+ MLK_ALLOC(p_reencoded, uint8_t, MLKEM_POLYVECBYTES, context);++ if (p == NULL || p_reencoded == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ mlk_polyvec_frombytes(p, pk);+ mlk_polyvec_reduce(p);+ mlk_polyvec_tobytes(p_reencoded, p);++ /* We use a constant-time memcmp here to avoid having to+ * declassify the PK before the PCT has succeeded. */+ ret = mlk_ct_memcmp(pk, p_reencoded, MLKEM_POLYVECBYTES) ? MLK_ERR_INVALID_PK+ : 0;++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(p_reencoded, uint8_t, MLKEM_POLYVECBYTES, context);+ MLK_FREE(p, mlk_polyvec, 1, context);+ return ret;+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API */+++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: Not implemented in the reference implementation @[REF]. */+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_check_sk(const uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(test, uint8_t, MLKEM_SYMBYTES, context);++ if (test == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /*+ * The parts of `sk` being hashed and compared here are public, so+ * no public information is leaked through the runtime or the return value+ * of this function.+ */++ /* Declassify the public part of the secret key */+ MLK_CT_TESTING_DECLASSIFY(sk + MLKEM_INDCPA_SECRETKEYBYTES,+ MLKEM_INDCCA_PUBLICKEYBYTES);+ MLK_CT_TESTING_DECLASSIFY(+ sk + MLKEM_INDCCA_SECRETKEYBYTES - 2 * MLKEM_SYMBYTES, MLKEM_SYMBYTES);++ mlk_hash_h(test, sk + MLKEM_INDCPA_SECRETKEYBYTES,+ MLKEM_INDCCA_PUBLICKEYBYTES);+ /* This doesn't have to be a constant-time memcmp, but it's the only place+ * in the library where a normal memcmp would be used otherwise, so for sake+ * of minimizing stdlib dependency, we use our constant-time one anyway. */+ ret = mlk_ct_memcmp(sk + MLKEM_INDCCA_SECRETKEYBYTES - 2 * MLKEM_SYMBYTES,+ test, MLKEM_SYMBYTES)+ ? MLK_ERR_INVALID_SK+ : 0;++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(test, uint8_t, MLKEM_SYMBYTES, context);+ return ret;+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+MLK_MUST_CHECK_RETURN_VALUE+static int mlk_check_pct(uint8_t const pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ uint8_t const sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLKEM_INDCCA_SECRETKEYBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_OUT_OF_MEMORY ||+ return_value == MLK_ERR_RNG_FAIL ||+ return_value == MLK_ERR_PCT_FAIL)+);++#if defined(MLK_CONFIG_KEYGEN_PCT)+/* Specification:+ * Partially implements 'Pairwise Consistency Test' @[FIPS140_3_IG, p.87] and+ * @[FIPS203, Section 7.1, Pairwise Consistency]. */++/* Reference: Not implemented in the reference implementation @[REF].+ *+ * @retval 0 Success.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ * @retval MLK_ERR_RNG_FAIL Random number generation failed.+ * @retval MLK_ERR_PCT_FAIL The consistency check failed. */+MLK_MUST_CHECK_RETURN_VALUE+static int mlk_check_pct(uint8_t const pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ uint8_t const sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(ct, uint8_t, MLKEM_INDCCA_CIPHERTEXTBYTES, context);+ MLK_ALLOC(ss_enc, uint8_t, MLKEM_SSBYTES, context);+ MLK_ALLOC(ss_dec, uint8_t, MLKEM_SSBYTES, context);++ if (ct == NULL || ss_enc == NULL || ss_dec == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ ret = mlk_kem_enc(ct, ss_enc, pk, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ ret = mlk_kem_dec(ss_dec, ct, sk, context);+ if (ret != 0)+ {+ goto cleanup;+ }++#if defined(MLK_CONFIG_KEYGEN_PCT_BREAKAGE_TEST)+ /* Deliberately break PCT for testing purposes */+ if (mlk_break_pct())+ {+ ss_enc[0] = ~ss_enc[0];+ }+#endif /* MLK_CONFIG_KEYGEN_PCT_BREAKAGE_TEST */++ ret = mlk_ct_memcmp(ss_enc, ss_dec, MLKEM_SSBYTES);+ /* The result of the PCT is public. */+ MLK_CT_TESTING_DECLASSIFY(&ret, sizeof(ret));++ if (ret != 0)+ {+ ret = MLK_ERR_PCT_FAIL;+ }++cleanup:++ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(ss_dec, uint8_t, MLKEM_SSBYTES, context);+ MLK_FREE(ss_enc, uint8_t, MLKEM_SSBYTES, context);+ MLK_FREE(ct, uint8_t, MLKEM_INDCCA_CIPHERTEXTBYTES, context);++ /* The key pair being tested was just generated by this library, so a key+ * check rejecting it hints at a faulty implementation rather than at bad+ * input. Report it as a PCT failure. */+ if (ret == MLK_ERR_INVALID_PK || ret == MLK_ERR_INVALID_SK)+ {+ ret = MLK_ERR_PCT_FAIL;+ }++ /* Other error codes, e.g. platform failures like out of memory or+ * randomness failure, are passed on unmodified. */++ return ret;+}+#else /* MLK_CONFIG_KEYGEN_PCT */+MLK_MUST_CHECK_RETURN_VALUE+static int mlk_check_pct(uint8_t const pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ uint8_t const sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ /* Skip PCT */+ ((void)pk);+ ((void)sk);+ MLK_CONTEXT_UNUSED(context);+ return 0;+}+#endif /* !MLK_CONFIG_KEYGEN_PCT */++/* Reference: `crypto_kem_keypair_derand()` in the reference implementation+ * @[REF].+ * - We optionally include PCT which is not present in+ * the reference code. */+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_keypair_derand(uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ const uint8_t coins[2 * MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret;++ ret = mlk_indcpa_keypair_derand(pk, sk, coins, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ mlk_memcpy(sk + MLKEM_INDCPA_SECRETKEYBYTES, pk, MLKEM_INDCCA_PUBLICKEYBYTES);+ mlk_hash_h(sk + MLKEM_INDCCA_SECRETKEYBYTES - 2 * MLKEM_SYMBYTES, pk,+ MLKEM_INDCCA_PUBLICKEYBYTES);+ /* Value z for pseudo-random output on reject */+ mlk_memcpy(sk + MLKEM_INDCCA_SECRETKEYBYTES - MLKEM_SYMBYTES,+ coins + MLKEM_SYMBYTES, MLKEM_SYMBYTES);++ /* Declassify public key */+ MLK_CT_TESTING_DECLASSIFY(pk, MLKEM_INDCCA_PUBLICKEYBYTES);++ /* Pairwise Consistency Test (PCT) @[FIPS140_3_IG, p.87] */+ ret = mlk_check_pct(pk, sk, context);+ if (ret != 0)+ {+ goto cleanup;+ }++cleanup:+ if (ret != 0)+ {+ mlk_zeroize(pk, MLKEM_INDCCA_PUBLICKEYBYTES);+ mlk_zeroize(sk, MLKEM_INDCCA_SECRETKEYBYTES);+ }++ return ret;+}++#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+/* Reference: `crypto_kem_keypair()` in the reference implementation @[REF]+ * - We zeroize the stack buffer */+MLK_EXTERNAL_API+int mlk_kem_keypair(uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(coins, uint8_t, 2 * MLKEM_SYMBYTES, context);++ if (coins == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Acquire necessary randomness, and mark it as secret. */+ if (mlk_randombytes(coins, 2 * MLKEM_SYMBYTES) != 0)+ {+ ret = MLK_ERR_RNG_FAIL;+ goto cleanup;+ }++ MLK_CT_TESTING_SECRET(coins, 2 * MLKEM_SYMBYTES);++ ret = mlk_kem_keypair_derand(pk, sk, coins, context);++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(coins, uint8_t, 2 * MLKEM_SYMBYTES, context);+ return ret;+}+#endif /* !MLK_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API)+/* Reference: `crypto_kem_enc_derand()` in the reference implementation @[REF]+ * - We include public key check+ * - We include stack buffer zeroization */+MLK_EXTERNAL_API+int mlk_kem_enc_derand(uint8_t ct[MLKEM_INDCCA_CIPHERTEXTBYTES],+ uint8_t ss[MLKEM_SSBYTES],+ const uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ const uint8_t coins[MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(buf, uint8_t, 2 * MLKEM_SYMBYTES, context);+ MLK_ALLOC(kr, uint8_t, 2 * MLKEM_SYMBYTES, context);++ if (buf == NULL || kr == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Specification: Implements @[FIPS203, Section 7.2, Modulus check] */+ ret = mlk_kem_check_pk(pk, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ mlk_memcpy(buf, coins, MLKEM_SYMBYTES);++ /* Multitarget countermeasure for coins + contributory KEM */+ mlk_hash_h(buf + MLKEM_SYMBYTES, pk, MLKEM_INDCCA_PUBLICKEYBYTES);+ mlk_hash_g(kr, buf, 2 * MLKEM_SYMBYTES);++ /* coins are in kr+MLKEM_SYMBYTES */+ ret = mlk_indcpa_enc(ct, buf, pk, kr + MLKEM_SYMBYTES, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ mlk_memcpy(ss, kr, MLKEM_SYMBYTES);++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(kr, uint8_t, 2 * MLKEM_SYMBYTES, context);+ MLK_FREE(buf, uint8_t, 2 * MLKEM_SYMBYTES, context);+ return ret;+}++#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+/* Reference: `crypto_kem_enc()` in the reference implementation @[REF]+ * - We include stack buffer zeroization */+MLK_EXTERNAL_API+int mlk_kem_enc(uint8_t ct[MLKEM_INDCCA_CIPHERTEXTBYTES],+ uint8_t ss[MLKEM_SSBYTES],+ const uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(coins, uint8_t, MLKEM_SYMBYTES, context);++ if (coins == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ if (mlk_randombytes(coins, MLKEM_SYMBYTES) != 0)+ {+ ret = MLK_ERR_RNG_FAIL;+ goto cleanup;+ }++ MLK_CT_TESTING_SECRET(coins, MLKEM_SYMBYTES);++ ret = mlk_kem_enc_derand(ct, ss, pk, coins, context);++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(coins, uint8_t, MLKEM_SYMBYTES, context);+ return ret;+}+#endif /* !MLK_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLK_CONFIG_NO_ENCAPS_API */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `crypto_kem_dec()` in the reference implementation @[REF]+ * - We include secret key check+ * - We include stack buffer zeroization */+MLK_EXTERNAL_API+int mlk_kem_dec(uint8_t ss[MLKEM_SSBYTES],+ const uint8_t ct[MLKEM_INDCCA_CIPHERTEXTBYTES],+ const uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ uint8_t fail;+ const uint8_t *pk = sk + MLKEM_INDCPA_SECRETKEYBYTES;+ MLK_ALLOC(buf, uint8_t, 2 * MLKEM_SYMBYTES, context);+ MLK_ALLOC(kr, uint8_t, 2 * MLKEM_SYMBYTES, context);+ MLK_ALLOC(tmp, uint8_t, MLKEM_SYMBYTES + MLKEM_INDCCA_CIPHERTEXTBYTES,+ context);++ if (buf == NULL || kr == NULL || tmp == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Specification: Implements @[FIPS203, Section 7.3, Hash check] */+ ret = mlk_kem_check_sk(sk, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ ret = mlk_indcpa_dec(buf, ct, sk, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ /* Multitarget countermeasure for coins + contributory KEM */+ mlk_memcpy(buf + MLKEM_SYMBYTES,+ sk + MLKEM_INDCCA_SECRETKEYBYTES - 2 * MLKEM_SYMBYTES,+ MLKEM_SYMBYTES);+ mlk_hash_g(kr, buf, 2 * MLKEM_SYMBYTES);++ /* Recompute and compare ciphertext */+ /* coins are in kr+MLKEM_SYMBYTES */+ ret = mlk_indcpa_enc(tmp, buf, pk, kr + MLKEM_SYMBYTES, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ fail = mlk_ct_memcmp(ct, tmp, MLKEM_INDCCA_CIPHERTEXTBYTES);++ /* Compute rejection key */+ mlk_memcpy(tmp, sk + MLKEM_INDCCA_SECRETKEYBYTES - MLKEM_SYMBYTES,+ MLKEM_SYMBYTES);+ mlk_memcpy(tmp + MLKEM_SYMBYTES, ct, MLKEM_INDCCA_CIPHERTEXTBYTES);+ mlk_hash_j(ss, tmp, MLKEM_SYMBYTES + MLKEM_INDCCA_CIPHERTEXTBYTES);++ /* Copy true key to return buffer if fail is 0 */+ mlk_ct_cmov_zero(ss, kr, MLKEM_SYMBYTES, fail);++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(tmp, uint8_t, MLKEM_SYMBYTES + MLKEM_INDCCA_CIPHERTEXTBYTES,+ context);+ MLK_FREE(kr, uint8_t, 2 * MLKEM_SYMBYTES, context);+ MLK_FREE(buf, uint8_t, 2 * MLKEM_SYMBYTES, context);++ return ret;+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef mlk_check_pct
+ cbits/mlkem/src/kem.h view
@@ -0,0 +1,353 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ *+ * - [REF]+ * CRYSTALS-Kyber C reference implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/ref+ */++#ifndef MLK_KEM_H+#define MLK_KEM_H++#include "cbmc.h"+#include "common.h"+#include "sys.h"++#if defined(MLK_CHECK_APIS)+/* Include to ensure consistency between internal kem.h+ * and external mlkem_native.h. */+#include "mlkem_native.h"++#if MLKEM_INDCCA_SECRETKEYBYTES != \+ MLKEM_SECRETKEYBYTES(MLK_CONFIG_PARAMETER_SET)+#error Mismatch for SECRETKEYBYTES between kem.h and mlkem_native.h+#endif++#if MLKEM_INDCCA_PUBLICKEYBYTES != \+ MLKEM_PUBLICKEYBYTES(MLK_CONFIG_PARAMETER_SET)+#error Mismatch for PUBLICKEYBYTES between kem.h and mlkem_native.h+#endif++#if MLKEM_INDCCA_CIPHERTEXTBYTES != \+ MLKEM_CIPHERTEXTBYTES(MLK_CONFIG_PARAMETER_SET)+#error Mismatch for CIPHERTEXTBYTES between kem.h and mlkem_native.h+#endif++#endif /* MLK_CHECK_APIS */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+#define mlk_kem_keypair_derand \+ MLK_NAMESPACE_K(keypair_derand) MLK_CONTEXT_PARAMETERS_3+#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+#define mlk_kem_keypair MLK_NAMESPACE_K(keypair) MLK_CONTEXT_PARAMETERS_2+#endif+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */+#if !defined(MLK_CONFIG_NO_ENCAPS_API)+#define mlk_kem_enc_derand MLK_NAMESPACE_K(enc_derand) MLK_CONTEXT_PARAMETERS_4+#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+#define mlk_kem_enc MLK_NAMESPACE_K(enc) MLK_CONTEXT_PARAMETERS_3+#endif+#define mlk_kem_check_pk MLK_NAMESPACE_K(check_pk) MLK_CONTEXT_PARAMETERS_1+#endif /* !MLK_CONFIG_NO_ENCAPS_API */+#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_kem_dec MLK_NAMESPACE_K(dec) MLK_CONTEXT_PARAMETERS_3+#define mlk_kem_check_sk MLK_NAMESPACE_K(check_sk) MLK_CONTEXT_PARAMETERS_1+#endif++/**+ * Implements modulus check mandated by FIPS 203, i.e., ensures that+ * coefficients are in [0,q-1].+ *+ * @spec{Implements @[FIPS203, Section 7.2, 'modulus check'].}+ *+ * @reference{Not implemented in the reference implementation @[REF].}+ *+ * @param[in] pk Input public key (an already allocated array of+ * MLKEM_INDCCA_PUBLICKEYBYTES bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_PK Modulus check failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+#if !defined(MLK_CONFIG_NO_ENCAPS_API)+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_check_pk(const uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_INVALID_PK ||+ return_value == MLK_ERR_OUT_OF_MEMORY)+);+#endif /* !MLK_CONFIG_NO_ENCAPS_API */+++/**+ * Implements public key hash check mandated by FIPS 203, i.e., ensures that+ * sk[768𝑘+32 ∶ 768𝑘+64] = H(pk) = H(sk[384𝑘 : 768𝑘+32]).+ *+ * @spec{Implements @[FIPS203, Section 7.3, 'hash check'].}+ *+ * @reference{Not implemented in the reference implementation @[REF].}+ *+ * @param[in] sk Input private key (an already allocated array of+ * MLKEM_INDCCA_SECRETKEYBYTES bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_SK Public key hash check failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_check_sk(const uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(sk, MLKEM_INDCCA_SECRETKEYBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_INVALID_SK ||+ return_value == MLK_ERR_OUT_OF_MEMORY)+);+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+/**+ * Generate a public/private keypair for the ML-KEM key encapsulation mechanism.+ *+ * @spec{Implements @[FIPS203, Algorithm 16, ML-KEM.KeyGen_Internal].}+ *+ * @param[out] pk Output public key (an already allocated array of+ * MLKEM_INDCCA_PUBLICKEYBYTES bytes).+ * @param[out] sk Output private key (an already allocated array of+ * MLKEM_INDCCA_SECRETKEYBYTES bytes).+ * @param[in] coins Input randomness (an already allocated array filled+ * with 2*MLKEM_SYMBYTES random bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_PCT_FAIL MLK_CONFIG_KEYGEN_PCT enabled and PCT failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ * @retval MLK_ERR_RNG_FAIL MLK_CONFIG_KEYGEN_PCT enabled and random+ * number generation failed within the PCT.+ */+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_keypair_derand(uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ const uint8_t coins[2 * MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLKEM_INDCCA_SECRETKEYBYTES))+ requires(memory_no_alias(coins, 2 * MLKEM_SYMBYTES))+ assigns(memory_slice(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ assigns(memory_slice(sk, MLKEM_INDCCA_SECRETKEYBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_PCT_FAIL ||+ return_value == MLK_ERR_OUT_OF_MEMORY ||+ return_value == MLK_ERR_RNG_FAIL)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sk, MLKEM_INDCCA_SECRETKEYBYTES))+);++#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+/**+ * Generate a public/private keypair for the ML-KEM key encapsulation mechanism.+ *+ * @spec{Implements @[FIPS203, Algorithm 19, ML-KEM.KeyGen].}+ *+ * @param[out] pk Output public key (an already allocated array of+ * MLKEM_INDCCA_PUBLICKEYBYTES bytes).+ * @param[out] sk Output private key (an already allocated array of+ * MLKEM_INDCCA_SECRETKEYBYTES bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ * @retval MLK_ERR_RNG_FAIL Random number generation failed.+ * @retval MLK_ERR_PCT_FAIL MLK_CONFIG_KEYGEN_PCT enabled and PCT failed.+ */+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_keypair(uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLKEM_INDCCA_SECRETKEYBYTES))+ assigns(memory_slice(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ assigns(memory_slice(sk, MLKEM_INDCCA_SECRETKEYBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_PCT_FAIL ||+ return_value == MLK_ERR_OUT_OF_MEMORY ||+ return_value == MLK_ERR_RNG_FAIL)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sk, MLKEM_INDCCA_SECRETKEYBYTES))+);+#endif /* !MLK_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++/**+ * Generate ciphertext and shared secret for a given public key.+ *+ * @spec{Implements @[FIPS203, Algorithm 17, ML-KEM.Encaps_Internal].}+ *+ * @param[out] ct Output ciphertext (an already allocated array of+ * MLKEM_INDCCA_CIPHERTEXTBYTES bytes).+ * @param[out] ss Output shared secret (an already allocated array of+ * MLKEM_SSBYTES bytes).+ * @param[in] pk Input public key (an already allocated array of+ * MLKEM_INDCCA_PUBLICKEYBYTES bytes).+ * @param[in] coins Input randomness (an already allocated array filled+ * with MLKEM_SYMBYTES random bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_PK The 'modulus check' @[FIPS203, Section 7.2]+ * for the public key failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+#if !defined(MLK_CONFIG_NO_ENCAPS_API)+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_enc_derand(uint8_t ct[MLKEM_INDCCA_CIPHERTEXTBYTES],+ uint8_t ss[MLKEM_SSBYTES],+ const uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ const uint8_t coins[MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(ct, MLKEM_INDCCA_CIPHERTEXTBYTES))+ requires(memory_no_alias(ss, MLKEM_SSBYTES))+ requires(memory_no_alias(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ requires(memory_no_alias(coins, MLKEM_SYMBYTES))+ assigns(memory_slice(ct, MLKEM_INDCCA_CIPHERTEXTBYTES))+ assigns(memory_slice(ss, MLKEM_SSBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_INVALID_PK ||+ return_value == MLK_ERR_OUT_OF_MEMORY)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(ct, MLKEM_INDCCA_CIPHERTEXTBYTES))+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(ss, MLKEM_SSBYTES))+);++#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+/**+ * Generate ciphertext and shared secret for a given public key.+ *+ * @spec{Implements @[FIPS203, Algorithm 20, ML-KEM.Encaps].}+ *+ * @param[out] ct Output ciphertext (an already allocated array of+ * MLKEM_INDCCA_CIPHERTEXTBYTES bytes).+ * @param[out] ss Output shared secret (an already allocated array of+ * MLKEM_SSBYTES bytes).+ * @param[in] pk Input public key (an already allocated array of+ * MLKEM_INDCCA_PUBLICKEYBYTES bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ * @retval MLK_ERR_RNG_FAIL Random number generation failed.+ * @retval MLK_ERR_INVALID_PK The 'modulus check' @[FIPS203, Section 7.2]+ * for the public key failed.+ */+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_enc(uint8_t ct[MLKEM_INDCCA_CIPHERTEXTBYTES],+ uint8_t ss[MLKEM_SSBYTES],+ const uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(ct, MLKEM_INDCCA_CIPHERTEXTBYTES))+ requires(memory_no_alias(ss, MLKEM_SSBYTES))+ requires(memory_no_alias(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ assigns(memory_slice(ct, MLKEM_INDCCA_CIPHERTEXTBYTES))+ assigns(memory_slice(ss, MLKEM_SSBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_INVALID_PK ||+ return_value == MLK_ERR_OUT_OF_MEMORY ||+ return_value == MLK_ERR_RNG_FAIL)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(ct, MLKEM_INDCCA_CIPHERTEXTBYTES))+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(ss, MLKEM_SSBYTES))+);+#endif /* !MLK_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLK_CONFIG_NO_ENCAPS_API */++/**+ * Generate shared secret for a given ciphertext and private key.+ *+ * @spec{Implements @[FIPS203, Algorithm 21, ML-KEM.Decaps].}+ *+ * @param[out] ss Output shared secret (an already allocated array of+ * MLKEM_SSBYTES bytes).+ * @param[in] ct Input ciphertext (an already allocated array of+ * MLKEM_INDCCA_CIPHERTEXTBYTES bytes).+ * @param[in] sk Input private key (an already allocated array of+ * MLKEM_INDCCA_SECRETKEYBYTES bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_SK The 'hash check' @[FIPS203, Section 7.3]+ * for the secret key failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_dec(uint8_t ss[MLKEM_SSBYTES],+ const uint8_t ct[MLKEM_INDCCA_CIPHERTEXTBYTES],+ const uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(ss, MLKEM_SSBYTES))+ requires(memory_no_alias(ct, MLKEM_INDCCA_CIPHERTEXTBYTES))+ requires(memory_no_alias(sk, MLKEM_INDCCA_SECRETKEYBYTES))+ assigns(memory_slice(ss, MLKEM_SSBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_INVALID_SK ||+ return_value == MLK_ERR_OUT_OF_MEMORY)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(ss, MLKEM_SSBYTES))+);+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#endif /* !MLK_KEM_H */
+ cbits/mlkem/src/native/aarch64/README.md view
@@ -0,0 +1,16 @@+[//]: # (SPDX-License-Identifier: CC-BY-4.0)++# AArch64 backend (little endian)++This directory contains a native backend for little endian AArch64 systems. It is derived from [^NeonNTT] [^SLOTHY_Paper].++The code in this directory is auto-generated from the 'clean' assembly in [dev/aarch64_clean](../../../../dev/aarch64_clean)+in a two-step fashion: First, it is superoptimized using the [SLOTHY](https://github.com/slothy-optimizer/slothy) superoptimizer,+giving the assembly in [dev/aarch64_opt](../../../../dev/aarch64_opt). Then, it is stripped of remaining register aliases, macros+and most preprocessor directives by [`scripts/simpasm`](../../../../scripts/simpasm).++If you want to understand how the assembly works, and/or make changes to it, consult [dev/](../../../../dev).++<!--- bibliography --->+[^NeonNTT]: Becker, Hwang, Kannwischer, Yang, Yang: Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1, [https://eprint.iacr.org/2021/986](https://eprint.iacr.org/2021/986)+[^SLOTHY_Paper]: Abdulrahman, Becker, Kannwischer, Klein: Fast and Clean: Auditable high-performance assembly via constraint solving, [https://eprint.iacr.org/2022/1303](https://eprint.iacr.org/2022/1303)
+ cbits/mlkem/src/native/aarch64/meta.h view
@@ -0,0 +1,166 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_NATIVE_AARCH64_META_H+#define MLK_NATIVE_AARCH64_META_H++/* Set of primitives that this backend replaces */+#define MLK_USE_NATIVE_NTT+#define MLK_USE_NATIVE_INTT+#define MLK_USE_NATIVE_POLY_REDUCE+#define MLK_USE_NATIVE_POLY_TOMONT+#define MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#define MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#define MLK_USE_NATIVE_POLY_TOBYTES+#define MLK_USE_NATIVE_REJ_UNIFORM++/* Identifier for this backend so that source and assembly files+ * in the build can be appropriately guarded. */+#define MLK_ARITH_BACKEND_AARCH64+++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/arith_native_aarch64.h"++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_ntt_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_ntt_aarch64_asm(data, mlk_aarch64_ntt_zetas_layer12345,+ mlk_aarch64_ntt_zetas_layer67);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_intt_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_intt_aarch64_asm(data, mlk_aarch64_invntt_zetas_layer12345,+ mlk_aarch64_invntt_zetas_layer67);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_reduce_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_poly_reduce_aarch64_asm(data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_tomont_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_poly_tomont_aarch64_asm(data);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_mulcache_compute_native(int16_t x[MLKEM_N / 2],+ const int16_t y[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_poly_mulcache_compute_aarch64_asm(+ x, y, mlk_aarch64_zetas_mulcache_native,+ mlk_aarch64_zetas_mulcache_twisted_native);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k2_native(+ int16_t r[MLKEM_N], const int16_t a[2 * MLKEM_N],+ const int16_t b[2 * MLKEM_N], const int16_t b_cache[2 * (MLKEM_N / 2)])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm(r, a, b, b_cache);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 3+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k3_native(+ int16_t r[MLKEM_N], const int16_t a[3 * MLKEM_N],+ const int16_t b[3 * MLKEM_N], const int16_t b_cache[3 * (MLKEM_N / 2)])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm(r, a, b, b_cache);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 3 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k4_native(+ int16_t r[MLKEM_N], const int16_t a[4 * MLKEM_N],+ const int16_t b[4 * MLKEM_N], const int16_t b_cache[4 * (MLKEM_N / 2)])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm(r, a, b, b_cache);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API) || !defined(MLK_CONFIG_NO_ENCAPS_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_tobytes_native(uint8_t r[MLKEM_POLYBYTES],+ const int16_t a[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_poly_tobytes_aarch64_asm(r, a);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API */++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_rej_uniform_native(int16_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON) || len != MLKEM_N ||+ buflen % 24 != 0)+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ return (int)mlk_rej_uniform_aarch64_asm(r, buf, buflen,+ mlk_rej_uniform_table);+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLK_NATIVE_AARCH64_META_H */
+ cbits/mlkem/src/native/aarch64/src/aarch64_zetas.c view
@@ -0,0 +1,184 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLK_ARITH_BACKEND_AARCH64) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#include "arith_native_aarch64.h"++/*+ * Table of zeta values used in the AArch64 forward NTT+ * See autogen for details.+ */+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t+ mlk_aarch64_ntt_zetas_layer12345[80] = {+ -1600, -15749, -749, -7373, -40, -394, -687, -6762, 630, 6201,+ -1432, -14095, 848, 8347, 0, 0, 1062, 10453, 296, 2914,+ -882, -8682, 0, 0, -1410, -13879, 1339, 13180, 1476, 14529,+ 0, 0, 193, 1900, -283, -2786, 56, 551, 0, 0,+ 797, 7845, -1089, -10719, 1333, 13121, 0, 0, -543, -5345,+ 1426, 14036, -1235, -12156, 0, 0, -69, -679, 535, 5266,+ -447, -4400, 0, 0, 569, 5601, -936, -9213, -450, -4429,+ 0, 0, -1583, -15582, -1355, -13338, 821, 8081, 0, 0,+};++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t+ mlk_aarch64_ntt_zetas_layer67[384] = {+ 289, 289, 331, 331, -76, -76, -1573, -1573, 2845,+ 2845, 3258, 3258, -748, -748, -15483, -15483, 17, 17,+ 583, 583, 1637, 1637, -1041, -1041, 167, 167, 5739,+ 5739, 16113, 16113, -10247, -10247, -568, -568, -680, -680,+ 723, 723, 1100, 1100, -5591, -5591, -6693, -6693, 7117,+ 7117, 10828, 10828, 1197, 1197, -1025, -1025, -1052, -1052,+ -1274, -1274, 11782, 11782, -10089, -10089, -10355, -10355, -12540,+ -12540, 1409, 1409, -48, -48, 756, 756, -314, -314,+ 13869, 13869, -472, -472, 7441, 7441, -3091, -3091, -667,+ -667, 233, 233, -1173, -1173, -279, -279, -6565, -6565,+ 2293, 2293, -11546, -11546, -2746, -2746, 650, 650, -1352,+ -1352, -816, -816, 632, 632, 6398, 6398, -13308, -13308,+ -8032, -8032, 6221, 6221, -1626, -1626, -540, -540, -1482,+ -1482, 1461, 1461, -16005, -16005, -5315, -5315, -14588, -14588,+ 14381, 14381, 1651, 1651, -1540, -1540, 952, 952, -642,+ -642, 16251, 16251, -15159, -15159, 9371, 9371, -6319, -6319,+ -464, -464, 33, 33, 1320, 1320, -1414, -1414, -4567,+ -4567, 325, 325, 12993, 12993, -13918, -13918, 939, 939,+ -892, -892, 733, 733, 268, 268, 9243, 9243, -8780,+ -8780, 7215, 7215, 2638, 2638, -1021, -1021, -941, -941,+ -992, -992, 641, 641, -10050, -10050, -9262, -9262, -9764,+ -9764, 6309, 6309, -1010, -1010, 1435, 1435, 807, 807,+ 452, 452, -9942, -9942, 14125, 14125, 7943, 7943, 4449,+ 4449, 1584, 1584, -1292, -1292, 375, 375, -1239, -1239,+ 15592, 15592, -12717, -12717, 3691, 3691, -12196, -12196, -1031,+ -1031, -109, -109, -780, -780, 1645, 1645, -10148, -10148,+ -1073, -1073, -7678, -7678, 16192, 16192, 1438, 1438, -461,+ -461, 1534, 1534, -927, -927, 14155, 14155, -4538, -4538,+ 15099, 15099, -9125, -9125, 1063, 1063, -556, -556, -1230,+ -1230, -863, -863, 10463, 10463, -5473, -5473, -12107, -12107,+ -8495, -8495, 319, 319, 757, 757, 561, 561, -735,+ -735, 3140, 3140, 7451, 7451, 5522, 5522, -7235, -7235,+ -682, -682, -712, -712, 1481, 1481, 648, 648, -6713,+ -6713, -7008, -7008, 14578, 14578, 6378, 6378, -525, -525,+ 403, 403, 1143, 1143, -554, -554, -5168, -5168, 3967,+ 3967, 11251, 11251, -5453, -5453, 1092, 1092, 1026, 1026,+ -1179, -1179, 886, 886, 10749, 10749, 10099, 10099, -11605,+ -11605, 8721, 8721, -855, -855, -219, -219, 1227, 1227,+ 910, 910, -8416, -8416, -2156, -2156, 12078, 12078, 8957,+ 8957, -1607, -1607, -1455, -1455, -1219, -1219, 885, 885,+ -15818, -15818, -14322, -14322, -11999, -11999, 8711, 8711, 1212,+ 1212, 1029, 1029, -394, -394, -1175, -1175, 11930, 11930,+ 10129, 10129, -3878, -3878, -11566, -11566,+};++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t+ mlk_aarch64_invntt_zetas_layer12345[80] = {+ 1583, 15582, -821, -8081, 1355, 13338, 0, 0, -569,+ -5601, 450, 4429, 936, 9213, 0, 0, 69, 679,+ 447, 4400, -535, -5266, 0, 0, 543, 5345, 1235,+ 12156, -1426, -14036, 0, 0, -797, -7845, -1333, -13121,+ 1089, 10719, 0, 0, -193, -1900, -56, -551, 283,+ 2786, 0, 0, 1410, 13879, -1476, -14529, -1339, -13180,+ 0, 0, -1062, -10453, 882, 8682, -296, -2914, 0,+ 0, 1600, 15749, 40, 394, 749, 7373, -848, -8347,+ 1432, 14095, -630, -6201, 687, 6762, 0, 0,+};++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t+ mlk_aarch64_invntt_zetas_layer67[384] = {+ -910, -910, -1227, -1227, 219, 219, 855, 855, -8957,+ -8957, -12078, -12078, 2156, 2156, 8416, 8416, 1175, 1175,+ 394, 394, -1029, -1029, -1212, -1212, 11566, 11566, 3878,+ 3878, -10129, -10129, -11930, -11930, -885, -885, 1219, 1219,+ 1455, 1455, 1607, 1607, -8711, -8711, 11999, 11999, 14322,+ 14322, 15818, 15818, -648, -648, -1481, -1481, 712, 712,+ 682, 682, -6378, -6378, -14578, -14578, 7008, 7008, 6713,+ 6713, -886, -886, 1179, 1179, -1026, -1026, -1092, -1092,+ -8721, -8721, 11605, 11605, -10099, -10099, -10749, -10749, 554,+ 554, -1143, -1143, -403, -403, 525, 525, 5453, 5453,+ -11251, -11251, -3967, -3967, 5168, 5168, 927, 927, -1534,+ -1534, 461, 461, -1438, -1438, 9125, 9125, -15099, -15099,+ 4538, 4538, -14155, -14155, 735, 735, -561, -561, -757,+ -757, -319, -319, 7235, 7235, -5522, -5522, -7451, -7451,+ -3140, -3140, 863, 863, 1230, 1230, 556, 556, -1063,+ -1063, 8495, 8495, 12107, 12107, 5473, 5473, -10463, -10463,+ -452, -452, -807, -807, -1435, -1435, 1010, 1010, -4449,+ -4449, -7943, -7943, -14125, -14125, 9942, 9942, -1645, -1645,+ 780, 780, 109, 109, 1031, 1031, -16192, -16192, 7678,+ 7678, 1073, 1073, 10148, 10148, 1239, 1239, -375, -375,+ 1292, 1292, -1584, -1584, 12196, 12196, -3691, -3691, 12717,+ 12717, -15592, -15592, 1414, 1414, -1320, -1320, -33, -33,+ 464, 464, 13918, 13918, -12993, -12993, -325, -325, 4567,+ 4567, -641, -641, 992, 992, 941, 941, 1021, 1021,+ -6309, -6309, 9764, 9764, 9262, 9262, 10050, 10050, -268,+ -268, -733, -733, 892, 892, -939, -939, -2638, -2638,+ -7215, -7215, 8780, 8780, -9243, -9243, -632, -632, 816,+ 816, 1352, 1352, -650, -650, -6221, -6221, 8032, 8032,+ 13308, 13308, -6398, -6398, 642, 642, -952, -952, 1540,+ 1540, -1651, -1651, 6319, 6319, -9371, -9371, 15159, 15159,+ -16251, -16251, -1461, -1461, 1482, 1482, 540, 540, 1626,+ 1626, -14381, -14381, 14588, 14588, 5315, 5315, 16005, 16005,+ 1274, 1274, 1052, 1052, 1025, 1025, -1197, -1197, 12540,+ 12540, 10355, 10355, 10089, 10089, -11782, -11782, 279, 279,+ 1173, 1173, -233, -233, 667, 667, 2746, 2746, 11546,+ 11546, -2293, -2293, 6565, 6565, 314, 314, -756, -756,+ 48, 48, -1409, -1409, 3091, 3091, -7441, -7441, 472,+ 472, -13869, -13869, 1573, 1573, 76, 76, -331, -331,+ -289, -289, 15483, 15483, 748, 748, -3258, -3258, -2845,+ -2845, -1100, -1100, -723, -723, 680, 680, 568, 568,+ -10828, -10828, -7117, -7117, 6693, 6693, 5591, 5591, 1041,+ 1041, -1637, -1637, -583, -583, -17, -17, 10247, 10247,+ -16113, -16113, -5739, -5739, -167, -167,+};+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t+ mlk_aarch64_zetas_mulcache_native[128] = {+ 17, -17, -568, 568, 583, -583, -680, 680, 1637, -1637,+ 723, -723, -1041, 1041, 1100, -1100, 1409, -1409, -667, 667,+ -48, 48, 233, -233, 756, -756, -1173, 1173, -314, 314,+ -279, 279, -1626, 1626, 1651, -1651, -540, 540, -1540, 1540,+ -1482, 1482, 952, -952, 1461, -1461, -642, 642, 939, -939,+ -1021, 1021, -892, 892, -941, 941, 733, -733, -992, 992,+ 268, -268, 641, -641, 1584, -1584, -1031, 1031, -1292, 1292,+ -109, 109, 375, -375, -780, 780, -1239, 1239, 1645, -1645,+ 1063, -1063, 319, -319, -556, 556, 757, -757, -1230, 1230,+ 561, -561, -863, 863, -735, 735, -525, 525, 1092, -1092,+ 403, -403, 1026, -1026, 1143, -1143, -1179, 1179, -554, 554,+ 886, -886, -1607, 1607, 1212, -1212, -1455, 1455, 1029, -1029,+ -1219, 1219, -394, 394, 885, -885, -1175, 1175,+};++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t+ mlk_aarch64_zetas_mulcache_twisted_native[128] = {+ 167, -167, -5591, 5591, 5739, -5739, -6693, 6693, 16113,+ -16113, 7117, -7117, -10247, 10247, 10828, -10828, 13869, -13869,+ -6565, 6565, -472, 472, 2293, -2293, 7441, -7441, -11546,+ 11546, -3091, 3091, -2746, 2746, -16005, 16005, 16251, -16251,+ -5315, 5315, -15159, 15159, -14588, 14588, 9371, -9371, 14381,+ -14381, -6319, 6319, 9243, -9243, -10050, 10050, -8780, 8780,+ -9262, 9262, 7215, -7215, -9764, 9764, 2638, -2638, 6309,+ -6309, 15592, -15592, -10148, 10148, -12717, 12717, -1073, 1073,+ 3691, -3691, -7678, 7678, -12196, 12196, 16192, -16192, 10463,+ -10463, 3140, -3140, -5473, 5473, 7451, -7451, -12107, 12107,+ 5522, -5522, -8495, 8495, -7235, 7235, -5168, 5168, 10749,+ -10749, 3967, -3967, 10099, -10099, 11251, -11251, -11605, 11605,+ -5453, 5453, 8721, -8721, -15818, 15818, 11930, -11930, -14322,+ 14322, 10129, -10129, -11999, 11999, -3878, 3878, 8711, -8711,+ -11566, 11566,+};++#else /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(aarch64_zetas)++#endif /* !(MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mlkem/src/native/aarch64/src/arith_native_aarch64.h view
@@ -0,0 +1,184 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H+#define MLK_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H++#include "../../../cbmc.h"+#include "../../../common.h"++#define mlk_aarch64_ntt_zetas_layer12345 \+ MLK_NAMESPACE(aarch64_ntt_zetas_layer12345)+#define mlk_aarch64_ntt_zetas_layer67 MLK_NAMESPACE(aarch64_ntt_zetas_layer67)+#define mlk_aarch64_invntt_zetas_layer12345 \+ MLK_NAMESPACE(aarch64_invntt_zetas_layer12345)+#define mlk_aarch64_invntt_zetas_layer67 \+ MLK_NAMESPACE(aarch64_invntt_zetas_layer67)+#define mlk_aarch64_zetas_mulcache_native \+ MLK_NAMESPACE(aarch64_zetas_mulcache_native)+#define mlk_aarch64_zetas_mulcache_twisted_native \+ MLK_NAMESPACE(aarch64_zetas_mulcache_twisted_native)+#define mlk_rej_uniform_table MLK_NAMESPACE(rej_uniform_table)++MLK_INTERNAL_DATA_DECLARATION const int16_t+ mlk_aarch64_ntt_zetas_layer12345[80];+MLK_INTERNAL_DATA_DECLARATION const int16_t mlk_aarch64_ntt_zetas_layer67[384];+MLK_INTERNAL_DATA_DECLARATION const int16_t+ mlk_aarch64_invntt_zetas_layer12345[80];+MLK_INTERNAL_DATA_DECLARATION const int16_t+ mlk_aarch64_invntt_zetas_layer67[384];+MLK_INTERNAL_DATA_DECLARATION const int16_t+ mlk_aarch64_zetas_mulcache_native[128];+MLK_INTERNAL_DATA_DECLARATION const int16_t+ mlk_aarch64_zetas_mulcache_twisted_native[128];+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_rej_uniform_table[4096];++#define mlk_ntt_aarch64_asm MLK_NAMESPACE(ntt_aarch64_asm)+void mlk_ntt_aarch64_asm(int16_t p[256], const int16_t twiddles12345[80],+ const int16_t twiddles56[384])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mlkem_ntt_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ requires(array_abs_bound(p, 0, MLKEM_N, 8192))+ requires(twiddles12345 == mlk_aarch64_ntt_zetas_layer12345)+ requires(twiddles56 == mlk_aarch64_ntt_zetas_layer67)+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ /* check-magic: off */+ ensures(array_abs_bound(p, 0, MLKEM_N, 23595))+ /* check-magic: on */+);++#define mlk_intt_aarch64_asm MLK_NAMESPACE(intt_aarch64_asm)+void mlk_intt_aarch64_asm(int16_t p[256], const int16_t twiddles12345[80],+ const int16_t twiddles56[384])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mlkem_intt_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ requires(twiddles12345 == mlk_aarch64_invntt_zetas_layer12345)+ requires(twiddles56 == mlk_aarch64_invntt_zetas_layer67)+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ /* check-magic: off */+ ensures(array_abs_bound(p, 0, MLKEM_N, 26625))+ /* check-magic: on */+);++#define mlk_poly_reduce_aarch64_asm MLK_NAMESPACE(poly_reduce_aarch64_asm)+void mlk_poly_reduce_aarch64_asm(int16_t p[256])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mlkem_poly_reduce_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(p, 0, MLKEM_N, 0, MLKEM_Q))+);++#define mlk_poly_tomont_aarch64_asm MLK_NAMESPACE(poly_tomont_aarch64_asm)+void mlk_poly_tomont_aarch64_asm(int16_t p[256])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mlkem_poly_tomont_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ ensures(array_abs_bound(p, 0, MLKEM_N, MLKEM_Q))+);++#define mlk_poly_mulcache_compute_aarch64_asm \+ MLK_NAMESPACE(poly_mulcache_compute_aarch64_asm)+void mlk_poly_mulcache_compute_aarch64_asm(int16_t cache[128],+ const int16_t mlk_poly[256],+ const int16_t zetas[128],+ const int16_t zetas_twisted[128])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mlkem_poly_mulcache_compute_aarch64_asm.ml+ */+__contract__(+ requires(memory_no_alias(cache, sizeof(int16_t) * (MLKEM_N / 2)))+ requires(memory_no_alias(mlk_poly, sizeof(int16_t) * MLKEM_N))+ requires(zetas == mlk_aarch64_zetas_mulcache_native)+ requires(zetas_twisted == mlk_aarch64_zetas_mulcache_twisted_native)+ assigns(memory_slice(cache, sizeof(int16_t) * (MLKEM_N / 2)))+ ensures(array_abs_bound(cache, 0, MLKEM_N/2, MLKEM_Q))+);++#define mlk_poly_tobytes_aarch64_asm MLK_NAMESPACE(poly_tobytes_aarch64_asm)+void mlk_poly_tobytes_aarch64_asm(uint8_t r[384], const int16_t a[256])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mlkem_poly_tobytes_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYBYTES))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT))+ assigns(memory_slice(r, MLKEM_POLYBYTES))+);++#define mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm \+ MLK_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm)+void mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm(+ int16_t r[256], const int16_t a[512], const int16_t b[512],+ const int16_t b_cache[256])+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/aarch64/proofs/mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 2 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 2 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 2 * (MLKEM_N / 2)))+ requires(array_abs_bound(a, 0, 2 * MLKEM_N, MLKEM_UINT12_LIMIT + 1))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+);++#define mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm \+ MLK_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm)+void mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm(+ int16_t r[256], const int16_t a[768], const int16_t b[768],+ const int16_t b_cache[384])+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/aarch64/proofs/mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 3 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 3 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 3 * (MLKEM_N / 2)))+ requires(array_abs_bound(a, 0, 3 * MLKEM_N, MLKEM_UINT12_LIMIT + 1))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+);++#define mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm \+ MLK_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm)+void mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm(+ int16_t r[256], const int16_t a[1024], const int16_t b[1024],+ const int16_t b_cache[512])+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/aarch64/proofs/mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 4 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 4 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 4 * (MLKEM_N / 2)))+ requires(array_abs_bound(a, 0, 4 * MLKEM_N, MLKEM_UINT12_LIMIT + 1))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+);++#define mlk_rej_uniform_aarch64_asm MLK_NAMESPACE(rej_uniform_aarch64_asm)+MLK_MUST_CHECK_RETURN_VALUE+uint64_t mlk_rej_uniform_aarch64_asm(int16_t r[256], const uint8_t *buf,+ unsigned buflen, const uint8_t table[4096])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mlkem_rej_uniform_aarch64_asm.ml. */+__contract__(+ requires(buflen % 24 == 0)+ requires(memory_no_alias(buf, buflen))+ requires(table == mlk_rej_uniform_table)+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value <= MLKEM_N)+ ensures(array_bound(r, 0, (unsigned) return_value, 0, MLKEM_Q))+);++#endif /* !MLK_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H */
+ cbits/mlkem/src/native/aarch64/src/mlkem_intt_aarch64_asm.S view
@@ -0,0 +1,635 @@+/* Copyright (c) 2022 Arm Limited+ * Copyright (c) 2022 Hanno Becker+ * Copyright (c) 2023 Amin Abdulrahman, Matthias Kannwischer+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ *+ * - [SLOTHY_Paper]+ * Fast and Clean: Auditable high-performance assembly via constraint solving+ * Abdulrahman, Becker, Kannwischer, Klein+ * https://eprint.iacr.org/2022/1303+ */++/*yaml+ Name: intt_aarch64_asm+ Description: AArch64 ML-KEM inverse NTT following @[NeonNTT] and @[SLOTHY_Paper]+ Signature: void mlk_intt_aarch64_asm(int16_t p[256], const int16_t twiddles12345[80], const int16_t twiddles56[384])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t p[256]+ description: Input/output polynomial+ x1:+ type: buffer+ size_bytes: 160+ permissions: read-only+ c_parameter: const int16_t twiddles12345[80]+ description: Twiddle factors for layers 1-5+ x2:+ type: buffer+ size_bytes: 768+ permissions: read-only+ c_parameter: const int16_t twiddles56[384]+ description: Twiddle factors for layers 6-7+ Stack:+ bytes: 64+ description: saving callee-saved Neon registers+*/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API))++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_intt_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(intt_aarch64_asm)+MLK_ASM_FN_SYMBOL(intt_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ mov w5, #0xd01 // =3329+ mov v7.h[0], w5+ mov w5, #0x4ebf // =20159+ mov v7.h[1], w5+ mov w5, #0x200 // =512+ dup v29.8h, w5+ mov w5, #0x13b0 // =5040+ dup v30.8h, w5+ mov x3, x0+ mov x4, #0x8 // =8+ ldr q13, [x3, #0x20]+ ldr q8, [x3, #0x30]+ ldr q6, [x3]+ ldr q16, [x3, #0x10]+ ldr q4, [x3, #0x50]+ ldr q11, [x3, #0x40]+ ldr q3, [x3, #0x70]+ trn1 v23.4s, v13.4s, v8.4s+ ldr q0, [x3, #0x60]+ trn2 v19.4s, v6.4s, v16.4s+ trn2 v21.4s, v13.4s, v8.4s+ trn1 v6.4s, v6.4s, v16.4s+ ldr q24, [x2, #0x20]+ trn1 v10.2d, v19.2d, v21.2d+ ldr q16, [x2], #0x60+ trn1 v5.2d, v6.2d, v23.2d+ trn1 v28.4s, v0.4s, v3.4s+ trn2 v18.2d, v6.2d, v23.2d+ mul v31.8h, v10.8h, v29.8h+ trn2 v13.4s, v0.4s, v3.4s+ ldur q14, [x2, #-0x50]+ sqrdmulh v26.8h, v18.8h, v30.8h+ ldur q20, [x2, #-0x20]+ mul v17.8h, v18.8h, v29.8h+ trn2 v18.2d, v19.2d, v21.2d+ mul v9.8h, v18.8h, v29.8h+ trn1 v12.4s, v11.4s, v4.4s+ sqrdmulh v22.8h, v18.8h, v30.8h+ sqrdmulh v3.8h, v10.8h, v30.8h+ sqrdmulh v25.8h, v5.8h, v30.8h+ mls v9.8h, v22.8h, v7.h[0]+ mls v17.8h, v26.8h, v7.h[0]+ trn2 v26.4s, v11.4s, v4.4s+ mul v8.8h, v5.8h, v29.8h+ trn1 v10.2d, v26.2d, v13.2d+ ldur q11, [x2, #-0x10]+ mls v31.8h, v3.8h, v7.h[0]+ trn1 v6.2d, v12.2d, v28.2d+ trn2 v3.2d, v26.2d, v13.2d+ ldur q4, [x2, #-0x30]+ mls v8.8h, v25.8h, v7.h[0]+ sub v19.8h, v17.8h, v9.8h+ trn2 v13.2d, v12.2d, v28.2d+ sqrdmulh v1.8h, v3.8h, v30.8h+ add v9.8h, v17.8h, v9.8h+ mul v18.8h, v19.8h, v20.8h+ add v28.8h, v8.8h, v31.8h+ sqrdmulh v20.8h, v19.8h, v11.8h+ sub v12.8h, v28.8h, v9.8h+ sub v23.8h, v8.8h, v31.8h+ sqrdmulh v11.8h, v13.8h, v30.8h+ sqrdmulh v5.8h, v23.8h, v4.8h+ mul v0.8h, v23.8h, v24.8h+ mul v2.8h, v13.8h, v29.8h+ mls v0.8h, v5.8h, v7.h[0]+ add v24.8h, v28.8h, v9.8h+ mls v18.8h, v20.8h, v7.h[0]+ sqrdmulh v15.8h, v6.8h, v30.8h+ sqrdmulh v25.8h, v12.8h, v14.8h+ mul v21.8h, v12.8h, v16.8h+ sub v23.8h, v0.8h, v18.8h+ sqrdmulh v8.8h, v23.8h, v14.8h+ mul v23.8h, v23.8h, v16.8h+ mls v21.8h, v25.8h, v7.h[0]+ mls v23.8h, v8.8h, v7.h[0]+ mul v14.8h, v3.8h, v29.8h+ add v3.8h, v0.8h, v18.8h+ trn2 v4.4s, v24.4s, v3.4s+ mls v14.8h, v1.8h, v7.h[0]+ trn1 v9.4s, v24.4s, v3.4s+ trn2 v12.4s, v21.4s, v23.4s+ mls v2.8h, v11.8h, v7.h[0]+ trn1 v28.4s, v21.4s, v23.4s+ ldr q11, [x1], #0x10+ mul v31.8h, v10.8h, v29.8h+ trn1 v25.2d, v4.2d, v12.2d+ trn1 v20.2d, v9.2d, v28.2d+ ldr q23, [x2, #0x50]+ trn2 v13.2d, v4.2d, v12.2d+ sqrdmulh v21.8h, v10.8h, v30.8h+ trn2 v4.2d, v9.2d, v28.2d+ ldr q9, [x2, #0x40]+ mul v27.8h, v6.8h, v29.8h+ add v26.8h, v20.8h, v25.8h+ sub v3.8h, v2.8h, v14.8h+ sqdmulh v12.8h, v26.8h, v7.h[1]+ add v5.8h, v4.8h, v13.8h+ sub v8.8h, v4.8h, v13.8h+ add v10.8h, v2.8h, v14.8h+ sqdmulh v6.8h, v5.8h, v7.h[1]+ ldr q2, [x2, #0x10]+ mls v27.8h, v15.8h, v7.h[0]+ ldr q15, [x2, #0x20]+ srshr v12.8h, v12.8h, #0xb+ mls v31.8h, v21.8h, v7.h[0]+ srshr v6.8h, v6.8h, #0xb+ sqrdmulh v23.8h, v3.8h, v23.8h+ mls v26.8h, v12.8h, v7.h[0]+ add v21.8h, v27.8h, v31.8h+ mls v5.8h, v6.8h, v7.h[0]+ sub v6.8h, v27.8h, v31.8h+ sub v14.8h, v21.8h, v10.8h+ ldr q27, [x2], #0x60+ mul v3.8h, v3.8h, v9.8h+ mls v3.8h, v23.8h, v7.h[0]+ ldur q13, [x2, #-0x30]+ sub v12.8h, v26.8h, v5.8h+ add v5.8h, v26.8h, v5.8h+ sqrdmulh v31.8h, v8.8h, v11.h[5]+ sqrdmulh v19.8h, v12.8h, v11.h[1]+ mul v24.8h, v12.8h, v11.h[0]+ sqrdmulh v13.8h, v6.8h, v13.8h+ mls v24.8h, v19.8h, v7.h[0]+ sub x4, x4, #0x2++Lmlk_intt_layer4567_start:+ add v16.8h, v21.8h, v10.8h+ mul v18.8h, v6.8h, v15.8h+ sub v19.8h, v20.8h, v25.8h+ ldr q21, [x3, #0xa0]+ str q5, [x3], #0x40+ mls v18.8h, v13.8h, v7.h[0]+ sqrdmulh v15.8h, v14.8h, v2.8h+ ldr q10, [x3, #0x50]+ ldr q12, [x3, #0x40]+ stur q24, [x3, #-0x20]+ mul v5.8h, v8.8h, v11.h[4]+ sub v0.8h, v18.8h, v3.8h+ ldr q24, [x3, #0x70]+ mls v5.8h, v31.8h, v7.h[0]+ ldr q26, [x2, #0x50]+ trn2 v1.4s, v12.4s, v10.4s+ add v6.8h, v18.8h, v3.8h+ sqrdmulh v20.8h, v0.8h, v2.8h+ trn1 v13.4s, v12.4s, v10.4s+ trn1 v18.4s, v16.4s, v6.4s+ mul v22.8h, v0.8h, v27.8h+ trn1 v17.4s, v21.4s, v24.4s+ sqrdmulh v0.8h, v19.8h, v11.h[3]+ trn1 v25.2d, v13.2d, v17.2d+ mls v22.8h, v20.8h, v7.h[0]+ trn2 v21.4s, v21.4s, v24.4s+ mul v24.8h, v25.8h, v29.8h+ trn2 v28.2d, v13.2d, v17.2d+ sqrdmulh v4.8h, v25.8h, v30.8h+ trn2 v3.2d, v1.2d, v21.2d+ mul v17.8h, v28.8h, v29.8h+ sqrdmulh v31.8h, v28.8h, v30.8h+ ldr q2, [x2, #0x10]+ mls v24.8h, v4.8h, v7.h[0]+ mul v4.8h, v19.8h, v11.h[2]+ ldr q19, [x2, #0x40]+ mls v4.8h, v0.8h, v7.h[0]+ mul v0.8h, v14.8h, v27.8h+ mls v0.8h, v15.8h, v7.h[0]+ sub v8.8h, v4.8h, v5.8h+ mul v12.8h, v3.8h, v29.8h+ mul v23.8h, v8.8h, v11.h[0]+ trn2 v28.4s, v16.4s, v6.4s+ sqrdmulh v10.8h, v8.8h, v11.h[1]+ trn1 v9.4s, v0.4s, v22.4s+ trn2 v22.4s, v0.4s, v22.4s+ ldr q11, [x1], #0x10+ mls v17.8h, v31.8h, v7.h[0]+ trn1 v20.2d, v18.2d, v9.2d+ trn2 v14.2d, v18.2d, v9.2d+ ldr q15, [x2, #0x20]+ trn1 v6.2d, v1.2d, v21.2d+ sqrdmulh v9.8h, v3.8h, v30.8h+ trn1 v25.2d, v28.2d, v22.2d+ trn2 v16.2d, v28.2d, v22.2d+ mls v23.8h, v10.8h, v7.h[0]+ add v1.8h, v20.8h, v25.8h+ sqrdmulh v21.8h, v6.8h, v30.8h+ add v8.8h, v14.8h, v16.8h+ ldr q27, [x2], #0x60+ sqdmulh v28.8h, v8.8h, v7.h[1]+ mls v12.8h, v9.8h, v7.h[0]+ sqdmulh v31.8h, v1.8h, v7.h[1]+ mul v0.8h, v6.8h, v29.8h+ sub v10.8h, v17.8h, v12.8h+ mls v0.8h, v21.8h, v7.h[0]+ srshr v21.8h, v28.8h, #0xb+ srshr v13.8h, v31.8h, #0xb+ sqrdmulh v22.8h, v10.8h, v26.8h+ mls v8.8h, v21.8h, v7.h[0]+ mls v1.8h, v13.8h, v7.h[0]+ add v21.8h, v24.8h, v0.8h+ stur q23, [x3, #-0x10]+ sub v6.8h, v24.8h, v0.8h+ mul v3.8h, v10.8h, v19.8h+ add v0.8h, v4.8h, v5.8h+ sqdmulh v13.8h, v0.8h, v7.h[1]+ ldur q10, [x2, #-0x30]+ add v5.8h, v1.8h, v8.8h+ mls v3.8h, v22.8h, v7.h[0]+ sub v8.8h, v1.8h, v8.8h+ mul v24.8h, v8.8h, v11.h[0]+ sqrdmulh v8.8h, v8.8h, v11.h[1]+ srshr v1.8h, v13.8h, #0xb+ sqrdmulh v13.8h, v6.8h, v10.8h+ mls v0.8h, v1.8h, v7.h[0]+ add v10.8h, v17.8h, v12.8h+ mls v24.8h, v8.8h, v7.h[0]+ sub v8.8h, v14.8h, v16.8h+ sqrdmulh v31.8h, v8.8h, v11.h[5]+ sub v14.8h, v21.8h, v10.8h+ stur q0, [x3, #-0x30]+ sub x4, x4, #0x1+ cbnz x4, Lmlk_intt_layer4567_start+ mul v15.8h, v6.8h, v15.8h+ sub v22.8h, v20.8h, v25.8h+ add v4.8h, v21.8h, v10.8h+ str q24, [x3, #0x20]+ mls v15.8h, v13.8h, v7.h[0]+ str q5, [x3], #0x40+ ldr q9, [x1], #0x10+ sqrdmulh v28.8h, v14.8h, v2.8h+ mul v16.8h, v14.8h, v27.8h+ sub v18.8h, v15.8h, v3.8h+ add v15.8h, v15.8h, v3.8h+ sqrdmulh v0.8h, v18.8h, v2.8h+ trn2 v24.4s, v4.4s, v15.4s+ trn1 v2.4s, v4.4s, v15.4s+ mul v18.8h, v18.8h, v27.8h+ mls v16.8h, v28.8h, v7.h[0]+ mls v18.8h, v0.8h, v7.h[0]+ mul v23.8h, v8.8h, v11.h[4]+ sqrdmulh v12.8h, v22.8h, v11.h[3]+ trn1 v17.4s, v16.4s, v18.4s+ trn2 v4.4s, v16.4s, v18.4s+ mls v23.8h, v31.8h, v7.h[0]+ trn2 v3.2d, v2.2d, v17.2d+ trn2 v6.2d, v24.2d, v4.2d+ mul v26.8h, v22.8h, v11.h[2]+ trn1 v28.2d, v2.2d, v17.2d+ mls v26.8h, v12.8h, v7.h[0]+ add v25.8h, v3.8h, v6.8h+ sub v18.8h, v3.8h, v6.8h+ trn1 v24.2d, v24.2d, v4.2d+ sqdmulh v1.8h, v25.8h, v7.h[1]+ sub v27.8h, v28.8h, v24.8h+ sqrdmulh v2.8h, v18.8h, v9.h[5]+ add v28.8h, v28.8h, v24.8h+ mul v24.8h, v27.8h, v9.h[2]+ sqdmulh v12.8h, v28.8h, v7.h[1]+ mul v20.8h, v18.8h, v9.h[4]+ mls v20.8h, v2.8h, v7.h[0]+ srshr v1.8h, v1.8h, #0xb+ sqrdmulh v19.8h, v27.8h, v9.h[3]+ srshr v15.8h, v12.8h, #0xb+ mls v25.8h, v1.8h, v7.h[0]+ add v8.8h, v26.8h, v23.8h+ sub v4.8h, v26.8h, v23.8h+ mls v28.8h, v15.8h, v7.h[0]+ mls v24.8h, v19.8h, v7.h[0]+ mul v2.8h, v4.8h, v11.h[0]+ sub v19.8h, v28.8h, v25.8h+ sqrdmulh v15.8h, v4.8h, v11.h[1]+ add v25.8h, v28.8h, v25.8h+ sub v10.8h, v24.8h, v20.8h+ str q25, [x3], #0x40+ sqrdmulh v22.8h, v19.8h, v9.h[1]+ add v28.8h, v24.8h, v20.8h+ sqrdmulh v25.8h, v10.8h, v9.h[1]+ mul v27.8h, v19.8h, v9.h[0]+ mul v26.8h, v10.8h, v9.h[0]+ sqdmulh v20.8h, v28.8h, v7.h[1]+ sqdmulh v16.8h, v8.8h, v7.h[1]+ mls v26.8h, v25.8h, v7.h[0]+ mls v2.8h, v15.8h, v7.h[0]+ srshr v15.8h, v20.8h, #0xb+ srshr v1.8h, v16.8h, #0xb+ mls v27.8h, v22.8h, v7.h[0]+ mls v28.8h, v15.8h, v7.h[0]+ mls v8.8h, v1.8h, v7.h[0]+ stur q27, [x3, #-0x20]+ stur q2, [x3, #-0x50]+ stur q28, [x3, #-0x30]+ stur q26, [x3, #-0x10]+ stur q8, [x3, #-0x70]+ mov x4, #0x4 // =4+ ldr q0, [x1], #0x20+ ldur q1, [x1, #-0x10]+ ldr q26, [x0]+ ldr q13, [x0, #0x40]+ ldr q28, [x0, #0xc0]+ ldr q2, [x0, #0x140]+ ldr q6, [x0, #0x80]+ ldr q9, [x0, #0x100]+ ldr q29, [x0, #0x1c0]+ ldr q23, [x0, #0x180]+ sub v17.8h, v26.8h, v13.8h+ add v4.8h, v26.8h, v13.8h+ ldr q25, [x0, #0xd0]+ ldr q24, [x0, #0x50]+ add v5.8h, v6.8h, v28.8h+ mul v19.8h, v17.8h, v0.h[6]+ sub v10.8h, v6.8h, v28.8h+ ldr q30, [x0, #0x150]+ sqrdmulh v12.8h, v17.8h, v0.h[7]+ add v17.8h, v9.8h, v2.8h+ sub v28.8h, v9.8h, v2.8h+ ldr q2, [x0, #0x90]+ sub v26.8h, v23.8h, v29.8h+ sqrdmulh v31.8h, v10.8h, v1.h[1]+ add v22.8h, v23.8h, v29.8h+ ldr q3, [x0, #0x110]+ sqrdmulh v9.8h, v28.8h, v1.h[3]+ sub v20.8h, v4.8h, v5.8h+ sub v27.8h, v17.8h, v22.8h+ ldr q29, [x0, #0x10]+ add v16.8h, v4.8h, v5.8h+ sqrdmulh v4.8h, v26.8h, v1.h[5]+ add v6.8h, v17.8h, v22.8h+ ldr q22, [x0, #0x1d0]+ mul v8.8h, v28.8h, v1.h[2]+ sub v21.8h, v2.8h, v25.8h+ sub v5.8h, v16.8h, v6.8h+ mul v17.8h, v26.8h, v1.h[4]+ mul v26.8h, v10.8h, v1.h[0]+ mls v26.8h, v31.8h, v7.h[0]+ mls v17.8h, v4.8h, v7.h[0]+ mls v19.8h, v12.8h, v7.h[0]+ mls v8.8h, v9.8h, v7.h[0]+ sqrdmulh v10.8h, v27.8h, v0.h[5]+ sub v12.8h, v19.8h, v26.8h+ add v9.8h, v19.8h, v26.8h+ sqrdmulh v26.8h, v20.8h, v0.h[3]+ sub v11.8h, v8.8h, v17.8h+ add v14.8h, v8.8h, v17.8h+ sqrdmulh v13.8h, v12.8h, v0.h[3]+ add v23.8h, v9.8h, v14.8h+ sqrdmulh v28.8h, v11.8h, v0.h[5]+ sub v19.8h, v9.8h, v14.8h+ mul v17.8h, v27.8h, v0.h[4]+ str q23, [x0, #0x40]+ mul v14.8h, v20.8h, v0.h[2]+ mul v8.8h, v11.8h, v0.h[4]+ mul v4.8h, v12.8h, v0.h[2]+ mls v14.8h, v26.8h, v7.h[0]+ mls v17.8h, v10.8h, v7.h[0]+ mls v8.8h, v28.8h, v7.h[0]+ mls v4.8h, v13.8h, v7.h[0]+ sub v10.8h, v14.8h, v17.8h+ add v20.8h, v14.8h, v17.8h+ sqrdmulh v28.8h, v5.8h, v0.h[1]+ mul v18.8h, v5.8h, v0.h[0]+ str q20, [x0, #0x80]+ sub v13.8h, v4.8h, v8.8h+ mul v23.8h, v10.8h, v0.h[0]+ mul v17.8h, v19.8h, v0.h[0]+ sqrdmulh v9.8h, v13.8h, v0.h[1]+ mls v18.8h, v28.8h, v7.h[0]+ sqrdmulh v10.8h, v10.8h, v0.h[1]+ sub x4, x4, #0x2++Lmlk_intt_layer123_start:+ sub v12.8h, v3.8h, v30.8h+ mul v11.8h, v21.8h, v1.h[0]+ add v28.8h, v4.8h, v8.8h+ ldr q20, [x0, #0x190]+ add v27.8h, v16.8h, v6.8h+ sqrdmulh v8.8h, v12.8h, v1.h[3]+ add v16.8h, v29.8h, v24.8h+ str q28, [x0, #0xc0]+ mls v23.8h, v10.8h, v7.h[0]+ str q27, [x0], #0x10+ add v15.8h, v20.8h, v22.8h+ str q18, [x0, #0xf0]+ mul v14.8h, v13.8h, v0.h[0]+ add v2.8h, v2.8h, v25.8h+ sub v26.8h, v20.8h, v22.8h+ mul v4.8h, v12.8h, v1.h[2]+ sub v5.8h, v16.8h, v2.8h+ str q23, [x0, #0x170]+ add v20.8h, v3.8h, v30.8h+ sqrdmulh v27.8h, v26.8h, v1.h[5]+ add v16.8h, v16.8h, v2.8h+ mul v18.8h, v26.8h, v1.h[4]+ sub v31.8h, v20.8h, v15.8h+ mls v4.8h, v8.8h, v7.h[0]+ sub v28.8h, v29.8h, v24.8h+ mls v18.8h, v27.8h, v7.h[0]+ ldr q22, [x0, #0x1d0]+ mul v26.8h, v28.8h, v0.h[6]+ mul v2.8h, v5.8h, v0.h[2]+ sub v12.8h, v4.8h, v18.8h+ sqrdmulh v24.8h, v28.8h, v0.h[7]+ mls v14.8h, v9.8h, v7.h[0]+ sqrdmulh v10.8h, v12.8h, v0.h[5]+ mls v26.8h, v24.8h, v7.h[0]+ ldr q24, [x0, #0x50]+ mul v8.8h, v12.8h, v0.h[4]+ str q14, [x0, #0x1b0]+ add v28.8h, v4.8h, v18.8h+ sqrdmulh v5.8h, v5.8h, v0.h[3]+ add v6.8h, v20.8h, v15.8h+ sqrdmulh v3.8h, v19.8h, v0.h[1]+ sub v13.8h, v16.8h, v6.8h+ sqrdmulh v12.8h, v21.8h, v1.h[1]+ sqrdmulh v21.8h, v13.8h, v0.h[1]+ sqrdmulh v27.8h, v31.8h, v0.h[5]+ ldr q25, [x0, #0xd0]+ mls v11.8h, v12.8h, v7.h[0]+ mul v23.8h, v31.8h, v0.h[4]+ mul v18.8h, v13.8h, v0.h[0]+ add v30.8h, v26.8h, v11.8h+ sub v13.8h, v26.8h, v11.8h+ mls v23.8h, v27.8h, v7.h[0]+ add v12.8h, v30.8h, v28.8h+ sub v19.8h, v30.8h, v28.8h+ mls v2.8h, v5.8h, v7.h[0]+ str q12, [x0, #0x40]+ sqrdmulh v26.8h, v13.8h, v0.h[3]+ mls v8.8h, v10.8h, v7.h[0]+ ldr q30, [x0, #0x150]+ sub v20.8h, v2.8h, v23.8h+ mul v4.8h, v13.8h, v0.h[2]+ add v13.8h, v2.8h, v23.8h+ mls v4.8h, v26.8h, v7.h[0]+ ldr q2, [x0, #0x90]+ mul v23.8h, v20.8h, v0.h[0]+ ldr q29, [x0, #0x10]+ sqrdmulh v10.8h, v20.8h, v0.h[1]+ str q13, [x0, #0x80]+ sub v13.8h, v4.8h, v8.8h+ mls v17.8h, v3.8h, v7.h[0]+ ldr q3, [x0, #0x110]+ mls v18.8h, v21.8h, v7.h[0]+ sub v21.8h, v2.8h, v25.8h+ sqrdmulh v9.8h, v13.8h, v0.h[1]+ str q17, [x0, #0x130]+ mul v17.8h, v19.8h, v0.h[0]+ sub x4, x4, #0x1+ cbnz x4, Lmlk_intt_layer123_start+ mls v23.8h, v10.8h, v7.h[0]+ ldr q11, [x0, #0x190]+ str q18, [x0, #0x100]+ add v27.8h, v3.8h, v30.8h+ mul v13.8h, v13.8h, v0.h[0]+ sub v5.8h, v29.8h, v24.8h+ add v14.8h, v16.8h, v6.8h+ mls v13.8h, v9.8h, v7.h[0]+ add v10.8h, v11.8h, v22.8h+ str q23, [x0, #0x180]+ sub v20.8h, v11.8h, v22.8h+ sub v23.8h, v27.8h, v10.8h+ sqrdmulh v16.8h, v21.8h, v1.h[1]+ sqrdmulh v31.8h, v23.8h, v0.h[5]+ str q13, [x0, #0x1c0]+ add v13.8h, v4.8h, v8.8h+ mul v18.8h, v21.8h, v1.h[0]+ str q13, [x0, #0xc0]+ sqrdmulh v13.8h, v19.8h, v0.h[1]+ sqrdmulh v28.8h, v20.8h, v1.h[5]+ str q14, [x0], #0x10+ mul v4.8h, v20.8h, v1.h[4]+ mls v17.8h, v13.8h, v7.h[0]+ sub v13.8h, v3.8h, v30.8h+ sqrdmulh v8.8h, v13.8h, v1.h[3]+ mul v12.8h, v13.8h, v1.h[2]+ mls v4.8h, v28.8h, v7.h[0]+ mls v12.8h, v8.8h, v7.h[0]+ mls v18.8h, v16.8h, v7.h[0]+ str q17, [x0, #0x130]+ sqrdmulh v15.8h, v5.8h, v0.h[7]+ add v11.8h, v27.8h, v10.8h+ mul v16.8h, v5.8h, v0.h[6]+ sub v8.8h, v12.8h, v4.8h+ sqrdmulh v28.8h, v8.8h, v0.h[5]+ add v13.8h, v2.8h, v25.8h+ mls v16.8h, v15.8h, v7.h[0]+ add v26.8h, v12.8h, v4.8h+ mul v8.8h, v8.8h, v0.h[4]+ add v4.8h, v29.8h, v24.8h+ mls v8.8h, v28.8h, v7.h[0]+ sub v20.8h, v4.8h, v13.8h+ add v14.8h, v4.8h, v13.8h+ add v12.8h, v16.8h, v18.8h+ sqrdmulh v22.8h, v20.8h, v0.h[3]+ add v27.8h, v14.8h, v11.8h+ sub v13.8h, v16.8h, v18.8h+ mul v4.8h, v20.8h, v0.h[2]+ str q27, [x0], #0x10+ sub v24.8h, v12.8h, v26.8h+ sqrdmulh v3.8h, v13.8h, v0.h[3]+ mul v13.8h, v13.8h, v0.h[2]+ sqrdmulh v27.8h, v24.8h, v0.h[1]+ mls v13.8h, v3.8h, v7.h[0]+ mul v9.8h, v24.8h, v0.h[0]+ mls v9.8h, v27.8h, v7.h[0]+ add v30.8h, v13.8h, v8.8h+ sub v13.8h, v13.8h, v8.8h+ mls v4.8h, v22.8h, v7.h[0]+ str q30, [x0, #0xb0]+ sqrdmulh v16.8h, v13.8h, v0.h[1]+ str q9, [x0, #0x130]+ mul v9.8h, v13.8h, v0.h[0]+ add v13.8h, v12.8h, v26.8h+ str q13, [x0, #0x30]+ mul v13.8h, v23.8h, v0.h[4]+ sub v23.8h, v14.8h, v11.8h+ mls v13.8h, v31.8h, v7.h[0]+ mls v9.8h, v16.8h, v7.h[0]+ mul v30.8h, v23.8h, v0.h[0]+ sub v24.8h, v4.8h, v13.8h+ add v13.8h, v4.8h, v13.8h+ sqrdmulh v23.8h, v23.8h, v0.h[1]+ str q9, [x0, #0x1b0]+ str q13, [x0, #0x70]+ sqrdmulh v13.8h, v24.8h, v0.h[1]+ mul v21.8h, v24.8h, v0.h[0]+ mls v30.8h, v23.8h, v7.h[0]+ mls v21.8h, v13.8h, v7.h[0]+ str q30, [x0, #0xf0]+ str q21, [x0, #0x170]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(intt_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/aarch64/src/mlkem_ntt_aarch64_asm.S view
@@ -0,0 +1,565 @@+/* Copyright (c) 2022 Arm Limited+ * Copyright (c) 2022 Hanno Becker+ * Copyright (c) 2023 Amin Abdulrahman, Matthias Kannwischer+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ *+ * - [SLOTHY_Paper]+ * Fast and Clean: Auditable high-performance assembly via constraint solving+ * Abdulrahman, Becker, Kannwischer, Klein+ * https://eprint.iacr.org/2022/1303+ */++/*yaml+ Name: ntt_aarch64_asm+ Description: AArch64 ML-KEM forward NTT following @[NeonNTT] and @[SLOTHY_Paper]+ Signature: void mlk_ntt_aarch64_asm(int16_t p[256], const int16_t twiddles12345[80], const int16_t twiddles56[384])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t p[256]+ description: Input/output polynomial+ x1:+ type: buffer+ size_bytes: 160+ permissions: read-only+ c_parameter: const int16_t twiddles12345[80]+ description: Twiddle factors for layers 1-5+ x2:+ type: buffer+ size_bytes: 768+ permissions: read-only+ c_parameter: const int16_t twiddles56[384]+ description: Twiddle factors for layers 6-7+ Stack:+ bytes: 64+ description: saving callee-saved Neon registers+*/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_ntt_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(ntt_aarch64_asm)+MLK_ASM_FN_SYMBOL(ntt_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ mov w5, #0xd01 // =3329+ mov v7.h[0], w5+ mov w5, #0x4ebf // =20159+ mov v7.h[1], w5+ mov x3, x0+ mov x4, #0x4 // =4+ ldr q0, [x1], #0x20+ ldur q1, [x1, #-0x10]+ ldr q21, [x0, #0x40]+ ldr q5, [x0, #0x1c0]+ ldr q30, [x0, #0x110]+ ldr q24, [x0, #0x140]+ ldr q12, [x0, #0x80]+ sqrdmulh v9.8h, v5.8h, v0.h[1]+ mul v23.8h, v5.8h, v0.h[0]+ sqrdmulh v17.8h, v24.8h, v0.h[1]+ ldr q13, [x0, #0xc0]+ mls v23.8h, v9.8h, v7.h[0]+ mul v8.8h, v24.8h, v0.h[0]+ mls v8.8h, v17.8h, v7.h[0]+ add v9.8h, v13.8h, v23.8h+ sub v10.8h, v13.8h, v23.8h+ mul v11.8h, v30.8h, v0.h[0]+ ldr q13, [x0, #0x180]+ sqrdmulh v28.8h, v9.8h, v0.h[3]+ sub v29.8h, v21.8h, v8.8h+ mul v26.8h, v9.8h, v0.h[2]+ add v8.8h, v21.8h, v8.8h+ mul v2.8h, v13.8h, v0.h[0]+ mls v26.8h, v28.8h, v7.h[0]+ mul v28.8h, v10.8h, v0.h[4]+ sqrdmulh v23.8h, v10.8h, v0.h[5]+ add v22.8h, v8.8h, v26.8h+ sqrdmulh v10.8h, v13.8h, v0.h[1]+ sqrdmulh v21.8h, v22.8h, v0.h[7]+ ldr q13, [x0, #0x100]+ mul v16.8h, v22.8h, v0.h[6]+ mls v28.8h, v23.8h, v7.h[0]+ mls v2.8h, v10.8h, v7.h[0]+ sqrdmulh v23.8h, v13.8h, v0.h[1]+ sub v10.8h, v29.8h, v28.8h+ add v17.8h, v29.8h, v28.8h+ mls v16.8h, v21.8h, v7.h[0]+ sub v18.8h, v12.8h, v2.8h+ ldr q29, [x0]+ sqrdmulh v14.8h, v17.8h, v1.h[3]+ add v22.8h, v12.8h, v2.8h+ sqrdmulh v9.8h, v18.8h, v0.h[5]+ mul v21.8h, v13.8h, v0.h[0]+ ldr q13, [x0, #0x150]+ mul v5.8h, v18.8h, v0.h[4]+ mls v5.8h, v9.8h, v7.h[0]+ mul v18.8h, v13.8h, v0.h[0]+ mls v21.8h, v23.8h, v7.h[0]+ sqrdmulh v2.8h, v13.8h, v0.h[1]+ mul v13.8h, v17.8h, v1.h[2]+ sub v4.8h, v29.8h, v21.8h+ mls v13.8h, v14.8h, v7.h[0]+ add v25.8h, v29.8h, v21.8h+ add v6.8h, v4.8h, v5.8h+ sqrdmulh v15.8h, v22.8h, v0.h[3]+ sub v21.8h, v4.8h, v5.8h+ sub v5.8h, v8.8h, v26.8h+ mul v23.8h, v22.8h, v0.h[2]+ add v28.8h, v6.8h, v13.8h+ sub v13.8h, v6.8h, v13.8h+ mul v4.8h, v5.8h, v1.h[0]+ sub x4, x4, #0x2++Lmlk_ntt_layer123_start:+ mls v23.8h, v15.8h, v7.h[0]+ ldr q6, [x0, #0x190]+ ldr q15, [x0, #0x90]+ ldr q19, [x0, #0x10]+ mul v22.8h, v10.8h, v1.h[4]+ ldr q24, [x0, #0x50]+ str q13, [x0, #0x140]+ sqrdmulh v13.8h, v6.8h, v0.h[1]+ sub v20.8h, v25.8h, v23.8h+ sqrdmulh v3.8h, v30.8h, v0.h[1]+ str q28, [x0, #0x100]+ ldr q30, [x0, #0x120]+ mul v8.8h, v6.8h, v0.h[0]+ sqrdmulh v27.8h, v10.8h, v1.h[5]+ mls v11.8h, v3.8h, v7.h[0]+ mls v18.8h, v2.8h, v7.h[0]+ ldr q31, [x0, #0x160]+ sqrdmulh v10.8h, v5.8h, v1.h[1]+ mls v8.8h, v13.8h, v7.h[0]+ ldr q13, [x0, #0x1d0]+ sub v14.8h, v24.8h, v18.8h+ add v9.8h, v24.8h, v18.8h+ sqrdmulh v2.8h, v31.8h, v0.h[1]+ mls v4.8h, v10.8h, v7.h[0]+ add v10.8h, v25.8h, v23.8h+ sub v24.8h, v19.8h, v11.8h+ add v25.8h, v19.8h, v11.8h+ sqrdmulh v28.8h, v13.8h, v0.h[1]+ mul v11.8h, v30.8h, v0.h[0]+ mul v17.8h, v13.8h, v0.h[0]+ sub v13.8h, v10.8h, v16.8h+ sub v6.8h, v15.8h, v8.8h+ mls v17.8h, v28.8h, v7.h[0]+ str q13, [x0, #0x40]+ mls v22.8h, v27.8h, v7.h[0]+ ldr q13, [x0, #0xd0]+ add v26.8h, v20.8h, v4.8h+ mul v18.8h, v31.8h, v0.h[0]+ add v27.8h, v10.8h, v16.8h+ str q26, [x0, #0x80]+ sqrdmulh v31.8h, v6.8h, v0.h[5]+ add v3.8h, v21.8h, v22.8h+ str q27, [x0], #0x10+ mul v26.8h, v6.8h, v0.h[4]+ add v6.8h, v13.8h, v17.8h+ sub v5.8h, v13.8h, v17.8h+ str q3, [x0, #0x170]+ sub v17.8h, v21.8h, v22.8h+ sqrdmulh v10.8h, v6.8h, v0.h[3]+ sub v13.8h, v20.8h, v4.8h+ add v20.8h, v15.8h, v8.8h+ sqrdmulh v12.8h, v5.8h, v0.h[5]+ str q13, [x0, #0xb0]+ mul v8.8h, v6.8h, v0.h[2]+ str q17, [x0, #0x1b0]+ mls v8.8h, v10.8h, v7.h[0]+ mul v29.8h, v5.8h, v0.h[4]+ mls v29.8h, v12.8h, v7.h[0]+ sub v5.8h, v9.8h, v8.8h+ add v3.8h, v9.8h, v8.8h+ sqrdmulh v15.8h, v20.8h, v0.h[3]+ mul v4.8h, v5.8h, v1.h[0]+ add v6.8h, v14.8h, v29.8h+ sqrdmulh v9.8h, v3.8h, v0.h[7]+ sqrdmulh v12.8h, v6.8h, v1.h[3]+ sub v10.8h, v14.8h, v29.8h+ mul v23.8h, v6.8h, v1.h[2]+ mls v26.8h, v31.8h, v7.h[0]+ mls v23.8h, v12.8h, v7.h[0]+ mul v16.8h, v3.8h, v0.h[6]+ add v13.8h, v24.8h, v26.8h+ sub v21.8h, v24.8h, v26.8h+ mls v16.8h, v9.8h, v7.h[0]+ add v28.8h, v13.8h, v23.8h+ sub v13.8h, v13.8h, v23.8h+ mul v23.8h, v20.8h, v0.h[2]+ sub x4, x4, #0x1+ cbnz x4, Lmlk_ntt_layer123_start+ sqrdmulh v3.8h, v5.8h, v1.h[1]+ mls v23.8h, v15.8h, v7.h[0]+ ldr q5, [x0, #0x190]+ mul v29.8h, v10.8h, v1.h[4]+ mls v4.8h, v3.8h, v7.h[0]+ sub v19.8h, v25.8h, v23.8h+ sqrdmulh v31.8h, v5.8h, v0.h[1]+ sqrdmulh v6.8h, v30.8h, v0.h[1]+ sub v3.8h, v19.8h, v4.8h+ mul v5.8h, v5.8h, v0.h[0]+ str q3, [x0, #0xc0]+ sqrdmulh v12.8h, v10.8h, v1.h[5]+ mls v18.8h, v2.8h, v7.h[0]+ ldr q3, [x0, #0x1d0]+ mls v5.8h, v31.8h, v7.h[0]+ sqrdmulh v10.8h, v3.8h, v0.h[1]+ mls v11.8h, v6.8h, v7.h[0]+ ldr q31, [x0, #0x90]+ mul v30.8h, v3.8h, v0.h[0]+ mls v30.8h, v10.8h, v7.h[0]+ sub v10.8h, v31.8h, v5.8h+ mls v29.8h, v12.8h, v7.h[0]+ ldr q6, [x0, #0xd0]+ sqrdmulh v15.8h, v10.8h, v0.h[5]+ mul v17.8h, v10.8h, v0.h[4]+ add v10.8h, v6.8h, v30.8h+ sub v6.8h, v6.8h, v30.8h+ sqrdmulh v12.8h, v10.8h, v0.h[3]+ sub v27.8h, v21.8h, v29.8h+ sqrdmulh v3.8h, v6.8h, v0.h[5]+ mul v10.8h, v10.8h, v0.h[2]+ ldr q20, [x0, #0x50]+ mls v10.8h, v12.8h, v7.h[0]+ mul v2.8h, v6.8h, v0.h[4]+ add v6.8h, v20.8h, v18.8h+ add v5.8h, v31.8h, v5.8h+ mls v2.8h, v3.8h, v7.h[0]+ sub v31.8h, v6.8h, v10.8h+ sqrdmulh v12.8h, v5.8h, v0.h[3]+ sub v22.8h, v20.8h, v18.8h+ add v6.8h, v6.8h, v10.8h+ mul v20.8h, v31.8h, v1.h[0]+ add v30.8h, v22.8h, v2.8h+ sqrdmulh v3.8h, v6.8h, v0.h[7]+ sqrdmulh v10.8h, v30.8h, v1.h[3]+ mul v9.8h, v30.8h, v1.h[2]+ ldr q30, [x0, #0x10]+ mls v17.8h, v15.8h, v7.h[0]+ mls v9.8h, v10.8h, v7.h[0]+ mul v15.8h, v6.8h, v0.h[6]+ add v24.8h, v30.8h, v11.8h+ sub v10.8h, v22.8h, v2.8h+ mls v15.8h, v3.8h, v7.h[0]+ add v6.8h, v19.8h, v4.8h+ add v22.8h, v25.8h, v23.8h+ sqrdmulh v3.8h, v10.8h, v1.h[5]+ str q13, [x0, #0x140]+ sub v19.8h, v30.8h, v11.8h+ add v25.8h, v22.8h, v16.8h+ mul v5.8h, v5.8h, v0.h[2]+ sub v13.8h, v22.8h, v16.8h+ str q28, [x0, #0x100]+ mls v5.8h, v12.8h, v7.h[0]+ str q13, [x0, #0x40]+ str q6, [x0, #0x80]+ add v21.8h, v21.8h, v29.8h+ sqrdmulh v13.8h, v31.8h, v1.h[1]+ str q25, [x0], #0x10+ add v12.8h, v19.8h, v17.8h+ sub v31.8h, v19.8h, v17.8h+ mul v30.8h, v10.8h, v1.h[4]+ str q21, [x0, #0x170]+ add v21.8h, v24.8h, v5.8h+ add v6.8h, v12.8h, v9.8h+ mls v30.8h, v3.8h, v7.h[0]+ str q27, [x0, #0x1b0]+ sub v10.8h, v21.8h, v15.8h+ sub v12.8h, v12.8h, v9.8h+ mls v20.8h, v13.8h, v7.h[0]+ str q6, [x0, #0x100]+ str q10, [x0, #0x40]+ sub v13.8h, v24.8h, v5.8h+ add v3.8h, v21.8h, v15.8h+ str q12, [x0, #0x140]+ sub v10.8h, v31.8h, v30.8h+ add v21.8h, v31.8h, v30.8h+ str q3, [x0], #0x10+ add v12.8h, v13.8h, v20.8h+ sub v13.8h, v13.8h, v20.8h+ str q21, [x0, #0x170]+ str q10, [x0, #0x1b0]+ str q12, [x0, #0x70]+ str q13, [x0, #0xb0]+ mov x0, x3+ mov x4, #0x8 // =8+ ldr q2, [x0, #0x20]+ ldr q13, [x1], #0x10+ ldr q30, [x0, #0x30]+ ldr q25, [x2, #0x40]+ ldr q5, [x0]+ ldr q18, [x0, #0x60]+ ldr q12, [x0, #0x70]+ sqrdmulh v17.8h, v2.8h, v13.h[1]+ ldr q4, [x1], #0x10+ ldr q23, [x0, #0x10]+ sqrdmulh v21.8h, v30.8h, v13.h[1]+ ldr q24, [x2, #0x20]+ ldr q9, [x2], #0x60+ mul v10.8h, v30.8h, v13.h[0]+ mul v11.8h, v2.8h, v13.h[0]+ mls v10.8h, v21.8h, v7.h[0]+ sqrdmulh v29.8h, v12.8h, v4.h[1]+ mul v1.8h, v12.8h, v4.h[0]+ add v21.8h, v23.8h, v10.8h+ sub v10.8h, v23.8h, v10.8h+ mul v8.8h, v18.8h, v4.h[0]+ sqrdmulh v23.8h, v21.8h, v13.h[3]+ mul v2.8h, v21.8h, v13.h[2]+ mls v1.8h, v29.8h, v7.h[0]+ mls v2.8h, v23.8h, v7.h[0]+ ldur q15, [x2, #-0x50]+ sqrdmulh v0.8h, v10.8h, v13.h[5]+ mls v11.8h, v17.8h, v7.h[0]+ ldr q29, [x0, #0x50]+ mul v23.8h, v10.8h, v13.h[4]+ mls v23.8h, v0.8h, v7.h[0]+ sub v16.8h, v29.8h, v1.8h+ add v3.8h, v5.8h, v11.8h+ sub v31.8h, v5.8h, v11.8h+ sqrdmulh v22.8h, v16.8h, v4.h[5]+ add v30.8h, v3.8h, v2.8h+ sub v0.8h, v3.8h, v2.8h+ sqrdmulh v28.8h, v18.8h, v4.h[1]+ add v21.8h, v31.8h, v23.8h+ sub v19.8h, v31.8h, v23.8h+ mul v26.8h, v16.8h, v4.h[4]+ trn2 v3.4s, v30.4s, v0.4s+ ldur q23, [x2, #-0x10]+ trn2 v18.4s, v21.4s, v19.4s+ mls v26.8h, v22.8h, v7.h[0]+ trn1 v13.4s, v30.4s, v0.4s+ mls v8.8h, v28.8h, v7.h[0]+ trn2 v31.2d, v3.2d, v18.2d+ trn1 v11.4s, v21.4s, v19.4s+ add v27.8h, v29.8h, v1.8h+ sqrdmulh v6.8h, v31.8h, v15.8h+ trn1 v2.2d, v13.2d, v11.2d+ trn2 v13.2d, v13.2d, v11.2d+ mul v1.8h, v31.8h, v9.8h+ ldr q11, [x0, #0x40]+ sqrdmulh v29.8h, v13.8h, v15.8h+ mls v1.8h, v6.8h, v7.h[0]+ trn1 v6.2d, v3.2d, v18.2d+ mul v17.8h, v13.8h, v9.8h+ sub v13.8h, v11.8h, v8.8h+ sqrdmulh v10.8h, v27.8h, v4.h[3]+ sub v12.8h, v13.8h, v26.8h+ sub v18.8h, v6.8h, v1.8h+ mls v17.8h, v29.8h, v7.h[0]+ add v30.8h, v6.8h, v1.8h+ add v6.8h, v13.8h, v26.8h+ ldur q13, [x2, #-0x30]+ sqrdmulh v16.8h, v18.8h, v23.8h+ trn1 v28.4s, v6.4s, v12.4s+ mul v23.8h, v18.8h, v25.8h+ ldr q25, [x2, #0x10]+ add v20.8h, v2.8h, v17.8h+ mul v0.8h, v30.8h, v24.8h+ sqrdmulh v29.8h, v30.8h, v13.8h+ sub v30.8h, v2.8h, v17.8h+ mls v23.8h, v16.8h, v7.h[0]+ sub x4, x4, #0x2++Lmlk_ntt_layer4567_start:+ ldr q19, [x2, #0x50]+ sub v31.8h, v30.8h, v23.8h+ mls v0.8h, v29.8h, v7.h[0]+ add v16.8h, v11.8h, v8.8h+ ldr q18, [x0, #0xa0]+ trn2 v14.4s, v6.4s, v12.4s+ mul v26.8h, v27.8h, v4.h[2]+ ldr q4, [x1], #0x10+ ldr q24, [x2, #0x40]+ ldr q21, [x0, #0xb0]+ mls v26.8h, v10.8h, v7.h[0]+ add v23.8h, v30.8h, v23.8h+ sub v15.8h, v20.8h, v0.8h+ ldr q9, [x0, #0x90]+ add v10.8h, v20.8h, v0.8h+ mul v8.8h, v18.8h, v4.h[0]+ ldr q1, [x2], #0x60+ trn1 v27.4s, v23.4s, v31.4s+ sqrdmulh v12.8h, v18.8h, v4.h[1]+ trn1 v5.4s, v10.4s, v15.4s+ sub v30.8h, v16.8h, v26.8h+ trn2 v13.2d, v5.2d, v27.2d+ sqrdmulh v2.8h, v21.8h, v4.h[1]+ add v29.8h, v16.8h, v26.8h+ mul v0.8h, v21.8h, v4.h[0]+ str q13, [x0, #0x20]+ trn1 v11.4s, v29.4s, v30.4s+ mls v8.8h, v12.8h, v7.h[0]+ trn2 v26.4s, v29.4s, v30.4s+ trn2 v6.2d, v11.2d, v28.2d+ mls v0.8h, v2.8h, v7.h[0]+ trn2 v16.2d, v26.2d, v14.2d+ trn1 v26.2d, v26.2d, v14.2d+ trn1 v20.2d, v5.2d, v27.2d+ sqrdmulh v29.8h, v6.8h, v25.8h+ trn2 v15.4s, v10.4s, v15.4s+ sqrdmulh v13.8h, v16.8h, v25.8h+ str q20, [x0], #0x40+ sub v30.8h, v9.8h, v0.8h+ add v27.8h, v9.8h, v0.8h+ mul v17.8h, v6.8h, v1.8h+ sqrdmulh v22.8h, v30.8h, v4.h[5]+ mul v18.8h, v16.8h, v1.8h+ mls v18.8h, v13.8h, v7.h[0]+ mul v2.8h, v30.8h, v4.h[4]+ mls v2.8h, v22.8h, v7.h[0]+ trn2 v22.4s, v23.4s, v31.4s+ sub v3.8h, v26.8h, v18.8h+ ldur q25, [x2, #-0x30]+ mls v17.8h, v29.8h, v7.h[0]+ trn2 v31.2d, v15.2d, v22.2d+ trn1 v20.2d, v15.2d, v22.2d+ add v16.8h, v26.8h, v18.8h+ sqrdmulh v26.8h, v3.8h, v19.8h+ trn1 v21.2d, v11.2d, v28.2d+ ldr q11, [x0, #0x40]+ sqrdmulh v29.8h, v16.8h, v25.8h+ stur q20, [x0, #-0x30]+ add v20.8h, v21.8h, v17.8h+ stur q31, [x0, #-0x10]+ mul v23.8h, v3.8h, v24.8h+ ldr q25, [x2, #0x10]+ sub v13.8h, v11.8h, v8.8h+ mls v23.8h, v26.8h, v7.h[0]+ ldur q1, [x2, #-0x40]+ sub v12.8h, v13.8h, v2.8h+ add v6.8h, v13.8h, v2.8h+ sqrdmulh v10.8h, v27.8h, v4.h[3]+ sub v30.8h, v21.8h, v17.8h+ mul v0.8h, v16.8h, v1.8h+ trn1 v28.4s, v6.4s, v12.4s+ sub x4, x4, #0x1+ cbnz x4, Lmlk_ntt_layer4567_start+ add v22.8h, v11.8h, v8.8h+ mul v27.8h, v27.8h, v4.h[2]+ trn2 v17.4s, v6.4s, v12.4s+ ldr q15, [x2], #0x60+ mls v27.8h, v10.8h, v7.h[0]+ add v4.8h, v30.8h, v23.8h+ sub v18.8h, v30.8h, v23.8h+ ldur q6, [x2, #-0x30]+ mls v0.8h, v29.8h, v7.h[0]+ ldur q12, [x2, #-0x40]+ ldur q24, [x2, #-0x20]+ ldur q2, [x2, #-0x10]+ trn1 v9.4s, v4.4s, v18.4s+ add v10.8h, v22.8h, v27.8h+ sub v13.8h, v22.8h, v27.8h+ sub v1.8h, v20.8h, v0.8h+ trn2 v21.4s, v10.4s, v13.4s+ add v27.8h, v20.8h, v0.8h+ trn2 v3.2d, v21.2d, v17.2d+ trn1 v13.4s, v10.4s, v13.4s+ trn1 v31.4s, v27.4s, v1.4s+ sqrdmulh v10.8h, v3.8h, v25.8h+ trn2 v5.2d, v13.2d, v28.2d+ trn1 v13.2d, v13.2d, v28.2d+ trn1 v21.2d, v21.2d, v17.2d+ sqrdmulh v17.8h, v5.8h, v25.8h+ trn2 v30.2d, v31.2d, v9.2d+ mul v25.8h, v3.8h, v15.8h+ str q30, [x0, #0x20]+ trn2 v30.4s, v4.4s, v18.4s+ mls v25.8h, v10.8h, v7.h[0]+ trn2 v3.4s, v27.4s, v1.4s+ mul v20.8h, v5.8h, v15.8h+ trn2 v10.2d, v3.2d, v30.2d+ mls v20.8h, v17.8h, v7.h[0]+ str q10, [x0, #0x30]+ sub v18.8h, v21.8h, v25.8h+ add v10.8h, v21.8h, v25.8h+ trn1 v3.2d, v3.2d, v30.2d+ sqrdmulh v30.8h, v18.8h, v2.8h+ mul v12.8h, v10.8h, v12.8h+ sqrdmulh v6.8h, v10.8h, v6.8h+ str q3, [x0, #0x10]+ add v21.8h, v13.8h, v20.8h+ mul v10.8h, v18.8h, v24.8h+ sub v13.8h, v13.8h, v20.8h+ mls v10.8h, v30.8h, v7.h[0]+ mls v12.8h, v6.8h, v7.h[0]+ trn1 v30.2d, v31.2d, v9.2d+ sub v3.8h, v13.8h, v10.8h+ add v6.8h, v13.8h, v10.8h+ add v10.8h, v21.8h, v12.8h+ sub v21.8h, v21.8h, v12.8h+ trn2 v13.4s, v6.4s, v3.4s+ trn1 v12.4s, v10.4s, v21.4s+ trn2 v21.4s, v10.4s, v21.4s+ trn1 v3.4s, v6.4s, v3.4s+ str q30, [x0], #0x40+ trn2 v10.2d, v21.2d, v13.2d+ trn1 v13.2d, v21.2d, v13.2d+ trn2 v21.2d, v12.2d, v3.2d+ trn1 v3.2d, v12.2d, v3.2d+ str q10, [x0, #0x30]+ str q13, [x0, #0x10]+ str q3, [x0], #0x40+ stur q21, [x0, #-0x20]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(ntt_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/aarch64/src/mlkem_poly_mulcache_compute_aarch64_asm.S view
@@ -0,0 +1,130 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: poly_mulcache_compute_aarch64_asm+ Description: Compute multiplication cache for polynomial+ Signature: void mlk_poly_mulcache_compute_aarch64_asm(int16_t cache[128], const int16_t mlk_poly[256], const int16_t zetas[128], const int16_t zetas_twisted[128])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 256+ permissions: write-only+ c_parameter: int16_t cache[128]+ description: Output cache+ x1:+ type: buffer+ size_bytes: 512+ permissions: read-only+ c_parameter: const int16_t mlk_poly[256]+ description: Input polynomial+ x2:+ type: buffer+ size_bytes: 256+ permissions: read-only+ c_parameter: const int16_t zetas[128]+ description: Zeta values+ x3:+ type: buffer+ size_bytes: 256+ permissions: read-only+ c_parameter: const int16_t zetas_twisted[128]+ description: Twisted zeta values+ Stack:+ bytes: 0+*/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_poly_mulcache_compute_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_mulcache_compute_aarch64_asm)+MLK_ASM_FN_SYMBOL(poly_mulcache_compute_aarch64_asm)++ .cfi_startproc+ mov w5, #0xd01 // =3329+ dup v6.8h, w5+ mov w5, #0x4ebf // =20159+ dup v7.8h, w5+ mov x4, #0x10 // =16+ ldr q0, [x1], #0x20+ ldur q2, [x1, #-0x10]+ ldr q19, [x1], #0x20+ ldr q29, [x3], #0x10+ ldur q16, [x1, #-0x10]+ ldr q18, [x2], #0x10+ ldr q26, [x1], #0x20+ ldr q25, [x2], #0x10+ uzp2 v5.8h, v0.8h, v2.8h+ ldr q28, [x3], #0x10+ ldur q7, [x1, #-0x10]+ ldr q2, [x1], #0x20+ uzp2 v27.8h, v19.8h, v16.8h+ sqrdmulh v16.8h, v5.8h, v29.8h+ ldr q17, [x3], #0x10+ ldr q19, [x3], #0x10+ mul v5.8h, v5.8h, v18.8h+ uzp2 v29.8h, v26.8h, v7.8h+ mul v26.8h, v27.8h, v25.8h+ sqrdmulh v4.8h, v27.8h, v28.8h+ mls v5.8h, v16.8h, v6.h[0]+ lsr x4, x4, #1+ sub x4, x4, #0x2++Lmlk_poly_mulcache_compute_loop_start:+ str q5, [x0], #0x10+ sqrdmulh v22.8h, v29.8h, v17.8h+ ldr q28, [x2], #0x10+ ldur q24, [x1, #-0x10]+ ldr q0, [x1], #0x20+ mls v26.8h, v4.8h, v6.h[0]+ ldur q16, [x1, #-0x10]+ ldr q17, [x3], #0x10+ mul v5.8h, v29.8h, v28.8h+ uzp2 v23.8h, v2.8h, v24.8h+ ldr q18, [x2], #0x10+ mls v5.8h, v22.8h, v6.h[0]+ uzp2 v29.8h, v0.8h, v16.8h+ sqrdmulh v4.8h, v23.8h, v19.8h+ ldr q2, [x1], #0x20+ ldr q19, [x3], #0x10+ str q26, [x0], #0x10+ mul v26.8h, v23.8h, v18.8h+ subs x4, x4, #0x1+ cbnz x4, Lmlk_poly_mulcache_compute_loop_start+ mls v26.8h, v4.8h, v6.h[0]+ str q5, [x0], #0x10+ ldr q5, [x2], #0x10+ ldur q4, [x1, #-0x10]+ sqrdmulh v16.8h, v29.8h, v17.8h+ ldr q0, [x2], #0x10+ mul v29.8h, v29.8h, v5.8h+ uzp2 v18.8h, v2.8h, v4.8h+ str q26, [x0], #0x10+ sqrdmulh v17.8h, v18.8h, v19.8h+ mls v29.8h, v16.8h, v6.h[0]+ mul v26.8h, v18.8h, v0.8h+ mls v26.8h, v17.8h, v6.h[0]+ str q29, [x0], #0x10+ str q26, [x0], #0x10+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_mulcache_compute_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/aarch64/src/mlkem_poly_reduce_aarch64_asm.S view
@@ -0,0 +1,153 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: poly_reduce_aarch64_asm+ Description: Barrett reduction of polynomial coefficients+ Signature: void mlk_poly_reduce_aarch64_asm(int16_t p[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t p[256]+ description: Input/output polynomial+ Stack:+ bytes: 0+*/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_poly_reduce_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_reduce_aarch64_asm)+MLK_ASM_FN_SYMBOL(poly_reduce_aarch64_asm)++ .cfi_startproc+ mov w2, #0xd01 // =3329+ dup v3.8h, w2+ mov w2, #0x4ebf // =20159+ dup v4.8h, w2+ mov x1, #0x8 // =8+ ldr q21, [x0], #0x40+ ldur q18, [x0, #-0x20]+ ldur q0, [x0, #-0x30]+ ldur q5, [x0, #-0x10]+ ldr q26, [x0], #0x40+ sqdmulh v17.8h, v21.8h, v4.h[0]+ sqdmulh v27.8h, v18.8h, v4.h[0]+ sqdmulh v22.8h, v0.8h, v4.h[0]+ srshr v17.8h, v17.8h, #0xb+ sqdmulh v23.8h, v5.8h, v4.h[0]+ srshr v29.8h, v27.8h, #0xb+ mls v21.8h, v17.8h, v3.h[0]+ srshr v17.8h, v22.8h, #0xb+ mls v18.8h, v29.8h, v3.h[0]+ srshr v22.8h, v23.8h, #0xb+ mls v0.8h, v17.8h, v3.h[0]+ sshr v2.8h, v21.8h, #0xf+ mls v5.8h, v22.8h, v3.h[0]+ sshr v29.8h, v18.8h, #0xf+ and v19.16b, v3.16b, v2.16b+ sqdmulh v2.8h, v26.8h, v4.h[0]+ sshr v31.8h, v0.8h, #0xf+ add v17.8h, v21.8h, v19.8h+ and v21.16b, v3.16b, v29.16b+ and v31.16b, v3.16b, v31.16b+ sub x1, x1, #0x2++Lmlk_poly_reduce_loop_start:+ add v21.8h, v18.8h, v21.8h+ ldur q18, [x0, #-0x20]+ add v25.8h, v0.8h, v31.8h+ ldur q0, [x0, #-0x30]+ stur q21, [x0, #-0x60]+ sshr v28.8h, v5.8h, #0xf+ stur q17, [x0, #-0x80]+ srshr v23.8h, v2.8h, #0xb+ sqdmulh v30.8h, v18.8h, v4.h[0]+ stur q25, [x0, #-0x70]+ and v22.16b, v3.16b, v28.16b+ sqdmulh v7.8h, v0.8h, v4.h[0]+ add v16.8h, v5.8h, v22.8h+ ldur q5, [x0, #-0x10]+ mls v26.8h, v23.8h, v3.h[0]+ stur q16, [x0, #-0x50]+ srshr v6.8h, v30.8h, #0xb+ srshr v1.8h, v7.8h, #0xb+ sqdmulh v19.8h, v5.8h, v4.h[0]+ mls v18.8h, v6.8h, v3.h[0]+ sshr v24.8h, v26.8h, #0xf+ mls v0.8h, v1.8h, v3.h[0]+ and v27.16b, v3.16b, v24.16b+ srshr v29.8h, v19.8h, #0xb+ add v17.8h, v26.8h, v27.8h+ ldr q26, [x0], #0x40+ sshr v1.8h, v18.8h, #0xf+ mls v5.8h, v29.8h, v3.h[0]+ sshr v20.8h, v0.8h, #0xf+ and v21.16b, v3.16b, v1.16b+ and v31.16b, v3.16b, v20.16b+ sqdmulh v2.8h, v26.8h, v4.h[0]+ subs x1, x1, #0x1+ cbnz x1, Lmlk_poly_reduce_loop_start+ add v28.8h, v0.8h, v31.8h+ ldur q29, [x0, #-0x10]+ add v21.8h, v18.8h, v21.8h+ srshr v18.8h, v2.8h, #0xb+ sshr v2.8h, v5.8h, #0xf+ ldur q16, [x0, #-0x20]+ stur q17, [x0, #-0x80]+ ldur q0, [x0, #-0x30]+ and v2.16b, v3.16b, v2.16b+ sqdmulh v24.8h, v29.8h, v4.h[0]+ stur q28, [x0, #-0x70]+ stur q21, [x0, #-0x60]+ add v31.8h, v5.8h, v2.8h+ sqdmulh v6.8h, v16.8h, v4.h[0]+ stur q31, [x0, #-0x50]+ sqdmulh v17.8h, v0.8h, v4.h[0]+ srshr v22.8h, v24.8h, #0xb+ mls v26.8h, v18.8h, v3.h[0]+ srshr v31.8h, v6.8h, #0xb+ mls v29.8h, v22.8h, v3.h[0]+ srshr v19.8h, v17.8h, #0xb+ mls v16.8h, v31.8h, v3.h[0]+ sshr v7.8h, v26.8h, #0xf+ mls v0.8h, v19.8h, v3.h[0]+ and v5.16b, v3.16b, v7.16b+ sshr v22.8h, v29.8h, #0xf+ add v27.8h, v26.8h, v5.8h+ and v26.16b, v3.16b, v22.16b+ sshr v20.8h, v16.8h, #0xf+ stur q27, [x0, #-0x40]+ and v2.16b, v3.16b, v20.16b+ sshr v23.8h, v0.8h, #0xf+ add v18.8h, v29.8h, v26.8h+ add v31.8h, v16.8h, v2.8h+ and v29.16b, v3.16b, v23.16b+ stur q18, [x0, #-0x10]+ add v25.8h, v0.8h, v29.8h+ stur q31, [x0, #-0x20]+ stur q25, [x0, #-0x30]+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_reduce_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/aarch64/src/mlkem_poly_tobytes_aarch64_asm.S view
@@ -0,0 +1,124 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: poly_tobytes_aarch64_asm+ Description: Convert polynomial to byte representation+ Signature: void mlk_poly_tobytes_aarch64_asm(uint8_t r[384], const int16_t a[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 384+ permissions: write-only+ c_parameter: uint8_t r[384]+ description: Output byte array+ x1:+ type: buffer+ size_bytes: 512+ permissions: read-only+ c_parameter: const int16_t a[256]+ description: Input polynomial+ Stack:+ bytes: 0+*/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (!defined(MLK_CONFIG_NO_KEYPAIR_API) || \+ !defined(MLK_CONFIG_NO_ENCAPS_API))++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_poly_tobytes_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_tobytes_aarch64_asm)+MLK_ASM_FN_SYMBOL(poly_tobytes_aarch64_asm)++ .cfi_startproc+ mov x2, #0x10 // =16+ ldr q5, [x1, #0x10]+ ldr q3, [x1], #0x20+ ldr q29, [x1], #0x20+ ldur q2, [x1, #-0x10]+ ldr q27, [x1, #0x10]+ ldr q23, [x1, #0x30]+ ldr q17, [x1], #0x20+ ldr q16, [x1], #0x20+ uzp2 v26.8h, v3.8h, v5.8h+ uzp1 v19.8h, v3.8h, v5.8h+ uzp2 v0.8h, v29.8h, v2.8h+ uzp1 v1.8h, v29.8h, v2.8h+ xtn v5.8b, v26.8h+ shrn v3.8b, v19.8h, #0x8+ shrn v4.8b, v26.8h, #0x4+ xtn v18.8b, v0.8h+ shrn v30.8b, v0.8h, #0x4+ xtn v28.8b, v1.8h+ shrn v29.8b, v1.8h, #0x8+ sli v3.8b, v5.8b, #0x4+ xtn v2.8b, v19.8h+ sli v29.8b, v18.8b, #0x4+ lsr x2, x2, #1+ sub x2, x2, #0x2++Lmlk_poly_tobytes_loop_start:+ uzp1 v25.8h, v17.8h, v27.8h+ uzp2 v31.8h, v17.8h, v27.8h+ uzp1 v24.8h, v16.8h, v23.8h+ uzp2 v6.8h, v16.8h, v23.8h+ st3 { v2.8b, v3.8b, v4.8b }, [x0], #24+ shrn v3.8b, v25.8h, #0x8+ ldr q17, [x1], #0x20+ shrn v4.8b, v31.8h, #0x4+ xtn v21.8b, v6.8h+ ldr q23, [x1, #0x10]+ st3 { v28.8b, v29.8b, v30.8b }, [x0], #24+ shrn v29.8b, v24.8h, #0x8+ ldur q27, [x1, #-0x10]+ xtn v20.8b, v31.8h+ ldr q16, [x1], #0x20+ sli v29.8b, v21.8b, #0x4+ xtn v2.8b, v25.8h+ sli v3.8b, v20.8b, #0x4+ xtn v28.8b, v24.8h+ shrn v30.8b, v6.8h, #0x4+ subs x2, x2, #0x1+ cbnz x2, Lmlk_poly_tobytes_loop_start+ uzp2 v7.8h, v17.8h, v27.8h+ uzp1 v25.8h, v17.8h, v27.8h+ uzp2 v0.8h, v16.8h, v23.8h+ st3 { v2.8b, v3.8b, v4.8b }, [x0], #24+ st3 { v28.8b, v29.8b, v30.8b }, [x0], #24+ shrn v21.8b, v25.8h, #0x8+ uzp1 v2.8h, v16.8h, v23.8h+ shrn v22.8b, v7.8h, #0x4+ shrn v4.8b, v0.8h, #0x4+ xtn v28.8b, v7.8h+ xtn v27.8b, v0.8h+ shrn v3.8b, v2.8h, #0x8+ sli v21.8b, v28.8b, #0x4+ xtn v2.8b, v2.8h+ sli v3.8b, v27.8b, #0x4+ xtn v20.8b, v25.8h+ st3 { v20.8b, v21.8b, v22.8b }, [x0], #24+ st3 { v2.8b, v3.8b, v4.8b }, [x0], #24+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_tobytes_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ (!MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/aarch64/src/mlkem_poly_tomont_aarch64_asm.S view
@@ -0,0 +1,102 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: poly_tomont_aarch64_asm+ Description: Convert polynomial to Montgomery domain+ Signature: void mlk_poly_tomont_aarch64_asm(int16_t p[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t p[256]+ description: Input/output polynomial+ Stack:+ bytes: 0+*/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ !defined(MLK_CONFIG_NO_KEYPAIR_API)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_poly_tomont_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_tomont_aarch64_asm)+MLK_ASM_FN_SYMBOL(poly_tomont_aarch64_asm)++ .cfi_startproc+ mov w2, #0xd01 // =3329+ dup v4.8h, w2+ mov w2, #-0x414 // =-1044+ dup v2.8h, w2+ mov w2, #-0x2824 // =-10276+ dup v3.8h, w2+ mov x1, #0x8 // =8+ ldr q18, [x0, #0x20]+ ldr q0, [x0, #0x10]+ ldr q16, [x0], #0x40+ sqrdmulh v23.8h, v0.8h, v3.8h+ mul v26.8h, v0.8h, v2.8h+ sqrdmulh v19.8h, v16.8h, v3.8h+ mls v26.8h, v23.8h, v4.h[0]+ mul v29.8h, v16.8h, v2.8h+ ldur q16, [x0, #-0x10]+ mls v29.8h, v19.8h, v4.h[0]+ stur q26, [x0, #-0x30]+ sqrdmulh v26.8h, v18.8h, v3.8h+ mul v18.8h, v18.8h, v2.8h+ stur q29, [x0, #-0x40]+ sqrdmulh v29.8h, v16.8h, v3.8h+ mls v18.8h, v26.8h, v4.h[0]+ sub x1, x1, #0x1++Lmlk_poly_tomont_loop:+ ldr q19, [x0, #0x10]+ mul v26.8h, v16.8h, v2.8h+ ldr q23, [x0, #0x20]+ ldr q17, [x0], #0x40+ mls v26.8h, v29.8h, v4.h[0]+ ldur q16, [x0, #-0x10]+ sqrdmulh v28.8h, v19.8h, v3.8h+ stur q18, [x0, #-0x60]+ mul v0.8h, v19.8h, v2.8h+ stur q26, [x0, #-0x50]+ sqrdmulh v24.8h, v23.8h, v3.8h+ mul v18.8h, v23.8h, v2.8h+ sqrdmulh v22.8h, v17.8h, v3.8h+ mul v26.8h, v17.8h, v2.8h+ mls v0.8h, v28.8h, v4.h[0]+ mls v26.8h, v22.8h, v4.h[0]+ sqrdmulh v29.8h, v16.8h, v3.8h+ stur q0, [x0, #-0x30]+ mls v18.8h, v24.8h, v4.h[0]+ stur q26, [x0, #-0x40]+ sub x1, x1, #0x1+ cbnz x1, Lmlk_poly_tomont_loop+ mul v16.8h, v16.8h, v2.8h+ stur q18, [x0, #-0x20]+ mls v16.8h, v29.8h, v4.h[0]+ stur q16, [x0, #-0x10]+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_tomont_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ !MLK_CONFIG_NO_KEYPAIR_API */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/aarch64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S view
@@ -0,0 +1,264 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ */++/*yaml+ Name: polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm+ Description: Re-implementation of asymmetric base multiplication following @[NeonNTT] for k=2+ Signature: void mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm(int16_t r[256], const int16_t a[512], const int16_t b[512], const int16_t b_cache[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: write-only+ c_parameter: int16_t r[256]+ description: Output polynomial+ x1:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int16_t a[512]+ description: Input polynomial vector a+ x2:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int16_t b[512]+ description: Input polynomial vector b+ x3:+ type: buffer+ size_bytes: 512+ permissions: read-only+ c_parameter: const int16_t b_cache[256]+ description: Cached values for b+ Stack:+ bytes: 64+ description: saving callee-saved Neon registers+*/++/* Re-implementation of asymmetric base multiplication following @[NeonNTT] */++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm)+MLK_ASM_FN_SYMBOL(polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ mov w14, #0xd01 // =3329+ dup v0.8h, w14+ mov w14, #0xcff // =3327+ dup v2.8h, w14+ add x4, x1, #0x200+ add x5, x2, #0x200+ add x6, x3, #0x100+ mov x13, #0x10 // =16+ ldr q12, [x1], #0x20+ ldur q9, [x1, #-0x10]+ ldr q22, [x2], #0x20+ ldur q30, [x2, #-0x10]+ ldr q6, [x5], #0x20+ ldr q7, [x4, #0x10]+ ldr q8, [x4], #0x20+ ldur q23, [x5, #-0x10]+ uzp1 v16.8h, v12.8h, v9.8h+ uzp2 v14.8h, v12.8h, v9.8h+ uzp2 v13.8h, v22.8h, v30.8h+ uzp1 v18.8h, v22.8h, v30.8h+ ld1 { v27.8h }, [x3], #16+ ld1 { v17.8h }, [x6], #16+ smull2 v4.4s, v16.8h, v18.8h+ ldr q31, [x1, #0x10]+ smull v19.4s, v16.4h, v13.4h+ ldr q24, [x1], #0x20+ smlal v19.4s, v14.4h, v18.4h+ ldr q22, [x2], #0x20+ smlal2 v4.4s, v14.8h, v27.8h+ uzp2 v5.8h, v6.8h, v23.8h+ smull2 v29.4s, v16.8h, v13.8h+ uzp2 v26.8h, v8.8h, v7.8h+ smlal2 v29.4s, v14.8h, v18.8h+ uzp1 v30.8h, v24.8h, v31.8h+ uzp1 v8.8h, v8.8h, v7.8h+ smull v11.4s, v16.4h, v18.4h+ smlal v11.4s, v14.4h, v27.4h+ ldur q1, [x2, #-0x10]+ uzp1 v28.8h, v6.8h, v23.8h+ smlal2 v29.4s, v8.8h, v5.8h+ ldr q25, [x5], #0x20+ smlal v19.4s, v8.4h, v5.4h+ ldr q3, [x4, #0x10]+ smlal2 v29.4s, v26.8h, v28.8h+ uzp1 v27.8h, v22.8h, v1.8h+ smlal v19.4s, v26.4h, v28.4h+ ldr q12, [x4], #0x20+ smlal2 v4.4s, v8.8h, v28.8h+ ldur q21, [x5, #-0x10]+ smlal2 v4.4s, v26.8h, v17.8h+ smlal v11.4s, v8.4h, v28.4h+ ld1 { v15.8h }, [x6], #16+ smlal v11.4s, v26.4h, v17.4h+ ld1 { v20.8h }, [x3], #16+ uzp1 v28.8h, v19.8h, v29.8h+ smull2 v23.4s, v30.8h, v27.8h+ smull v26.4s, v30.4h, v27.4h+ uzp2 v16.8h, v22.8h, v1.8h+ mul v28.8h, v28.8h, v2.8h+ uzp1 v10.8h, v11.8h, v4.8h+ smull2 v8.4s, v30.8h, v16.8h+ mul v13.8h, v10.8h, v2.8h+ smlal v19.4s, v28.4h, v0.4h+ smlal2 v29.4s, v28.8h, v0.8h+ smull v18.4s, v30.4h, v16.4h+ uzp1 v30.8h, v25.8h, v21.8h+ smlal v11.4s, v13.4h, v0.4h+ uzp2 v6.8h, v24.8h, v31.8h+ uzp1 v16.8h, v12.8h, v3.8h+ smlal2 v4.4s, v13.8h, v0.8h+ uzp2 v17.8h, v25.8h, v21.8h+ smlal2 v8.4s, v6.8h, v27.8h+ uzp2 v12.8h, v12.8h, v3.8h+ smlal v18.4s, v6.4h, v27.4h+ uzp2 v9.8h, v19.8h, v29.8h+ smlal2 v8.4s, v16.8h, v17.8h+ smlal2 v8.4s, v12.8h, v30.8h+ uzp2 v19.8h, v11.8h, v4.8h+ sub x13, x13, #0x2++Lmlk_polyvec_basemul_acc_montgomery_cached_k2_loop_start:+ smlal v18.4s, v16.4h, v17.4h+ ldr q7, [x4], #0x20+ ldr q10, [x2, #0x10]+ smlal v18.4s, v12.4h, v30.4h+ smlal2 v23.4s, v6.8h, v20.8h+ ldr q14, [x2], #0x20+ smlal2 v23.4s, v16.8h, v30.8h+ zip1 v25.8h, v19.8h, v9.8h+ zip2 v3.8h, v19.8h, v9.8h+ smlal2 v23.4s, v12.8h, v15.8h+ smlal v26.4s, v6.4h, v20.4h+ uzp1 v5.8h, v18.8h, v8.8h+ uzp2 v21.8h, v14.8h, v10.8h+ smlal v26.4s, v16.4h, v30.4h+ str q25, [x0], #0x20+ mul v29.8h, v5.8h, v2.8h+ uzp1 v24.8h, v14.8h, v10.8h+ stur q3, [x0, #-0x10]+ smlal v26.4s, v12.4h, v15.4h+ ld1 { v15.8h }, [x6], #16+ ldr q28, [x1, #0x10]+ ldr q11, [x1], #0x20+ ldr q13, [x5], #0x20+ ldur q27, [x4, #-0x10]+ smlal2 v8.4s, v29.8h, v0.8h+ ldur q22, [x5, #-0x10]+ smlal v18.4s, v29.4h, v0.4h+ uzp1 v4.8h, v26.8h, v23.8h+ uzp1 v1.8h, v11.8h, v28.8h+ uzp2 v6.8h, v11.8h, v28.8h+ uzp1 v16.8h, v7.8h, v27.8h+ mul v31.8h, v4.8h, v2.8h+ uzp2 v17.8h, v13.8h, v22.8h+ ld1 { v20.8h }, [x3], #16+ uzp2 v9.8h, v18.8h, v8.8h+ smull2 v8.4s, v1.8h, v21.8h+ uzp1 v30.8h, v13.8h, v22.8h+ smlal2 v8.4s, v6.8h, v24.8h+ smlal2 v8.4s, v16.8h, v17.8h+ uzp2 v12.8h, v7.8h, v27.8h+ smlal v26.4s, v31.4h, v0.4h+ smlal2 v23.4s, v31.8h, v0.8h+ smull v18.4s, v1.4h, v21.4h+ smlal v18.4s, v6.4h, v24.4h+ smlal2 v8.4s, v12.8h, v30.8h+ uzp2 v19.8h, v26.8h, v23.8h+ smull2 v23.4s, v1.8h, v24.8h+ smull v26.4s, v1.4h, v24.4h+ subs x13, x13, #0x1+ cbnz x13, Lmlk_polyvec_basemul_acc_montgomery_cached_k2_loop_start+ smlal v26.4s, v6.4h, v20.4h+ smlal2 v23.4s, v6.8h, v20.8h+ smlal v26.4s, v16.4h, v30.4h+ smlal2 v23.4s, v16.8h, v30.8h+ smlal v26.4s, v12.4h, v15.4h+ smlal2 v23.4s, v12.8h, v15.8h+ smlal v18.4s, v16.4h, v17.4h+ smlal v18.4s, v12.4h, v30.4h+ zip1 v12.8h, v19.8h, v9.8h+ str q12, [x0], #0x20+ uzp1 v12.8h, v26.8h, v23.8h+ mul v6.8h, v12.8h, v2.8h+ uzp1 v12.8h, v18.8h, v8.8h+ mul v12.8h, v12.8h, v2.8h+ smlal v26.4s, v6.4h, v0.4h+ smlal2 v23.4s, v6.8h, v0.8h+ smlal2 v8.4s, v12.8h, v0.8h+ smlal v18.4s, v12.4h, v0.4h+ zip2 v12.8h, v19.8h, v9.8h+ uzp2 v6.8h, v26.8h, v23.8h+ stur q12, [x0, #-0x10]+ uzp2 v12.8h, v18.8h, v8.8h+ zip2 v1.8h, v6.8h, v12.8h+ zip1 v12.8h, v6.8h, v12.8h+ str q1, [x0, #0x10]+ str q12, [x0], #0x20+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm)+++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/aarch64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S view
@@ -0,0 +1,317 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ */++/*yaml+ Name: polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm+ Description: Re-implementation of asymmetric base multiplication following @[NeonNTT] for k=3+ Signature: void mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm(int16_t r[256], const int16_t a[768], const int16_t b[768], const int16_t b_cache[384])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: write-only+ c_parameter: int16_t r[256]+ description: Output polynomial+ x1:+ type: buffer+ size_bytes: 1536+ permissions: read-only+ c_parameter: const int16_t a[768]+ description: Input polynomial vector a+ x2:+ type: buffer+ size_bytes: 1536+ permissions: read-only+ c_parameter: const int16_t b[768]+ description: Input polynomial vector b+ x3:+ type: buffer+ size_bytes: 768+ permissions: read-only+ c_parameter: const int16_t b_cache[384]+ description: Cached values for b+ Stack:+ bytes: 64+ description: saving callee-saved Neon registers+*/++/* Re-implementation of asymmetric base multiplication following @[NeonNTT] */++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 3)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm)+MLK_ASM_FN_SYMBOL(polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ mov w14, #0xd01 // =3329+ dup v0.8h, w14+ mov w14, #0xcff // =3327+ dup v2.8h, w14+ add x4, x1, #0x200+ add x5, x2, #0x200+ add x6, x3, #0x100+ add x7, x1, #0x400+ add x8, x2, #0x400+ add x9, x3, #0x200+ mov x13, #0x10 // =16+ ldr q6, [x7], #0x20+ ldr q19, [x2, #0x10]+ ldr q23, [x1], #0x20+ ldur q14, [x1, #-0x10]+ ldr q17, [x2], #0x20+ ldr q11, [x4, #0x10]+ ldur q28, [x7, #-0x10]+ ld1 { v30.8h }, [x3], #16+ ldr q26, [x4], #0x20+ ldr q16, [x8, #0x10]+ uzp1 v8.8h, v23.8h, v14.8h+ ldr q22, [x5, #0x10]+ ldr q18, [x5], #0x20+ uzp1 v20.8h, v17.8h, v19.8h+ uzp2 v24.8h, v23.8h, v14.8h+ ldr q31, [x8], #0x20+ smull2 v4.4s, v8.8h, v20.8h+ uzp1 v25.8h, v26.8h, v11.8h+ smull v13.4s, v8.4h, v20.4h+ ld1 { v23.8h }, [x6], #16+ uzp1 v1.8h, v18.8h, v22.8h+ smlal v13.4s, v24.4h, v30.4h+ smlal2 v4.4s, v24.8h, v30.8h+ uzp2 v5.8h, v26.8h, v11.8h+ smlal2 v4.4s, v25.8h, v1.8h+ uzp1 v29.8h, v6.8h, v28.8h+ smlal2 v4.4s, v5.8h, v23.8h+ ld1 { v7.8h }, [x9], #16+ smlal v13.4s, v25.4h, v1.4h+ uzp2 v17.8h, v17.8h, v19.8h+ uzp1 v27.8h, v31.8h, v16.8h+ smlal v13.4s, v5.4h, v23.4h+ uzp2 v22.8h, v18.8h, v22.8h+ smull v18.4s, v8.4h, v17.4h+ uzp2 v28.8h, v6.8h, v28.8h+ smlal v13.4s, v29.4h, v27.4h+ smlal2 v4.4s, v29.8h, v27.8h+ uzp2 v26.8h, v31.8h, v16.8h+ smlal2 v4.4s, v28.8h, v7.8h+ ldr q3, [x7, #0x10]+ smlal v13.4s, v28.4h, v7.4h+ ldr q7, [x1], #0x20+ smlal v18.4s, v24.4h, v20.4h+ ldr q15, [x2], #0x20+ smlal v18.4s, v25.4h, v22.4h+ smull2 v8.4s, v8.8h, v17.8h+ ldur q17, [x1, #-0x10]+ uzp1 v23.8h, v13.8h, v4.8h+ smlal v18.4s, v5.4h, v1.4h+ smlal2 v8.4s, v24.8h, v20.8h+ ld1 { v16.8h }, [x3], #16+ mul v23.8h, v23.8h, v2.8h+ ldr q19, [x5, #0x10]+ ldr q14, [x4, #0x10]+ ldr q11, [x4], #0x20+ ldur q20, [x2, #-0x10]+ smlal2 v8.4s, v25.8h, v22.8h+ smlal2 v8.4s, v5.8h, v1.8h+ ldr q22, [x5], #0x20+ uzp1 v1.8h, v7.8h, v17.8h+ smlal v18.4s, v29.4h, v26.4h+ smlal v13.4s, v23.4h, v0.4h+ uzp2 v31.8h, v11.8h, v14.8h+ uzp1 v21.8h, v15.8h, v20.8h+ smlal2 v4.4s, v23.8h, v0.8h+ ld1 { v9.8h }, [x6], #16+ smlal v18.4s, v28.4h, v27.4h+ smlal2 v8.4s, v29.8h, v26.8h+ ldr q25, [x7], #0x20+ smull v26.4s, v1.4h, v21.4h+ uzp1 v24.8h, v22.8h, v19.8h+ smlal2 v8.4s, v28.8h, v27.8h+ uzp2 v28.8h, v7.8h, v17.8h+ uzp1 v29.8h, v11.8h, v14.8h+ smull2 v23.4s, v1.8h, v21.8h+ ldr q27, [x8], #0x20+ smlal2 v23.4s, v28.8h, v16.8h+ ldur q11, [x8, #-0x10]+ smlal2 v23.4s, v29.8h, v24.8h+ uzp2 v7.8h, v13.8h, v4.8h+ uzp2 v19.8h, v22.8h, v19.8h+ ld1 { v4.8h }, [x9], #16+ smlal2 v23.4s, v31.8h, v9.8h+ uzp1 v13.8h, v25.8h, v3.8h+ uzp1 v14.8h, v18.8h, v8.8h+ smlal v26.4s, v28.4h, v16.4h+ uzp2 v17.8h, v27.8h, v11.8h+ uzp2 v20.8h, v15.8h, v20.8h+ mul v14.8h, v14.8h, v2.8h+ sub x13, x13, #0x2++Lmlk_polyvec_basemul_acc_montgomery_cached_k3_loop_start:+ uzp1 v6.8h, v27.8h, v11.8h+ smlal v26.4s, v29.4h, v24.4h+ uzp2 v16.8h, v25.8h, v3.8h+ smlal v26.4s, v31.4h, v9.4h+ ldr q3, [x7, #0x10]+ smlal v26.4s, v13.4h, v6.4h+ smlal2 v8.4s, v14.8h, v0.8h+ ldr q27, [x8], #0x20+ smlal v18.4s, v14.4h, v0.4h+ ldr q25, [x7], #0x20+ smlal2 v23.4s, v13.8h, v6.8h+ ldr q11, [x1], #0x20+ smlal2 v23.4s, v16.8h, v4.8h+ smlal v26.4s, v16.4h, v4.4h+ ldur q22, [x1, #-0x10]+ uzp2 v30.8h, v18.8h, v8.8h+ smull v18.4s, v1.4h, v20.4h+ smlal v18.4s, v28.4h, v21.4h+ ldr q14, [x2], #0x20+ smlal v18.4s, v29.4h, v19.4h+ zip1 v5.8h, v7.8h, v30.8h+ uzp1 v4.8h, v26.8h, v23.8h+ smull2 v8.4s, v1.8h, v20.8h+ zip2 v10.8h, v7.8h, v30.8h+ smlal v18.4s, v31.4h, v24.4h+ mul v12.8h, v4.8h, v2.8h+ ldr q4, [x5, #0x10]+ ldr q20, [x4, #0x10]+ ldr q1, [x4], #0x20+ ldur q30, [x2, #-0x10]+ smlal2 v8.4s, v28.8h, v21.8h+ smlal2 v8.4s, v29.8h, v19.8h+ ldr q19, [x5], #0x20+ smlal2 v8.4s, v31.8h, v24.8h+ ld1 { v15.8h }, [x3], #16+ uzp2 v31.8h, v1.8h, v20.8h+ smlal v26.4s, v12.4h, v0.4h+ smlal2 v23.4s, v12.8h, v0.8h+ uzp1 v21.8h, v14.8h, v30.8h+ uzp1 v29.8h, v1.8h, v20.8h+ uzp1 v1.8h, v11.8h, v22.8h+ smlal2 v8.4s, v13.8h, v17.8h+ ld1 { v9.8h }, [x6], #16+ smlal v18.4s, v13.4h, v17.4h+ uzp1 v24.8h, v19.8h, v4.8h+ uzp2 v7.8h, v26.8h, v23.8h+ smull v26.4s, v1.4h, v21.4h+ smlal v18.4s, v16.4h, v6.4h+ uzp2 v19.8h, v19.8h, v4.8h+ smlal2 v8.4s, v16.8h, v6.8h+ uzp2 v28.8h, v11.8h, v22.8h+ smull2 v23.4s, v1.8h, v21.8h+ uzp1 v13.8h, v25.8h, v3.8h+ smlal2 v23.4s, v28.8h, v15.8h+ ldur q11, [x8, #-0x10]+ smlal2 v23.4s, v29.8h, v24.8h+ ld1 { v4.8h }, [x9], #16+ smlal2 v23.4s, v31.8h, v9.8h+ uzp1 v12.8h, v18.8h, v8.8h+ uzp2 v20.8h, v14.8h, v30.8h+ smlal v26.4s, v28.4h, v15.4h+ str q5, [x0], #0x20+ mul v14.8h, v12.8h, v2.8h+ stur q10, [x0, #-0x10]+ uzp2 v17.8h, v27.8h, v11.8h+ subs x13, x13, #0x1+ cbnz x13, Lmlk_polyvec_basemul_acc_montgomery_cached_k3_loop_start+ uzp2 v3.8h, v25.8h, v3.8h+ smull2 v16.4s, v1.8h, v20.8h+ smull v25.4s, v1.4h, v20.4h+ uzp1 v22.8h, v27.8h, v11.8h+ smlal2 v16.4s, v28.8h, v21.8h+ smlal v25.4s, v28.4h, v21.4h+ smlal2 v16.4s, v29.8h, v19.8h+ smlal v25.4s, v29.4h, v19.4h+ smlal2 v16.4s, v31.8h, v24.8h+ smlal v25.4s, v31.4h, v24.4h+ smlal v25.4s, v13.4h, v17.4h+ smlal2 v16.4s, v13.8h, v17.8h+ smlal2 v16.4s, v3.8h, v22.8h+ smlal v25.4s, v3.4h, v22.4h+ smlal2 v23.4s, v13.8h, v22.8h+ smlal v26.4s, v29.4h, v24.4h+ smlal v26.4s, v31.4h, v9.4h+ smlal v26.4s, v13.4h, v22.4h+ uzp1 v10.8h, v25.8h, v16.8h+ smlal2 v23.4s, v3.8h, v4.8h+ smlal v26.4s, v3.4h, v4.4h+ mul v13.8h, v10.8h, v2.8h+ smlal v18.4s, v14.4h, v0.4h+ smlal2 v8.4s, v14.8h, v0.8h+ uzp1 v3.8h, v26.8h, v23.8h+ mul v24.8h, v3.8h, v2.8h+ uzp2 v17.8h, v18.8h, v8.8h+ smlal v25.4s, v13.4h, v0.4h+ smlal2 v16.4s, v13.8h, v0.8h+ zip1 v21.8h, v7.8h, v17.8h+ zip2 v20.8h, v7.8h, v17.8h+ smlal2 v23.4s, v24.8h, v0.8h+ str q21, [x0], #0x20+ smlal v26.4s, v24.4h, v0.4h+ uzp2 v13.8h, v25.8h, v16.8h+ stur q20, [x0, #-0x10]+ uzp2 v23.8h, v26.8h, v23.8h+ zip1 v18.8h, v23.8h, v13.8h+ zip2 v13.8h, v23.8h, v13.8h+ str q18, [x0], #0x20+ stur q13, [x0, #-0x10]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 3) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/aarch64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S view
@@ -0,0 +1,371 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ */++/*yaml+ Name: polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm+ Description: Re-implementation of asymmetric base multiplication following @[NeonNTT] for k=4+ Signature: void mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm(int16_t r[256], const int16_t a[1024], const int16_t b[1024], const int16_t b_cache[512])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: write-only+ c_parameter: int16_t r[256]+ description: Output polynomial+ x1:+ type: buffer+ size_bytes: 2048+ permissions: read-only+ c_parameter: const int16_t a[1024]+ description: Input polynomial vector a+ x2:+ type: buffer+ size_bytes: 2048+ permissions: read-only+ c_parameter: const int16_t b[1024]+ description: Input polynomial vector b+ x3:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int16_t b_cache[512]+ description: Cached values for b+ Stack:+ bytes: 64+ description: saving callee-saved Neon registers+*/++/* Re-implementation of asymmetric base multiplication following @[NeonNTT] */++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm)+MLK_ASM_FN_SYMBOL(polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ mov w14, #0xd01 // =3329+ dup v0.8h, w14+ mov w14, #0xcff // =3327+ dup v2.8h, w14+ add x4, x1, #0x200+ add x5, x2, #0x200+ add x6, x3, #0x100+ add x7, x1, #0x400+ add x8, x2, #0x400+ add x9, x3, #0x200+ add x10, x1, #0x600+ add x11, x2, #0x600+ add x12, x3, #0x300+ mov x13, #0x10 // =16+ ldr q28, [x1], #0x20+ ldur q5, [x1, #-0x10]+ ldr q31, [x2], #0x20+ ldur q27, [x2, #-0x10]+ ldr q7, [x5], #0x20+ ldr q10, [x4], #0x20+ ldur q18, [x5, #-0x10]+ ldur q9, [x4, #-0x10]+ uzp1 v11.8h, v28.8h, v5.8h+ uzp2 v19.8h, v28.8h, v5.8h+ uzp2 v4.8h, v31.8h, v27.8h+ uzp1 v1.8h, v31.8h, v27.8h+ ldr q29, [x7], #0x20+ ldr q28, [x8, #0x10]+ uzp1 v24.8h, v10.8h, v9.8h+ uzp1 v17.8h, v7.8h, v18.8h+ uzp2 v7.8h, v7.8h, v18.8h+ ldr q21, [x8], #0x20+ uzp2 v27.8h, v10.8h, v9.8h+ ldur q6, [x7, #-0x10]+ smull v18.4s, v11.4h, v4.4h+ ld1 { v9.8h }, [x3], #16+ smull2 v8.4s, v11.8h, v4.8h+ ldr q16, [x11], #0x20+ smlal2 v8.4s, v19.8h, v1.8h+ ldur q14, [x11, #-0x10]+ smlal v18.4s, v19.4h, v1.4h+ uzp1 v10.8h, v21.8h, v28.8h+ smlal v18.4s, v24.4h, v7.4h+ ldr q4, [x10], #0x20+ smlal2 v8.4s, v24.8h, v7.8h+ ld1 { v12.8h }, [x6], #16+ smull2 v23.4s, v11.8h, v1.8h+ uzp2 v13.8h, v29.8h, v6.8h+ smull v26.4s, v11.4h, v1.4h+ uzp1 v29.8h, v29.8h, v6.8h+ smlal v26.4s, v19.4h, v9.4h+ ldur q15, [x10, #-0x10]+ smlal2 v23.4s, v19.8h, v9.8h+ uzp2 v9.8h, v21.8h, v28.8h+ smlal v18.4s, v27.4h, v17.4h+ uzp2 v6.8h, v16.8h, v14.8h+ uzp1 v21.8h, v16.8h, v14.8h+ smlal2 v8.4s, v27.8h, v17.8h+ smlal2 v8.4s, v29.8h, v9.8h+ uzp1 v30.8h, v4.8h, v15.8h+ uzp2 v16.8h, v4.8h, v15.8h+ smlal v18.4s, v29.4h, v9.4h+ smlal2 v8.4s, v13.8h, v10.8h+ ld1 { v15.8h }, [x9], #16+ smlal v18.4s, v13.4h, v10.4h+ ldr q11, [x4], #0x20+ smlal v18.4s, v30.4h, v6.4h+ ldr q7, [x2], #0x20+ smlal2 v8.4s, v30.8h, v6.8h+ ld1 { v9.8h }, [x12], #16+ smlal2 v23.4s, v24.8h, v17.8h+ ldur q4, [x2, #-0x10]+ smlal v26.4s, v24.4h, v17.4h+ ldur q25, [x4, #-0x10]+ smlal2 v8.4s, v16.8h, v21.8h+ ldr q5, [x5], #0x20+ smlal v18.4s, v16.4h, v21.4h+ ldur q22, [x5, #-0x10]+ smlal v26.4s, v27.4h, v12.4h+ ldr q19, [x1, #0x10]+ smlal v26.4s, v29.4h, v10.4h+ ld1 { v20.8h }, [x3], #16+ smlal v26.4s, v13.4h, v15.4h+ uzp1 v24.8h, v7.8h, v4.8h+ smlal2 v23.4s, v27.8h, v12.8h+ uzp1 v28.8h, v18.8h, v8.8h+ smlal v26.4s, v30.4h, v21.4h+ uzp2 v27.8h, v11.8h, v25.8h+ smlal2 v23.4s, v29.8h, v10.8h+ uzp2 v31.8h, v7.8h, v4.8h+ smlal2 v23.4s, v13.8h, v15.8h+ uzp1 v14.8h, v5.8h, v22.8h+ uzp1 v17.8h, v11.8h, v25.8h+ smlal v26.4s, v16.4h, v9.4h+ mul v29.8h, v28.8h, v2.8h+ sub x13, x13, #0x2++Lmlk_polyvec_basemul_acc_montgomery_cached_k4_loop_start:+ smlal2 v23.4s, v30.8h, v21.8h+ ldr q11, [x1], #0x20+ uzp2 v15.8h, v5.8h, v22.8h+ smlal v18.4s, v29.4h, v0.4h+ ldr q12, [x7], #0x20+ smlal2 v8.4s, v29.8h, v0.8h+ ldur q3, [x7, #-0x10]+ ldr q21, [x8], #0x20+ uzp1 v29.8h, v11.8h, v19.8h+ ldur q13, [x8, #-0x10]+ uzp2 v5.8h, v11.8h, v19.8h+ smlal2 v23.4s, v16.8h, v9.8h+ uzp2 v28.8h, v18.8h, v8.8h+ smull2 v8.4s, v29.8h, v31.8h+ smlal2 v8.4s, v5.8h, v24.8h+ uzp1 v7.8h, v12.8h, v3.8h+ smlal2 v8.4s, v17.8h, v15.8h+ uzp2 v11.8h, v21.8h, v13.8h+ uzp1 v4.8h, v26.8h, v23.8h+ smlal2 v8.4s, v27.8h, v14.8h+ smlal2 v8.4s, v7.8h, v11.8h+ mul v6.8h, v4.8h, v2.8h+ ldr q19, [x11], #0x20+ uzp2 v25.8h, v12.8h, v3.8h+ ldr q12, [x10], #0x20+ smull v18.4s, v29.4h, v31.4h+ ldur q3, [x10, #-0x10]+ smlal v18.4s, v5.4h, v24.4h+ uzp1 v4.8h, v21.8h, v13.8h+ smlal v18.4s, v17.4h, v15.4h+ ldur q13, [x11, #-0x10]+ ld1 { v1.8h }, [x6], #16+ smlal v26.4s, v6.4h, v0.4h+ smlal2 v23.4s, v6.8h, v0.8h+ ld1 { v10.8h }, [x9], #16+ smlal v18.4s, v27.4h, v14.4h+ uzp1 v30.8h, v12.8h, v3.8h+ smlal2 v8.4s, v25.8h, v4.8h+ uzp2 v31.8h, v19.8h, v13.8h+ smlal v18.4s, v7.4h, v11.4h+ ld1 { v9.8h }, [x12], #16+ smlal v18.4s, v25.4h, v4.4h+ uzp1 v21.8h, v19.8h, v13.8h+ uzp2 v16.8h, v12.8h, v3.8h+ smlal v18.4s, v30.4h, v31.4h+ smlal2 v8.4s, v30.8h, v31.8h+ uzp2 v31.8h, v26.8h, v23.8h+ smlal2 v8.4s, v16.8h, v21.8h+ smlal v18.4s, v16.4h, v21.4h+ zip1 v15.8h, v31.8h, v28.8h+ ldr q19, [x1, #0x10]+ smull2 v23.4s, v29.8h, v24.8h+ smull v26.4s, v29.4h, v24.4h+ ldr q3, [x2, #0x10]+ smlal v26.4s, v5.4h, v20.4h+ ldr q11, [x2], #0x20+ uzp1 v6.8h, v18.8h, v8.8h+ smlal v26.4s, v17.4h, v14.4h+ smlal v26.4s, v27.4h, v1.4h+ zip2 v13.8h, v31.8h, v28.8h+ smlal v26.4s, v7.4h, v4.4h+ str q15, [x0], #0x20+ smlal v26.4s, v25.4h, v10.4h+ stur q13, [x0, #-0x10]+ mul v29.8h, v6.8h, v2.8h+ uzp1 v24.8h, v11.8h, v3.8h+ uzp2 v31.8h, v11.8h, v3.8h+ ldr q11, [x4], #0x20+ smlal2 v23.4s, v5.8h, v20.8h+ ldur q28, [x4, #-0x10]+ smlal2 v23.4s, v17.8h, v14.8h+ ldr q5, [x5], #0x20+ smlal2 v23.4s, v27.8h, v1.8h+ ldur q22, [x5, #-0x10]+ smlal v26.4s, v30.4h, v21.4h+ ld1 { v20.8h }, [x3], #16+ smlal v26.4s, v16.4h, v9.4h+ uzp1 v17.8h, v11.8h, v28.8h+ smlal2 v23.4s, v7.8h, v4.8h+ uzp2 v27.8h, v11.8h, v28.8h+ smlal2 v23.4s, v25.8h, v10.8h+ uzp1 v14.8h, v5.8h, v22.8h+ subs x13, x13, #0x1+ cbnz x13, Lmlk_polyvec_basemul_acc_montgomery_cached_k4_loop_start+ smlal v18.4s, v29.4h, v0.4h+ ldr q11, [x1], #0x20+ uzp2 v28.8h, v5.8h, v22.8h+ smlal2 v23.4s, v30.8h, v21.8h+ smlal2 v8.4s, v29.8h, v0.8h+ ldr q15, [x8, #0x10]+ smlal2 v23.4s, v16.8h, v9.8h+ ldr q21, [x8], #0x20+ uzp1 v22.8h, v11.8h, v19.8h+ uzp2 v12.8h, v11.8h, v19.8h+ ldr q1, [x7, #0x10]+ ld1 { v6.8h }, [x6], #16+ uzp2 v3.8h, v18.8h, v8.8h+ smull v9.4s, v22.4h, v31.4h+ smull2 v18.4s, v22.8h, v31.8h+ ldr q16, [x7], #0x20+ smull v19.4s, v22.4h, v24.4h+ uzp1 v30.8h, v21.8h, v15.8h+ uzp2 v25.8h, v21.8h, v15.8h+ smull2 v8.4s, v22.8h, v24.8h+ smlal v19.4s, v12.4h, v20.4h+ ldr q13, [x10, #0x10]+ smlal2 v8.4s, v12.8h, v20.8h+ uzp1 v29.8h, v16.8h, v1.8h+ smlal2 v18.4s, v12.8h, v24.8h+ ldr q5, [x10], #0x20+ smlal v9.4s, v12.4h, v24.4h+ ldr q4, [x11], #0x20+ smlal v9.4s, v17.4h, v28.4h+ ldur q22, [x11, #-0x10]+ smlal2 v18.4s, v17.8h, v28.8h+ uzp2 v16.8h, v16.8h, v1.8h+ smlal v19.4s, v17.4h, v14.4h+ ld1 { v28.8h }, [x9], #16+ smlal2 v8.4s, v17.8h, v14.8h+ uzp1 v7.8h, v5.8h, v13.8h+ smlal v9.4s, v27.4h, v14.4h+ uzp1 v17.8h, v4.8h, v22.8h+ smlal2 v18.4s, v27.8h, v14.8h+ uzp2 v12.8h, v5.8h, v13.8h+ uzp2 v21.8h, v4.8h, v22.8h+ smlal v19.4s, v27.4h, v6.4h+ smlal2 v8.4s, v27.8h, v6.8h+ ld1 { v15.8h }, [x12], #16+ smlal v19.4s, v29.4h, v30.4h+ uzp1 v20.8h, v26.8h, v23.8h+ smlal v9.4s, v29.4h, v25.4h+ smlal2 v18.4s, v29.8h, v25.8h+ smlal2 v8.4s, v29.8h, v30.8h+ smlal v19.4s, v16.4h, v28.4h+ smlal2 v8.4s, v16.8h, v28.8h+ smlal2 v18.4s, v16.8h, v30.8h+ smlal v9.4s, v16.4h, v30.4h+ smlal v9.4s, v7.4h, v21.4h+ smlal2 v18.4s, v7.8h, v21.8h+ smlal2 v8.4s, v7.8h, v17.8h+ smlal v19.4s, v7.4h, v17.4h+ smlal v19.4s, v12.4h, v15.4h+ smlal2 v8.4s, v12.8h, v15.8h+ smlal2 v18.4s, v12.8h, v17.8h+ smlal v9.4s, v12.4h, v17.4h+ mul v6.8h, v20.8h, v2.8h+ uzp1 v4.8h, v19.8h, v8.8h+ mul v17.8h, v4.8h, v2.8h+ uzp1 v12.8h, v9.8h, v18.8h+ smlal v26.4s, v6.4h, v0.4h+ mul v21.8h, v12.8h, v2.8h+ smlal2 v23.4s, v6.8h, v0.8h+ smlal2 v8.4s, v17.8h, v0.8h+ smlal v19.4s, v17.4h, v0.4h+ smlal2 v18.4s, v21.8h, v0.8h+ uzp2 v23.8h, v26.8h, v23.8h+ smlal v9.4s, v21.4h, v0.4h+ zip2 v12.8h, v23.8h, v3.8h+ zip1 v22.8h, v23.8h, v3.8h+ uzp2 v14.8h, v19.8h, v8.8h+ uzp2 v18.8h, v9.8h, v18.8h+ str q12, [x0, #0x10]+ str q22, [x0], #0x20+ zip2 v24.8h, v14.8h, v18.8h+ zip1 v21.8h, v14.8h, v18.8h+ str q24, [x0, #0x10]+ str q21, [x0], #0x20+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/aarch64/src/mlkem_rej_uniform_aarch64_asm.S view
@@ -0,0 +1,226 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: rej_uniform_aarch64_asm+ Description: Run rejection sampling on uniform random bytes to generate uniform random integers mod q+ Signature: uint64_t mlk_rej_uniform_aarch64_asm(int16_t r[256], const uint8_t *buf, unsigned buflen, const uint8_t table[4096])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: write-only+ c_parameter: int16_t r[256]+ description: Output buffer+ x1:+ type: buffer+ size_bytes: x2+ permissions: read-only+ c_parameter: const uint8_t *buf+ description: Input buffer+ x2:+ type: scalar+ c_parameter: unsigned buflen+ description: Length of input buffer (must be multiple of 24)+ test_with: 504 # MLKEM_GEN_MATRIX_NBLOCKS * MLK_XOF_RATE+ x3:+ type: buffer+ size_bytes: 4096+ permissions: read-only+ c_parameter: const uint8_t table[4096]+ description: Lookup table+ Stack:+ bytes: 576+ description: register preservation and temporary storage+*/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_rej_uniform_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(rej_uniform_aarch64_asm)+MLK_ASM_FN_SYMBOL(rej_uniform_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x240+ .cfi_adjust_cfa_offset 0x240+ mov x7, #0x1 // =1+ movk x7, #0x2, lsl #16+ movk x7, #0x4, lsl #32+ movk x7, #0x8, lsl #48+ mov v31.d[0], x7+ mov x7, #0x10 // =16+ movk x7, #0x20, lsl #16+ movk x7, #0x40, lsl #32+ movk x7, #0x80, lsl #48+ mov v31.d[1], x7+ mov w11, #0xd01 // =3329+ dup v30.8h, w11+ mov x8, sp+ mov x7, x8+ mov x11, #0x0 // =0+ eor v16.16b, v16.16b, v16.16b++Lmlk_rej_uniform_initial_zero:+ str q16, [x7], #0x40+ stur q16, [x7, #-0x30]+ stur q16, [x7, #-0x20]+ stur q16, [x7, #-0x10]+ add x11, x11, #0x20+ cmp x11, #0x100+ b.lt Lmlk_rej_uniform_initial_zero+ mov x7, x8+ mov x9, #0x0 // =0+ mov x4, #0x100 // =256+ cmp x2, #0x30+ b.lo Lmlk_rej_uniform_loop48_end++Lmlk_rej_uniform_loop48:+ cmp x9, x4+ b.hs Lmlk_rej_uniform_memory_copy+ sub x2, x2, #0x30+ ld3 { v0.16b, v1.16b, v2.16b }, [x1], #48+ zip1 v4.16b, v0.16b, v1.16b+ zip2 v5.16b, v0.16b, v1.16b+ zip1 v6.16b, v1.16b, v2.16b+ zip2 v7.16b, v1.16b, v2.16b+ bic v4.8h, #0xf0, lsl #8+ bic v5.8h, #0xf0, lsl #8+ ushr v6.8h, v6.8h, #0x4+ ushr v7.8h, v7.8h, #0x4+ zip1 v16.8h, v4.8h, v6.8h+ zip2 v17.8h, v4.8h, v6.8h+ zip1 v18.8h, v5.8h, v7.8h+ zip2 v19.8h, v5.8h, v7.8h+ cmhi v4.8h, v30.8h, v16.8h+ cmhi v5.8h, v30.8h, v17.8h+ cmhi v6.8h, v30.8h, v18.8h+ cmhi v7.8h, v30.8h, v19.8h+ and v4.16b, v4.16b, v31.16b+ and v5.16b, v5.16b, v31.16b+ and v6.16b, v6.16b, v31.16b+ and v7.16b, v7.16b, v31.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ uaddlv s22, v6.8h+ uaddlv s23, v7.8h+ fmov w12, s20+ fmov w13, s21+ fmov w14, s22+ fmov w15, s23+ ldr q24, [x3, x12, lsl #4]+ ldr q25, [x3, x13, lsl #4]+ ldr q26, [x3, x14, lsl #4]+ ldr q27, [x3, x15, lsl #4]+ cnt v4.16b, v4.16b+ cnt v5.16b, v5.16b+ cnt v6.16b, v6.16b+ cnt v7.16b, v7.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ uaddlv s22, v6.8h+ uaddlv s23, v7.8h+ fmov w12, s20+ fmov w13, s21+ fmov w14, s22+ fmov w15, s23+ tbl v16.16b, { v16.16b }, v24.16b+ tbl v17.16b, { v17.16b }, v25.16b+ tbl v18.16b, { v18.16b }, v26.16b+ tbl v19.16b, { v19.16b }, v27.16b+ st1 { v16.8h }, [x7]+ add x7, x7, x12, lsl #1+ st1 { v17.8h }, [x7]+ add x7, x7, x13, lsl #1+ st1 { v18.8h }, [x7]+ add x7, x7, x14, lsl #1+ st1 { v19.8h }, [x7]+ add x7, x7, x15, lsl #1+ add x12, x12, x13+ add x14, x14, x15+ add x9, x9, x12+ add x9, x9, x14+ cmp x2, #0x30+ b.hs Lmlk_rej_uniform_loop48++Lmlk_rej_uniform_loop48_end:+ cmp x9, x4+ b.hs Lmlk_rej_uniform_memory_copy+ cmp x2, #0x18+ b.lo Lmlk_rej_uniform_memory_copy+ ld3 { v0.8b, v1.8b, v2.8b }, [x1], #24+ zip1 v4.16b, v0.16b, v1.16b+ zip1 v5.16b, v1.16b, v2.16b+ bic v4.8h, #0xf0, lsl #8+ ushr v5.8h, v5.8h, #0x4+ zip1 v16.8h, v4.8h, v5.8h+ zip2 v17.8h, v4.8h, v5.8h+ cmhi v4.8h, v30.8h, v16.8h+ cmhi v5.8h, v30.8h, v17.8h+ and v4.16b, v4.16b, v31.16b+ and v5.16b, v5.16b, v31.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ fmov w12, s20+ fmov w13, s21+ ldr q24, [x3, x12, lsl #4]+ ldr q25, [x3, x13, lsl #4]+ cnt v4.16b, v4.16b+ cnt v5.16b, v5.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ fmov w12, s20+ fmov w13, s21+ tbl v16.16b, { v16.16b }, v24.16b+ tbl v17.16b, { v17.16b }, v25.16b+ st1 { v16.8h }, [x7]+ add x7, x7, x12, lsl #1+ st1 { v17.8h }, [x7]+ add x9, x9, x12+ add x9, x9, x13++Lmlk_rej_uniform_memory_copy:+ cmp x9, x4+ csel x9, x9, x4, lo+ mov x11, #0x0 // =0+ mov x7, x8++Lmlk_rej_uniform_final_copy:+ ldr q16, [x7], #0x40+ ldur q17, [x7, #-0x30]+ ldur q18, [x7, #-0x20]+ ldur q19, [x7, #-0x10]+ str q16, [x0], #0x40+ stur q17, [x0, #-0x30]+ stur q18, [x0, #-0x20]+ stur q19, [x0, #-0x10]+ add x11, x11, #0x20+ cmp x11, #0x100+ b.lt Lmlk_rej_uniform_final_copy+ mov x0, x9++Lmlk_rej_uniform_return:+ add sp, sp, #0x240+ .cfi_adjust_cfa_offset -0x240+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(rej_uniform_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/aarch64/src/rej_uniform_table.c view
@@ -0,0 +1,543 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLK_ARITH_BACKEND_AARCH64) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#include "arith_native_aarch64.h"++/*+ * Lookup table used by rejection sampling of the public matrix.+ * See autogen for details.+ */+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_rej_uniform_table[4096] = {+ 255, 255, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 0 */,+ 0, 1, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 1 */,+ 2, 3, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 2 */,+ 0, 1, 2, 3, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 3 */,+ 4, 5, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 4 */,+ 0, 1, 4, 5, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 5 */,+ 2, 3, 4, 5, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 6 */,+ 0, 1, 2, 3, 4, 5, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 7 */,+ 6, 7, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 8 */,+ 0, 1, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 9 */,+ 2, 3, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 10 */,+ 0, 1, 2, 3, 6, 7, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 11 */,+ 4, 5, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 12 */,+ 0, 1, 4, 5, 6, 7, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 13 */,+ 2, 3, 4, 5, 6, 7, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 14 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 15 */,+ 8, 9, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 16 */,+ 0, 1, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 17 */,+ 2, 3, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 18 */,+ 0, 1, 2, 3, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 19 */,+ 4, 5, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 20 */,+ 0, 1, 4, 5, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 21 */,+ 2, 3, 4, 5, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 22 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 23 */,+ 6, 7, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 24 */,+ 0, 1, 6, 7, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 25 */,+ 2, 3, 6, 7, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 26 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 27 */,+ 4, 5, 6, 7, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 28 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 29 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 30 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 255, 255, 255, 255, 255, 255 /* 31 */,+ 10, 11, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 32 */,+ 0, 1, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 33 */,+ 2, 3, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 34 */,+ 0, 1, 2, 3, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 35 */,+ 4, 5, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 36 */,+ 0, 1, 4, 5, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 37 */,+ 2, 3, 4, 5, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 38 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 39 */,+ 6, 7, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 40 */,+ 0, 1, 6, 7, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 41 */,+ 2, 3, 6, 7, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 42 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 43 */,+ 4, 5, 6, 7, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 44 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 45 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 46 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 47 */,+ 8, 9, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 48 */,+ 0, 1, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 49 */,+ 2, 3, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 50 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 51 */,+ 4, 5, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 52 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 53 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 54 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 55 */,+ 6, 7, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 56 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 57 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 58 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 59 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 60 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 61 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 62 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 255, 255, 255, 255 /* 63 */,+ 12, 13, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 64 */,+ 0, 1, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 65 */,+ 2, 3, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 66 */,+ 0, 1, 2, 3, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 67 */,+ 4, 5, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 68 */,+ 0, 1, 4, 5, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 69 */,+ 2, 3, 4, 5, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 70 */,+ 0, 1, 2, 3, 4, 5, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 71 */,+ 6, 7, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 72 */,+ 0, 1, 6, 7, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 73 */,+ 2, 3, 6, 7, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 74 */,+ 0, 1, 2, 3, 6, 7, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 75 */,+ 4, 5, 6, 7, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 76 */,+ 0, 1, 4, 5, 6, 7, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 77 */,+ 2, 3, 4, 5, 6, 7, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 78 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 79 */,+ 8, 9, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 80 */,+ 0, 1, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 81 */,+ 2, 3, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 82 */,+ 0, 1, 2, 3, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 83 */,+ 4, 5, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 84 */,+ 0, 1, 4, 5, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 85 */,+ 2, 3, 4, 5, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 86 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 87 */,+ 6, 7, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 88 */,+ 0, 1, 6, 7, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 89 */,+ 2, 3, 6, 7, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 90 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 91 */,+ 4, 5, 6, 7, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 92 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 93 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 94 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 12, 13, 255, 255, 255, 255 /* 95 */,+ 10, 11, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 96 */,+ 0, 1, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 97 */,+ 2, 3, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 98 */,+ 0, 1, 2, 3, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 99 */,+ 4, 5, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 100 */,+ 0, 1, 4, 5, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 101 */,+ 2, 3, 4, 5, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 102 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 103 */,+ 6, 7, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 104 */,+ 0, 1, 6, 7, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 105 */,+ 2, 3, 6, 7, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 106 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 107 */,+ 4, 5, 6, 7, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 108 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 109 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 110 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 111 */,+ 8, 9, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 112 */,+ 0, 1, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 113 */,+ 2, 3, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 114 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 115 */,+ 4, 5, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 116 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 117 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 118 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 119 */,+ 6, 7, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 120 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 121 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 122 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 123 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 124 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 125 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 126 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 12, 13, 255, 255 /* 127 */,+ 14, 15, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 128 */,+ 0, 1, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 129 */,+ 2, 3, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 130 */,+ 0, 1, 2, 3, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 131 */,+ 4, 5, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 132 */,+ 0, 1, 4, 5, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 133 */,+ 2, 3, 4, 5, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 134 */,+ 0, 1, 2, 3, 4, 5, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 135 */,+ 6, 7, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 136 */,+ 0, 1, 6, 7, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 137 */,+ 2, 3, 6, 7, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 138 */,+ 0, 1, 2, 3, 6, 7, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 139 */,+ 4, 5, 6, 7, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 140 */,+ 0, 1, 4, 5, 6, 7, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 141 */,+ 2, 3, 4, 5, 6, 7, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 142 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 143 */,+ 8, 9, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 144 */,+ 0, 1, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 145 */,+ 2, 3, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 146 */,+ 0, 1, 2, 3, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 147 */,+ 4, 5, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 148 */,+ 0, 1, 4, 5, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 149 */,+ 2, 3, 4, 5, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 150 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 151 */,+ 6, 7, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 152 */,+ 0, 1, 6, 7, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 153 */,+ 2, 3, 6, 7, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 154 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 155 */,+ 4, 5, 6, 7, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 156 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 157 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 158 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 14, 15, 255, 255, 255, 255 /* 159 */,+ 10, 11, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 160 */,+ 0, 1, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 161 */,+ 2, 3, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 162 */,+ 0, 1, 2, 3, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 163 */,+ 4, 5, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 164 */,+ 0, 1, 4, 5, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 165 */,+ 2, 3, 4, 5, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 166 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 167 */,+ 6, 7, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 168 */,+ 0, 1, 6, 7, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 169 */,+ 2, 3, 6, 7, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 170 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 171 */,+ 4, 5, 6, 7, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 172 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 173 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 174 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 175 */,+ 8, 9, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 176 */,+ 0, 1, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 177 */,+ 2, 3, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 178 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 179 */,+ 4, 5, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 180 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 181 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 182 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 183 */,+ 6, 7, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 184 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 185 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 186 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 187 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 188 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 189 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 190 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 14, 15, 255, 255 /* 191 */,+ 12, 13, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 192 */,+ 0, 1, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 193 */,+ 2, 3, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 194 */,+ 0, 1, 2, 3, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 195 */,+ 4, 5, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 196 */,+ 0, 1, 4, 5, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 197 */,+ 2, 3, 4, 5, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 198 */,+ 0, 1, 2, 3, 4, 5, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 199 */,+ 6, 7, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 200 */,+ 0, 1, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 201 */,+ 2, 3, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 202 */,+ 0, 1, 2, 3, 6, 7, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 203 */,+ 4, 5, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 204 */,+ 0, 1, 4, 5, 6, 7, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 205 */,+ 2, 3, 4, 5, 6, 7, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 206 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 207 */,+ 8, 9, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 208 */,+ 0, 1, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 209 */,+ 2, 3, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 210 */,+ 0, 1, 2, 3, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 211 */,+ 4, 5, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 212 */,+ 0, 1, 4, 5, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 213 */,+ 2, 3, 4, 5, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 214 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 215 */,+ 6, 7, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 216 */,+ 0, 1, 6, 7, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 217 */,+ 2, 3, 6, 7, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 218 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 219 */,+ 4, 5, 6, 7, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 220 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 221 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 222 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 12, 13, 14, 15, 255, 255 /* 223 */,+ 10, 11, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 224 */,+ 0, 1, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 225 */,+ 2, 3, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 226 */,+ 0, 1, 2, 3, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 227 */,+ 4, 5, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 228 */,+ 0, 1, 4, 5, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 229 */,+ 2, 3, 4, 5, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 230 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 231 */,+ 6, 7, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 232 */,+ 0, 1, 6, 7, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 233 */,+ 2, 3, 6, 7, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 234 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 235 */,+ 4, 5, 6, 7, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 236 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 237 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 238 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 239 */,+ 8, 9, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 240 */,+ 0, 1, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 241 */,+ 2, 3, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 242 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 243 */,+ 4, 5, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 244 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 245 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 246 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 247 */,+ 6, 7, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 248 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 249 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 250 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 251 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 252 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 253 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 254 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 12, 13, 14, 15 /* 255 */,+};++#else /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(aarch64_rej_uniform_table)++#endif /* !(MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mlkem/src/native/api.h view
@@ -0,0 +1,651 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#ifndef MLK_NATIVE_API_H+#define MLK_NATIVE_API_H+/*+ * Native arithmetic interface+ *+ * This header is primarily for documentation purposes.+ * It should not be included by backend implementations.+ *+ * To ensure consistency with backends, the header will be+ * included automatically after inclusion of the active+ * backend, to ensure consistency of function signatures,+ * and run sanity checks.+ */++#include "../cbmc.h"+#include "../common.h"++/* Backends must return MLK_NATIVE_FUNC_SUCCESS upon success. */+#define MLK_NATIVE_FUNC_SUCCESS (0)+/* Backends may return MLK_NATIVE_FUNC_FALLBACK to signal to the frontend that+ * the target/parameters are unsupported; typically, this would be because of+ * dependencies on CPU features not detected on the host CPU. In this case,+ * the frontend falls back to the default C implementation. */+#define MLK_NATIVE_FUNC_FALLBACK (-1)+++/* Absolute exclusive upper bound for the output of the inverse NTT+ *+ * NOTE: This is the same bound as in poly.h and has to be kept+ * in sync. */+#define MLK_INVNTT_BOUND (8 * MLKEM_Q)++/* Absolute exclusive upper bound for the output of the forward NTT+ *+ * NOTE: This is the same bound as in poly.h and has to be kept+ * in sync. */+#define MLK_NTT_BOUND (8 * MLKEM_Q)++/*+ * This is the C<->native interface allowing for the drop-in of+ * native code for performance critical arithmetic components of ML-KEM.+ *+ * A _backend_ is a specific implementation of (part of) this interface.+ *+ * To add a function to a backend, define MLK_USE_NATIVE_XXX and+ * implement `static inline xxx(...)` in the profile header.+ *+ * The only exception is MLK_USE_NATIVE_NTT_CUSTOM_ORDER. This option can+ * be set if there are native implementations for all of NTT, invNTT, and+ * base multiplication, and allows the native implementation to use a+ * custom order of polynomial coefficients in NTT domain -- the use of such+ * custom order is not an implementation-detail since the public matrix+ * is generated in NTT domain. In this case, a permutation function+ * mlk_poly_permute_bitrev_to_custom() needs to be provided that permutes+ * polynomials in NTT domain from bitreversed to the custom order.+ */++/*+ * Those functions are meant to be trivial wrappers around the chosen native+ * implementation. The are static inline to avoid unnecessary calls.+ * The macro before each declaration controls whether a native+ * implementation is present.+ */++#if defined(MLK_USE_NATIVE_NTT)+/**+ * Compute the negacyclic number-theoretic transform (NTT) of a polynomial+ * in place.+ *+ * The input polynomial is assumed to be in normal order. The output+ * polynomial is in bitreversed order, or of a custom order if+ * MLK_USE_NATIVE_NTT_CUSTOM_ORDER is set. See the documentation of+ * MLK_USE_NATIVE_NTT_CUSTOM_ORDER for more information.+ *+ * @param[in,out] p Input/output polynomial.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_ntt_native(int16_t p[MLKEM_N])+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ requires(array_abs_bound(p, 0, MLKEM_N, MLKEM_Q))+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(p, 0, MLKEM_N, MLK_NTT_BOUND))+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_abs_bound(p, 0, MLKEM_N, MLKEM_Q))+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged(p, MLKEM_N))+);+#endif /* MLK_USE_NATIVE_NTT */++#if defined(MLK_USE_NATIVE_NTT_CUSTOM_ORDER)+/*+ * This must only be set if NTT, invNTT, basemul, mulcache, and+ * to/from byte stream conversions all have native implementations+ * that are adapted to the custom order.+ */+#if !defined(MLK_USE_NATIVE_NTT) || !defined(MLK_USE_NATIVE_INTT) || \+ !defined(MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE) || \+ !defined(MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED) || \+ !defined(MLK_USE_NATIVE_POLY_TOBYTES) || \+ !defined(MLK_USE_NATIVE_POLY_FROMBYTES)+#error \+ "Invalid native profile: MLK_USE_NATIVE_NTT_CUSTOM_ORDER can only be \+set if there are native implementations for NTT, invNTT, mulcache, basemul, \+and to/from bytes conversions."+#endif /* !MLK_USE_NATIVE_NTT || !MLK_USE_NATIVE_INTT || \+ !MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE || \+ !MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED || \+ !MLK_USE_NATIVE_POLY_TOBYTES || !MLK_USE_NATIVE_POLY_FROMBYTES */++/**+ * When MLK_USE_NATIVE_NTT_CUSTOM_ORDER is defined, convert a polynomial in+ * NTT domain from bitreversed order to the custom order output by the+ * native NTT.+ *+ * This must only be defined if there is native code for all of (a) NTT,+ * (b) invNTT, (c) basemul, (d) mulcache.+ *+ * @param[in,out] p Input/output polynomial.+ */+static MLK_INLINE void mlk_poly_permute_bitrev_to_custom(int16_t p[MLKEM_N])+__contract__(+ /* We don't specify that this should be a permutation, but only+ * that it does not change the bound established at the end of mlk_gen_matrix. */+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(p, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(p, 0, MLKEM_N, 0, MLKEM_Q)));+#endif /* MLK_USE_NATIVE_NTT_CUSTOM_ORDER */++#if defined(MLK_USE_NATIVE_INTT) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API))+/**+ * Compute the inverse negacyclic number-theoretic transform (NTT) of a+ * polynomial in place.+ *+ * The input polynomial is in bitreversed order, or of a custom order if+ * MLK_USE_NATIVE_NTT_CUSTOM_ORDER is set. See the documentation of+ * MLK_USE_NATIVE_NTT_CUSTOM_ORDER for more information. The output+ * polynomial is assumed to be in normal order.+ *+ * @param[in,out] p Input/output polynomial.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_intt_native(int16_t p[MLKEM_N])+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(p, 0, MLKEM_N, MLK_INVNTT_BOUND))+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged(p, MLKEM_N))+);+#endif /* MLK_USE_NATIVE_INTT && (!MLK_CONFIG_NO_ENCAPS_API || \+ !MLK_CONFIG_NO_DECAPS_API) */++#if defined(MLK_USE_NATIVE_POLY_REDUCE)+/**+ * Apply modular reduction to all coefficients of a polynomial, mapping them+ * to unsigned canonical representatives in [0,..,MLKEM_Q-1].+ *+ * @param[in,out] p Input/output polynomial.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_reduce_native(int16_t p[MLKEM_N])+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_bound(p, 0, MLKEM_N, 0, MLKEM_Q))+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged(p, MLKEM_N))+);+#endif /* MLK_USE_NATIVE_POLY_REDUCE */++#if defined(MLK_USE_NATIVE_POLY_TOMONT) && !defined(MLK_CONFIG_NO_KEYPAIR_API)+/**+ * In-place conversion of all coefficients of a polynomial from the normal+ * domain to the Montgomery domain.+ *+ * @param[in,out] p Input/output polynomial.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_tomont_native(int16_t p[MLKEM_N])+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(p, 0, MLKEM_N, MLKEM_Q))+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged(p, MLKEM_N))+);+#endif /* MLK_USE_NATIVE_POLY_TOMONT && !MLK_CONFIG_NO_KEYPAIR_API */++#if defined(MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE)+/**+ * Compute multiplication cache for a polynomial in NTT domain.+ *+ * The purpose of the multiplication cache is to cache repeated computations+ * required during a base multiplication of polynomials in NTT domain. The+ * structure of the multiplication-cache is implementation defined.+ *+ * @param[out] cache Multiplication cache.+ * @param[in] mlk_poly Input polynomial. Must be in NTT domain and in+ * bitreversed order, or of a custom order if+ * MLK_USE_NATIVE_NTT_CUSTOM_ORDER is set. See the+ * documentation of MLK_USE_NATIVE_NTT_CUSTOM_ORDER for+ * more information.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_mulcache_compute_native(+ int16_t cache[MLKEM_N / 2], const int16_t mlk_poly[MLKEM_N])+__contract__(+ requires(memory_no_alias(cache, sizeof(int16_t) * (MLKEM_N / 2)))+ requires(memory_no_alias(mlk_poly, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(cache, sizeof(int16_t) * (MLKEM_N / 2)))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(cache, 0, MLKEM_N/2, MLKEM_Q))+);+#endif /* MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE */++#if defined(MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED)+#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2+/**+ * Compute scalar product of length-2 polynomial vectors in NTT domain.+ *+ * @param[out] r Result of the scalar product. Again in NTT domain, of+ * the same ordering as @p a and @p b.+ * @param[in] a First polynomial vector operand. Must be in NTT+ * domain and in bitreversed order, or of a custom order+ * if MLK_USE_NATIVE_NTT_CUSTOM_ORDER is set. See the+ * documentation of MLK_USE_NATIVE_NTT_CUSTOM_ORDER for+ * more information.+ * @param[in] b Second polynomial vector operand. As for @p a.+ * @param[in] b_cache Multiplication-cache for @p b.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k2_native(+ int16_t r[MLKEM_N], const int16_t a[2 * MLKEM_N],+ const int16_t b[2 * MLKEM_N], const int16_t b_cache[2 * (MLKEM_N / 2)])+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 2 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 2 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 2 * (MLKEM_N / 2)))+ requires(array_bound(a, 0, 2 * MLKEM_N, 0, MLKEM_UINT12_LIMIT))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+);+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 3+/**+ * Compute scalar product of length-3 polynomial vectors in NTT domain.+ *+ * @param[out] r Result of the scalar product. Again in NTT domain, of+ * the same ordering as @p a and @p b.+ * @param[in] a First polynomial vector operand. Must be in NTT+ * domain and in bitreversed order, or of a custom order+ * if MLK_USE_NATIVE_NTT_CUSTOM_ORDER is set. See the+ * documentation of MLK_USE_NATIVE_NTT_CUSTOM_ORDER for+ * more information.+ * @param[in] b Second polynomial vector operand. As for @p a.+ * @param[in] b_cache Multiplication-cache for @p b.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k3_native(+ int16_t r[MLKEM_N], const int16_t a[3 * MLKEM_N],+ const int16_t b[3 * MLKEM_N], const int16_t b_cache[3 * (MLKEM_N / 2)])+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 3 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 3 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 3 * (MLKEM_N / 2)))+ requires(array_bound(a, 0, 3 * MLKEM_N, 0, MLKEM_UINT12_LIMIT))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+);+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 3 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4+/**+ * Compute scalar product of length-4 polynomial vectors in NTT domain.+ *+ * @param[out] r Result of the scalar product. Again in NTT domain, of+ * the same ordering as @p a and @p b.+ * @param[in] a First polynomial vector operand. Must be in NTT+ * domain and in bitreversed order, or of a custom order+ * if MLK_USE_NATIVE_NTT_CUSTOM_ORDER is set. See the+ * documentation of MLK_USE_NATIVE_NTT_CUSTOM_ORDER for+ * more information.+ * @param[in] b Second polynomial vector operand. As for @p a.+ * @param[in] b_cache Multiplication-cache for @p b.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k4_native(+ int16_t r[MLKEM_N], const int16_t a[4 * MLKEM_N],+ const int16_t b[4 * MLKEM_N], const int16_t b_cache[4 * (MLKEM_N / 2)])+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 4 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 4 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 4 * (MLKEM_N / 2)))+ requires(array_bound(a, 0, 4 * MLKEM_N, 0, MLKEM_UINT12_LIMIT))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+);+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */+#endif /* MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED */++#if defined(MLK_USE_NATIVE_POLY_TOBYTES) && \+ (!defined(MLK_CONFIG_NO_KEYPAIR_API) || \+ !defined(MLK_CONFIG_NO_ENCAPS_API))+/**+ * Serialization of a polynomial with unsigned canonical coefficients.+ *+ * @spec{Implements ByteEncode_12 from @[FIPS203, Algorithm 5].}+ *+ * @param[out] r Output byte array (of MLKEM_POLYBYTES bytes).+ * @param[in] a Input polynomial, with each coefficient in [0,..,MLKEM_Q-1].+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_tobytes_native(uint8_t r[MLKEM_POLYBYTES],+ const int16_t a[MLKEM_N])+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYBYTES))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYBYTES))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)+);+#endif /* MLK_USE_NATIVE_POLY_TOBYTES && (!MLK_CONFIG_NO_KEYPAIR_API || \+ !MLK_CONFIG_NO_ENCAPS_API) */++#if defined(MLK_USE_NATIVE_POLY_FROMBYTES) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API))+/**+ * Deserialization of a polynomial.+ *+ * @spec{Implements ByteDecode_12 from @[FIPS203, Algorithm 6].}+ *+ * @param[out] a Output polynomial in NTT domain.+ * @param[in] r Input byte array (of MLKEM_POLYBYTES bytes).+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_frombytes_native(+ int16_t a[MLKEM_N], const uint8_t r[MLKEM_POLYBYTES])+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYBYTES))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(a, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_bound(a, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT))+);+#endif /* MLK_USE_NATIVE_POLY_FROMBYTES && (!MLK_CONFIG_NO_ENCAPS_API || \+ !MLK_CONFIG_NO_DECAPS_API) */++#if defined(MLK_USE_NATIVE_REJ_UNIFORM)+/**+ * Run rejection sampling on uniformly random bytes to generate uniformly random+ * integers mod MLKEM_Q, represented in [0,..,MLKEM_Q-1].+ *+ * @param[out] r Output buffer.+ * @param len Requested number of 16-bit integers (uniform mod MLKEM_Q).+ * @param[in] buf Input buffer (assumed to be uniform random bytes).+ * @param buflen Length of input buffer in bytes.+ *+ * @retval MLK_NATIVE_FUNC_FALLBACK Native implementation does not support+ * the input lengths.+ * @retval other Non-negative number of sampled 16-bit+ * integers (at most @p len).+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_rej_uniform_native(int16_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+__contract__(+ requires(len <= 4096 && buflen <= 4096 && buflen % 3 == 0)+ requires(memory_no_alias(r, sizeof(int16_t) * len))+ requires(memory_no_alias(buf, buflen))+ assigns(memory_slice(r, sizeof(int16_t) * len))+ ensures(return_value != MLK_NATIVE_FUNC_FALLBACK+ ==> (0 <= return_value && return_value <= len))+ ensures(return_value != MLK_NATIVE_FUNC_FALLBACK+ ==> array_bound(r, 0, (unsigned) return_value, 0, MLKEM_Q))+);+#endif /* MLK_USE_NATIVE_REJ_UNIFORM */++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || (MLKEM_K == 2 || MLKEM_K == 3)+#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D4)+/**+ * Compression (4 bits) and subsequent serialization of a polynomial.+ *+ * @spec{Compress_4 from @[FIPS203, Eq (4.7)].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D4+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d4_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D4], const int16_t a[MLKEM_N])+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK));+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D4 */++#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D10)+/**+ * Compression (10 bits) and subsequent serialization of a polynomial.+ *+ * @spec{Compress_10 from @[FIPS203, Eq (4.7)].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D10+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d10_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D10], const int16_t a[MLKEM_N])+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK));+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D10 */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D4)+/**+ * De-serialization and subsequent decompression (4 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d4.+ *+ * @spec{Decompress_4 from @[FIPS203, Eq (4.8)].}+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D4+ * bytes).+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d4_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D4])+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D4))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_bound(r, 0, MLKEM_N, 0, MLKEM_Q)));+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D4 */++#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D10)+/**+ * De-serialization and subsequent decompression (10 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d10.+ *+ * @spec{Decompress_10 from @[FIPS203, Eq (4.8)].}+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D10+ * bytes).+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d10_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D10])+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D10))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_bound(r, 0, MLKEM_N, 0, MLKEM_Q)));+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D10 */+#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4+#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D5)+/**+ * Compression (5 bits) and subsequent serialization of a polynomial.+ *+ * @spec{Compress_5 from @[FIPS203, Eq (4.7)].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D5+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d5_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D5], const int16_t a[MLKEM_N])+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK));+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D5 */++#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D11)+/**+ * Compression (11 bits) and subsequent serialization of a polynomial.+ *+ * @spec{Compress_11 from @[FIPS203, Eq (4.7)].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D11+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d11_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D11], const int16_t a[MLKEM_N])+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK));+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D11 */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D5)+/**+ * De-serialization and subsequent decompression (5 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d5.+ *+ * @spec{Decompress_5 from @[FIPS203, Eq (4.8)].}+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D5+ * bytes).+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d5_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D5])+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D5))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_bound(r, 0, MLKEM_N, 0, MLKEM_Q)));+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D5 */++#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D11)+/**+ * De-serialization and subsequent decompression (11 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d11.+ *+ * @spec{Decompress_11 from @[FIPS203, Eq (4.8)].}+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D11+ * bytes).+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d11_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D11])+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D11))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_bound(r, 0, MLKEM_N, 0, MLKEM_Q)));+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D11 */+#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#endif /* !MLK_NATIVE_API_H */
+ cbits/mlkem/src/native/meta.h view
@@ -0,0 +1,30 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_NATIVE_META_H+#define MLK_NATIVE_META_H++/*+ * Default arithmetic backend+ */+#include "../sys.h"++#ifdef MLK_SYS_AARCH64_NEON+#include "aarch64/meta.h"+#endif++/* The x86_64 backend requires toolchain support for the SysV ABI */+#if defined(MLK_SYS_X86_64_AVX2) && defined(MLK_SYSV_ABI_SUPPORTED)+#include "x86_64/meta.h"+#endif++#if defined(MLK_SYS_RISCV64_RVV)+#include "riscv64/meta.h"+#endif++#ifdef MLK_SYS_PPC64LE+#include "ppc64le/meta.h"+#endif++#endif /* !MLK_NATIVE_META_H */
+ cbits/mlkem/src/native/x86_64/README.md view
@@ -0,0 +1,4 @@+[//]: # (SPDX-License-Identifier: CC-BY-4.0)++This directory contains the native x86_64 arithmetic backend for ML-KEM provided by the official [AVX2+implementation](https://github.com/pq-crystals/kyber/tree/main/avx2) of the Kyber team.
+ cbits/mlkem/src/native/x86_64/meta.h view
@@ -0,0 +1,324 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_NATIVE_X86_64_META_H+#define MLK_NATIVE_X86_64_META_H++/* Identifier for this backend so that source and assembly files+ * in the build can be appropriately guarded. */+#define MLK_ARITH_BACKEND_X86_64_DEFAULT++#define MLK_USE_NATIVE_NTT_CUSTOM_ORDER+#define MLK_USE_NATIVE_REJ_UNIFORM+#define MLK_USE_NATIVE_NTT+#define MLK_USE_NATIVE_INTT+#define MLK_USE_NATIVE_POLY_REDUCE+#define MLK_USE_NATIVE_POLY_TOMONT+#define MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#define MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#define MLK_USE_NATIVE_POLY_TOBYTES+#define MLK_USE_NATIVE_POLY_FROMBYTES+#define MLK_USE_NATIVE_POLY_COMPRESS_D4+#define MLK_USE_NATIVE_POLY_COMPRESS_D5+#define MLK_USE_NATIVE_POLY_COMPRESS_D10+#define MLK_USE_NATIVE_POLY_COMPRESS_D11+#define MLK_USE_NATIVE_POLY_DECOMPRESS_D4+#define MLK_USE_NATIVE_POLY_DECOMPRESS_D5+#define MLK_USE_NATIVE_POLY_DECOMPRESS_D10+#define MLK_USE_NATIVE_POLY_DECOMPRESS_D11++#if !defined(__ASSEMBLER__)+#include "../../common.h"+#include "../api.h"+#include "src/arith_native_x86_64.h"+#include "src/compress_consts.h"++static MLK_INLINE void mlk_poly_permute_bitrev_to_custom(int16_t data[MLKEM_N])+{+ if (mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ mlk_nttunpack_avx2_asm(data);+ }+}++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_rej_uniform_native(int16_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2) || len != MLKEM_N ||+ buflen % 12 != 0)+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ return (int)mlk_rej_uniform_avx2_asm(r, buf, buflen, mlk_rej_uniform_table);+}++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_ntt_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_ntt_avx2_asm(data, mlk_qdata);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_intt_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_invntt_avx2_asm(data, mlk_qdata);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_reduce_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_reduce_avx2_asm(data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_tomont_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_tomont_avx2_asm(data);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_mulcache_compute_native(int16_t x[MLKEM_N / 2],+ const int16_t y[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_mulcache_compute_avx2_asm(x, y, mlk_qdata);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k2_native(+ int16_t r[MLKEM_N], const int16_t a[2 * MLKEM_N],+ const int16_t b[2 * MLKEM_N], const int16_t b_cache[2 * (MLKEM_N / 2)])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm(r, a, b, b_cache);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 3+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k3_native(+ int16_t r[MLKEM_N], const int16_t a[3 * MLKEM_N],+ const int16_t b[3 * MLKEM_N], const int16_t b_cache[3 * (MLKEM_N / 2)])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm(r, a, b, b_cache);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 3 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k4_native(+ int16_t r[MLKEM_N], const int16_t a[4 * MLKEM_N],+ const int16_t b[4 * MLKEM_N], const int16_t b_cache[4 * (MLKEM_N / 2)])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm(r, a, b, b_cache);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API) || !defined(MLK_CONFIG_NO_ENCAPS_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_tobytes_native(uint8_t r[MLKEM_POLYBYTES],+ const int16_t a[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_ntttobytes_avx2_asm(r, a);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_frombytes_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYBYTES])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_nttfrombytes_avx2_asm(r, a);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API)) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2 || \+ MLKEM_K == 3)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d4_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D4], const int16_t a[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_compress_d4_avx2_asm(r, a, mlk_compress_d4_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d10_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D10], const int16_t a[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_compress_d10_avx2_asm(r, a, mlk_compress_d10_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d4_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D4])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_decompress_d4_avx2_asm(r, a, mlk_decompress_d4_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d10_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D10])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_decompress_d10_avx2_asm(r, a, mlk_decompress_d10_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3) \+ */++#if (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API)) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d5_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D5], const int16_t a[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_compress_d5_avx2_asm(r, a, mlk_compress_d5_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d11_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D11], const int16_t a[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_compress_d11_avx2_asm(r, a, mlk_compress_d11_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d5_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D5])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_decompress_d5_avx2_asm(r, a, mlk_decompress_d5_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d11_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D11])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_decompress_d11_avx2_asm(r, a, mlk_decompress_d11_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4) */++#endif /* !__ASSEMBLER__ */++#endif /* !MLK_NATIVE_X86_64_META_H */
+ cbits/mlkem/src/native/x86_64/src/arith_native_x86_64.h view
@@ -0,0 +1,327 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H+#define MLK_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H++#include "../../../common.h"++#include <stdint.h>+#include "compress_consts.h"+#include "consts.h"++#define MLK_AVX2_REJ_UNIFORM_BUFLEN \+ (3 * 168) /* REJ_UNIFORM_NBLOCKS * SHAKE128_RATE */++#define mlk_rej_uniform_table MLK_NAMESPACE(rej_uniform_table)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_rej_uniform_table[4096];++#define mlk_rej_uniform_avx2_asm MLK_NAMESPACE(rej_uniform_avx2_asm)+MLK_MUST_CHECK_RETURN_VALUE MLK_SYSV_ABI+uint64_t mlk_rej_uniform_avx2_asm(int16_t *r, const uint8_t *buf,+ unsigned buflen, const uint8_t *table)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mlkem_rej_uniform_avx2_asm.ml. */+__contract__(+ requires(buflen % 12 == 0)+ requires(memory_no_alias(buf, buflen))+ requires(table == mlk_rej_uniform_table)+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value <= MLKEM_N)+ ensures(array_bound(r, 0, (unsigned) return_value, 0, MLKEM_Q))+);++#define mlk_ntt_avx2_asm MLK_NAMESPACE(ntt_avx2_asm)+MLK_SYSV_ABI+void mlk_ntt_avx2_asm(int16_t *r, const int16_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mlkem_ntt_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(array_abs_bound(r, 0, MLKEM_N, 8192))+ requires(qdata == mlk_qdata)+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ /* check-magic: off */+ ensures(array_abs_bound(r, 0, MLKEM_N, 23595))+ /* check-magic: on */+);++#define mlk_invntt_avx2_asm MLK_NAMESPACE(invntt_avx2_asm)+MLK_SYSV_ABI+void mlk_invntt_avx2_asm(int16_t *r, const int16_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mlkem_intt_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(qdata == mlk_qdata)+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ /* check-magic: off */+ ensures(array_abs_bound(r, 0, MLKEM_N, 26632))+ /* check-magic: on */+);++#define mlk_nttunpack_avx2_asm MLK_NAMESPACE(nttunpack_avx2_asm)+MLK_SYSV_ABI+void mlk_nttunpack_avx2_asm(int16_t *r)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mlkem_nttunpack_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(r, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ /* Output is a permutation of input: every output coefficient+ * is some input coefficient */+ ensures(forall(i, 0, MLKEM_N, exists(j, 0, MLKEM_N,+ r[i] == old(*(int16_t (*)[MLKEM_N])r)[j])))+);++#define mlk_reduce_avx2_asm MLK_NAMESPACE(reduce_avx2_asm)+MLK_SYSV_ABI+void mlk_reduce_avx2_asm(int16_t *r)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mlkem_reduce_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(r, 0, MLKEM_N, 0, MLKEM_Q))+);++#define mlk_poly_mulcache_compute_avx2_asm \+ MLK_NAMESPACE(poly_mulcache_compute_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_mulcache_compute_avx2_asm(int16_t *out, const int16_t *in,+ const int16_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mlkem_poly_mulcache_compute_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(out, sizeof(int16_t) * (MLKEM_N / 2)))+ requires(memory_no_alias(in, sizeof(int16_t) * MLKEM_N))+ requires(qdata == mlk_qdata)+ assigns(memory_slice(out, sizeof(int16_t) * (MLKEM_N / 2)))+ ensures(array_abs_bound(out, 0, MLKEM_N/2, MLKEM_Q))+);++#define mlk_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm \+ MLK_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k2_avx2_asm)+MLK_SYSV_ABI+void mlk_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm(+ int16_t *r, const int16_t *a, const int16_t *b, const int16_t *b_cache)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 2 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 2 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 2 * (MLKEM_N / 2)))+ requires(array_abs_bound(a, 0, 2 * MLKEM_N, MLKEM_UINT12_LIMIT + 1))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+);++#define mlk_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm \+ MLK_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k3_avx2_asm)+MLK_SYSV_ABI+void mlk_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm(+ int16_t *r, const int16_t *a, const int16_t *b, const int16_t *b_cache)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 3 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 3 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 3 * (MLKEM_N / 2)))+ requires(array_abs_bound(a, 0, 3 * MLKEM_N, MLKEM_UINT12_LIMIT + 1))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+);++#define mlk_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm \+ MLK_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k4_avx2_asm)+MLK_SYSV_ABI+void mlk_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm(+ int16_t *r, const int16_t *a, const int16_t *b, const int16_t *b_cache)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 4 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 4 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 4 * (MLKEM_N / 2)))+ requires(array_abs_bound(a, 0, 4 * MLKEM_N, MLKEM_UINT12_LIMIT + 1))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+);++#define mlk_ntttobytes_avx2_asm MLK_NAMESPACE(ntttobytes_avx2_asm)+MLK_SYSV_ABI+void mlk_ntttobytes_avx2_asm(uint8_t *r, const int16_t *a)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_ntttobytes_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYBYTES))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYBYTES))+);++#define mlk_nttfrombytes_avx2_asm MLK_NAMESPACE(nttfrombytes_avx2_asm)+MLK_SYSV_ABI+void mlk_nttfrombytes_avx2_asm(int16_t *r, const uint8_t *a)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_nttfrombytes_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(a, MLKEM_POLYBYTES))+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(r, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT))+);++#define mlk_tomont_avx2_asm MLK_NAMESPACE(tomont_avx2_asm)+MLK_SYSV_ABI+void mlk_tomont_avx2_asm(int16_t *r)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_tomont_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_abs_bound(r, 0, MLKEM_N, MLKEM_Q))+);++#define mlk_poly_compress_d4_avx2_asm MLK_NAMESPACE(poly_compress_d4_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_compress_d4_avx2_asm(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D4],+ const int16_t *MLK_RESTRICT a,+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_compress_d4_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ requires(data == mlk_compress_d4_data)+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+);++#define mlk_poly_decompress_d4_avx2_asm \+ MLK_NAMESPACE(poly_decompress_d4_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_decompress_d4_avx2_asm(+ int16_t *MLK_RESTRICT r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D4],+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_decompress_d4_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D4))+ requires(data == mlk_decompress_d4_data)+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(r, 0, MLKEM_N, 0, MLKEM_Q))+);++#define mlk_poly_compress_d10_avx2_asm MLK_NAMESPACE(poly_compress_d10_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_compress_d10_avx2_asm(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D10],+ const int16_t *MLK_RESTRICT a,+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_compress_d10_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ requires(data == mlk_compress_d10_data)+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+);++#define mlk_poly_decompress_d10_avx2_asm \+ MLK_NAMESPACE(poly_decompress_d10_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_decompress_d10_avx2_asm(+ int16_t *MLK_RESTRICT r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D10],+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_decompress_d10_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D10))+ requires(data == mlk_decompress_d10_data)+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(r, 0, MLKEM_N, 0, MLKEM_Q))+);++#define mlk_poly_compress_d5_avx2_asm MLK_NAMESPACE(poly_compress_d5_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_compress_d5_avx2_asm(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D5],+ const int16_t *MLK_RESTRICT a,+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_compress_d5_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ requires(data == mlk_compress_d5_data)+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+);++#define mlk_poly_decompress_d5_avx2_asm \+ MLK_NAMESPACE(poly_decompress_d5_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_decompress_d5_avx2_asm(+ int16_t *MLK_RESTRICT r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D5],+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_decompress_d5_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D5))+ requires(data == mlk_decompress_d5_data)+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(r, 0, MLKEM_N, 0, MLKEM_Q))+);++#define mlk_poly_compress_d11_avx2_asm MLK_NAMESPACE(poly_compress_d11_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_compress_d11_avx2_asm(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D11],+ const int16_t *MLK_RESTRICT a,+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_compress_d11_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ requires(data == mlk_compress_d11_data)+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+);++#define mlk_poly_decompress_d11_avx2_asm \+ MLK_NAMESPACE(poly_decompress_d11_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_decompress_d11_avx2_asm(+ int16_t *MLK_RESTRICT r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D11],+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_decompress_d11_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D11))+ requires(data == mlk_decompress_d11_data)+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(r, 0, MLKEM_N, 0, MLKEM_Q))+);++#endif /* !MLK_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H */
+ cbits/mlkem/src/native/x86_64/src/compress_consts.c view
@@ -0,0 +1,115 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API))++#include "compress_consts.h"++#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2 || \+ MLKEM_K == 3)++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_compress_d4_data[32] = {+ 0, 0, 0, 0, 4, 0, 0, 0, 1, 0, 0, 0, 5, 0, 0, 0,+ 2, 0, 0, 0, 6, 0, 0, 0, 3, 0, 0, 0, 7, 0, 0, 0, /* permdidx */+};++#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_decompress_d4_data[32] = {+ 0, 0, 0, 0, 1, 1, 1, 1, 2, 2, 2, 2, 3, 3, 3, 3,+ 4, 4, 4, 4, 5, 5, 5, 5, 6, 6, 6, 6, 7, 7, 7, 7, /* shufbidx */+};+#endif /* !MLK_CONFIG_NO_DECAPS_API */++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_compress_d10_data[32] = {+ 0, 1, 2, 3, 4, 8, 9, 10, 11, 12, 255,+ 255, 255, 255, 255, 255, 9, 10, 11, 12, 255, 255,+ 255, 255, 255, 255, 0, 1, 2, 3, 4, 8, /* shufbidx */+};++#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_decompress_d10_data[32] = {+ 0, 1, 1, 2, 2, 3, 3, 4, 5, 6, 6, 7, 7, 8, 8, 9,+ 2, 3, 3, 4, 4, 5, 5, 6, 7, 8, 8, 9, 9, 10, 10, 11, /* shufbidx */+};+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#endif /* !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3) \+ */++#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4)++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_compress_d5_data[32] = {+ 0, 1, 2, 3, 4, 255, 255, 255, 255, 255, 8,+ 9, 10, 11, 12, 255, 9, 10, 11, 12, 255, 0,+ 1, 2, 3, 4, 255, 255, 255, 255, 255, 8, /* shufbidx */+};++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* shufbidx[0:32], mask[32:64], shift[64:96] */+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_decompress_d5_data[96] = {+ 0, 0, 0, 1, 1, 1, 1, 2, 2, 3, 3, 3, 3, 4, 4, 4, 5, 5,+ 5, 6, 6, 6, 6, 7, 7, 8, 8, 8, 8, 9, 9, 9, /* shufbidx */+ 31, 0, 224, 3, 124, 0, 128, 15, 240, 1, 62, 0, 192, 7, 248, 0, 31, 0,+ 224, 3, 124, 0, 128, 15, 240, 1, 62, 0, 192, 7, 248, 0, /* mask */+ 0, 4, 32, 0, 0, 1, 8, 0, 64, 0, 0, 2, 16, 0, 128, 0, 0, 4,+ 32, 0, 0, 1, 8, 0, 64, 0, 0, 2, 16, 0, 128, 0, /* shift */+};+#endif /* !MLK_CONFIG_NO_DECAPS_API */++/* srlvqidx[0:32], shufbidx[32:64] */+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_compress_d11_data[64] =+ {+ 10, 0, 0, 0, 0, 0, 0, 0, 30, 0, 0,+ 0, 0, 0, 0, 0, 10, 0, 0, 0, 0, 0,+ 0, 0, 30, 0, 0, 0, 0, 0, 0, 0, /* srlvqidx */+ 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10,+ 255, 255, 255, 255, 255, 5, 6, 7, 8, 9, 10,+ 255, 255, 255, 255, 0, 0, 1, 2, 3, 4, /* shufbidx */+};++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* shufbidx[0:32], srlvdidx[32:64], srlvqidx[64:96], shift[96:128] */+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_decompress_d11_data[128] = {+ 0, 1, 1, 2, 2, 3, 4, 5, 5, 6, 6, 7, 8, 9, 9, 10,+ 3, 4, 4, 5, 5, 6, 7, 8, 8, 9, 9, 10, 11, 12, 12, 13, /* shufbidx */+ 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,+ 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, /* srlvdidx */+ 0, 0, 0, 0, 0, 0, 0, 0, 2, 0, 0, 0, 0, 0, 0, 0,+ 0, 0, 0, 0, 0, 0, 0, 0, 2, 0, 0, 0, 0, 0, 0, 0, /* srlvqidx */+ 32, 0, 4, 0, 1, 0, 32, 0, 8, 0, 1, 0, 32, 0, 4, 0,+ 32, 0, 4, 0, 1, 0, 32, 0, 8, 0, 1, 0, 32, 0, 4, 0, /* shift */+};+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#endif /* !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4) */++#else /* MLK_ARITH_BACKEND_X86_64_DEFAULT && (!MLK_CONFIG_NO_ENCAPS_API || \+ !MLK_CONFIG_NO_DECAPS_API) */++MLK_EMPTY_CU(avx2_compress_consts)++#endif /* !(MLK_ARITH_BACKEND_X86_64_DEFAULT && (!MLK_CONFIG_NO_ENCAPS_API || \+ !MLK_CONFIG_NO_DECAPS_API)) */
+ cbits/mlkem/src/native/x86_64/src/compress_consts.h view
@@ -0,0 +1,53 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#ifndef MLK_NATIVE_X86_64_SRC_COMPRESS_CONSTS_H+#define MLK_NATIVE_X86_64_SRC_COMPRESS_CONSTS_H++#include "../../../common.h"++#ifndef __ASSEMBLER__++#define mlk_compress_d4_data MLK_NAMESPACE(compress_d4_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_compress_d4_data[32];++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_decompress_d4_data MLK_NAMESPACE(decompress_d4_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_decompress_d4_data[32];+#endif++#define mlk_compress_d10_data MLK_NAMESPACE(compress_d10_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_compress_d10_data[32];++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_decompress_d10_data MLK_NAMESPACE(decompress_d10_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_decompress_d10_data[32];+#endif++#define mlk_compress_d5_data MLK_NAMESPACE(compress_d5_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_compress_d5_data[32];++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_decompress_d5_data MLK_NAMESPACE(decompress_d5_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_decompress_d5_data[96];+#endif++#define mlk_compress_d11_data MLK_NAMESPACE(compress_d11_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_compress_d11_data[64];++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_decompress_d11_data MLK_NAMESPACE(decompress_d11_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_decompress_d11_data[128];+#endif++#endif /* !__ASSEMBLER__ */++#endif /* !MLK_NATIVE_X86_64_SRC_COMPRESS_CONSTS_H */
+ cbits/mlkem/src/native/x86_64/src/consts.c view
@@ -0,0 +1,102 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#include "consts.h"++/*+ * Table of zeta values used in the AVX2 NTTs+ * See autogen for details.+ */+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t mlk_qdata[624] = {+ 3854, 3340, 2826, 2312, 1798, 1284, 770, 256, 3854,+ 3340, 2826, 2312, 1798, 1284, 770, 256, 7, 0,+ 6, 0, 5, 0, 4, 0, 3, 0, 2,+ 0, 1, 0, 0, 0, 31498, 31498, 31498, 31498,+ -758, -758, -758, -758, 0, 0, 0, 0, 0,+ 0, 0, 0, 14745, 14745, 14745, 14745, 14745, 14745,+ 14745, 14745, 14745, 14745, 14745, 14745, 14745, 14745, 14745,+ 14745, -359, -359, -359, -359, -359, -359, -359, -359,+ -359, -359, -359, -359, -359, -359, -359, -359, 13525,+ 13525, 13525, 13525, 13525, 13525, 13525, 13525, -12402, -12402,+ -12402, -12402, -12402, -12402, -12402, -12402, 1493, 1493, 1493,+ 1493, 1493, 1493, 1493, 1493, 1422, 1422, 1422, 1422,+ 1422, 1422, 1422, 1422, -20907, -20907, -20907, -20907, 27758,+ 27758, 27758, 27758, -3799, -3799, -3799, -3799, -15690, -15690,+ -15690, -15690, -171, -171, -171, -171, 622, 622, 622,+ 622, 1577, 1577, 1577, 1577, 182, 182, 182, 182,+ -5827, -5827, 17363, 17363, -26360, -26360, -29057, -29057, 5571,+ 5571, -1102, -1102, 21438, 21438, -26242, -26242, 573, 573,+ -1325, -1325, 264, 264, 383, 383, -829, -829, 1458,+ 1458, -1602, -1602, -130, -130, -5689, -6516, 1496, 30967,+ -23565, 20179, 20710, 25080, -12796, 26616, 16064, -12442, 9134,+ -650, -25986, 27837, 1223, 652, -552, 1015, -1293, 1491,+ -282, -1544, 516, -8, -320, -666, -1618, -1162, 126,+ 1469, -335, -11477, -32227, 20494, -27738, 945, -14883, 6182,+ 32010, 10631, 29175, -28762, -18486, 17560, -14430, -5276, -1103,+ 555, -1251, 1550, 422, 177, -291, 1574, -246, 1159,+ -777, -602, -1590, -872, 418, -156, 11182, 13387, -14233,+ -21655, 13131, -4587, 23092, 5493, -32502, 30317, -18741, 12639,+ 20100, 18525, 19529, -12619, 430, 843, 871, 105, 587,+ -235, -460, 1653, 778, -147, 1483, 1119, 644, 349,+ 329, -75, 787, 787, 787, 787, 787, 787, 787,+ 787, 787, 787, 787, 787, 787, 787, 787, 787,+ -1517, -1517, -1517, -1517, -1517, -1517, -1517, -1517, -1517,+ -1517, -1517, -1517, -1517, -1517, -1517, -1517, 28191, 28191,+ 28191, 28191, 28191, 28191, 28191, 28191, -16694, -16694, -16694,+ -16694, -16694, -16694, -16694, -16694, 287, 287, 287, 287,+ 287, 287, 287, 287, 202, 202, 202, 202, 202,+ 202, 202, 202, 10690, 10690, 10690, 10690, 1358, 1358,+ 1358, 1358, -11202, -11202, -11202, -11202, 31164, 31164, 31164,+ 31164, 962, 962, 962, 962, -1202, -1202, -1202, -1202,+ -1474, -1474, -1474, -1474, 1468, 1468, 1468, 1468, -28073,+ -28073, 24313, 24313, -10532, -10532, 8800, 8800, 18426, 18426,+ 8859, 8859, 26675, 26675, -16163, -16163, -681, -681, 1017,+ 1017, 732, 732, 608, 608, -1542, -1542, 411, 411,+ -205, -205, -1571, -1571, 19883, -28250, -15887, -8898, -28309,+ 9075, -30199, 18249, 13426, 14017, -29156, -12757, 16832, 4311,+ -24155, -17915, -853, -90, -271, 830, 107, -1421, -247,+ -951, -398, 961, -1508, -725, 448, -1065, 677, -1275,+ -31183, 25435, -7382, 24391, -20927, 10946, 24214, 16989, 10335,+ -7934, -22502, 10906, 31636, 28644, 23998, -17422, 817, 603,+ 1322, -1465, -1215, 1218, -874, -1187, -1185, -1278, -1510,+ -870, -108, 996, 958, 1522, 20297, 2146, 15355, -32384,+ -6280, -14903, -11044, 14469, -21498, -20198, 23210, -17442, -23860,+ -20257, 7756, 23132, 1097, 610, -1285, 384, -136, -1335,+ 220, -1659, -1530, 794, -854, 478, -308, 991, -1460,+ 1628, -1103, 555, -1251, 1550, 422, 177, -291, 1574,+ -246, 1159, -777, -602, -1590, -872, 418, -156, 430,+ 843, 871, 105, 587, -235, -460, 1653, 778, -147,+ 1483, 1119, 644, 349, 329, -75, 817, 603, 1322,+ -1465, -1215, 1218, -874, -1187, -1185, -1278, -1510, -870,+ -108, 996, 958, 1522, 1097, 610, -1285, 384, -136,+ -1335, 220, -1659, -1530, 794, -854, 478, -308, 991,+ -1460, 1628, -335, -11477, -32227, 20494, -27738, 945, -14883,+ 6182, 32010, 10631, 29175, -28762, -18486, 17560, -14430, -5276,+ 11182, 13387, -14233, -21655, 13131, -4587, 23092, 5493, -32502,+ 30317, -18741, 12639, 20100, 18525, 19529, -12619, -31183, 25435,+ -7382, 24391, -20927, 10946, 24214, 16989, 10335, -7934, -22502,+ 10906, 31636, 28644, 23998, -17422, 20297, 2146, 15355, -32384,+ -6280, -14903, -11044, 14469, -21498, -20198, 23210, -17442, -23860,+ -20257, 7756, 23132,+};++#else /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ */++MLK_EMPTY_CU(avx2_consts)++#endif /* !(MLK_ARITH_BACKEND_X86_64_DEFAULT && \+ !MLK_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mlkem/src/native/x86_64/src/consts.h view
@@ -0,0 +1,25 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#ifndef MLK_NATIVE_X86_64_SRC_CONSTS_H+#define MLK_NATIVE_X86_64_SRC_CONSTS_H+#include "../../../common.h"+#define MLK_AVX2_BACKEND_DATA_OFFSET_REVIDXB 0+#define MLK_AVX2_BACKEND_DATA_OFFSET_REVIDXD 16+#define MLK_AVX2_BACKEND_DATA_OFFSET_ZETAS_EXP 32+#define MLK_AVX2_BACKEND_DATA_OFFSET_MULCACHE_TWIDDLES 496++#ifndef __ASSEMBLER__+#define mlk_qdata MLK_NAMESPACE(qdata)+MLK_INTERNAL_DATA_DECLARATION const int16_t mlk_qdata[624];+#endif++#endif /* !MLK_NATIVE_X86_64_SRC_CONSTS_H */
+ cbits/mlkem/src/native/x86_64/src/mlkem_intt_avx2_asm.S view
@@ -0,0 +1,743 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [AVX2_NTT]+ * Faster AVX2 optimized NTT multiplication for Ring-LWE lattice cryptography.+ * Gregor Seiler+ * https://eprint.iacr.org/2018/039+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Kyber implementation @[REF_AVX2].+ *+ * The core ideas behind the implementation are described in @[AVX2_NTT].+ *+ * Changes:+ * - Different placement of modular reductions to simplify+ * reasoning of non-overflow+ */++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API))+/*yaml+ Name: invntt_avx2_asm+ Description: x86_64 AVX2 inverse NTT+ Signature: void mlk_invntt_avx2_asm(int16_t *r, const int16_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t *r+ description: Input/output polynomial (256 x int16_t)+ rsi:+ type: buffer+ size_bytes: 1248+ permissions: read-only+ c_parameter: const int16_t *qdata+ description: Precomputed constants (624 x int16_t)+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_intt_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(invntt_avx2_asm)+MLK_ASM_FN_SYMBOL(invntt_avx2_asm)++ .cfi_startproc+ movl $0xd010d01, %eax # imm = 0xD010D01+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ movl $0xd8a1d8a1, %eax # imm = 0xD8A1D8A1+ vmovd %eax, %xmm2+ vpbroadcastd %xmm2, %ymm2+ movl $0x5a105a1, %eax # imm = 0x5A105A1+ vmovd %eax, %xmm3+ vpbroadcastd %xmm3, %ymm3+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x40(%rdi), %ymm6+ vmovdqa 0x20(%rdi), %ymm5+ vmovdqa 0x60(%rdi), %ymm7+ vpmullw %ymm2, %ymm4, %ymm12+ vpmulhw %ymm3, %ymm4, %ymm4+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm4, %ymm4+ vpmullw %ymm2, %ymm6, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm6+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm6, %ymm6+ vpmullw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm5, %ymm5+ vpmullw %ymm2, %ymm7, %ymm12+ vpmulhw %ymm3, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm7, %ymm7+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xc0(%rdi), %ymm10+ vmovdqa 0xa0(%rdi), %ymm9+ vmovdqa 0xe0(%rdi), %ymm11+ vpmullw %ymm2, %ymm8, %ymm12+ vpmulhw %ymm3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm8, %ymm8+ vpmullw %ymm2, %ymm10, %ymm12+ vpmulhw %ymm3, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm10, %ymm10+ vpmullw %ymm2, %ymm9, %ymm12+ vpmulhw %ymm3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm9, %ymm9+ vpmullw %ymm2, %ymm11, %ymm12+ vpmulhw %ymm3, %ymm11, %ymm11+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm11, %ymm11+ vpermq $0x4e, 0x3a0(%rsi), %ymm15 # ymm15 = mem[2,3,0,1]+ vpermq $0x4e, 0x360(%rsi), %ymm1 # ymm1 = mem[2,3,0,1]+ vpermq $0x4e, 0x3c0(%rsi), %ymm2 # ymm2 = mem[2,3,0,1]+ vpermq $0x4e, 0x380(%rsi), %ymm3 # ymm3 = mem[2,3,0,1]+ vmovdqa (%rsi), %ymm12+ vpshufb %ymm12, %ymm15, %ymm15+ vpshufb %ymm12, %ymm1, %ymm1+ vpshufb %ymm12, %ymm2, %ymm2+ vpshufb %ymm12, %ymm3, %ymm3+ vpsubw %ymm4, %ymm6, %ymm12+ vpaddw %ymm6, %ymm4, %ymm4+ vpsubw %ymm5, %ymm7, %ymm13+ vpmullw %ymm15, %ymm12, %ymm6+ vpaddw %ymm7, %ymm5, %ymm5+ vpsubw %ymm8, %ymm10, %ymm14+ vpmullw %ymm15, %ymm13, %ymm7+ vpaddw %ymm10, %ymm8, %ymm8+ vpsubw %ymm9, %ymm11, %ymm15+ vpmullw %ymm1, %ymm14, %ymm10+ vpaddw %ymm11, %ymm9, %ymm9+ vpmullw %ymm1, %ymm15, %ymm11+ vpmulhw %ymm2, %ymm12, %ymm12+ vpmulhw %ymm2, %ymm13, %ymm13+ vpmulhw %ymm3, %ymm14, %ymm14+ vpmulhw %ymm3, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm6, %ymm6+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm6, %ymm12, %ymm6+ vpsubw %ymm7, %ymm13, %ymm7+ vpsubw %ymm10, %ymm14, %ymm10+ vpsubw %ymm11, %ymm15, %ymm11+ vpermq $0x4e, 0x320(%rsi), %ymm2 # ymm2 = mem[2,3,0,1]+ vpermq $0x4e, 0x340(%rsi), %ymm3 # ymm3 = mem[2,3,0,1]+ vmovdqa (%rsi), %ymm1+ vpshufb %ymm1, %ymm2, %ymm2+ vpshufb %ymm1, %ymm3, %ymm3+ vpsubw %ymm4, %ymm8, %ymm12+ vpaddw %ymm8, %ymm4, %ymm4+ vpsubw %ymm5, %ymm9, %ymm13+ vpmullw %ymm2, %ymm12, %ymm8+ vpaddw %ymm9, %ymm5, %ymm5+ vpsubw %ymm6, %ymm10, %ymm14+ vpmullw %ymm2, %ymm13, %ymm9+ vpaddw %ymm10, %ymm6, %ymm6+ vpsubw %ymm7, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm10+ vpaddw %ymm11, %ymm7, %ymm7+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm3, %ymm12, %ymm12+ vpmulhw %ymm3, %ymm13, %ymm13+ vpmulhw %ymm3, %ymm14, %ymm14+ vpmulhw %ymm3, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm8, %ymm12, %ymm8+ vpsubw %ymm9, %ymm13, %ymm9+ vpsubw %ymm10, %ymm14, %ymm10+ vpsubw %ymm11, %ymm15, %ymm11+ vpslld $0x10, %ymm5, %ymm3+ vpblendw $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7],ymm4[8],ymm3[9],ymm4[10],ymm3[11],ymm4[12],ymm3[13],ymm4[14],ymm3[15]+ vpsrld $0x10, %ymm4, %ymm4+ vpblendw $0xaa, %ymm5, %ymm4, %ymm5 # ymm5 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7],ymm4[8],ymm5[9],ymm4[10],ymm5[11],ymm4[12],ymm5[13],ymm4[14],ymm5[15]+ vpslld $0x10, %ymm7, %ymm4+ vpblendw $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7],ymm6[8],ymm4[9],ymm6[10],ymm4[11],ymm6[12],ymm4[13],ymm6[14],ymm4[15]+ vpsrld $0x10, %ymm6, %ymm6+ vpblendw $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7],ymm6[8],ymm7[9],ymm6[10],ymm7[11],ymm6[12],ymm7[13],ymm6[14],ymm7[15]+ vpslld $0x10, %ymm9, %ymm6+ vpblendw $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7],ymm8[8],ymm6[9],ymm8[10],ymm6[11],ymm8[12],ymm6[13],ymm8[14],ymm6[15]+ vpsrld $0x10, %ymm8, %ymm8+ vpblendw $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7],ymm8[8],ymm9[9],ymm8[10],ymm9[11],ymm8[12],ymm9[13],ymm8[14],ymm9[15]+ vpslld $0x10, %ymm11, %ymm8+ vpblendw $0xaa, %ymm8, %ymm10, %ymm8 # ymm8 = ymm10[0],ymm8[1],ymm10[2],ymm8[3],ymm10[4],ymm8[5],ymm10[6],ymm8[7],ymm10[8],ymm8[9],ymm10[10],ymm8[11],ymm10[12],ymm8[13],ymm10[14],ymm8[15]+ vpsrld $0x10, %ymm10, %ymm10+ vpblendw $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7],ymm10[8],ymm11[9],ymm10[10],ymm11[11],ymm10[12],ymm11[13],ymm10[14],ymm11[15]+ vmovdqa 0x20(%rsi), %ymm12+ vpermd 0x2e0(%rsi), %ymm12, %ymm2+ vpermd 0x300(%rsi), %ymm12, %ymm10+ vpsubw %ymm3, %ymm5, %ymm12+ vpaddw %ymm5, %ymm3, %ymm3+ vpsubw %ymm4, %ymm7, %ymm13+ vpmullw %ymm2, %ymm12, %ymm5+ vpaddw %ymm7, %ymm4, %ymm4+ vpsubw %ymm6, %ymm9, %ymm14+ vpmullw %ymm2, %ymm13, %ymm7+ vpaddw %ymm9, %ymm6, %ymm6+ vpsubw %ymm8, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm9+ vpaddw %ymm11, %ymm8, %ymm8+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm10, %ymm12, %ymm12+ vpmulhw %ymm10, %ymm13, %ymm13+ vpmulhw %ymm10, %ymm14, %ymm14+ vpmulhw %ymm10, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm5, %ymm12, %ymm5+ vpsubw %ymm7, %ymm13, %ymm7+ vpsubw %ymm9, %ymm14, %ymm9+ vpsubw %ymm11, %ymm15, %ymm11+ movl $0x4ebf4ebf, %eax # imm = 0x4EBF4EBF+ vmovd %eax, %xmm1+ vpbroadcastd %xmm1, %ymm1+ vpmulhw %ymm1, %ymm3, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm3, %ymm3+ vmovsldup %ymm4, %ymm10 # ymm10 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm10, %ymm3, %ymm10 # ymm10 = ymm3[0],ymm10[1],ymm3[2],ymm10[3],ymm3[4],ymm10[5],ymm3[6],ymm10[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm8, %ymm3 # ymm3 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm6, %ymm3 # ymm3 = ymm6[0],ymm3[1],ymm6[2],ymm3[3],ymm6[4],ymm3[5],ymm6[6],ymm3[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm8, %ymm6, %ymm8 # ymm8 = ymm6[0],ymm8[1],ymm6[2],ymm8[3],ymm6[4],ymm8[5],ymm6[6],ymm8[7]+ vmovsldup %ymm7, %ymm6 # ymm6 = ymm7[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vmovsldup %ymm11, %ymm5 # ymm5 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm5[1],ymm9[2],ymm5[3],ymm9[4],ymm5[5],ymm9[6],ymm5[7]+ vpsrlq $0x20, %ymm9, %ymm9+ vpblendd $0xaa, %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[0],ymm11[1],ymm9[2],ymm11[3],ymm9[4],ymm11[5],ymm9[6],ymm11[7]+ vpermq $0x1b, 0x2a0(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpermq $0x1b, 0x2c0(%rsi), %ymm9 # ymm9 = mem[3,2,1,0]+ vpsubw %ymm10, %ymm4, %ymm12+ vpaddw %ymm4, %ymm10, %ymm10+ vpsubw %ymm3, %ymm8, %ymm13+ vpmullw %ymm2, %ymm12, %ymm4+ vpaddw %ymm8, %ymm3, %ymm3+ vpsubw %ymm6, %ymm7, %ymm14+ vpmullw %ymm2, %ymm13, %ymm8+ vpaddw %ymm7, %ymm6, %ymm6+ vpsubw %ymm5, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm7+ vpaddw %ymm11, %ymm5, %ymm5+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm9, %ymm12, %ymm12+ vpmulhw %ymm9, %ymm13, %ymm13+ vpmulhw %ymm9, %ymm14, %ymm14+ vpmulhw %ymm9, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm4, %ymm4+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm4, %ymm12, %ymm4+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm7, %ymm14, %ymm7+ vpsubw %ymm11, %ymm15, %ymm11+ vpmulhw %ymm1, %ymm10, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm10, %ymm10+ vpunpcklqdq %ymm3, %ymm10, %ymm9 # ymm9 = ymm10[0],ymm3[0],ymm10[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[1],ymm3[1],ymm10[3],ymm3[3]+ vpunpcklqdq %ymm5, %ymm6, %ymm10 # ymm10 = ymm6[0],ymm5[0],ymm6[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[1],ymm5[1],ymm6[3],ymm5[3]+ vpunpcklqdq %ymm8, %ymm4, %ymm6 # ymm6 = ymm4[0],ymm8[0],ymm4[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[1],ymm8[1],ymm4[3],ymm8[3]+ vpunpcklqdq %ymm11, %ymm7, %ymm4 # ymm4 = ymm7[0],ymm11[0],ymm7[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[1],ymm11[1],ymm7[3],ymm11[3]+ vpermq $0x4e, 0x260(%rsi), %ymm2 # ymm2 = mem[2,3,0,1]+ vpermq $0x4e, 0x280(%rsi), %ymm7 # ymm7 = mem[2,3,0,1]+ vpsubw %ymm9, %ymm3, %ymm12+ vpaddw %ymm3, %ymm9, %ymm9+ vpsubw %ymm10, %ymm5, %ymm13+ vpmullw %ymm2, %ymm12, %ymm3+ vpaddw %ymm5, %ymm10, %ymm10+ vpsubw %ymm6, %ymm8, %ymm14+ vpmullw %ymm2, %ymm13, %ymm5+ vpaddw %ymm8, %ymm6, %ymm6+ vpsubw %ymm4, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm8+ vpaddw %ymm11, %ymm4, %ymm4+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm7, %ymm12, %ymm12+ vpmulhw %ymm7, %ymm13, %ymm13+ vpmulhw %ymm7, %ymm14, %ymm14+ vpmulhw %ymm7, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm3, %ymm3+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm3, %ymm12, %ymm3+ vpsubw %ymm5, %ymm13, %ymm5+ vpsubw %ymm8, %ymm14, %ymm8+ vpsubw %ymm11, %ymm15, %ymm11+ vpmulhw %ymm1, %ymm9, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm9, %ymm9+ vperm2i128 $0x20, %ymm10, %ymm9, %ymm7 # ymm7 = ymm9[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm9, %ymm10 # ymm10 = ymm9[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm4, %ymm6, %ymm9 # ymm9 = ymm6[0,1],ymm4[0,1]+ vperm2i128 $0x31, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[2,3],ymm4[2,3]+ vperm2i128 $0x20, %ymm5, %ymm3, %ymm6 # ymm6 = ymm3[0,1],ymm5[0,1]+ vperm2i128 $0x31, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[2,3],ymm5[2,3]+ vperm2i128 $0x20, %ymm11, %ymm8, %ymm3 # ymm3 = ymm8[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm8, %ymm11 # ymm11 = ymm8[2,3],ymm11[2,3]+ vmovdqa 0x220(%rsi), %ymm2+ vmovdqa 0x240(%rsi), %ymm8+ vpsubw %ymm7, %ymm10, %ymm12+ vpaddw %ymm10, %ymm7, %ymm7+ vpsubw %ymm9, %ymm4, %ymm13+ vpmullw %ymm2, %ymm12, %ymm10+ vpaddw %ymm4, %ymm9, %ymm9+ vpsubw %ymm6, %ymm5, %ymm14+ vpmullw %ymm2, %ymm13, %ymm4+ vpaddw %ymm5, %ymm6, %ymm6+ vpsubw %ymm3, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm5+ vpaddw %ymm11, %ymm3, %ymm3+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm8, %ymm12, %ymm12+ vpmulhw %ymm8, %ymm13, %ymm13+ vpmulhw %ymm8, %ymm14, %ymm14+ vpmulhw %ymm8, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm4, %ymm4+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm10, %ymm12, %ymm10+ vpsubw %ymm4, %ymm13, %ymm4+ vpsubw %ymm5, %ymm14, %ymm5+ vpsubw %ymm11, %ymm15, %ymm11+ vpmulhw %ymm1, %ymm7, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm7, %ymm7+ vmovdqa %ymm7, (%rdi)+ vmovdqa %ymm9, 0x20(%rdi)+ vmovdqa %ymm6, 0x40(%rdi)+ vmovdqa %ymm3, 0x60(%rdi)+ vmovdqa %ymm10, 0x80(%rdi)+ vmovdqa %ymm4, 0xa0(%rdi)+ vmovdqa %ymm5, 0xc0(%rdi)+ vmovdqa %ymm11, 0xe0(%rdi)+ movl $0xd8a1d8a1, %eax # imm = 0xD8A1D8A1+ vmovd %eax, %xmm2+ vpbroadcastd %xmm2, %ymm2+ movl $0x5a105a1, %eax # imm = 0x5A105A1+ vmovd %eax, %xmm3+ vpbroadcastd %xmm3, %ymm3+ vmovdqa 0x100(%rdi), %ymm4+ vmovdqa 0x140(%rdi), %ymm6+ vmovdqa 0x120(%rdi), %ymm5+ vmovdqa 0x160(%rdi), %ymm7+ vpmullw %ymm2, %ymm4, %ymm12+ vpmulhw %ymm3, %ymm4, %ymm4+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm4, %ymm4+ vpmullw %ymm2, %ymm6, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm6+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm6, %ymm6+ vpmullw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm5, %ymm5+ vpmullw %ymm2, %ymm7, %ymm12+ vpmulhw %ymm3, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm7, %ymm7+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1c0(%rdi), %ymm10+ vmovdqa 0x1a0(%rdi), %ymm9+ vmovdqa 0x1e0(%rdi), %ymm11+ vpmullw %ymm2, %ymm8, %ymm12+ vpmulhw %ymm3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm8, %ymm8+ vpmullw %ymm2, %ymm10, %ymm12+ vpmulhw %ymm3, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm10, %ymm10+ vpmullw %ymm2, %ymm9, %ymm12+ vpmulhw %ymm3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm9, %ymm9+ vpmullw %ymm2, %ymm11, %ymm12+ vpmulhw %ymm3, %ymm11, %ymm11+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm11, %ymm11+ vpermq $0x4e, 0x1e0(%rsi), %ymm15 # ymm15 = mem[2,3,0,1]+ vpermq $0x4e, 0x1a0(%rsi), %ymm1 # ymm1 = mem[2,3,0,1]+ vpermq $0x4e, 0x200(%rsi), %ymm2 # ymm2 = mem[2,3,0,1]+ vpermq $0x4e, 0x1c0(%rsi), %ymm3 # ymm3 = mem[2,3,0,1]+ vmovdqa (%rsi), %ymm12+ vpshufb %ymm12, %ymm15, %ymm15+ vpshufb %ymm12, %ymm1, %ymm1+ vpshufb %ymm12, %ymm2, %ymm2+ vpshufb %ymm12, %ymm3, %ymm3+ vpsubw %ymm4, %ymm6, %ymm12+ vpaddw %ymm6, %ymm4, %ymm4+ vpsubw %ymm5, %ymm7, %ymm13+ vpmullw %ymm15, %ymm12, %ymm6+ vpaddw %ymm7, %ymm5, %ymm5+ vpsubw %ymm8, %ymm10, %ymm14+ vpmullw %ymm15, %ymm13, %ymm7+ vpaddw %ymm10, %ymm8, %ymm8+ vpsubw %ymm9, %ymm11, %ymm15+ vpmullw %ymm1, %ymm14, %ymm10+ vpaddw %ymm11, %ymm9, %ymm9+ vpmullw %ymm1, %ymm15, %ymm11+ vpmulhw %ymm2, %ymm12, %ymm12+ vpmulhw %ymm2, %ymm13, %ymm13+ vpmulhw %ymm3, %ymm14, %ymm14+ vpmulhw %ymm3, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm6, %ymm6+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm6, %ymm12, %ymm6+ vpsubw %ymm7, %ymm13, %ymm7+ vpsubw %ymm10, %ymm14, %ymm10+ vpsubw %ymm11, %ymm15, %ymm11+ vpermq $0x4e, 0x160(%rsi), %ymm2 # ymm2 = mem[2,3,0,1]+ vpermq $0x4e, 0x180(%rsi), %ymm3 # ymm3 = mem[2,3,0,1]+ vmovdqa (%rsi), %ymm1+ vpshufb %ymm1, %ymm2, %ymm2+ vpshufb %ymm1, %ymm3, %ymm3+ vpsubw %ymm4, %ymm8, %ymm12+ vpaddw %ymm8, %ymm4, %ymm4+ vpsubw %ymm5, %ymm9, %ymm13+ vpmullw %ymm2, %ymm12, %ymm8+ vpaddw %ymm9, %ymm5, %ymm5+ vpsubw %ymm6, %ymm10, %ymm14+ vpmullw %ymm2, %ymm13, %ymm9+ vpaddw %ymm10, %ymm6, %ymm6+ vpsubw %ymm7, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm10+ vpaddw %ymm11, %ymm7, %ymm7+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm3, %ymm12, %ymm12+ vpmulhw %ymm3, %ymm13, %ymm13+ vpmulhw %ymm3, %ymm14, %ymm14+ vpmulhw %ymm3, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm8, %ymm12, %ymm8+ vpsubw %ymm9, %ymm13, %ymm9+ vpsubw %ymm10, %ymm14, %ymm10+ vpsubw %ymm11, %ymm15, %ymm11+ vpslld $0x10, %ymm5, %ymm3+ vpblendw $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7],ymm4[8],ymm3[9],ymm4[10],ymm3[11],ymm4[12],ymm3[13],ymm4[14],ymm3[15]+ vpsrld $0x10, %ymm4, %ymm4+ vpblendw $0xaa, %ymm5, %ymm4, %ymm5 # ymm5 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7],ymm4[8],ymm5[9],ymm4[10],ymm5[11],ymm4[12],ymm5[13],ymm4[14],ymm5[15]+ vpslld $0x10, %ymm7, %ymm4+ vpblendw $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7],ymm6[8],ymm4[9],ymm6[10],ymm4[11],ymm6[12],ymm4[13],ymm6[14],ymm4[15]+ vpsrld $0x10, %ymm6, %ymm6+ vpblendw $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7],ymm6[8],ymm7[9],ymm6[10],ymm7[11],ymm6[12],ymm7[13],ymm6[14],ymm7[15]+ vpslld $0x10, %ymm9, %ymm6+ vpblendw $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7],ymm8[8],ymm6[9],ymm8[10],ymm6[11],ymm8[12],ymm6[13],ymm8[14],ymm6[15]+ vpsrld $0x10, %ymm8, %ymm8+ vpblendw $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7],ymm8[8],ymm9[9],ymm8[10],ymm9[11],ymm8[12],ymm9[13],ymm8[14],ymm9[15]+ vpslld $0x10, %ymm11, %ymm8+ vpblendw $0xaa, %ymm8, %ymm10, %ymm8 # ymm8 = ymm10[0],ymm8[1],ymm10[2],ymm8[3],ymm10[4],ymm8[5],ymm10[6],ymm8[7],ymm10[8],ymm8[9],ymm10[10],ymm8[11],ymm10[12],ymm8[13],ymm10[14],ymm8[15]+ vpsrld $0x10, %ymm10, %ymm10+ vpblendw $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7],ymm10[8],ymm11[9],ymm10[10],ymm11[11],ymm10[12],ymm11[13],ymm10[14],ymm11[15]+ vmovdqa 0x20(%rsi), %ymm12+ vpermd 0x120(%rsi), %ymm12, %ymm2+ vpermd 0x140(%rsi), %ymm12, %ymm10+ vpsubw %ymm3, %ymm5, %ymm12+ vpaddw %ymm5, %ymm3, %ymm3+ vpsubw %ymm4, %ymm7, %ymm13+ vpmullw %ymm2, %ymm12, %ymm5+ vpaddw %ymm7, %ymm4, %ymm4+ vpsubw %ymm6, %ymm9, %ymm14+ vpmullw %ymm2, %ymm13, %ymm7+ vpaddw %ymm9, %ymm6, %ymm6+ vpsubw %ymm8, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm9+ vpaddw %ymm11, %ymm8, %ymm8+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm10, %ymm12, %ymm12+ vpmulhw %ymm10, %ymm13, %ymm13+ vpmulhw %ymm10, %ymm14, %ymm14+ vpmulhw %ymm10, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm5, %ymm12, %ymm5+ vpsubw %ymm7, %ymm13, %ymm7+ vpsubw %ymm9, %ymm14, %ymm9+ vpsubw %ymm11, %ymm15, %ymm11+ movl $0x4ebf4ebf, %eax # imm = 0x4EBF4EBF+ vmovd %eax, %xmm1+ vpbroadcastd %xmm1, %ymm1+ vpmulhw %ymm1, %ymm3, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm3, %ymm3+ vmovsldup %ymm4, %ymm10 # ymm10 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm10, %ymm3, %ymm10 # ymm10 = ymm3[0],ymm10[1],ymm3[2],ymm10[3],ymm3[4],ymm10[5],ymm3[6],ymm10[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm8, %ymm3 # ymm3 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm6, %ymm3 # ymm3 = ymm6[0],ymm3[1],ymm6[2],ymm3[3],ymm6[4],ymm3[5],ymm6[6],ymm3[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm8, %ymm6, %ymm8 # ymm8 = ymm6[0],ymm8[1],ymm6[2],ymm8[3],ymm6[4],ymm8[5],ymm6[6],ymm8[7]+ vmovsldup %ymm7, %ymm6 # ymm6 = ymm7[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vmovsldup %ymm11, %ymm5 # ymm5 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm5[1],ymm9[2],ymm5[3],ymm9[4],ymm5[5],ymm9[6],ymm5[7]+ vpsrlq $0x20, %ymm9, %ymm9+ vpblendd $0xaa, %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[0],ymm11[1],ymm9[2],ymm11[3],ymm9[4],ymm11[5],ymm9[6],ymm11[7]+ vpermq $0x1b, 0xe0(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpermq $0x1b, 0x100(%rsi), %ymm9 # ymm9 = mem[3,2,1,0]+ vpsubw %ymm10, %ymm4, %ymm12+ vpaddw %ymm4, %ymm10, %ymm10+ vpsubw %ymm3, %ymm8, %ymm13+ vpmullw %ymm2, %ymm12, %ymm4+ vpaddw %ymm8, %ymm3, %ymm3+ vpsubw %ymm6, %ymm7, %ymm14+ vpmullw %ymm2, %ymm13, %ymm8+ vpaddw %ymm7, %ymm6, %ymm6+ vpsubw %ymm5, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm7+ vpaddw %ymm11, %ymm5, %ymm5+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm9, %ymm12, %ymm12+ vpmulhw %ymm9, %ymm13, %ymm13+ vpmulhw %ymm9, %ymm14, %ymm14+ vpmulhw %ymm9, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm4, %ymm4+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm4, %ymm12, %ymm4+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm7, %ymm14, %ymm7+ vpsubw %ymm11, %ymm15, %ymm11+ vpmulhw %ymm1, %ymm10, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm10, %ymm10+ vpunpcklqdq %ymm3, %ymm10, %ymm9 # ymm9 = ymm10[0],ymm3[0],ymm10[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[1],ymm3[1],ymm10[3],ymm3[3]+ vpunpcklqdq %ymm5, %ymm6, %ymm10 # ymm10 = ymm6[0],ymm5[0],ymm6[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[1],ymm5[1],ymm6[3],ymm5[3]+ vpunpcklqdq %ymm8, %ymm4, %ymm6 # ymm6 = ymm4[0],ymm8[0],ymm4[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[1],ymm8[1],ymm4[3],ymm8[3]+ vpunpcklqdq %ymm11, %ymm7, %ymm4 # ymm4 = ymm7[0],ymm11[0],ymm7[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[1],ymm11[1],ymm7[3],ymm11[3]+ vpermq $0x4e, 0xa0(%rsi), %ymm2 # ymm2 = mem[2,3,0,1]+ vpermq $0x4e, 0xc0(%rsi), %ymm7 # ymm7 = mem[2,3,0,1]+ vpsubw %ymm9, %ymm3, %ymm12+ vpaddw %ymm3, %ymm9, %ymm9+ vpsubw %ymm10, %ymm5, %ymm13+ vpmullw %ymm2, %ymm12, %ymm3+ vpaddw %ymm5, %ymm10, %ymm10+ vpsubw %ymm6, %ymm8, %ymm14+ vpmullw %ymm2, %ymm13, %ymm5+ vpaddw %ymm8, %ymm6, %ymm6+ vpsubw %ymm4, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm8+ vpaddw %ymm11, %ymm4, %ymm4+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm7, %ymm12, %ymm12+ vpmulhw %ymm7, %ymm13, %ymm13+ vpmulhw %ymm7, %ymm14, %ymm14+ vpmulhw %ymm7, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm3, %ymm3+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm3, %ymm12, %ymm3+ vpsubw %ymm5, %ymm13, %ymm5+ vpsubw %ymm8, %ymm14, %ymm8+ vpsubw %ymm11, %ymm15, %ymm11+ vpmulhw %ymm1, %ymm9, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm9, %ymm9+ vperm2i128 $0x20, %ymm10, %ymm9, %ymm7 # ymm7 = ymm9[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm9, %ymm10 # ymm10 = ymm9[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm4, %ymm6, %ymm9 # ymm9 = ymm6[0,1],ymm4[0,1]+ vperm2i128 $0x31, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[2,3],ymm4[2,3]+ vperm2i128 $0x20, %ymm5, %ymm3, %ymm6 # ymm6 = ymm3[0,1],ymm5[0,1]+ vperm2i128 $0x31, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[2,3],ymm5[2,3]+ vperm2i128 $0x20, %ymm11, %ymm8, %ymm3 # ymm3 = ymm8[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm8, %ymm11 # ymm11 = ymm8[2,3],ymm11[2,3]+ vmovdqa 0x60(%rsi), %ymm2+ vmovdqa 0x80(%rsi), %ymm8+ vpsubw %ymm7, %ymm10, %ymm12+ vpaddw %ymm10, %ymm7, %ymm7+ vpsubw %ymm9, %ymm4, %ymm13+ vpmullw %ymm2, %ymm12, %ymm10+ vpaddw %ymm4, %ymm9, %ymm9+ vpsubw %ymm6, %ymm5, %ymm14+ vpmullw %ymm2, %ymm13, %ymm4+ vpaddw %ymm5, %ymm6, %ymm6+ vpsubw %ymm3, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm5+ vpaddw %ymm11, %ymm3, %ymm3+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm8, %ymm12, %ymm12+ vpmulhw %ymm8, %ymm13, %ymm13+ vpmulhw %ymm8, %ymm14, %ymm14+ vpmulhw %ymm8, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm4, %ymm4+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm10, %ymm12, %ymm10+ vpsubw %ymm4, %ymm13, %ymm4+ vpsubw %ymm5, %ymm14, %ymm5+ vpsubw %ymm11, %ymm15, %ymm11+ vpmulhw %ymm1, %ymm7, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm7, %ymm7+ vmovdqa %ymm7, 0x100(%rdi)+ vmovdqa %ymm9, 0x120(%rdi)+ vmovdqa %ymm6, 0x140(%rdi)+ vmovdqa %ymm3, 0x160(%rdi)+ vmovdqa %ymm10, 0x180(%rdi)+ vmovdqa %ymm4, 0x1a0(%rdi)+ vmovdqa %ymm5, 0x1c0(%rdi)+ vmovdqa %ymm11, 0x1e0(%rdi)+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x100(%rdi), %ymm8+ vmovdqa 0x20(%rdi), %ymm5+ vmovdqa 0x120(%rdi), %ymm9+ vpbroadcastq 0x40(%rsi), %ymm2+ vmovdqa 0x40(%rdi), %ymm6+ vmovdqa 0x140(%rdi), %ymm10+ vmovdqa 0x60(%rdi), %ymm7+ vmovdqa 0x160(%rdi), %ymm11+ vpbroadcastq 0x48(%rsi), %ymm3+ vpsubw %ymm4, %ymm8, %ymm12+ vpaddw %ymm8, %ymm4, %ymm4+ vpsubw %ymm5, %ymm9, %ymm13+ vpmullw %ymm2, %ymm12, %ymm8+ vpaddw %ymm9, %ymm5, %ymm5+ vpsubw %ymm6, %ymm10, %ymm14+ vpmullw %ymm2, %ymm13, %ymm9+ vpaddw %ymm10, %ymm6, %ymm6+ vpsubw %ymm7, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm10+ vpaddw %ymm11, %ymm7, %ymm7+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm3, %ymm12, %ymm12+ vpmulhw %ymm3, %ymm13, %ymm13+ vpmulhw %ymm3, %ymm14, %ymm14+ vpmulhw %ymm3, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm8, %ymm12, %ymm8+ vpsubw %ymm9, %ymm13, %ymm9+ vpsubw %ymm10, %ymm14, %ymm10+ vpsubw %ymm11, %ymm15, %ymm11+ vmovdqa %ymm4, (%rdi)+ vmovdqa %ymm5, 0x20(%rdi)+ vmovdqa %ymm6, 0x40(%rdi)+ vmovdqa %ymm7, 0x60(%rdi)+ vmovdqa %ymm8, 0x100(%rdi)+ vmovdqa %ymm9, 0x120(%rdi)+ vmovdqa %ymm10, 0x140(%rdi)+ vmovdqa %ymm11, 0x160(%rdi)+ vmovdqa 0x80(%rdi), %ymm4+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm5+ vmovdqa 0x1a0(%rdi), %ymm9+ vpbroadcastq 0x40(%rsi), %ymm2+ vmovdqa 0xc0(%rdi), %ymm6+ vmovdqa 0x1c0(%rdi), %ymm10+ vmovdqa 0xe0(%rdi), %ymm7+ vmovdqa 0x1e0(%rdi), %ymm11+ vpbroadcastq 0x48(%rsi), %ymm3+ vpsubw %ymm4, %ymm8, %ymm12+ vpaddw %ymm8, %ymm4, %ymm4+ vpsubw %ymm5, %ymm9, %ymm13+ vpmullw %ymm2, %ymm12, %ymm8+ vpaddw %ymm9, %ymm5, %ymm5+ vpsubw %ymm6, %ymm10, %ymm14+ vpmullw %ymm2, %ymm13, %ymm9+ vpaddw %ymm10, %ymm6, %ymm6+ vpsubw %ymm7, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm10+ vpaddw %ymm11, %ymm7, %ymm7+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm3, %ymm12, %ymm12+ vpmulhw %ymm3, %ymm13, %ymm13+ vpmulhw %ymm3, %ymm14, %ymm14+ vpmulhw %ymm3, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm8, %ymm12, %ymm8+ vpsubw %ymm9, %ymm13, %ymm9+ vpsubw %ymm10, %ymm14, %ymm10+ vpsubw %ymm11, %ymm15, %ymm11+ vmovdqa %ymm4, 0x80(%rdi)+ vmovdqa %ymm5, 0xa0(%rdi)+ vmovdqa %ymm6, 0xc0(%rdi)+ vmovdqa %ymm7, 0xe0(%rdi)+ vmovdqa %ymm8, 0x180(%rdi)+ vmovdqa %ymm9, 0x1a0(%rdi)+ vmovdqa %ymm10, 0x1c0(%rdi)+ vmovdqa %ymm11, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(invntt_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_ntt_avx2_asm.S view
@@ -0,0 +1,661 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [AVX2_NTT]+ * Faster AVX2 optimized NTT multiplication for Ring-LWE lattice cryptography.+ * Gregor Seiler+ * https://eprint.iacr.org/2018/039+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Kyber implementation @[REF_AVX2].+ *+ * The core ideas behind the implementation are described in @[AVX2_NTT].+ */++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)+/*yaml+ Name: ntt_avx2_asm+ Description: x86_64 AVX2 forward NTT+ Signature: void mlk_ntt_avx2_asm(int16_t *r, const int16_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t *r+ description: Input/output polynomial (256 x int16_t)+ rsi:+ type: buffer+ size_bytes: 1248+ permissions: read-only+ c_parameter: const int16_t *qdata+ description: Precomputed constants (624 x int16_t)+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_ntt_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(ntt_avx2_asm)+MLK_ASM_FN_SYMBOL(ntt_avx2_asm)++ .cfi_startproc+ movl $0xd010d01, %eax # imm = 0xD010D01+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ vpbroadcastq 0x40(%rsi), %ymm15+ vmovdqa 0x100(%rdi), %ymm8+ vmovdqa 0x120(%rdi), %ymm9+ vmovdqa 0x140(%rdi), %ymm10+ vmovdqa 0x160(%rdi), %ymm11+ vpbroadcastq 0x48(%rsi), %ymm2+ vpmullw %ymm15, %ymm8, %ymm12+ vpmullw %ymm15, %ymm9, %ymm13+ vpmullw %ymm15, %ymm10, %ymm14+ vpmullw %ymm15, %ymm11, %ymm15+ vpmulhw %ymm2, %ymm8, %ymm8+ vpmulhw %ymm2, %ymm9, %ymm9+ vpmulhw %ymm2, %ymm10, %ymm10+ vpmulhw %ymm2, %ymm11, %ymm11+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x20(%rdi), %ymm5+ vmovdqa 0x40(%rdi), %ymm6+ vmovdqa 0x60(%rdi), %ymm7+ vpmulhw %ymm0, %ymm12, %ymm12+ vpmulhw %ymm0, %ymm13, %ymm13+ vpmulhw %ymm0, %ymm14, %ymm14+ vpmulhw %ymm0, %ymm15, %ymm15+ vpaddw %ymm8, %ymm4, %ymm3+ vpsubw %ymm8, %ymm4, %ymm8+ vpaddw %ymm9, %ymm5, %ymm4+ vpsubw %ymm9, %ymm5, %ymm9+ vpaddw %ymm10, %ymm6, %ymm5+ vpsubw %ymm10, %ymm6, %ymm10+ vpaddw %ymm11, %ymm7, %ymm6+ vpsubw %ymm11, %ymm7, %ymm11+ vpsubw %ymm12, %ymm3, %ymm3+ vpaddw %ymm12, %ymm8, %ymm8+ vpsubw %ymm13, %ymm4, %ymm4+ vpaddw %ymm13, %ymm9, %ymm9+ vpsubw %ymm14, %ymm5, %ymm5+ vpaddw %ymm14, %ymm10, %ymm10+ vpsubw %ymm15, %ymm6, %ymm6+ vpaddw %ymm15, %ymm11, %ymm11+ vmovdqa %ymm3, (%rdi)+ vmovdqa %ymm4, 0x20(%rdi)+ vmovdqa %ymm5, 0x40(%rdi)+ vmovdqa %ymm6, 0x60(%rdi)+ vmovdqa %ymm8, 0x100(%rdi)+ vmovdqa %ymm9, 0x120(%rdi)+ vmovdqa %ymm10, 0x140(%rdi)+ vmovdqa %ymm11, 0x160(%rdi)+ vpbroadcastq 0x40(%rsi), %ymm15+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1a0(%rdi), %ymm9+ vmovdqa 0x1c0(%rdi), %ymm10+ vmovdqa 0x1e0(%rdi), %ymm11+ vpbroadcastq 0x48(%rsi), %ymm2+ vpmullw %ymm15, %ymm8, %ymm12+ vpmullw %ymm15, %ymm9, %ymm13+ vpmullw %ymm15, %ymm10, %ymm14+ vpmullw %ymm15, %ymm11, %ymm15+ vpmulhw %ymm2, %ymm8, %ymm8+ vpmulhw %ymm2, %ymm9, %ymm9+ vpmulhw %ymm2, %ymm10, %ymm10+ vpmulhw %ymm2, %ymm11, %ymm11+ vmovdqa 0x80(%rdi), %ymm4+ vmovdqa 0xa0(%rdi), %ymm5+ vmovdqa 0xc0(%rdi), %ymm6+ vmovdqa 0xe0(%rdi), %ymm7+ vpmulhw %ymm0, %ymm12, %ymm12+ vpmulhw %ymm0, %ymm13, %ymm13+ vpmulhw %ymm0, %ymm14, %ymm14+ vpmulhw %ymm0, %ymm15, %ymm15+ vpaddw %ymm8, %ymm4, %ymm3+ vpsubw %ymm8, %ymm4, %ymm8+ vpaddw %ymm9, %ymm5, %ymm4+ vpsubw %ymm9, %ymm5, %ymm9+ vpaddw %ymm10, %ymm6, %ymm5+ vpsubw %ymm10, %ymm6, %ymm10+ vpaddw %ymm11, %ymm7, %ymm6+ vpsubw %ymm11, %ymm7, %ymm11+ vpsubw %ymm12, %ymm3, %ymm3+ vpaddw %ymm12, %ymm8, %ymm8+ vpsubw %ymm13, %ymm4, %ymm4+ vpaddw %ymm13, %ymm9, %ymm9+ vpsubw %ymm14, %ymm5, %ymm5+ vpaddw %ymm14, %ymm10, %ymm10+ vpsubw %ymm15, %ymm6, %ymm6+ vpaddw %ymm15, %ymm11, %ymm11+ vmovdqa %ymm3, 0x80(%rdi)+ vmovdqa %ymm4, 0xa0(%rdi)+ vmovdqa %ymm5, 0xc0(%rdi)+ vmovdqa %ymm6, 0xe0(%rdi)+ vmovdqa %ymm8, 0x180(%rdi)+ vmovdqa %ymm9, 0x1a0(%rdi)+ vmovdqa %ymm10, 0x1c0(%rdi)+ vmovdqa %ymm11, 0x1e0(%rdi)+ vmovdqa 0x60(%rsi), %ymm15+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm9+ vmovdqa 0xc0(%rdi), %ymm10+ vmovdqa 0xe0(%rdi), %ymm11+ vmovdqa 0x80(%rsi), %ymm2+ vpmullw %ymm15, %ymm8, %ymm12+ vpmullw %ymm15, %ymm9, %ymm13+ vpmullw %ymm15, %ymm10, %ymm14+ vpmullw %ymm15, %ymm11, %ymm15+ vpmulhw %ymm2, %ymm8, %ymm8+ vpmulhw %ymm2, %ymm9, %ymm9+ vpmulhw %ymm2, %ymm10, %ymm10+ vpmulhw %ymm2, %ymm11, %ymm11+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x20(%rdi), %ymm5+ vmovdqa 0x40(%rdi), %ymm6+ vmovdqa 0x60(%rdi), %ymm7+ vpmulhw %ymm0, %ymm12, %ymm12+ vpmulhw %ymm0, %ymm13, %ymm13+ vpmulhw %ymm0, %ymm14, %ymm14+ vpmulhw %ymm0, %ymm15, %ymm15+ vpaddw %ymm8, %ymm4, %ymm3+ vpsubw %ymm8, %ymm4, %ymm8+ vpaddw %ymm9, %ymm5, %ymm4+ vpsubw %ymm9, %ymm5, %ymm9+ vpaddw %ymm10, %ymm6, %ymm5+ vpsubw %ymm10, %ymm6, %ymm10+ vpaddw %ymm11, %ymm7, %ymm6+ vpsubw %ymm11, %ymm7, %ymm11+ vpsubw %ymm12, %ymm3, %ymm3+ vpaddw %ymm12, %ymm8, %ymm8+ vpsubw %ymm13, %ymm4, %ymm4+ vpaddw %ymm13, %ymm9, %ymm9+ vpsubw %ymm14, %ymm5, %ymm5+ vpaddw %ymm14, %ymm10, %ymm10+ vpsubw %ymm15, %ymm6, %ymm6+ vpaddw %ymm15, %ymm11, %ymm11+ vperm2i128 $0x20, %ymm10, %ymm5, %ymm7 # ymm7 = ymm5[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm5, %ymm10 # ymm10 = ymm5[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm6, %ymm11 # ymm11 = ymm6[2,3],ymm11[2,3]+ vmovdqa 0xa0(%rsi), %ymm15+ vmovdqa 0xc0(%rsi), %ymm2+ vpmullw %ymm15, %ymm7, %ymm12+ vpmullw %ymm15, %ymm10, %ymm13+ vpmullw %ymm15, %ymm5, %ymm14+ vpmullw %ymm15, %ymm11, %ymm15+ vpmulhw %ymm2, %ymm7, %ymm7+ vpmulhw %ymm2, %ymm10, %ymm10+ vpmulhw %ymm2, %ymm5, %ymm5+ vpmulhw %ymm2, %ymm11, %ymm11+ vperm2i128 $0x20, %ymm8, %ymm3, %ymm6 # ymm6 = ymm3[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm3, %ymm8 # ymm8 = ymm3[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm4, %ymm9 # ymm9 = ymm4[2,3],ymm9[2,3]+ vpmulhw %ymm0, %ymm12, %ymm12+ vpmulhw %ymm0, %ymm13, %ymm13+ vpmulhw %ymm0, %ymm14, %ymm14+ vpmulhw %ymm0, %ymm15, %ymm15+ vpaddw %ymm7, %ymm6, %ymm4+ vpsubw %ymm7, %ymm6, %ymm7+ vpaddw %ymm10, %ymm8, %ymm6+ vpsubw %ymm10, %ymm8, %ymm10+ vpaddw %ymm5, %ymm3, %ymm8+ vpsubw %ymm5, %ymm3, %ymm5+ vpaddw %ymm11, %ymm9, %ymm3+ vpsubw %ymm11, %ymm9, %ymm11+ vpsubw %ymm12, %ymm4, %ymm4+ vpaddw %ymm12, %ymm7, %ymm7+ vpsubw %ymm13, %ymm6, %ymm6+ vpaddw %ymm13, %ymm10, %ymm10+ vpsubw %ymm14, %ymm8, %ymm8+ vpaddw %ymm14, %ymm5, %ymm5+ vpsubw %ymm15, %ymm3, %ymm3+ vpaddw %ymm15, %ymm11, %ymm11+ vpunpcklqdq %ymm5, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm5[0],ymm8[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm8, %ymm5 # ymm5 = ymm8[1],ymm5[1],ymm8[3],ymm5[3]+ vpunpcklqdq %ymm11, %ymm3, %ymm8 # ymm8 = ymm3[0],ymm11[0],ymm3[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm3, %ymm11 # ymm11 = ymm3[1],ymm11[1],ymm3[3],ymm11[3]+ vmovdqa 0xe0(%rsi), %ymm15+ vmovdqa 0x100(%rsi), %ymm2+ vpmullw %ymm15, %ymm9, %ymm12+ vpmullw %ymm15, %ymm5, %ymm13+ vpmullw %ymm15, %ymm8, %ymm14+ vpmullw %ymm15, %ymm11, %ymm15+ vpmulhw %ymm2, %ymm9, %ymm9+ vpmulhw %ymm2, %ymm5, %ymm5+ vpmulhw %ymm2, %ymm8, %ymm8+ vpmulhw %ymm2, %ymm11, %ymm11+ vpunpcklqdq %ymm7, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm7[0],ymm4[2],ymm7[2]+ vpunpckhqdq %ymm7, %ymm4, %ymm7 # ymm7 = ymm4[1],ymm7[1],ymm4[3],ymm7[3]+ vpunpcklqdq %ymm10, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm10[0],ymm6[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[1],ymm10[1],ymm6[3],ymm10[3]+ vpmulhw %ymm0, %ymm12, %ymm12+ vpmulhw %ymm0, %ymm13, %ymm13+ vpmulhw %ymm0, %ymm14, %ymm14+ vpmulhw %ymm0, %ymm15, %ymm15+ vpaddw %ymm9, %ymm3, %ymm6+ vpsubw %ymm9, %ymm3, %ymm9+ vpaddw %ymm5, %ymm7, %ymm3+ vpsubw %ymm5, %ymm7, %ymm5+ vpaddw %ymm8, %ymm4, %ymm7+ vpsubw %ymm8, %ymm4, %ymm8+ vpaddw %ymm11, %ymm10, %ymm4+ vpsubw %ymm11, %ymm10, %ymm11+ vpsubw %ymm12, %ymm6, %ymm6+ vpaddw %ymm12, %ymm9, %ymm9+ vpsubw %ymm13, %ymm3, %ymm3+ vpaddw %ymm13, %ymm5, %ymm5+ vpsubw %ymm14, %ymm7, %ymm7+ vpaddw %ymm14, %ymm8, %ymm8+ vpsubw %ymm15, %ymm4, %ymm4+ vpaddw %ymm15, %ymm11, %ymm11+ vmovsldup %ymm8, %ymm10 # ymm10 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm10, %ymm7, %ymm10 # ymm10 = ymm7[0],ymm10[1],ymm7[2],ymm10[3],ymm7[4],ymm10[5],ymm7[6],ymm10[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm11, %ymm7 # ymm7 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm4, %ymm7 # ymm7 = ymm4[0],ymm7[1],ymm4[2],ymm7[3],ymm4[4],ymm7[5],ymm4[6],ymm7[7]+ vpsrlq $0x20, %ymm4, %ymm4+ vpblendd $0xaa, %ymm11, %ymm4, %ymm11 # ymm11 = ymm4[0],ymm11[1],ymm4[2],ymm11[3],ymm4[4],ymm11[5],ymm4[6],ymm11[7]+ vmovdqa 0x120(%rsi), %ymm15+ vmovdqa 0x140(%rsi), %ymm2+ vpmullw %ymm15, %ymm10, %ymm12+ vpmullw %ymm15, %ymm8, %ymm13+ vpmullw %ymm15, %ymm7, %ymm14+ vpmullw %ymm15, %ymm11, %ymm15+ vpmulhw %ymm2, %ymm10, %ymm10+ vpmulhw %ymm2, %ymm8, %ymm8+ vpmulhw %ymm2, %ymm7, %ymm7+ vpmulhw %ymm2, %ymm11, %ymm11+ vmovsldup %ymm9, %ymm4 # ymm4 = ymm9[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm9, %ymm6, %ymm9 # ymm9 = ymm6[0],ymm9[1],ymm6[2],ymm9[3],ymm6[4],ymm9[5],ymm6[6],ymm9[7]+ vmovsldup %ymm5, %ymm6 # ymm6 = ymm5[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm3, %ymm6 # ymm6 = ymm3[0],ymm6[1],ymm3[2],ymm6[3],ymm3[4],ymm6[5],ymm3[6],ymm6[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpmulhw %ymm0, %ymm12, %ymm12+ vpmulhw %ymm0, %ymm13, %ymm13+ vpmulhw %ymm0, %ymm14, %ymm14+ vpmulhw %ymm0, %ymm15, %ymm15+ vpaddw %ymm10, %ymm4, %ymm3+ vpsubw %ymm10, %ymm4, %ymm10+ vpaddw %ymm8, %ymm9, %ymm4+ vpsubw %ymm8, %ymm9, %ymm8+ vpaddw %ymm7, %ymm6, %ymm9+ vpsubw %ymm7, %ymm6, %ymm7+ vpaddw %ymm11, %ymm5, %ymm6+ vpsubw %ymm11, %ymm5, %ymm11+ vpsubw %ymm12, %ymm3, %ymm3+ vpaddw %ymm12, %ymm10, %ymm10+ vpsubw %ymm13, %ymm4, %ymm4+ vpaddw %ymm13, %ymm8, %ymm8+ vpsubw %ymm14, %ymm9, %ymm9+ vpaddw %ymm14, %ymm7, %ymm7+ vpsubw %ymm15, %ymm6, %ymm6+ vpaddw %ymm15, %ymm11, %ymm11+ vpslld $0x10, %ymm7, %ymm5+ vpblendw $0xaa, %ymm5, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm5[1],ymm9[2],ymm5[3],ymm9[4],ymm5[5],ymm9[6],ymm5[7],ymm9[8],ymm5[9],ymm9[10],ymm5[11],ymm9[12],ymm5[13],ymm9[14],ymm5[15]+ vpsrld $0x10, %ymm9, %ymm9+ vpblendw $0xaa, %ymm7, %ymm9, %ymm7 # ymm7 = ymm9[0],ymm7[1],ymm9[2],ymm7[3],ymm9[4],ymm7[5],ymm9[6],ymm7[7],ymm9[8],ymm7[9],ymm9[10],ymm7[11],ymm9[12],ymm7[13],ymm9[14],ymm7[15]+ vpslld $0x10, %ymm11, %ymm9+ vpblendw $0xaa, %ymm9, %ymm6, %ymm9 # ymm9 = ymm6[0],ymm9[1],ymm6[2],ymm9[3],ymm6[4],ymm9[5],ymm6[6],ymm9[7],ymm6[8],ymm9[9],ymm6[10],ymm9[11],ymm6[12],ymm9[13],ymm6[14],ymm9[15]+ vpsrld $0x10, %ymm6, %ymm6+ vpblendw $0xaa, %ymm11, %ymm6, %ymm11 # ymm11 = ymm6[0],ymm11[1],ymm6[2],ymm11[3],ymm6[4],ymm11[5],ymm6[6],ymm11[7],ymm6[8],ymm11[9],ymm6[10],ymm11[11],ymm6[12],ymm11[13],ymm6[14],ymm11[15]+ vmovdqa 0x160(%rsi), %ymm15+ vmovdqa 0x180(%rsi), %ymm2+ vpmullw %ymm15, %ymm5, %ymm12+ vpmullw %ymm15, %ymm7, %ymm13+ vpmullw %ymm15, %ymm9, %ymm14+ vpmullw %ymm15, %ymm11, %ymm15+ vpmulhw %ymm2, %ymm5, %ymm5+ vpmulhw %ymm2, %ymm7, %ymm7+ vpmulhw %ymm2, %ymm9, %ymm9+ vpmulhw %ymm2, %ymm11, %ymm11+ vpslld $0x10, %ymm10, %ymm6+ vpblendw $0xaa, %ymm6, %ymm3, %ymm6 # ymm6 = ymm3[0],ymm6[1],ymm3[2],ymm6[3],ymm3[4],ymm6[5],ymm3[6],ymm6[7],ymm3[8],ymm6[9],ymm3[10],ymm6[11],ymm3[12],ymm6[13],ymm3[14],ymm6[15]+ vpsrld $0x10, %ymm3, %ymm3+ vpblendw $0xaa, %ymm10, %ymm3, %ymm10 # ymm10 = ymm3[0],ymm10[1],ymm3[2],ymm10[3],ymm3[4],ymm10[5],ymm3[6],ymm10[7],ymm3[8],ymm10[9],ymm3[10],ymm10[11],ymm3[12],ymm10[13],ymm3[14],ymm10[15]+ vpslld $0x10, %ymm8, %ymm3+ vpblendw $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7],ymm4[8],ymm3[9],ymm4[10],ymm3[11],ymm4[12],ymm3[13],ymm4[14],ymm3[15]+ vpsrld $0x10, %ymm4, %ymm4+ vpblendw $0xaa, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm8[1],ymm4[2],ymm8[3],ymm4[4],ymm8[5],ymm4[6],ymm8[7],ymm4[8],ymm8[9],ymm4[10],ymm8[11],ymm4[12],ymm8[13],ymm4[14],ymm8[15]+ vpmulhw %ymm0, %ymm12, %ymm12+ vpmulhw %ymm0, %ymm13, %ymm13+ vpmulhw %ymm0, %ymm14, %ymm14+ vpmulhw %ymm0, %ymm15, %ymm15+ vpaddw %ymm5, %ymm6, %ymm4+ vpsubw %ymm5, %ymm6, %ymm5+ vpaddw %ymm7, %ymm10, %ymm6+ vpsubw %ymm7, %ymm10, %ymm7+ vpaddw %ymm9, %ymm3, %ymm10+ vpsubw %ymm9, %ymm3, %ymm9+ vpaddw %ymm11, %ymm8, %ymm3+ vpsubw %ymm11, %ymm8, %ymm11+ vpsubw %ymm12, %ymm4, %ymm4+ vpaddw %ymm12, %ymm5, %ymm5+ vpsubw %ymm13, %ymm6, %ymm6+ vpaddw %ymm13, %ymm7, %ymm7+ vpsubw %ymm14, %ymm10, %ymm10+ vpaddw %ymm14, %ymm9, %ymm9+ vpsubw %ymm15, %ymm3, %ymm3+ vpaddw %ymm15, %ymm11, %ymm11+ vmovdqa 0x1a0(%rsi), %ymm14+ vmovdqa 0x1e0(%rsi), %ymm15+ vmovdqa 0x1c0(%rsi), %ymm8+ vmovdqa 0x200(%rsi), %ymm2+ vpmullw %ymm14, %ymm10, %ymm12+ vpmullw %ymm14, %ymm3, %ymm13+ vpmullw %ymm15, %ymm9, %ymm14+ vpmullw %ymm15, %ymm11, %ymm15+ vpmulhw %ymm8, %ymm10, %ymm10+ vpmulhw %ymm8, %ymm3, %ymm3+ vpmulhw %ymm2, %ymm9, %ymm9+ vpmulhw %ymm2, %ymm11, %ymm11+ vpmulhw %ymm0, %ymm12, %ymm12+ vpmulhw %ymm0, %ymm13, %ymm13+ vpmulhw %ymm0, %ymm14, %ymm14+ vpmulhw %ymm0, %ymm15, %ymm15+ vpaddw %ymm10, %ymm4, %ymm8+ vpsubw %ymm10, %ymm4, %ymm10+ vpaddw %ymm3, %ymm6, %ymm4+ vpsubw %ymm3, %ymm6, %ymm3+ vpaddw %ymm9, %ymm5, %ymm6+ vpsubw %ymm9, %ymm5, %ymm9+ vpaddw %ymm11, %ymm7, %ymm5+ vpsubw %ymm11, %ymm7, %ymm11+ vpsubw %ymm12, %ymm8, %ymm8+ vpaddw %ymm12, %ymm10, %ymm10+ vpsubw %ymm13, %ymm4, %ymm4+ vpaddw %ymm13, %ymm3, %ymm3+ vpsubw %ymm14, %ymm6, %ymm6+ vpaddw %ymm14, %ymm9, %ymm9+ vpsubw %ymm15, %ymm5, %ymm5+ vpaddw %ymm15, %ymm11, %ymm11+ vmovdqa %ymm8, (%rdi)+ vmovdqa %ymm4, 0x20(%rdi)+ vmovdqa %ymm10, 0x40(%rdi)+ vmovdqa %ymm3, 0x60(%rdi)+ vmovdqa %ymm6, 0x80(%rdi)+ vmovdqa %ymm5, 0xa0(%rdi)+ vmovdqa %ymm9, 0xc0(%rdi)+ vmovdqa %ymm11, 0xe0(%rdi)+ vmovdqa 0x220(%rsi), %ymm15+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1a0(%rdi), %ymm9+ vmovdqa 0x1c0(%rdi), %ymm10+ vmovdqa 0x1e0(%rdi), %ymm11+ vmovdqa 0x240(%rsi), %ymm2+ vpmullw %ymm15, %ymm8, %ymm12+ vpmullw %ymm15, %ymm9, %ymm13+ vpmullw %ymm15, %ymm10, %ymm14+ vpmullw %ymm15, %ymm11, %ymm15+ vpmulhw %ymm2, %ymm8, %ymm8+ vpmulhw %ymm2, %ymm9, %ymm9+ vpmulhw %ymm2, %ymm10, %ymm10+ vpmulhw %ymm2, %ymm11, %ymm11+ vmovdqa 0x100(%rdi), %ymm4+ vmovdqa 0x120(%rdi), %ymm5+ vmovdqa 0x140(%rdi), %ymm6+ vmovdqa 0x160(%rdi), %ymm7+ vpmulhw %ymm0, %ymm12, %ymm12+ vpmulhw %ymm0, %ymm13, %ymm13+ vpmulhw %ymm0, %ymm14, %ymm14+ vpmulhw %ymm0, %ymm15, %ymm15+ vpaddw %ymm8, %ymm4, %ymm3+ vpsubw %ymm8, %ymm4, %ymm8+ vpaddw %ymm9, %ymm5, %ymm4+ vpsubw %ymm9, %ymm5, %ymm9+ vpaddw %ymm10, %ymm6, %ymm5+ vpsubw %ymm10, %ymm6, %ymm10+ vpaddw %ymm11, %ymm7, %ymm6+ vpsubw %ymm11, %ymm7, %ymm11+ vpsubw %ymm12, %ymm3, %ymm3+ vpaddw %ymm12, %ymm8, %ymm8+ vpsubw %ymm13, %ymm4, %ymm4+ vpaddw %ymm13, %ymm9, %ymm9+ vpsubw %ymm14, %ymm5, %ymm5+ vpaddw %ymm14, %ymm10, %ymm10+ vpsubw %ymm15, %ymm6, %ymm6+ vpaddw %ymm15, %ymm11, %ymm11+ vperm2i128 $0x20, %ymm10, %ymm5, %ymm7 # ymm7 = ymm5[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm5, %ymm10 # ymm10 = ymm5[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm6, %ymm11 # ymm11 = ymm6[2,3],ymm11[2,3]+ vmovdqa 0x260(%rsi), %ymm15+ vmovdqa 0x280(%rsi), %ymm2+ vpmullw %ymm15, %ymm7, %ymm12+ vpmullw %ymm15, %ymm10, %ymm13+ vpmullw %ymm15, %ymm5, %ymm14+ vpmullw %ymm15, %ymm11, %ymm15+ vpmulhw %ymm2, %ymm7, %ymm7+ vpmulhw %ymm2, %ymm10, %ymm10+ vpmulhw %ymm2, %ymm5, %ymm5+ vpmulhw %ymm2, %ymm11, %ymm11+ vperm2i128 $0x20, %ymm8, %ymm3, %ymm6 # ymm6 = ymm3[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm3, %ymm8 # ymm8 = ymm3[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm4, %ymm9 # ymm9 = ymm4[2,3],ymm9[2,3]+ vpmulhw %ymm0, %ymm12, %ymm12+ vpmulhw %ymm0, %ymm13, %ymm13+ vpmulhw %ymm0, %ymm14, %ymm14+ vpmulhw %ymm0, %ymm15, %ymm15+ vpaddw %ymm7, %ymm6, %ymm4+ vpsubw %ymm7, %ymm6, %ymm7+ vpaddw %ymm10, %ymm8, %ymm6+ vpsubw %ymm10, %ymm8, %ymm10+ vpaddw %ymm5, %ymm3, %ymm8+ vpsubw %ymm5, %ymm3, %ymm5+ vpaddw %ymm11, %ymm9, %ymm3+ vpsubw %ymm11, %ymm9, %ymm11+ vpsubw %ymm12, %ymm4, %ymm4+ vpaddw %ymm12, %ymm7, %ymm7+ vpsubw %ymm13, %ymm6, %ymm6+ vpaddw %ymm13, %ymm10, %ymm10+ vpsubw %ymm14, %ymm8, %ymm8+ vpaddw %ymm14, %ymm5, %ymm5+ vpsubw %ymm15, %ymm3, %ymm3+ vpaddw %ymm15, %ymm11, %ymm11+ vpunpcklqdq %ymm5, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm5[0],ymm8[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm8, %ymm5 # ymm5 = ymm8[1],ymm5[1],ymm8[3],ymm5[3]+ vpunpcklqdq %ymm11, %ymm3, %ymm8 # ymm8 = ymm3[0],ymm11[0],ymm3[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm3, %ymm11 # ymm11 = ymm3[1],ymm11[1],ymm3[3],ymm11[3]+ vmovdqa 0x2a0(%rsi), %ymm15+ vmovdqa 0x2c0(%rsi), %ymm2+ vpmullw %ymm15, %ymm9, %ymm12+ vpmullw %ymm15, %ymm5, %ymm13+ vpmullw %ymm15, %ymm8, %ymm14+ vpmullw %ymm15, %ymm11, %ymm15+ vpmulhw %ymm2, %ymm9, %ymm9+ vpmulhw %ymm2, %ymm5, %ymm5+ vpmulhw %ymm2, %ymm8, %ymm8+ vpmulhw %ymm2, %ymm11, %ymm11+ vpunpcklqdq %ymm7, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm7[0],ymm4[2],ymm7[2]+ vpunpckhqdq %ymm7, %ymm4, %ymm7 # ymm7 = ymm4[1],ymm7[1],ymm4[3],ymm7[3]+ vpunpcklqdq %ymm10, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm10[0],ymm6[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[1],ymm10[1],ymm6[3],ymm10[3]+ vpmulhw %ymm0, %ymm12, %ymm12+ vpmulhw %ymm0, %ymm13, %ymm13+ vpmulhw %ymm0, %ymm14, %ymm14+ vpmulhw %ymm0, %ymm15, %ymm15+ vpaddw %ymm9, %ymm3, %ymm6+ vpsubw %ymm9, %ymm3, %ymm9+ vpaddw %ymm5, %ymm7, %ymm3+ vpsubw %ymm5, %ymm7, %ymm5+ vpaddw %ymm8, %ymm4, %ymm7+ vpsubw %ymm8, %ymm4, %ymm8+ vpaddw %ymm11, %ymm10, %ymm4+ vpsubw %ymm11, %ymm10, %ymm11+ vpsubw %ymm12, %ymm6, %ymm6+ vpaddw %ymm12, %ymm9, %ymm9+ vpsubw %ymm13, %ymm3, %ymm3+ vpaddw %ymm13, %ymm5, %ymm5+ vpsubw %ymm14, %ymm7, %ymm7+ vpaddw %ymm14, %ymm8, %ymm8+ vpsubw %ymm15, %ymm4, %ymm4+ vpaddw %ymm15, %ymm11, %ymm11+ vmovsldup %ymm8, %ymm10 # ymm10 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm10, %ymm7, %ymm10 # ymm10 = ymm7[0],ymm10[1],ymm7[2],ymm10[3],ymm7[4],ymm10[5],ymm7[6],ymm10[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm11, %ymm7 # ymm7 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm4, %ymm7 # ymm7 = ymm4[0],ymm7[1],ymm4[2],ymm7[3],ymm4[4],ymm7[5],ymm4[6],ymm7[7]+ vpsrlq $0x20, %ymm4, %ymm4+ vpblendd $0xaa, %ymm11, %ymm4, %ymm11 # ymm11 = ymm4[0],ymm11[1],ymm4[2],ymm11[3],ymm4[4],ymm11[5],ymm4[6],ymm11[7]+ vmovdqa 0x2e0(%rsi), %ymm15+ vmovdqa 0x300(%rsi), %ymm2+ vpmullw %ymm15, %ymm10, %ymm12+ vpmullw %ymm15, %ymm8, %ymm13+ vpmullw %ymm15, %ymm7, %ymm14+ vpmullw %ymm15, %ymm11, %ymm15+ vpmulhw %ymm2, %ymm10, %ymm10+ vpmulhw %ymm2, %ymm8, %ymm8+ vpmulhw %ymm2, %ymm7, %ymm7+ vpmulhw %ymm2, %ymm11, %ymm11+ vmovsldup %ymm9, %ymm4 # ymm4 = ymm9[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm9, %ymm6, %ymm9 # ymm9 = ymm6[0],ymm9[1],ymm6[2],ymm9[3],ymm6[4],ymm9[5],ymm6[6],ymm9[7]+ vmovsldup %ymm5, %ymm6 # ymm6 = ymm5[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm3, %ymm6 # ymm6 = ymm3[0],ymm6[1],ymm3[2],ymm6[3],ymm3[4],ymm6[5],ymm3[6],ymm6[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpmulhw %ymm0, %ymm12, %ymm12+ vpmulhw %ymm0, %ymm13, %ymm13+ vpmulhw %ymm0, %ymm14, %ymm14+ vpmulhw %ymm0, %ymm15, %ymm15+ vpaddw %ymm10, %ymm4, %ymm3+ vpsubw %ymm10, %ymm4, %ymm10+ vpaddw %ymm8, %ymm9, %ymm4+ vpsubw %ymm8, %ymm9, %ymm8+ vpaddw %ymm7, %ymm6, %ymm9+ vpsubw %ymm7, %ymm6, %ymm7+ vpaddw %ymm11, %ymm5, %ymm6+ vpsubw %ymm11, %ymm5, %ymm11+ vpsubw %ymm12, %ymm3, %ymm3+ vpaddw %ymm12, %ymm10, %ymm10+ vpsubw %ymm13, %ymm4, %ymm4+ vpaddw %ymm13, %ymm8, %ymm8+ vpsubw %ymm14, %ymm9, %ymm9+ vpaddw %ymm14, %ymm7, %ymm7+ vpsubw %ymm15, %ymm6, %ymm6+ vpaddw %ymm15, %ymm11, %ymm11+ vpslld $0x10, %ymm7, %ymm5+ vpblendw $0xaa, %ymm5, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm5[1],ymm9[2],ymm5[3],ymm9[4],ymm5[5],ymm9[6],ymm5[7],ymm9[8],ymm5[9],ymm9[10],ymm5[11],ymm9[12],ymm5[13],ymm9[14],ymm5[15]+ vpsrld $0x10, %ymm9, %ymm9+ vpblendw $0xaa, %ymm7, %ymm9, %ymm7 # ymm7 = ymm9[0],ymm7[1],ymm9[2],ymm7[3],ymm9[4],ymm7[5],ymm9[6],ymm7[7],ymm9[8],ymm7[9],ymm9[10],ymm7[11],ymm9[12],ymm7[13],ymm9[14],ymm7[15]+ vpslld $0x10, %ymm11, %ymm9+ vpblendw $0xaa, %ymm9, %ymm6, %ymm9 # ymm9 = ymm6[0],ymm9[1],ymm6[2],ymm9[3],ymm6[4],ymm9[5],ymm6[6],ymm9[7],ymm6[8],ymm9[9],ymm6[10],ymm9[11],ymm6[12],ymm9[13],ymm6[14],ymm9[15]+ vpsrld $0x10, %ymm6, %ymm6+ vpblendw $0xaa, %ymm11, %ymm6, %ymm11 # ymm11 = ymm6[0],ymm11[1],ymm6[2],ymm11[3],ymm6[4],ymm11[5],ymm6[6],ymm11[7],ymm6[8],ymm11[9],ymm6[10],ymm11[11],ymm6[12],ymm11[13],ymm6[14],ymm11[15]+ vmovdqa 0x320(%rsi), %ymm15+ vmovdqa 0x340(%rsi), %ymm2+ vpmullw %ymm15, %ymm5, %ymm12+ vpmullw %ymm15, %ymm7, %ymm13+ vpmullw %ymm15, %ymm9, %ymm14+ vpmullw %ymm15, %ymm11, %ymm15+ vpmulhw %ymm2, %ymm5, %ymm5+ vpmulhw %ymm2, %ymm7, %ymm7+ vpmulhw %ymm2, %ymm9, %ymm9+ vpmulhw %ymm2, %ymm11, %ymm11+ vpslld $0x10, %ymm10, %ymm6+ vpblendw $0xaa, %ymm6, %ymm3, %ymm6 # ymm6 = ymm3[0],ymm6[1],ymm3[2],ymm6[3],ymm3[4],ymm6[5],ymm3[6],ymm6[7],ymm3[8],ymm6[9],ymm3[10],ymm6[11],ymm3[12],ymm6[13],ymm3[14],ymm6[15]+ vpsrld $0x10, %ymm3, %ymm3+ vpblendw $0xaa, %ymm10, %ymm3, %ymm10 # ymm10 = ymm3[0],ymm10[1],ymm3[2],ymm10[3],ymm3[4],ymm10[5],ymm3[6],ymm10[7],ymm3[8],ymm10[9],ymm3[10],ymm10[11],ymm3[12],ymm10[13],ymm3[14],ymm10[15]+ vpslld $0x10, %ymm8, %ymm3+ vpblendw $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7],ymm4[8],ymm3[9],ymm4[10],ymm3[11],ymm4[12],ymm3[13],ymm4[14],ymm3[15]+ vpsrld $0x10, %ymm4, %ymm4+ vpblendw $0xaa, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm8[1],ymm4[2],ymm8[3],ymm4[4],ymm8[5],ymm4[6],ymm8[7],ymm4[8],ymm8[9],ymm4[10],ymm8[11],ymm4[12],ymm8[13],ymm4[14],ymm8[15]+ vpmulhw %ymm0, %ymm12, %ymm12+ vpmulhw %ymm0, %ymm13, %ymm13+ vpmulhw %ymm0, %ymm14, %ymm14+ vpmulhw %ymm0, %ymm15, %ymm15+ vpaddw %ymm5, %ymm6, %ymm4+ vpsubw %ymm5, %ymm6, %ymm5+ vpaddw %ymm7, %ymm10, %ymm6+ vpsubw %ymm7, %ymm10, %ymm7+ vpaddw %ymm9, %ymm3, %ymm10+ vpsubw %ymm9, %ymm3, %ymm9+ vpaddw %ymm11, %ymm8, %ymm3+ vpsubw %ymm11, %ymm8, %ymm11+ vpsubw %ymm12, %ymm4, %ymm4+ vpaddw %ymm12, %ymm5, %ymm5+ vpsubw %ymm13, %ymm6, %ymm6+ vpaddw %ymm13, %ymm7, %ymm7+ vpsubw %ymm14, %ymm10, %ymm10+ vpaddw %ymm14, %ymm9, %ymm9+ vpsubw %ymm15, %ymm3, %ymm3+ vpaddw %ymm15, %ymm11, %ymm11+ vmovdqa 0x360(%rsi), %ymm14+ vmovdqa 0x3a0(%rsi), %ymm15+ vmovdqa 0x380(%rsi), %ymm8+ vmovdqa 0x3c0(%rsi), %ymm2+ vpmullw %ymm14, %ymm10, %ymm12+ vpmullw %ymm14, %ymm3, %ymm13+ vpmullw %ymm15, %ymm9, %ymm14+ vpmullw %ymm15, %ymm11, %ymm15+ vpmulhw %ymm8, %ymm10, %ymm10+ vpmulhw %ymm8, %ymm3, %ymm3+ vpmulhw %ymm2, %ymm9, %ymm9+ vpmulhw %ymm2, %ymm11, %ymm11+ vpmulhw %ymm0, %ymm12, %ymm12+ vpmulhw %ymm0, %ymm13, %ymm13+ vpmulhw %ymm0, %ymm14, %ymm14+ vpmulhw %ymm0, %ymm15, %ymm15+ vpaddw %ymm10, %ymm4, %ymm8+ vpsubw %ymm10, %ymm4, %ymm10+ vpaddw %ymm3, %ymm6, %ymm4+ vpsubw %ymm3, %ymm6, %ymm3+ vpaddw %ymm9, %ymm5, %ymm6+ vpsubw %ymm9, %ymm5, %ymm9+ vpaddw %ymm11, %ymm7, %ymm5+ vpsubw %ymm11, %ymm7, %ymm11+ vpsubw %ymm12, %ymm8, %ymm8+ vpaddw %ymm12, %ymm10, %ymm10+ vpsubw %ymm13, %ymm4, %ymm4+ vpaddw %ymm13, %ymm3, %ymm3+ vpsubw %ymm14, %ymm6, %ymm6+ vpaddw %ymm14, %ymm9, %ymm9+ vpsubw %ymm15, %ymm5, %ymm5+ vpaddw %ymm15, %ymm11, %ymm11+ vmovdqa %ymm8, 0x100(%rdi)+ vmovdqa %ymm4, 0x120(%rdi)+ vmovdqa %ymm10, 0x140(%rdi)+ vmovdqa %ymm3, 0x160(%rdi)+ vmovdqa %ymm6, 0x180(%rdi)+ vmovdqa %ymm5, 0x1a0(%rdi)+ vmovdqa %ymm9, 0x1c0(%rdi)+ vmovdqa %ymm11, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(ntt_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_nttfrombytes_avx2_asm.S view
@@ -0,0 +1,217 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Kyber implementation @[REF_AVX2].+ */++#include "../../../common.h"++#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API))+/*yaml+ Name: nttfrombytes_avx2_asm+ Description: x86_64 AVX2 polynomial deserialization in NTT domain+ Signature: void mlk_nttfrombytes_avx2_asm(int16_t *r, const uint8_t *a)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 512+ permissions: write-only+ c_parameter: int16_t *r+ description: Output polynomial (256 x int16_t)+ rsi:+ type: buffer+ size_bytes: 384+ permissions: read-only+ c_parameter: const uint8_t *a+ description: Input byte array (MLKEM_POLYBYTES = 384)+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_nttfrombytes_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(nttfrombytes_avx2_asm)+MLK_ASM_FN_SYMBOL(nttfrombytes_avx2_asm)++ .cfi_startproc+ movl $0xfff0fff, %eax # imm = 0xFFF0FFF+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ vmovdqu (%rsi), %ymm4+ vmovdqu 0x20(%rsi), %ymm5+ vmovdqu 0x40(%rsi), %ymm6+ vmovdqu 0x60(%rsi), %ymm7+ vmovdqu 0x80(%rsi), %ymm8+ vmovdqu 0xa0(%rsi), %ymm9+ vperm2i128 $0x20, %ymm7, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm7[0,1]+ vperm2i128 $0x31, %ymm7, %ymm4, %ymm7 # ymm7 = ymm4[2,3],ymm7[2,3]+ vperm2i128 $0x20, %ymm8, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm5, %ymm8 # ymm8 = ymm5[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm6, %ymm9 # ymm9 = ymm6[2,3],ymm9[2,3]+ vpunpcklqdq %ymm8, %ymm3, %ymm6 # ymm6 = ymm3[0],ymm8[0],ymm3[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm3, %ymm8 # ymm8 = ymm3[1],ymm8[1],ymm3[3],ymm8[3]+ vpunpcklqdq %ymm5, %ymm7, %ymm3 # ymm3 = ymm7[0],ymm5[0],ymm7[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm7, %ymm5 # ymm5 = ymm7[1],ymm5[1],ymm7[3],ymm5[3]+ vpunpcklqdq %ymm9, %ymm4, %ymm7 # ymm7 = ymm4[0],ymm9[0],ymm4[2],ymm9[2]+ vpunpckhqdq %ymm9, %ymm4, %ymm9 # ymm9 = ymm4[1],ymm9[1],ymm4[3],ymm9[3]+ vmovsldup %ymm5, %ymm4 # ymm4 = ymm5[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[0],ymm5[1],ymm6[2],ymm5[3],ymm6[4],ymm5[5],ymm6[6],ymm5[7]+ vmovsldup %ymm7, %ymm6 # ymm6 = ymm7[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7]+ vpsrlq $0x20, %ymm8, %ymm8+ vpblendd $0xaa, %ymm7, %ymm8, %ymm7 # ymm7 = ymm8[0],ymm7[1],ymm8[2],ymm7[3],ymm8[4],ymm7[5],ymm8[6],ymm7[7]+ vmovsldup %ymm9, %ymm8 # ymm8 = ymm9[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm8, %ymm3, %ymm8 # ymm8 = ymm3[0],ymm8[1],ymm3[2],ymm8[3],ymm3[4],ymm8[5],ymm3[6],ymm8[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm9, %ymm3, %ymm9 # ymm9 = ymm3[0],ymm9[1],ymm3[2],ymm9[3],ymm3[4],ymm9[5],ymm3[6],ymm9[7]+ vpslld $0x10, %ymm7, %ymm10+ vpblendw $0xaa, %ymm10, %ymm4, %ymm10 # ymm10 = ymm4[0],ymm10[1],ymm4[2],ymm10[3],ymm4[4],ymm10[5],ymm4[6],ymm10[7],ymm4[8],ymm10[9],ymm4[10],ymm10[11],ymm4[12],ymm10[13],ymm4[14],ymm10[15]+ vpsrld $0x10, %ymm4, %ymm4+ vpblendw $0xaa, %ymm7, %ymm4, %ymm7 # ymm7 = ymm4[0],ymm7[1],ymm4[2],ymm7[3],ymm4[4],ymm7[5],ymm4[6],ymm7[7],ymm4[8],ymm7[9],ymm4[10],ymm7[11],ymm4[12],ymm7[13],ymm4[14],ymm7[15]+ vpslld $0x10, %ymm8, %ymm4+ vpblendw $0xaa, %ymm4, %ymm5, %ymm4 # ymm4 = ymm5[0],ymm4[1],ymm5[2],ymm4[3],ymm5[4],ymm4[5],ymm5[6],ymm4[7],ymm5[8],ymm4[9],ymm5[10],ymm4[11],ymm5[12],ymm4[13],ymm5[14],ymm4[15]+ vpsrld $0x10, %ymm5, %ymm5+ vpblendw $0xaa, %ymm8, %ymm5, %ymm8 # ymm8 = ymm5[0],ymm8[1],ymm5[2],ymm8[3],ymm5[4],ymm8[5],ymm5[6],ymm8[7],ymm5[8],ymm8[9],ymm5[10],ymm8[11],ymm5[12],ymm8[13],ymm5[14],ymm8[15]+ vpslld $0x10, %ymm9, %ymm5+ vpblendw $0xaa, %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[0],ymm5[1],ymm6[2],ymm5[3],ymm6[4],ymm5[5],ymm6[6],ymm5[7],ymm6[8],ymm5[9],ymm6[10],ymm5[11],ymm6[12],ymm5[13],ymm6[14],ymm5[15]+ vpsrld $0x10, %ymm6, %ymm6+ vpblendw $0xaa, %ymm9, %ymm6, %ymm9 # ymm9 = ymm6[0],ymm9[1],ymm6[2],ymm9[3],ymm6[4],ymm9[5],ymm6[6],ymm9[7],ymm6[8],ymm9[9],ymm6[10],ymm9[11],ymm6[12],ymm9[13],ymm6[14],ymm9[15]+ vpsrlw $0xc, %ymm10, %ymm11+ vpsllw $0x4, %ymm7, %ymm12+ vpor %ymm11, %ymm12, %ymm11+ vpand %ymm0, %ymm10, %ymm10+ vpand %ymm0, %ymm11, %ymm11+ vpsrlw $0x8, %ymm7, %ymm12+ vpsllw $0x8, %ymm4, %ymm13+ vpor %ymm12, %ymm13, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpsrlw $0x4, %ymm4, %ymm13+ vpand %ymm0, %ymm13, %ymm13+ vpsrlw $0xc, %ymm8, %ymm14+ vpsllw $0x4, %ymm5, %ymm15+ vpor %ymm14, %ymm15, %ymm14+ vpand %ymm0, %ymm8, %ymm8+ vpand %ymm0, %ymm14, %ymm14+ vpsrlw $0x8, %ymm5, %ymm15+ vpsllw $0x8, %ymm9, %ymm1+ vpor %ymm15, %ymm1, %ymm15+ vpand %ymm0, %ymm15, %ymm15+ vpsrlw $0x4, %ymm9, %ymm1+ vpand %ymm0, %ymm1, %ymm1+ vmovdqa %ymm10, (%rdi)+ vmovdqa %ymm11, 0x20(%rdi)+ vmovdqa %ymm12, 0x40(%rdi)+ vmovdqa %ymm13, 0x60(%rdi)+ vmovdqa %ymm8, 0x80(%rdi)+ vmovdqa %ymm14, 0xa0(%rdi)+ vmovdqa %ymm15, 0xc0(%rdi)+ vmovdqa %ymm1, 0xe0(%rdi)+ vmovdqu 0xc0(%rsi), %ymm4+ vmovdqu 0xe0(%rsi), %ymm5+ vmovdqu 0x100(%rsi), %ymm6+ vmovdqu 0x120(%rsi), %ymm7+ vmovdqu 0x140(%rsi), %ymm8+ vmovdqu 0x160(%rsi), %ymm9+ vperm2i128 $0x20, %ymm7, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm7[0,1]+ vperm2i128 $0x31, %ymm7, %ymm4, %ymm7 # ymm7 = ymm4[2,3],ymm7[2,3]+ vperm2i128 $0x20, %ymm8, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm5, %ymm8 # ymm8 = ymm5[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm6, %ymm9 # ymm9 = ymm6[2,3],ymm9[2,3]+ vpunpcklqdq %ymm8, %ymm3, %ymm6 # ymm6 = ymm3[0],ymm8[0],ymm3[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm3, %ymm8 # ymm8 = ymm3[1],ymm8[1],ymm3[3],ymm8[3]+ vpunpcklqdq %ymm5, %ymm7, %ymm3 # ymm3 = ymm7[0],ymm5[0],ymm7[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm7, %ymm5 # ymm5 = ymm7[1],ymm5[1],ymm7[3],ymm5[3]+ vpunpcklqdq %ymm9, %ymm4, %ymm7 # ymm7 = ymm4[0],ymm9[0],ymm4[2],ymm9[2]+ vpunpckhqdq %ymm9, %ymm4, %ymm9 # ymm9 = ymm4[1],ymm9[1],ymm4[3],ymm9[3]+ vmovsldup %ymm5, %ymm4 # ymm4 = ymm5[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[0],ymm5[1],ymm6[2],ymm5[3],ymm6[4],ymm5[5],ymm6[6],ymm5[7]+ vmovsldup %ymm7, %ymm6 # ymm6 = ymm7[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7]+ vpsrlq $0x20, %ymm8, %ymm8+ vpblendd $0xaa, %ymm7, %ymm8, %ymm7 # ymm7 = ymm8[0],ymm7[1],ymm8[2],ymm7[3],ymm8[4],ymm7[5],ymm8[6],ymm7[7]+ vmovsldup %ymm9, %ymm8 # ymm8 = ymm9[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm8, %ymm3, %ymm8 # ymm8 = ymm3[0],ymm8[1],ymm3[2],ymm8[3],ymm3[4],ymm8[5],ymm3[6],ymm8[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm9, %ymm3, %ymm9 # ymm9 = ymm3[0],ymm9[1],ymm3[2],ymm9[3],ymm3[4],ymm9[5],ymm3[6],ymm9[7]+ vpslld $0x10, %ymm7, %ymm10+ vpblendw $0xaa, %ymm10, %ymm4, %ymm10 # ymm10 = ymm4[0],ymm10[1],ymm4[2],ymm10[3],ymm4[4],ymm10[5],ymm4[6],ymm10[7],ymm4[8],ymm10[9],ymm4[10],ymm10[11],ymm4[12],ymm10[13],ymm4[14],ymm10[15]+ vpsrld $0x10, %ymm4, %ymm4+ vpblendw $0xaa, %ymm7, %ymm4, %ymm7 # ymm7 = ymm4[0],ymm7[1],ymm4[2],ymm7[3],ymm4[4],ymm7[5],ymm4[6],ymm7[7],ymm4[8],ymm7[9],ymm4[10],ymm7[11],ymm4[12],ymm7[13],ymm4[14],ymm7[15]+ vpslld $0x10, %ymm8, %ymm4+ vpblendw $0xaa, %ymm4, %ymm5, %ymm4 # ymm4 = ymm5[0],ymm4[1],ymm5[2],ymm4[3],ymm5[4],ymm4[5],ymm5[6],ymm4[7],ymm5[8],ymm4[9],ymm5[10],ymm4[11],ymm5[12],ymm4[13],ymm5[14],ymm4[15]+ vpsrld $0x10, %ymm5, %ymm5+ vpblendw $0xaa, %ymm8, %ymm5, %ymm8 # ymm8 = ymm5[0],ymm8[1],ymm5[2],ymm8[3],ymm5[4],ymm8[5],ymm5[6],ymm8[7],ymm5[8],ymm8[9],ymm5[10],ymm8[11],ymm5[12],ymm8[13],ymm5[14],ymm8[15]+ vpslld $0x10, %ymm9, %ymm5+ vpblendw $0xaa, %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[0],ymm5[1],ymm6[2],ymm5[3],ymm6[4],ymm5[5],ymm6[6],ymm5[7],ymm6[8],ymm5[9],ymm6[10],ymm5[11],ymm6[12],ymm5[13],ymm6[14],ymm5[15]+ vpsrld $0x10, %ymm6, %ymm6+ vpblendw $0xaa, %ymm9, %ymm6, %ymm9 # ymm9 = ymm6[0],ymm9[1],ymm6[2],ymm9[3],ymm6[4],ymm9[5],ymm6[6],ymm9[7],ymm6[8],ymm9[9],ymm6[10],ymm9[11],ymm6[12],ymm9[13],ymm6[14],ymm9[15]+ vpsrlw $0xc, %ymm10, %ymm11+ vpsllw $0x4, %ymm7, %ymm12+ vpor %ymm11, %ymm12, %ymm11+ vpand %ymm0, %ymm10, %ymm10+ vpand %ymm0, %ymm11, %ymm11+ vpsrlw $0x8, %ymm7, %ymm12+ vpsllw $0x8, %ymm4, %ymm13+ vpor %ymm12, %ymm13, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpsrlw $0x4, %ymm4, %ymm13+ vpand %ymm0, %ymm13, %ymm13+ vpsrlw $0xc, %ymm8, %ymm14+ vpsllw $0x4, %ymm5, %ymm15+ vpor %ymm14, %ymm15, %ymm14+ vpand %ymm0, %ymm8, %ymm8+ vpand %ymm0, %ymm14, %ymm14+ vpsrlw $0x8, %ymm5, %ymm15+ vpsllw $0x8, %ymm9, %ymm1+ vpor %ymm15, %ymm1, %ymm15+ vpand %ymm0, %ymm15, %ymm15+ vpsrlw $0x4, %ymm9, %ymm1+ vpand %ymm0, %ymm1, %ymm1+ vmovdqa %ymm10, 0x100(%rdi)+ vmovdqa %ymm11, 0x120(%rdi)+ vmovdqa %ymm12, 0x140(%rdi)+ vmovdqa %ymm13, 0x160(%rdi)+ vmovdqa %ymm8, 0x180(%rdi)+ vmovdqa %ymm14, 0x1a0(%rdi)+ vmovdqa %ymm15, 0x1c0(%rdi)+ vmovdqa %ymm1, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(nttfrombytes_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_ntttobytes_avx2_asm.S view
@@ -0,0 +1,205 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Kyber implementation @[REF_AVX2].+ */++#include "../../../common.h"++#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (!defined(MLK_CONFIG_NO_KEYPAIR_API) || \+ !defined(MLK_CONFIG_NO_ENCAPS_API))+/*yaml+ Name: ntttobytes_avx2_asm+ Description: x86_64 AVX2 polynomial serialization in NTT domain+ Signature: void mlk_ntttobytes_avx2_asm(uint8_t *r, const int16_t *a)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 384+ permissions: write-only+ c_parameter: uint8_t *r+ description: Output byte array (MLKEM_POLYBYTES = 384)+ rsi:+ type: buffer+ size_bytes: 512+ permissions: read-only+ c_parameter: const int16_t *a+ description: Input polynomial (256 x int16_t)+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_ntttobytes_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(ntttobytes_avx2_asm)+MLK_ASM_FN_SYMBOL(ntttobytes_avx2_asm)++ .cfi_startproc+ movl $0xd010d01, %eax # imm = 0xD010D01+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ vmovdqa (%rsi), %ymm5+ vmovdqa 0x20(%rsi), %ymm6+ vmovdqa 0x40(%rsi), %ymm7+ vmovdqa 0x60(%rsi), %ymm8+ vmovdqa 0x80(%rsi), %ymm9+ vmovdqa 0xa0(%rsi), %ymm10+ vmovdqa 0xc0(%rsi), %ymm11+ vmovdqa 0xe0(%rsi), %ymm12+ vpsllw $0xc, %ymm6, %ymm4+ vpor %ymm4, %ymm5, %ymm4+ vpsrlw $0x4, %ymm6, %ymm5+ vpsllw $0x8, %ymm7, %ymm6+ vpor %ymm5, %ymm6, %ymm5+ vpsrlw $0x8, %ymm7, %ymm6+ vpsllw $0x4, %ymm8, %ymm7+ vpor %ymm6, %ymm7, %ymm6+ vpsllw $0xc, %ymm10, %ymm7+ vpor %ymm7, %ymm9, %ymm7+ vpsrlw $0x4, %ymm10, %ymm8+ vpsllw $0x8, %ymm11, %ymm9+ vpor %ymm8, %ymm9, %ymm8+ vpsrlw $0x8, %ymm11, %ymm9+ vpsllw $0x4, %ymm12, %ymm10+ vpor %ymm9, %ymm10, %ymm9+ vpslld $0x10, %ymm5, %ymm3+ vpblendw $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7],ymm4[8],ymm3[9],ymm4[10],ymm3[11],ymm4[12],ymm3[13],ymm4[14],ymm3[15]+ vpsrld $0x10, %ymm4, %ymm4+ vpblendw $0xaa, %ymm5, %ymm4, %ymm5 # ymm5 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7],ymm4[8],ymm5[9],ymm4[10],ymm5[11],ymm4[12],ymm5[13],ymm4[14],ymm5[15]+ vpslld $0x10, %ymm7, %ymm4+ vpblendw $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7],ymm6[8],ymm4[9],ymm6[10],ymm4[11],ymm6[12],ymm4[13],ymm6[14],ymm4[15]+ vpsrld $0x10, %ymm6, %ymm6+ vpblendw $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7],ymm6[8],ymm7[9],ymm6[10],ymm7[11],ymm6[12],ymm7[13],ymm6[14],ymm7[15]+ vpslld $0x10, %ymm9, %ymm6+ vpblendw $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7],ymm8[8],ymm6[9],ymm8[10],ymm6[11],ymm8[12],ymm6[13],ymm8[14],ymm6[15]+ vpsrld $0x10, %ymm8, %ymm8+ vpblendw $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7],ymm8[8],ymm9[9],ymm8[10],ymm9[11],ymm8[12],ymm9[13],ymm8[14],ymm9[15]+ vmovsldup %ymm4, %ymm8 # ymm8 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm8, %ymm3, %ymm8 # ymm8 = ymm3[0],ymm8[1],ymm3[2],ymm8[3],ymm3[4],ymm8[5],ymm3[6],ymm8[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm5, %ymm3 # ymm3 = ymm5[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm6, %ymm3 # ymm3 = ymm6[0],ymm3[1],ymm6[2],ymm3[3],ymm6[4],ymm3[5],ymm6[6],ymm3[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[0],ymm5[1],ymm6[2],ymm5[3],ymm6[4],ymm5[5],ymm6[6],ymm5[7]+ vmovsldup %ymm9, %ymm6 # ymm6 = ymm9[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm7, %ymm6 # ymm6 = ymm7[0],ymm6[1],ymm7[2],ymm6[3],ymm7[4],ymm6[5],ymm7[6],ymm6[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpunpcklqdq %ymm3, %ymm8, %ymm7 # ymm7 = ymm8[0],ymm3[0],ymm8[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm8, %ymm3 # ymm3 = ymm8[1],ymm3[1],ymm8[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm6, %ymm8 # ymm8 = ymm6[0],ymm4[0],ymm6[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[1],ymm4[1],ymm6[3],ymm4[3]+ vpunpcklqdq %ymm9, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm9[0],ymm5[2],ymm9[2]+ vpunpckhqdq %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[1],ymm9[1],ymm5[3],ymm9[3]+ vperm2i128 $0x20, %ymm8, %ymm7, %ymm5 # ymm5 = ymm7[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm3, %ymm6, %ymm7 # ymm7 = ymm6[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm6, %ymm3 # ymm3 = ymm6[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm9, %ymm4, %ymm6 # ymm6 = ymm4[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm4, %ymm9 # ymm9 = ymm4[2,3],ymm9[2,3]+ vmovdqu %ymm5, (%rdi)+ vmovdqu %ymm7, 0x20(%rdi)+ vmovdqu %ymm6, 0x40(%rdi)+ vmovdqu %ymm8, 0x60(%rdi)+ vmovdqu %ymm3, 0x80(%rdi)+ vmovdqu %ymm9, 0xa0(%rdi)+ vmovdqa 0x100(%rsi), %ymm5+ vmovdqa 0x120(%rsi), %ymm6+ vmovdqa 0x140(%rsi), %ymm7+ vmovdqa 0x160(%rsi), %ymm8+ vmovdqa 0x180(%rsi), %ymm9+ vmovdqa 0x1a0(%rsi), %ymm10+ vmovdqa 0x1c0(%rsi), %ymm11+ vmovdqa 0x1e0(%rsi), %ymm12+ vpsllw $0xc, %ymm6, %ymm4+ vpor %ymm4, %ymm5, %ymm4+ vpsrlw $0x4, %ymm6, %ymm5+ vpsllw $0x8, %ymm7, %ymm6+ vpor %ymm5, %ymm6, %ymm5+ vpsrlw $0x8, %ymm7, %ymm6+ vpsllw $0x4, %ymm8, %ymm7+ vpor %ymm6, %ymm7, %ymm6+ vpsllw $0xc, %ymm10, %ymm7+ vpor %ymm7, %ymm9, %ymm7+ vpsrlw $0x4, %ymm10, %ymm8+ vpsllw $0x8, %ymm11, %ymm9+ vpor %ymm8, %ymm9, %ymm8+ vpsrlw $0x8, %ymm11, %ymm9+ vpsllw $0x4, %ymm12, %ymm10+ vpor %ymm9, %ymm10, %ymm9+ vpslld $0x10, %ymm5, %ymm3+ vpblendw $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7],ymm4[8],ymm3[9],ymm4[10],ymm3[11],ymm4[12],ymm3[13],ymm4[14],ymm3[15]+ vpsrld $0x10, %ymm4, %ymm4+ vpblendw $0xaa, %ymm5, %ymm4, %ymm5 # ymm5 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7],ymm4[8],ymm5[9],ymm4[10],ymm5[11],ymm4[12],ymm5[13],ymm4[14],ymm5[15]+ vpslld $0x10, %ymm7, %ymm4+ vpblendw $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7],ymm6[8],ymm4[9],ymm6[10],ymm4[11],ymm6[12],ymm4[13],ymm6[14],ymm4[15]+ vpsrld $0x10, %ymm6, %ymm6+ vpblendw $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7],ymm6[8],ymm7[9],ymm6[10],ymm7[11],ymm6[12],ymm7[13],ymm6[14],ymm7[15]+ vpslld $0x10, %ymm9, %ymm6+ vpblendw $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7],ymm8[8],ymm6[9],ymm8[10],ymm6[11],ymm8[12],ymm6[13],ymm8[14],ymm6[15]+ vpsrld $0x10, %ymm8, %ymm8+ vpblendw $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7],ymm8[8],ymm9[9],ymm8[10],ymm9[11],ymm8[12],ymm9[13],ymm8[14],ymm9[15]+ vmovsldup %ymm4, %ymm8 # ymm8 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm8, %ymm3, %ymm8 # ymm8 = ymm3[0],ymm8[1],ymm3[2],ymm8[3],ymm3[4],ymm8[5],ymm3[6],ymm8[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm5, %ymm3 # ymm3 = ymm5[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm6, %ymm3 # ymm3 = ymm6[0],ymm3[1],ymm6[2],ymm3[3],ymm6[4],ymm3[5],ymm6[6],ymm3[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[0],ymm5[1],ymm6[2],ymm5[3],ymm6[4],ymm5[5],ymm6[6],ymm5[7]+ vmovsldup %ymm9, %ymm6 # ymm6 = ymm9[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm7, %ymm6 # ymm6 = ymm7[0],ymm6[1],ymm7[2],ymm6[3],ymm7[4],ymm6[5],ymm7[6],ymm6[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpunpcklqdq %ymm3, %ymm8, %ymm7 # ymm7 = ymm8[0],ymm3[0],ymm8[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm8, %ymm3 # ymm3 = ymm8[1],ymm3[1],ymm8[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm6, %ymm8 # ymm8 = ymm6[0],ymm4[0],ymm6[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[1],ymm4[1],ymm6[3],ymm4[3]+ vpunpcklqdq %ymm9, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm9[0],ymm5[2],ymm9[2]+ vpunpckhqdq %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[1],ymm9[1],ymm5[3],ymm9[3]+ vperm2i128 $0x20, %ymm8, %ymm7, %ymm5 # ymm5 = ymm7[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm3, %ymm6, %ymm7 # ymm7 = ymm6[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm6, %ymm3 # ymm3 = ymm6[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm9, %ymm4, %ymm6 # ymm6 = ymm4[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm4, %ymm9 # ymm9 = ymm4[2,3],ymm9[2,3]+ vmovdqu %ymm5, 0xc0(%rdi)+ vmovdqu %ymm7, 0xe0(%rdi)+ vmovdqu %ymm6, 0x100(%rdi)+ vmovdqu %ymm8, 0x120(%rdi)+ vmovdqu %ymm3, 0x140(%rdi)+ vmovdqu %ymm9, 0x160(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(ntttobytes_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && (!MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_nttunpack_avx2_asm.S view
@@ -0,0 +1,190 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Kyber implementation @[REF_AVX2].+ */++#include "../../../common.h"++#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)+/*yaml+ Name: nttunpack_avx2_asm+ Description: x86_64 AVX2 NTT unpack from custom to bitreversed order+ Signature: void mlk_nttunpack_avx2_asm(int16_t *r)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t *r+ description: Input/output polynomial (256 x int16_t)+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_nttunpack_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(nttunpack_avx2_asm)+MLK_ASM_FN_SYMBOL(nttunpack_avx2_asm)++ .cfi_startproc+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x20(%rdi), %ymm5+ vmovdqa 0x40(%rdi), %ymm6+ vmovdqa 0x60(%rdi), %ymm7+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm9+ vmovdqa 0xc0(%rdi), %ymm10+ vmovdqa 0xe0(%rdi), %ymm11+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vpslld $0x10, %ymm5, %ymm10+ vpblendw $0xaa, %ymm10, %ymm9, %ymm10 # ymm10 = ymm9[0],ymm10[1],ymm9[2],ymm10[3],ymm9[4],ymm10[5],ymm9[6],ymm10[7],ymm9[8],ymm10[9],ymm9[10],ymm10[11],ymm9[12],ymm10[13],ymm9[14],ymm10[15]+ vpsrld $0x10, %ymm9, %ymm9+ vpblendw $0xaa, %ymm5, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm5[1],ymm9[2],ymm5[3],ymm9[4],ymm5[5],ymm9[6],ymm5[7],ymm9[8],ymm5[9],ymm9[10],ymm5[11],ymm9[12],ymm5[13],ymm9[14],ymm5[15]+ vpslld $0x10, %ymm4, %ymm9+ vpblendw $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7],ymm8[8],ymm9[9],ymm8[10],ymm9[11],ymm8[12],ymm9[13],ymm8[14],ymm9[15]+ vpsrld $0x10, %ymm8, %ymm8+ vpblendw $0xaa, %ymm4, %ymm8, %ymm4 # ymm4 = ymm8[0],ymm4[1],ymm8[2],ymm4[3],ymm8[4],ymm4[5],ymm8[6],ymm4[7],ymm8[8],ymm4[9],ymm8[10],ymm4[11],ymm8[12],ymm4[13],ymm8[14],ymm4[15]+ vpslld $0x10, %ymm3, %ymm8+ vpblendw $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7],ymm7[8],ymm8[9],ymm7[10],ymm8[11],ymm7[12],ymm8[13],ymm7[14],ymm8[15]+ vpsrld $0x10, %ymm7, %ymm7+ vpblendw $0xaa, %ymm3, %ymm7, %ymm3 # ymm3 = ymm7[0],ymm3[1],ymm7[2],ymm3[3],ymm7[4],ymm3[5],ymm7[6],ymm3[7],ymm7[8],ymm3[9],ymm7[10],ymm3[11],ymm7[12],ymm3[13],ymm7[14],ymm3[15]+ vpslld $0x10, %ymm11, %ymm7+ vpblendw $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7],ymm6[8],ymm7[9],ymm6[10],ymm7[11],ymm6[12],ymm7[13],ymm6[14],ymm7[15]+ vpsrld $0x10, %ymm6, %ymm6+ vpblendw $0xaa, %ymm11, %ymm6, %ymm11 # ymm11 = ymm6[0],ymm11[1],ymm6[2],ymm11[3],ymm6[4],ymm11[5],ymm6[6],ymm11[7],ymm6[8],ymm11[9],ymm6[10],ymm11[11],ymm6[12],ymm11[13],ymm6[14],ymm11[15]+ vmovdqa %ymm10, (%rdi)+ vmovdqa %ymm5, 0x20(%rdi)+ vmovdqa %ymm9, 0x40(%rdi)+ vmovdqa %ymm4, 0x60(%rdi)+ vmovdqa %ymm8, 0x80(%rdi)+ vmovdqa %ymm3, 0xa0(%rdi)+ vmovdqa %ymm7, 0xc0(%rdi)+ vmovdqa %ymm11, 0xe0(%rdi)+ vmovdqa 0x100(%rdi), %ymm4+ vmovdqa 0x120(%rdi), %ymm5+ vmovdqa 0x140(%rdi), %ymm6+ vmovdqa 0x160(%rdi), %ymm7+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1a0(%rdi), %ymm9+ vmovdqa 0x1c0(%rdi), %ymm10+ vmovdqa 0x1e0(%rdi), %ymm11+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vpslld $0x10, %ymm5, %ymm10+ vpblendw $0xaa, %ymm10, %ymm9, %ymm10 # ymm10 = ymm9[0],ymm10[1],ymm9[2],ymm10[3],ymm9[4],ymm10[5],ymm9[6],ymm10[7],ymm9[8],ymm10[9],ymm9[10],ymm10[11],ymm9[12],ymm10[13],ymm9[14],ymm10[15]+ vpsrld $0x10, %ymm9, %ymm9+ vpblendw $0xaa, %ymm5, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm5[1],ymm9[2],ymm5[3],ymm9[4],ymm5[5],ymm9[6],ymm5[7],ymm9[8],ymm5[9],ymm9[10],ymm5[11],ymm9[12],ymm5[13],ymm9[14],ymm5[15]+ vpslld $0x10, %ymm4, %ymm9+ vpblendw $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7],ymm8[8],ymm9[9],ymm8[10],ymm9[11],ymm8[12],ymm9[13],ymm8[14],ymm9[15]+ vpsrld $0x10, %ymm8, %ymm8+ vpblendw $0xaa, %ymm4, %ymm8, %ymm4 # ymm4 = ymm8[0],ymm4[1],ymm8[2],ymm4[3],ymm8[4],ymm4[5],ymm8[6],ymm4[7],ymm8[8],ymm4[9],ymm8[10],ymm4[11],ymm8[12],ymm4[13],ymm8[14],ymm4[15]+ vpslld $0x10, %ymm3, %ymm8+ vpblendw $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7],ymm7[8],ymm8[9],ymm7[10],ymm8[11],ymm7[12],ymm8[13],ymm7[14],ymm8[15]+ vpsrld $0x10, %ymm7, %ymm7+ vpblendw $0xaa, %ymm3, %ymm7, %ymm3 # ymm3 = ymm7[0],ymm3[1],ymm7[2],ymm3[3],ymm7[4],ymm3[5],ymm7[6],ymm3[7],ymm7[8],ymm3[9],ymm7[10],ymm3[11],ymm7[12],ymm3[13],ymm7[14],ymm3[15]+ vpslld $0x10, %ymm11, %ymm7+ vpblendw $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7],ymm6[8],ymm7[9],ymm6[10],ymm7[11],ymm6[12],ymm7[13],ymm6[14],ymm7[15]+ vpsrld $0x10, %ymm6, %ymm6+ vpblendw $0xaa, %ymm11, %ymm6, %ymm11 # ymm11 = ymm6[0],ymm11[1],ymm6[2],ymm11[3],ymm6[4],ymm11[5],ymm6[6],ymm11[7],ymm6[8],ymm11[9],ymm6[10],ymm11[11],ymm6[12],ymm11[13],ymm6[14],ymm11[15]+ vmovdqa %ymm10, 0x100(%rdi)+ vmovdqa %ymm5, 0x120(%rdi)+ vmovdqa %ymm9, 0x140(%rdi)+ vmovdqa %ymm4, 0x160(%rdi)+ vmovdqa %ymm8, 0x180(%rdi)+ vmovdqa %ymm3, 0x1a0(%rdi)+ vmovdqa %ymm7, 0x1c0(%rdi)+ vmovdqa %ymm11, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(nttunpack_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_poly_compress_d10_avx2_asm.S view
@@ -0,0 +1,413 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Kyber implementation @[REF_AVX2].+ */++/*************************************************+ * Name: mlk_poly_compress_d10_avx2_asm+ *+ * Description: Compression of a polynomial to 10 bits per coefficient.+ *+ * Arguments: - uint8_t *r: pointer to output byte array+ * (of length MLKEM_POLYCOMPRESSEDBYTES_D10)+ * - const int16_t *a: pointer to input polynomial+ * - const uint8_t *data: pointer to shufbidx constant+ **************************************************/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API)) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2 || MLKEM_K == 3)+/*yaml+ Name: poly_compress_d10_avx2_asm+ Description: x86_64 AVX2 polynomial compression (d=10)+ Signature: void mlk_poly_compress_d10_avx2_asm(uint8_t *r, const int16_t *a, const uint8_t *data)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 320+ permissions: write-only+ c_parameter: uint8_t *r+ description: Output compressed polynomial+ rsi:+ type: buffer+ size_bytes: 512+ permissions: read-only+ c_parameter: const int16_t *a+ description: Input polynomial (256 x int16_t)+ rdx:+ type: buffer+ size_bytes: 32+ permissions: read-only+ c_parameter: const uint8_t *data+ description: Precomputed compression constants+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_poly_compress_d10_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_compress_d10_avx2_asm)+MLK_ASM_FN_SYMBOL(poly_compress_d10_avx2_asm)++ .cfi_startproc+ movl $0x4ebf4ebf, %eax # imm = 0x4EBF4EBF+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ vpsllw $0x3, %ymm0, %ymm1+ movl $0xf000f, %eax # imm = 0xF000F+ vmovd %eax, %xmm2+ vpbroadcastd %xmm2, %ymm2+ movl $0x10001000, %eax # imm = 0x10001000+ vmovd %eax, %xmm3+ vpbroadcastd %xmm3, %ymm3+ movl $0x3ff03ff, %eax # imm = 0x3FF03FF+ vmovd %eax, %xmm4+ vpbroadcastd %xmm4, %ymm4+ movabsq $0x400000104000001, %rax # imm = 0x400000104000001+ vmovq %rax, %xmm5+ vpbroadcastq %xmm5, %ymm5+ movl $0xc, %eax+ vmovq %rax, %xmm6+ vpbroadcastq %xmm6, %ymm6+ vmovdqa (%rdx), %ymm7+ vmovdqa (%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, (%rdi)+ vmovd %xmm9, 0x10(%rdi)+ vmovdqa 0x20(%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, 0x14(%rdi)+ vmovd %xmm9, 0x24(%rdi)+ vmovdqa 0x40(%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, 0x28(%rdi)+ vmovd %xmm9, 0x38(%rdi)+ vmovdqa 0x60(%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, 0x3c(%rdi)+ vmovd %xmm9, 0x4c(%rdi)+ vmovdqa 0x80(%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, 0x50(%rdi)+ vmovd %xmm9, 0x60(%rdi)+ vmovdqa 0xa0(%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, 0x64(%rdi)+ vmovd %xmm9, 0x74(%rdi)+ vmovdqa 0xc0(%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, 0x78(%rdi)+ vmovd %xmm9, 0x88(%rdi)+ vmovdqa 0xe0(%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, 0x8c(%rdi)+ vmovd %xmm9, 0x9c(%rdi)+ vmovdqa 0x100(%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, 0xa0(%rdi)+ vmovd %xmm9, 0xb0(%rdi)+ vmovdqa 0x120(%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, 0xb4(%rdi)+ vmovd %xmm9, 0xc4(%rdi)+ vmovdqa 0x140(%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, 0xc8(%rdi)+ vmovd %xmm9, 0xd8(%rdi)+ vmovdqa 0x160(%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, 0xdc(%rdi)+ vmovd %xmm9, 0xec(%rdi)+ vmovdqa 0x180(%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, 0xf0(%rdi)+ vmovd %xmm9, 0x100(%rdi)+ vmovdqa 0x1a0(%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, 0x104(%rdi)+ vmovd %xmm9, 0x114(%rdi)+ vmovdqa 0x1c0(%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, 0x118(%rdi)+ vmovd %xmm9, 0x128(%rdi)+ vmovdqa 0x1e0(%rsi), %ymm8+ vpmullw %ymm1, %ymm8, %ymm9+ vpaddw %ymm2, %ymm8, %ymm10+ vpsllw $0x3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm8, %ymm8+ vpsubw %ymm10, %ymm9, %ymm10+ vpandn %ymm10, %ymm9, %ymm9+ vpsrlw $0xf, %ymm9, %ymm9+ vpsubw %ymm9, %ymm8, %ymm8+ vpmulhrsw %ymm3, %ymm8, %ymm8+ vpand %ymm4, %ymm8, %ymm8+ vpmaddwd %ymm5, %ymm8, %ymm8+ vpsllvd %ymm6, %ymm8, %ymm8+ vpsrlq $0xc, %ymm8, %ymm8+ vpshufb %ymm7, %ymm8, %ymm8+ vextracti128 $0x1, %ymm8, %xmm9+ vpblendw $0xe0, %xmm9, %xmm8, %xmm8 # xmm8 = xmm8[0,1,2,3,4],xmm9[5,6,7]+ vmovdqu %xmm8, 0x12c(%rdi)+ vmovd %xmm9, 0x13c(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_compress_d10_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3) \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_poly_compress_d11_avx2_asm.S view
@@ -0,0 +1,479 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Kyber implementation @[REF_AVX2].+ */++/*************************************************+ * Name: mlk_poly_compress_d11_avx2_asm+ *+ * Description: Compression of a polynomial to 11 bits per coefficient.+ *+ * Arguments: - uint8_t *r: pointer to output byte array+ * (of length MLKEM_POLYCOMPRESSEDBYTES_D11)+ * - const int16_t *a: pointer to input polynomial+ * - const uint8_t *data: pointer to constants+ * (srlvqidx[0:32], shufbidx[32:64])+ **************************************************/++#include "../../../common.h"++#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API)) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4)+/*yaml+ Name: poly_compress_d11_avx2_asm+ Description: x86_64 AVX2 polynomial compression (d=11)+ Signature: void mlk_poly_compress_d11_avx2_asm(uint8_t *r, const int16_t *a, const uint8_t *data)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 352+ permissions: write-only+ c_parameter: uint8_t *r+ description: Output compressed polynomial+ rsi:+ type: buffer+ size_bytes: 512+ permissions: read-only+ c_parameter: const int16_t *a+ description: Input polynomial (256 x int16_t)+ rdx:+ type: buffer+ size_bytes: 64+ permissions: read-only+ c_parameter: const uint8_t *data+ description: Precomputed compression constants+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_poly_compress_d11_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_compress_d11_avx2_asm)+MLK_ASM_FN_SYMBOL(poly_compress_d11_avx2_asm)++ .cfi_startproc+ movl $0x4ebf4ebf, %eax # imm = 0x4EBF4EBF+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ vpsllw $0x3, %ymm0, %ymm1+ movl $0x240024, %eax # imm = 0x240024+ vmovd %eax, %xmm2+ vpbroadcastd %xmm2, %ymm2+ movl $0x20002000, %eax # imm = 0x20002000+ vmovd %eax, %xmm3+ vpbroadcastd %xmm3, %ymm3+ movl $0x7ff07ff, %eax # imm = 0x7FF07FF+ vmovd %eax, %xmm4+ vpbroadcastd %xmm4, %ymm4+ movabsq $0x800000108000001, %rax # imm = 0x800000108000001+ vmovq %rax, %xmm5+ vpbroadcastq %xmm5, %ymm5+ movl $0xa, %eax+ vmovq %rax, %xmm6+ vpbroadcastq %xmm6, %ymm6+ vmovdqa (%rdx), %ymm7+ vmovdqa 0x20(%rdx), %ymm8+ vmovdqa (%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, (%rdi)+ vmovd %xmm10, 0x10(%rdi)+ vpextrw $0x2, %xmm10, 0x14(%rdi)+ vmovdqa 0x20(%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, 0x16(%rdi)+ vmovd %xmm10, 0x26(%rdi)+ vpextrw $0x2, %xmm10, 0x2a(%rdi)+ vmovdqa 0x40(%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, 0x2c(%rdi)+ vmovd %xmm10, 0x3c(%rdi)+ vpextrw $0x2, %xmm10, 0x40(%rdi)+ vmovdqa 0x60(%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, 0x42(%rdi)+ vmovd %xmm10, 0x52(%rdi)+ vpextrw $0x2, %xmm10, 0x56(%rdi)+ vmovdqa 0x80(%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, 0x58(%rdi)+ vmovd %xmm10, 0x68(%rdi)+ vpextrw $0x2, %xmm10, 0x6c(%rdi)+ vmovdqa 0xa0(%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, 0x6e(%rdi)+ vmovd %xmm10, 0x7e(%rdi)+ vpextrw $0x2, %xmm10, 0x82(%rdi)+ vmovdqa 0xc0(%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, 0x84(%rdi)+ vmovd %xmm10, 0x94(%rdi)+ vpextrw $0x2, %xmm10, 0x98(%rdi)+ vmovdqa 0xe0(%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, 0x9a(%rdi)+ vmovd %xmm10, 0xaa(%rdi)+ vpextrw $0x2, %xmm10, 0xae(%rdi)+ vmovdqa 0x100(%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, 0xb0(%rdi)+ vmovd %xmm10, 0xc0(%rdi)+ vpextrw $0x2, %xmm10, 0xc4(%rdi)+ vmovdqa 0x120(%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, 0xc6(%rdi)+ vmovd %xmm10, 0xd6(%rdi)+ vpextrw $0x2, %xmm10, 0xda(%rdi)+ vmovdqa 0x140(%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, 0xdc(%rdi)+ vmovd %xmm10, 0xec(%rdi)+ vpextrw $0x2, %xmm10, 0xf0(%rdi)+ vmovdqa 0x160(%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, 0xf2(%rdi)+ vmovd %xmm10, 0x102(%rdi)+ vpextrw $0x2, %xmm10, 0x106(%rdi)+ vmovdqa 0x180(%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, 0x108(%rdi)+ vmovd %xmm10, 0x118(%rdi)+ vpextrw $0x2, %xmm10, 0x11c(%rdi)+ vmovdqa 0x1a0(%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, 0x11e(%rdi)+ vmovd %xmm10, 0x12e(%rdi)+ vpextrw $0x2, %xmm10, 0x132(%rdi)+ vmovdqa 0x1c0(%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, 0x134(%rdi)+ vmovd %xmm10, 0x144(%rdi)+ vpextrw $0x2, %xmm10, 0x148(%rdi)+ vmovdqa 0x1e0(%rsi), %ymm9+ vpmullw %ymm1, %ymm9, %ymm10+ vpaddw %ymm2, %ymm9, %ymm11+ vpsllw $0x3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm9, %ymm9+ vpsubw %ymm11, %ymm10, %ymm11+ vpandn %ymm11, %ymm10, %ymm10+ vpsrlw $0xf, %ymm10, %ymm10+ vpsubw %ymm10, %ymm9, %ymm9+ vpmulhrsw %ymm3, %ymm9, %ymm9+ vpand %ymm4, %ymm9, %ymm9+ vpmaddwd %ymm5, %ymm9, %ymm9+ vpsllvd %ymm6, %ymm9, %ymm9+ vpsrldq $0x8, %ymm9, %ymm10 # ymm10 = ymm9[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero,ymm9[24,25,26,27,28,29,30,31],zero,zero,zero,zero,zero,zero,zero,zero+ vpsrlvq %ymm7, %ymm9, %ymm9+ vpsllq $0x22, %ymm10, %ymm10+ vpaddq %ymm10, %ymm9, %ymm9+ vpshufb %ymm8, %ymm9, %ymm9+ vextracti128 $0x1, %ymm9, %xmm10+ vpblendvb %xmm8, %xmm10, %xmm9, %xmm9+ vmovdqu %xmm9, 0x14a(%rdi)+ vmovd %xmm10, 0x15a(%rdi)+ vpextrw $0x2, %xmm10, 0x15e(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_compress_d11_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_poly_compress_d4_avx2_asm.S view
@@ -0,0 +1,194 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Kyber implementation @[REF_AVX2].+ */++/*************************************************+ * Name: mlk_poly_compress_d4_avx2_asm+ *+ * Description: Compression of a polynomial to 4 bits per coefficient.+ *+ * Arguments: - uint8_t *r: pointer to output byte array+ * (of length MLKEM_POLYCOMPRESSEDBYTES_D4)+ * - const int16_t *a: pointer to input polynomial+ * - const uint8_t *data: pointer to permdidx constant+ **************************************************/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API)) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2 || MLKEM_K == 3)+/*yaml+ Name: poly_compress_d4_avx2_asm+ Description: x86_64 AVX2 polynomial compression (d=4)+ Signature: void mlk_poly_compress_d4_avx2_asm(uint8_t *r, const int16_t *a, const uint8_t *data)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 128+ permissions: write-only+ c_parameter: uint8_t *r+ description: Output compressed polynomial+ rsi:+ type: buffer+ size_bytes: 512+ permissions: read-only+ c_parameter: const int16_t *a+ description: Input polynomial (256 x int16_t)+ rdx:+ type: buffer+ size_bytes: 32+ permissions: read-only+ c_parameter: const uint8_t *data+ description: Precomputed compression constants+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_poly_compress_d4_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_compress_d4_avx2_asm)+MLK_ASM_FN_SYMBOL(poly_compress_d4_avx2_asm)++ .cfi_startproc+ movl $0x4ebf4ebf, %eax # imm = 0x4EBF4EBF+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ movl $0x2000200, %eax # imm = 0x2000200+ vmovd %eax, %xmm1+ vpbroadcastd %xmm1, %ymm1+ movl $0xf000f, %eax # imm = 0xF000F+ vmovd %eax, %xmm2+ vpbroadcastd %xmm2, %ymm2+ movl $0x10011001, %eax # imm = 0x10011001+ vmovd %eax, %xmm3+ vpbroadcastd %xmm3, %ymm3+ vmovdqa (%rdx), %ymm4+ vmovdqa (%rsi), %ymm5+ vmovdqa 0x20(%rsi), %ymm6+ vmovdqa 0x40(%rsi), %ymm7+ vmovdqa 0x60(%rsi), %ymm8+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm6, %ymm6+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhrsw %ymm1, %ymm5, %ymm5+ vpmulhrsw %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm1, %ymm7, %ymm7+ vpmulhrsw %ymm1, %ymm8, %ymm8+ vpand %ymm2, %ymm5, %ymm5+ vpand %ymm2, %ymm6, %ymm6+ vpand %ymm2, %ymm7, %ymm7+ vpand %ymm2, %ymm8, %ymm8+ vpackuswb %ymm6, %ymm5, %ymm5+ vpackuswb %ymm8, %ymm7, %ymm7+ vpmaddubsw %ymm3, %ymm5, %ymm5+ vpmaddubsw %ymm3, %ymm7, %ymm7+ vpackuswb %ymm7, %ymm5, %ymm5+ vpermd %ymm5, %ymm4, %ymm5+ vmovdqu %ymm5, (%rdi)+ vmovdqa 0x80(%rsi), %ymm5+ vmovdqa 0xa0(%rsi), %ymm6+ vmovdqa 0xc0(%rsi), %ymm7+ vmovdqa 0xe0(%rsi), %ymm8+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm6, %ymm6+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhrsw %ymm1, %ymm5, %ymm5+ vpmulhrsw %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm1, %ymm7, %ymm7+ vpmulhrsw %ymm1, %ymm8, %ymm8+ vpand %ymm2, %ymm5, %ymm5+ vpand %ymm2, %ymm6, %ymm6+ vpand %ymm2, %ymm7, %ymm7+ vpand %ymm2, %ymm8, %ymm8+ vpackuswb %ymm6, %ymm5, %ymm5+ vpackuswb %ymm8, %ymm7, %ymm7+ vpmaddubsw %ymm3, %ymm5, %ymm5+ vpmaddubsw %ymm3, %ymm7, %ymm7+ vpackuswb %ymm7, %ymm5, %ymm5+ vpermd %ymm5, %ymm4, %ymm5+ vmovdqu %ymm5, 0x20(%rdi)+ vmovdqa 0x100(%rsi), %ymm5+ vmovdqa 0x120(%rsi), %ymm6+ vmovdqa 0x140(%rsi), %ymm7+ vmovdqa 0x160(%rsi), %ymm8+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm6, %ymm6+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhrsw %ymm1, %ymm5, %ymm5+ vpmulhrsw %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm1, %ymm7, %ymm7+ vpmulhrsw %ymm1, %ymm8, %ymm8+ vpand %ymm2, %ymm5, %ymm5+ vpand %ymm2, %ymm6, %ymm6+ vpand %ymm2, %ymm7, %ymm7+ vpand %ymm2, %ymm8, %ymm8+ vpackuswb %ymm6, %ymm5, %ymm5+ vpackuswb %ymm8, %ymm7, %ymm7+ vpmaddubsw %ymm3, %ymm5, %ymm5+ vpmaddubsw %ymm3, %ymm7, %ymm7+ vpackuswb %ymm7, %ymm5, %ymm5+ vpermd %ymm5, %ymm4, %ymm5+ vmovdqu %ymm5, 0x40(%rdi)+ vmovdqa 0x180(%rsi), %ymm5+ vmovdqa 0x1a0(%rsi), %ymm6+ vmovdqa 0x1c0(%rsi), %ymm7+ vmovdqa 0x1e0(%rsi), %ymm8+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm6, %ymm6+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhrsw %ymm1, %ymm5, %ymm5+ vpmulhrsw %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm1, %ymm7, %ymm7+ vpmulhrsw %ymm1, %ymm8, %ymm8+ vpand %ymm2, %ymm5, %ymm5+ vpand %ymm2, %ymm6, %ymm6+ vpand %ymm2, %ymm7, %ymm7+ vpand %ymm2, %ymm8, %ymm8+ vpackuswb %ymm6, %ymm5, %ymm5+ vpackuswb %ymm8, %ymm7, %ymm7+ vpmaddubsw %ymm3, %ymm5, %ymm5+ vpmaddubsw %ymm3, %ymm7, %ymm7+ vpackuswb %ymm7, %ymm5, %ymm5+ vpermd %ymm5, %ymm4, %ymm5+ vmovdqu %ymm5, 0x60(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_compress_d4_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3) \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_poly_compress_d5_avx2_asm.S view
@@ -0,0 +1,251 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Kyber implementation @[REF_AVX2].+ */++/*************************************************+ * Name: mlk_poly_compress_d5_avx2_asm+ *+ * Description: Compression of a polynomial to 5 bits per coefficient.+ *+ * Arguments: - uint8_t *r: pointer to output byte array+ * (of length MLKEM_POLYCOMPRESSEDBYTES_D5)+ * - const int16_t *a: pointer to input polynomial+ * - const uint8_t *data: pointer to shufbidx constant+ **************************************************/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API)) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4)+/*yaml+ Name: poly_compress_d5_avx2_asm+ Description: x86_64 AVX2 polynomial compression (d=5)+ Signature: void mlk_poly_compress_d5_avx2_asm(uint8_t *r, const int16_t *a, const uint8_t *data)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 160+ permissions: write-only+ c_parameter: uint8_t *r+ description: Output compressed polynomial+ rsi:+ type: buffer+ size_bytes: 512+ permissions: read-only+ c_parameter: const int16_t *a+ description: Input polynomial (256 x int16_t)+ rdx:+ type: buffer+ size_bytes: 32+ permissions: read-only+ c_parameter: const uint8_t *data+ description: Precomputed compression constants+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_poly_compress_d5_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_compress_d5_avx2_asm)+MLK_ASM_FN_SYMBOL(poly_compress_d5_avx2_asm)++ .cfi_startproc+ movl $0x4ebf4ebf, %eax # imm = 0x4EBF4EBF+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ movl $0x4000400, %eax # imm = 0x4000400+ vmovd %eax, %xmm1+ vpbroadcastd %xmm1, %ymm1+ movl $0x1f001f, %eax # imm = 0x1F001F+ vmovd %eax, %xmm2+ vpbroadcastd %xmm2, %ymm2+ movl $0x20012001, %eax # imm = 0x20012001+ vmovd %eax, %xmm3+ vpbroadcastd %xmm3, %ymm3+ movl $0x4000001, %eax # imm = 0x4000001+ vmovd %eax, %xmm4+ vpbroadcastd %xmm4, %ymm4+ movl $0xc, %eax+ vmovq %rax, %xmm5+ vpbroadcastq %xmm5, %ymm5+ vmovdqa (%rdx), %ymm6+ vmovdqa (%rsi), %ymm7+ vmovdqa 0x20(%rsi), %ymm8+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhrsw %ymm1, %ymm7, %ymm7+ vpmulhrsw %ymm1, %ymm8, %ymm8+ vpand %ymm2, %ymm7, %ymm7+ vpand %ymm2, %ymm8, %ymm8+ vpackuswb %ymm8, %ymm7, %ymm7+ vpmaddubsw %ymm3, %ymm7, %ymm7+ vpmaddwd %ymm4, %ymm7, %ymm7+ vpsllvd %ymm5, %ymm7, %ymm7+ vpsrlvq %ymm5, %ymm7, %ymm7+ vpshufb %ymm6, %ymm7, %ymm7+ vextracti128 $0x1, %ymm7, %xmm8+ vpblendvb %xmm6, %xmm8, %xmm7, %xmm7+ vmovdqu %xmm7, (%rdi)+ vmovd %xmm8, 0x10(%rdi)+ vmovdqa 0x40(%rsi), %ymm7+ vmovdqa 0x60(%rsi), %ymm8+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhrsw %ymm1, %ymm7, %ymm7+ vpmulhrsw %ymm1, %ymm8, %ymm8+ vpand %ymm2, %ymm7, %ymm7+ vpand %ymm2, %ymm8, %ymm8+ vpackuswb %ymm8, %ymm7, %ymm7+ vpmaddubsw %ymm3, %ymm7, %ymm7+ vpmaddwd %ymm4, %ymm7, %ymm7+ vpsllvd %ymm5, %ymm7, %ymm7+ vpsrlvq %ymm5, %ymm7, %ymm7+ vpshufb %ymm6, %ymm7, %ymm7+ vextracti128 $0x1, %ymm7, %xmm8+ vpblendvb %xmm6, %xmm8, %xmm7, %xmm7+ vmovdqu %xmm7, 0x14(%rdi)+ vmovd %xmm8, 0x24(%rdi)+ vmovdqa 0x80(%rsi), %ymm7+ vmovdqa 0xa0(%rsi), %ymm8+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhrsw %ymm1, %ymm7, %ymm7+ vpmulhrsw %ymm1, %ymm8, %ymm8+ vpand %ymm2, %ymm7, %ymm7+ vpand %ymm2, %ymm8, %ymm8+ vpackuswb %ymm8, %ymm7, %ymm7+ vpmaddubsw %ymm3, %ymm7, %ymm7+ vpmaddwd %ymm4, %ymm7, %ymm7+ vpsllvd %ymm5, %ymm7, %ymm7+ vpsrlvq %ymm5, %ymm7, %ymm7+ vpshufb %ymm6, %ymm7, %ymm7+ vextracti128 $0x1, %ymm7, %xmm8+ vpblendvb %xmm6, %xmm8, %xmm7, %xmm7+ vmovdqu %xmm7, 0x28(%rdi)+ vmovd %xmm8, 0x38(%rdi)+ vmovdqa 0xc0(%rsi), %ymm7+ vmovdqa 0xe0(%rsi), %ymm8+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhrsw %ymm1, %ymm7, %ymm7+ vpmulhrsw %ymm1, %ymm8, %ymm8+ vpand %ymm2, %ymm7, %ymm7+ vpand %ymm2, %ymm8, %ymm8+ vpackuswb %ymm8, %ymm7, %ymm7+ vpmaddubsw %ymm3, %ymm7, %ymm7+ vpmaddwd %ymm4, %ymm7, %ymm7+ vpsllvd %ymm5, %ymm7, %ymm7+ vpsrlvq %ymm5, %ymm7, %ymm7+ vpshufb %ymm6, %ymm7, %ymm7+ vextracti128 $0x1, %ymm7, %xmm8+ vpblendvb %xmm6, %xmm8, %xmm7, %xmm7+ vmovdqu %xmm7, 0x3c(%rdi)+ vmovd %xmm8, 0x4c(%rdi)+ vmovdqa 0x100(%rsi), %ymm7+ vmovdqa 0x120(%rsi), %ymm8+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhrsw %ymm1, %ymm7, %ymm7+ vpmulhrsw %ymm1, %ymm8, %ymm8+ vpand %ymm2, %ymm7, %ymm7+ vpand %ymm2, %ymm8, %ymm8+ vpackuswb %ymm8, %ymm7, %ymm7+ vpmaddubsw %ymm3, %ymm7, %ymm7+ vpmaddwd %ymm4, %ymm7, %ymm7+ vpsllvd %ymm5, %ymm7, %ymm7+ vpsrlvq %ymm5, %ymm7, %ymm7+ vpshufb %ymm6, %ymm7, %ymm7+ vextracti128 $0x1, %ymm7, %xmm8+ vpblendvb %xmm6, %xmm8, %xmm7, %xmm7+ vmovdqu %xmm7, 0x50(%rdi)+ vmovd %xmm8, 0x60(%rdi)+ vmovdqa 0x140(%rsi), %ymm7+ vmovdqa 0x160(%rsi), %ymm8+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhrsw %ymm1, %ymm7, %ymm7+ vpmulhrsw %ymm1, %ymm8, %ymm8+ vpand %ymm2, %ymm7, %ymm7+ vpand %ymm2, %ymm8, %ymm8+ vpackuswb %ymm8, %ymm7, %ymm7+ vpmaddubsw %ymm3, %ymm7, %ymm7+ vpmaddwd %ymm4, %ymm7, %ymm7+ vpsllvd %ymm5, %ymm7, %ymm7+ vpsrlvq %ymm5, %ymm7, %ymm7+ vpshufb %ymm6, %ymm7, %ymm7+ vextracti128 $0x1, %ymm7, %xmm8+ vpblendvb %xmm6, %xmm8, %xmm7, %xmm7+ vmovdqu %xmm7, 0x64(%rdi)+ vmovd %xmm8, 0x74(%rdi)+ vmovdqa 0x180(%rsi), %ymm7+ vmovdqa 0x1a0(%rsi), %ymm8+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhrsw %ymm1, %ymm7, %ymm7+ vpmulhrsw %ymm1, %ymm8, %ymm8+ vpand %ymm2, %ymm7, %ymm7+ vpand %ymm2, %ymm8, %ymm8+ vpackuswb %ymm8, %ymm7, %ymm7+ vpmaddubsw %ymm3, %ymm7, %ymm7+ vpmaddwd %ymm4, %ymm7, %ymm7+ vpsllvd %ymm5, %ymm7, %ymm7+ vpsrlvq %ymm5, %ymm7, %ymm7+ vpshufb %ymm6, %ymm7, %ymm7+ vextracti128 $0x1, %ymm7, %xmm8+ vpblendvb %xmm6, %xmm8, %xmm7, %xmm7+ vmovdqu %xmm7, 0x78(%rdi)+ vmovd %xmm8, 0x88(%rdi)+ vmovdqa 0x1c0(%rsi), %ymm7+ vmovdqa 0x1e0(%rsi), %ymm8+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhrsw %ymm1, %ymm7, %ymm7+ vpmulhrsw %ymm1, %ymm8, %ymm8+ vpand %ymm2, %ymm7, %ymm7+ vpand %ymm2, %ymm8, %ymm8+ vpackuswb %ymm8, %ymm7, %ymm7+ vpmaddubsw %ymm3, %ymm7, %ymm7+ vpmaddwd %ymm4, %ymm7, %ymm7+ vpsllvd %ymm5, %ymm7, %ymm7+ vpsrlvq %ymm5, %ymm7, %ymm7+ vpshufb %ymm6, %ymm7, %ymm7+ vextracti128 $0x1, %ymm7, %xmm8+ vpblendvb %xmm6, %xmm8, %xmm7, %xmm7+ vmovdqu %xmm7, 0x8c(%rdi)+ vmovd %xmm8, 0x9c(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_compress_d5_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_poly_decompress_d10_avx2_asm.S view
@@ -0,0 +1,257 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Kyber implementation @[REF_AVX2].+ */++/*************************************************+ * Name: mlk_poly_decompress_d10_avx2_asm+ *+ * Description: Decompression of a polynomial from 10 bits per coefficient.+ *+ * Arguments: - int16_t *r: pointer to output polynomial+ * - const uint8_t *a: pointer to input byte array+ * (of length MLKEM_POLYCOMPRESSEDBYTES_D10)+ * - const uint8_t *data: pointer to shufbidx constant+ **************************************************/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ !defined(MLK_CONFIG_NO_DECAPS_API) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2 || MLKEM_K == 3)+/*yaml+ Name: poly_decompress_d10_avx2_asm+ Description: x86_64 AVX2 polynomial decompression (d=10)+ Signature: void mlk_poly_decompress_d10_avx2_asm(int16_t *r, const uint8_t *a, const uint8_t *data)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 512+ permissions: write-only+ c_parameter: int16_t *r+ description: Output polynomial (256 x int16_t)+ rsi:+ type: buffer+ size_bytes: 320+ permissions: read-only+ c_parameter: const uint8_t *a+ description: Input compressed polynomial+ rdx:+ type: buffer+ size_bytes: 32+ permissions: read-only+ c_parameter: const uint8_t *data+ description: Precomputed decompression constants+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_poly_decompress_d10_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_decompress_d10_avx2_asm)+MLK_ASM_FN_SYMBOL(poly_decompress_d10_avx2_asm)++ .cfi_startproc+ movl $0xd013404, %eax # imm = 0xD013404+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ movl $0x4, %eax+ vmovq %rax, %xmm1+ vpbroadcastq %xmm1, %ymm1+ movl $0x7fe01ff8, %eax # imm = 0x7FE01FF8+ vmovd %eax, %xmm2+ vpbroadcastd %xmm2, %ymm2+ vmovdqa (%rdx), %ymm3+ vmovdqu (%rsi), %xmm4+ vmovd 0x10(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, (%rdi)+ vmovdqu 0x14(%rsi), %xmm4+ vmovd 0x24(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x20(%rdi)+ vmovdqu 0x28(%rsi), %xmm4+ vmovd 0x38(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x40(%rdi)+ vmovdqu 0x3c(%rsi), %xmm4+ vmovd 0x4c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x60(%rdi)+ vmovdqu 0x50(%rsi), %xmm4+ vmovd 0x60(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x80(%rdi)+ vmovdqu 0x64(%rsi), %xmm4+ vmovd 0x74(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0xa0(%rdi)+ vmovdqu 0x78(%rsi), %xmm4+ vmovd 0x88(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0xc0(%rdi)+ vmovdqu 0x8c(%rsi), %xmm4+ vmovd 0x9c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0xe0(%rdi)+ vmovdqu 0xa0(%rsi), %xmm4+ vmovd 0xb0(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x100(%rdi)+ vmovdqu 0xb4(%rsi), %xmm4+ vmovd 0xc4(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x120(%rdi)+ vmovdqu 0xc8(%rsi), %xmm4+ vmovd 0xd8(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x140(%rdi)+ vmovdqu 0xdc(%rsi), %xmm4+ vmovd 0xec(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x160(%rdi)+ vmovdqu 0xf0(%rsi), %xmm4+ vmovd 0x100(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x180(%rdi)+ vmovdqu 0x104(%rsi), %xmm4+ vmovd 0x114(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x1a0(%rdi)+ vmovdqu 0x118(%rsi), %xmm4+ vmovd 0x128(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x1c0(%rdi)+ vmovdqu 0x12c(%rsi), %xmm4+ vmovd 0x13c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm4, %ymm4+ vpermq $0x94, %ymm4, %ymm4 # ymm4 = ymm4[0,1,1,2]+ vpshufb %ymm3, %ymm4, %ymm4+ vpsllvd %ymm1, %ymm4, %ymm4+ vpsrlw $0x1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_decompress_d10_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && !MLK_CONFIG_NO_DECAPS_API && (MLK_CONFIG_MULTILEVEL_WITH_SHARED || \+ MLKEM_K == 2 || MLKEM_K == 3) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_poly_decompress_d11_avx2_asm.S view
@@ -0,0 +1,307 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Kyber implementation @[REF_AVX2].+ */++/*************************************************+ * Name: mlk_poly_decompress_d11_avx2_asm+ *+ * Description: Decompression of a polynomial from 11 bits per coefficient.+ *+ * Arguments: - int16_t *r: pointer to output polynomial+ * - const uint8_t *a: pointer to input byte array+ * (of length MLKEM_POLYCOMPRESSEDBYTES_D11)+ * - const uint8_t *data: pointer to constants+ * (shufbidx[0:32], srlvdidx[32:64],+ * srlvqidx[64:96], shift[96:128])+ **************************************************/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ !defined(MLK_CONFIG_NO_DECAPS_API) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4)+/*yaml+ Name: poly_decompress_d11_avx2_asm+ Description: x86_64 AVX2 polynomial decompression (d=11)+ Signature: void mlk_poly_decompress_d11_avx2_asm(int16_t *r, const uint8_t *a, const uint8_t *data)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 512+ permissions: write-only+ c_parameter: int16_t *r+ description: Output polynomial (256 x int16_t)+ rsi:+ type: buffer+ size_bytes: 352+ permissions: read-only+ c_parameter: const uint8_t *a+ description: Input compressed polynomial+ rdx:+ type: buffer+ size_bytes: 128+ permissions: read-only+ c_parameter: const uint8_t *data+ description: Precomputed decompression constants+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_poly_decompress_d11_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_decompress_d11_avx2_asm)+MLK_ASM_FN_SYMBOL(poly_decompress_d11_avx2_asm)++ .cfi_startproc+ movl $0xd010d01, %eax # imm = 0xD010D01+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ movl $0x7ff07ff0, %eax # imm = 0x7FF07FF0+ vmovd %eax, %xmm1+ vpbroadcastd %xmm1, %ymm1+ vmovdqa (%rdx), %ymm2+ vmovdqa 0x20(%rdx), %ymm3+ vmovdqa 0x40(%rdx), %ymm4+ vmovdqa 0x60(%rdx), %ymm5+ vmovdqu (%rsi), %xmm6+ vmovd 0x10(%rsi), %xmm7+ vpinsrw $0x2, 0x14(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, (%rdi)+ vmovdqu 0x16(%rsi), %xmm6+ vmovd 0x26(%rsi), %xmm7+ vpinsrw $0x2, 0x2a(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, 0x20(%rdi)+ vmovdqu 0x2c(%rsi), %xmm6+ vmovd 0x3c(%rsi), %xmm7+ vpinsrw $0x2, 0x40(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, 0x40(%rdi)+ vmovdqu 0x42(%rsi), %xmm6+ vmovd 0x52(%rsi), %xmm7+ vpinsrw $0x2, 0x56(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, 0x60(%rdi)+ vmovdqu 0x58(%rsi), %xmm6+ vmovd 0x68(%rsi), %xmm7+ vpinsrw $0x2, 0x6c(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, 0x80(%rdi)+ vmovdqu 0x6e(%rsi), %xmm6+ vmovd 0x7e(%rsi), %xmm7+ vpinsrw $0x2, 0x82(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, 0xa0(%rdi)+ vmovdqu 0x84(%rsi), %xmm6+ vmovd 0x94(%rsi), %xmm7+ vpinsrw $0x2, 0x98(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, 0xc0(%rdi)+ vmovdqu 0x9a(%rsi), %xmm6+ vmovd 0xaa(%rsi), %xmm7+ vpinsrw $0x2, 0xae(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, 0xe0(%rdi)+ vmovdqu 0xb0(%rsi), %xmm6+ vmovd 0xc0(%rsi), %xmm7+ vpinsrw $0x2, 0xc4(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, 0x100(%rdi)+ vmovdqu 0xc6(%rsi), %xmm6+ vmovd 0xd6(%rsi), %xmm7+ vpinsrw $0x2, 0xda(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, 0x120(%rdi)+ vmovdqu 0xdc(%rsi), %xmm6+ vmovd 0xec(%rsi), %xmm7+ vpinsrw $0x2, 0xf0(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, 0x140(%rdi)+ vmovdqu 0xf2(%rsi), %xmm6+ vmovd 0x102(%rsi), %xmm7+ vpinsrw $0x2, 0x106(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, 0x160(%rdi)+ vmovdqu 0x108(%rsi), %xmm6+ vmovd 0x118(%rsi), %xmm7+ vpinsrw $0x2, 0x11c(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, 0x180(%rdi)+ vmovdqu 0x11e(%rsi), %xmm6+ vmovd 0x12e(%rsi), %xmm7+ vpinsrw $0x2, 0x132(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, 0x1a0(%rdi)+ vmovdqu 0x134(%rsi), %xmm6+ vmovd 0x144(%rsi), %xmm7+ vpinsrw $0x2, 0x148(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, 0x1c0(%rdi)+ vmovdqu 0x14a(%rsi), %xmm6+ vmovd 0x15a(%rsi), %xmm7+ vpinsrw $0x2, 0x15e(%rsi), %xmm7, %xmm7+ vinserti128 $0x1, %xmm7, %ymm6, %ymm6+ vpermq $0x94, %ymm6, %ymm6 # ymm6 = ymm6[0,1,1,2]+ vpshufb %ymm2, %ymm6, %ymm6+ vpsrlvd %ymm3, %ymm6, %ymm6+ vpsrlvq %ymm4, %ymm6, %ymm6+ vpmullw %ymm5, %ymm6, %ymm6+ vpsrlw $0x1, %ymm6, %ymm6+ vpand %ymm1, %ymm6, %ymm6+ vpmulhrsw %ymm0, %ymm6, %ymm6+ vmovdqu %ymm6, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_decompress_d11_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && !MLK_CONFIG_NO_DECAPS_API && (MLK_CONFIG_MULTILEVEL_WITH_SHARED || \+ MLKEM_K == 4) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_poly_decompress_d4_avx2_asm.S view
@@ -0,0 +1,209 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Kyber implementation @[REF_AVX2].+ */++/*************************************************+ * Name: mlk_poly_decompress_d4_avx2_asm+ *+ * Description: Decompression of a polynomial from 4 bits per coefficient.+ *+ * Arguments: - int16_t *r: pointer to output polynomial+ * - const uint8_t *a: pointer to input byte array+ * (of length MLKEM_POLYCOMPRESSEDBYTES_D4)+ * - const int8_t *data: pointer to shufbidx constant+ **************************************************/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ !defined(MLK_CONFIG_NO_DECAPS_API) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2 || MLKEM_K == 3)+/*yaml+ Name: poly_decompress_d4_avx2_asm+ Description: x86_64 AVX2 polynomial decompression (d=4)+ Signature: void mlk_poly_decompress_d4_avx2_asm(int16_t *r, const uint8_t *a, const uint8_t *data)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 512+ permissions: write-only+ c_parameter: int16_t *r+ description: Output polynomial (256 x int16_t)+ rsi:+ type: buffer+ size_bytes: 128+ permissions: read-only+ c_parameter: const uint8_t *a+ description: Input compressed polynomial+ rdx:+ type: buffer+ size_bytes: 32+ permissions: read-only+ c_parameter: const uint8_t *data+ description: Precomputed decompression constants+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_poly_decompress_d4_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_decompress_d4_avx2_asm)+MLK_ASM_FN_SYMBOL(poly_decompress_d4_avx2_asm)++ .cfi_startproc+ movl $0xd010d01, %eax # imm = 0xD010D01+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ movl $0xf0000f, %eax # imm = 0xF0000F+ vmovd %eax, %xmm1+ vpbroadcastd %xmm1, %ymm1+ movl $0x800800, %eax # imm = 0x800800+ vmovd %eax, %xmm2+ vpbroadcastd %xmm2, %ymm2+ vmovdqa (%rdx), %ymm3+ vmovq (%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, (%rdi)+ vmovq 0x8(%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x20(%rdi)+ vmovq 0x10(%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x40(%rdi)+ vmovq 0x18(%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x60(%rdi)+ vmovq 0x20(%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x80(%rdi)+ vmovq 0x28(%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0xa0(%rdi)+ vmovq 0x30(%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0xc0(%rdi)+ vmovq 0x38(%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0xe0(%rdi)+ vmovq 0x40(%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x100(%rdi)+ vmovq 0x48(%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x120(%rdi)+ vmovq 0x50(%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x140(%rdi)+ vmovq 0x58(%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x160(%rdi)+ vmovq 0x60(%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x180(%rdi)+ vmovq 0x68(%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x1a0(%rdi)+ vmovq 0x70(%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x1c0(%rdi)+ vmovq 0x78(%rsi), %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm3, %ymm4, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpmullw %ymm2, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_decompress_d4_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && !MLK_CONFIG_NO_DECAPS_API && (MLK_CONFIG_MULTILEVEL_WITH_SHARED || \+ MLKEM_K == 2 || MLKEM_K == 3) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_poly_decompress_d5_avx2_asm.S view
@@ -0,0 +1,222 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Kyber implementation @[REF_AVX2].+ */++/*************************************************+ * Name: mlk_poly_decompress_d5_avx2_asm+ *+ * Description: Decompression of a polynomial from 5 bits per coefficient.+ *+ * Arguments: - int16_t *r: pointer to output polynomial+ * - const uint8_t *a: pointer to input byte array+ * (of length MLKEM_POLYCOMPRESSEDBYTES_D5)+ * - const uint8_t *data: pointer to constants+ * (shufbidx[0:32], mask[32:64], shift[64:96])+ **************************************************/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ !defined(MLK_CONFIG_NO_DECAPS_API) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4)+/*yaml+ Name: poly_decompress_d5_avx2_asm+ Description: x86_64 AVX2 polynomial decompression (d=5)+ Signature: void mlk_poly_decompress_d5_avx2_asm(int16_t *r, const uint8_t *a, const uint8_t *data)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 512+ permissions: write-only+ c_parameter: int16_t *r+ description: Output polynomial (256 x int16_t)+ rsi:+ type: buffer+ size_bytes: 160+ permissions: read-only+ c_parameter: const uint8_t *a+ description: Input compressed polynomial+ rdx:+ type: buffer+ size_bytes: 96+ permissions: read-only+ c_parameter: const uint8_t *data+ description: Precomputed decompression constants+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_poly_decompress_d5_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_decompress_d5_avx2_asm)+MLK_ASM_FN_SYMBOL(poly_decompress_d5_avx2_asm)++ .cfi_startproc+ movl $0xd010d01, %eax # imm = 0xD010D01+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ vmovdqa (%rdx), %ymm1+ vmovdqa 0x20(%rdx), %ymm2+ vmovdqa 0x40(%rdx), %ymm3+ vmovq (%rsi), %xmm4+ vpinsrw $0x4, 0x8(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, (%rdi)+ vmovq 0xa(%rsi), %xmm4+ vpinsrw $0x4, 0x12(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x20(%rdi)+ vmovq 0x14(%rsi), %xmm4+ vpinsrw $0x4, 0x1c(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x40(%rdi)+ vmovq 0x1e(%rsi), %xmm4+ vpinsrw $0x4, 0x26(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x60(%rdi)+ vmovq 0x28(%rsi), %xmm4+ vpinsrw $0x4, 0x30(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x80(%rdi)+ vmovq 0x32(%rsi), %xmm4+ vpinsrw $0x4, 0x3a(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0xa0(%rdi)+ vmovq 0x3c(%rsi), %xmm4+ vpinsrw $0x4, 0x44(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0xc0(%rdi)+ vmovq 0x46(%rsi), %xmm4+ vpinsrw $0x4, 0x4e(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0xe0(%rdi)+ vmovq 0x50(%rsi), %xmm4+ vpinsrw $0x4, 0x58(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x100(%rdi)+ vmovq 0x5a(%rsi), %xmm4+ vpinsrw $0x4, 0x62(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x120(%rdi)+ vmovq 0x64(%rsi), %xmm4+ vpinsrw $0x4, 0x6c(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x140(%rdi)+ vmovq 0x6e(%rsi), %xmm4+ vpinsrw $0x4, 0x76(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x160(%rdi)+ vmovq 0x78(%rsi), %xmm4+ vpinsrw $0x4, 0x80(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x180(%rdi)+ vmovq 0x82(%rsi), %xmm4+ vpinsrw $0x4, 0x8a(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x1a0(%rdi)+ vmovq 0x8c(%rsi), %xmm4+ vpinsrw $0x4, 0x94(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x1c0(%rdi)+ vmovq 0x96(%rsi), %xmm4+ vpinsrw $0x4, 0x9e(%rsi), %xmm4, %xmm4+ vinserti128 $0x1, %xmm4, %ymm4, %ymm4+ vpshufb %ymm1, %ymm4, %ymm4+ vpand %ymm2, %ymm4, %ymm4+ vpmullw %ymm3, %ymm4, %ymm4+ vpmulhrsw %ymm0, %ymm4, %ymm4+ vmovdqu %ymm4, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_decompress_d5_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && !MLK_CONFIG_NO_DECAPS_API && (MLK_CONFIG_MULTILEVEL_WITH_SHARED || \+ MLKEM_K == 4) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_poly_mulcache_compute_avx2_asm.S view
@@ -0,0 +1,118 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)+/*yaml+ Name: poly_mulcache_compute_avx2_asm+ Description: x86_64 AVX2 mulcache computation+ Signature: void mlk_poly_mulcache_compute_avx2_asm(int16_t *out, const int16_t *in, const int16_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 256+ permissions: write-only+ c_parameter: int16_t *out+ description: Output mulcache (128 x int16_t)+ rsi:+ type: buffer+ size_bytes: 512+ permissions: read-only+ c_parameter: const int16_t *in+ description: Input polynomial (256 x int16_t)+ rdx:+ type: buffer+ size_bytes: 1248+ permissions: read-only+ c_parameter: const int16_t *qdata+ description: Precomputed constants (624 x int16_t)+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_poly_mulcache_compute_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_mulcache_compute_avx2_asm)+MLK_ASM_FN_SYMBOL(poly_mulcache_compute_avx2_asm)++ .cfi_startproc+ movl $0xd010d01, %eax # imm = 0xD010D01+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ vmovdqa 0x20(%rsi), %ymm2+ vmovdqa 0x60(%rsi), %ymm3+ vmovdqa 0x3e0(%rdx), %ymm4+ vmovdqa 0x460(%rdx), %ymm1+ vpmullw %ymm2, %ymm1, %ymm5+ vpmullw %ymm3, %ymm1, %ymm6+ vpmulhw %ymm2, %ymm4, %ymm7+ vpmulhw %ymm3, %ymm4, %ymm8+ vpmulhw %ymm5, %ymm0, %ymm9+ vpmulhw %ymm6, %ymm0, %ymm10+ vpsubw %ymm9, %ymm7, %ymm7+ vpsubw %ymm10, %ymm8, %ymm8+ vmovdqa %ymm7, (%rdi)+ vmovdqa %ymm8, 0x20(%rdi)+ vmovdqa 0xa0(%rsi), %ymm2+ vmovdqa 0xe0(%rsi), %ymm3+ vmovdqa 0x400(%rdx), %ymm4+ vmovdqa 0x480(%rdx), %ymm1+ vpmullw %ymm2, %ymm1, %ymm5+ vpmullw %ymm3, %ymm1, %ymm6+ vpmulhw %ymm2, %ymm4, %ymm7+ vpmulhw %ymm3, %ymm4, %ymm8+ vpmulhw %ymm5, %ymm0, %ymm9+ vpmulhw %ymm6, %ymm0, %ymm10+ vpsubw %ymm9, %ymm7, %ymm7+ vpsubw %ymm10, %ymm8, %ymm8+ vmovdqa %ymm7, 0x40(%rdi)+ vmovdqa %ymm8, 0x60(%rdi)+ vmovdqa 0x120(%rsi), %ymm2+ vmovdqa 0x160(%rsi), %ymm3+ vmovdqa 0x420(%rdx), %ymm4+ vmovdqa 0x4a0(%rdx), %ymm1+ vpmullw %ymm2, %ymm1, %ymm5+ vpmullw %ymm3, %ymm1, %ymm6+ vpmulhw %ymm2, %ymm4, %ymm7+ vpmulhw %ymm3, %ymm4, %ymm8+ vpmulhw %ymm5, %ymm0, %ymm9+ vpmulhw %ymm6, %ymm0, %ymm10+ vpsubw %ymm9, %ymm7, %ymm7+ vpsubw %ymm10, %ymm8, %ymm8+ vmovdqa %ymm7, 0x80(%rdi)+ vmovdqa %ymm8, 0xa0(%rdi)+ vmovdqa 0x1a0(%rsi), %ymm2+ vmovdqa 0x1e0(%rsi), %ymm3+ vmovdqa 0x440(%rdx), %ymm4+ vmovdqa 0x4c0(%rdx), %ymm1+ vpmullw %ymm2, %ymm1, %ymm5+ vpmullw %ymm3, %ymm1, %ymm6+ vpmulhw %ymm2, %ymm4, %ymm7+ vpmulhw %ymm3, %ymm4, %ymm8+ vpmulhw %ymm5, %ymm0, %ymm9+ vpmulhw %ymm6, %ymm0, %ymm10+ vpsubw %ymm9, %ymm7, %ymm7+ vpsubw %ymm10, %ymm8, %ymm8+ vmovdqa %ymm7, 0xc0(%rdi)+ vmovdqa %ymm8, 0xe0(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_mulcache_compute_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S view
@@ -0,0 +1,536 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2)+/*yaml+ Name: polyvec_basemul_acc_montgomery_cached_k2_avx2_asm+ Description: x86_64 AVX2 base multiplication with accumulation (k=2)+ Signature: void mlk_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm(int16_t *r, const int16_t *a, const int16_t *b, const int16_t *b_cache)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t *r+ description: Output polynomial (256 x int16_t)+ rsi:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int16_t *a+ description: Input polyvec a (2 x 256 x int16_t)+ rdx:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int16_t *b+ description: Input polyvec b (2 x 256 x int16_t)+ rcx:+ type: buffer+ size_bytes: 512+ permissions: read-only+ c_parameter: const int16_t *b_cache+ description: Mulcache for b (2 x 128 x int16_t)+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k2_avx2_asm)+MLK_ASM_FN_SYMBOL(polyvec_basemul_acc_montgomery_cached_k2_avx2_asm)++ .cfi_startproc+ movl $0xd010d01, %eax # imm = 0xD010D01+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ movl $0xf301f301, %eax # imm = 0xF301F301+ vmovd %eax, %xmm1+ vpbroadcastd %xmm1, %ymm1+ vmovdqa (%rsi), %ymm2+ vmovdqa 0x20(%rsi), %ymm3+ vmovdqa (%rdx), %ymm4+ vmovdqa 0x20(%rdx), %ymm5+ vmovdqa (%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, (%rdi)+ vmovdqa %ymm9, 0x20(%rdi)+ vmovdqa 0x40(%rsi), %ymm2+ vmovdqa 0x60(%rsi), %ymm3+ vmovdqa 0x40(%rdx), %ymm4+ vmovdqa 0x60(%rdx), %ymm5+ vmovdqa 0x20(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x40(%rdi)+ vmovdqa %ymm9, 0x60(%rdi)+ vmovdqa 0x80(%rsi), %ymm2+ vmovdqa 0xa0(%rsi), %ymm3+ vmovdqa 0x80(%rdx), %ymm4+ vmovdqa 0xa0(%rdx), %ymm5+ vmovdqa 0x40(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x80(%rdi)+ vmovdqa %ymm9, 0xa0(%rdi)+ vmovdqa 0xc0(%rsi), %ymm2+ vmovdqa 0xe0(%rsi), %ymm3+ vmovdqa 0xc0(%rdx), %ymm4+ vmovdqa 0xe0(%rdx), %ymm5+ vmovdqa 0x60(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0xc0(%rdi)+ vmovdqa %ymm9, 0xe0(%rdi)+ vmovdqa 0x100(%rsi), %ymm2+ vmovdqa 0x120(%rsi), %ymm3+ vmovdqa 0x100(%rdx), %ymm4+ vmovdqa 0x120(%rdx), %ymm5+ vmovdqa 0x80(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x100(%rdi)+ vmovdqa %ymm9, 0x120(%rdi)+ vmovdqa 0x140(%rsi), %ymm2+ vmovdqa 0x160(%rsi), %ymm3+ vmovdqa 0x140(%rdx), %ymm4+ vmovdqa 0x160(%rdx), %ymm5+ vmovdqa 0xa0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x140(%rdi)+ vmovdqa %ymm9, 0x160(%rdi)+ vmovdqa 0x180(%rsi), %ymm2+ vmovdqa 0x1a0(%rsi), %ymm3+ vmovdqa 0x180(%rdx), %ymm4+ vmovdqa 0x1a0(%rdx), %ymm5+ vmovdqa 0xc0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x180(%rdi)+ vmovdqa %ymm9, 0x1a0(%rdi)+ vmovdqa 0x1c0(%rsi), %ymm2+ vmovdqa 0x1e0(%rsi), %ymm3+ vmovdqa 0x1c0(%rdx), %ymm4+ vmovdqa 0x1e0(%rdx), %ymm5+ vmovdqa 0xe0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x1c0(%rdi)+ vmovdqa %ymm9, 0x1e0(%rdi)+ vmovdqa 0x200(%rsi), %ymm2+ vmovdqa 0x220(%rsi), %ymm3+ vmovdqa 0x200(%rdx), %ymm4+ vmovdqa 0x220(%rdx), %ymm5+ vmovdqa 0x100(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa (%rdi), %ymm8+ vmovdqa 0x20(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, (%rdi)+ vmovdqa %ymm9, 0x20(%rdi)+ vmovdqa 0x240(%rsi), %ymm2+ vmovdqa 0x260(%rsi), %ymm3+ vmovdqa 0x240(%rdx), %ymm4+ vmovdqa 0x260(%rdx), %ymm5+ vmovdqa 0x120(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x40(%rdi), %ymm8+ vmovdqa 0x60(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x40(%rdi)+ vmovdqa %ymm9, 0x60(%rdi)+ vmovdqa 0x280(%rsi), %ymm2+ vmovdqa 0x2a0(%rsi), %ymm3+ vmovdqa 0x280(%rdx), %ymm4+ vmovdqa 0x2a0(%rdx), %ymm5+ vmovdqa 0x140(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x80(%rdi)+ vmovdqa %ymm9, 0xa0(%rdi)+ vmovdqa 0x2c0(%rsi), %ymm2+ vmovdqa 0x2e0(%rsi), %ymm3+ vmovdqa 0x2c0(%rdx), %ymm4+ vmovdqa 0x2e0(%rdx), %ymm5+ vmovdqa 0x160(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0xc0(%rdi), %ymm8+ vmovdqa 0xe0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0xc0(%rdi)+ vmovdqa %ymm9, 0xe0(%rdi)+ vmovdqa 0x300(%rsi), %ymm2+ vmovdqa 0x320(%rsi), %ymm3+ vmovdqa 0x300(%rdx), %ymm4+ vmovdqa 0x320(%rdx), %ymm5+ vmovdqa 0x180(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x100(%rdi), %ymm8+ vmovdqa 0x120(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x100(%rdi)+ vmovdqa %ymm9, 0x120(%rdi)+ vmovdqa 0x340(%rsi), %ymm2+ vmovdqa 0x360(%rsi), %ymm3+ vmovdqa 0x340(%rdx), %ymm4+ vmovdqa 0x360(%rdx), %ymm5+ vmovdqa 0x1a0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x140(%rdi), %ymm8+ vmovdqa 0x160(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x140(%rdi)+ vmovdqa %ymm9, 0x160(%rdi)+ vmovdqa 0x380(%rsi), %ymm2+ vmovdqa 0x3a0(%rsi), %ymm3+ vmovdqa 0x380(%rdx), %ymm4+ vmovdqa 0x3a0(%rdx), %ymm5+ vmovdqa 0x1c0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1a0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x180(%rdi)+ vmovdqa %ymm9, 0x1a0(%rdi)+ vmovdqa 0x3c0(%rsi), %ymm2+ vmovdqa 0x3e0(%rsi), %ymm3+ vmovdqa 0x3c0(%rdx), %ymm4+ vmovdqa 0x3e0(%rdx), %ymm5+ vmovdqa 0x1e0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x1c0(%rdi), %ymm8+ vmovdqa 0x1e0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x1c0(%rdi)+ vmovdqa %ymm9, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(polyvec_basemul_acc_montgomery_cached_k2_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S view
@@ -0,0 +1,784 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 3)+/*yaml+ Name: polyvec_basemul_acc_montgomery_cached_k3_avx2_asm+ Description: x86_64 AVX2 base multiplication with accumulation (k=3)+ Signature: void mlk_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm(int16_t *r, const int16_t *a, const int16_t *b, const int16_t *b_cache)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t *r+ description: Output polynomial (256 x int16_t)+ rsi:+ type: buffer+ size_bytes: 1536+ permissions: read-only+ c_parameter: const int16_t *a+ description: Input polyvec a (3 x 256 x int16_t)+ rdx:+ type: buffer+ size_bytes: 1536+ permissions: read-only+ c_parameter: const int16_t *b+ description: Input polyvec b (3 x 256 x int16_t)+ rcx:+ type: buffer+ size_bytes: 768+ permissions: read-only+ c_parameter: const int16_t *b_cache+ description: Mulcache for b (3 x 128 x int16_t)+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k3_avx2_asm)+MLK_ASM_FN_SYMBOL(polyvec_basemul_acc_montgomery_cached_k3_avx2_asm)++ .cfi_startproc+ movl $0xd010d01, %eax # imm = 0xD010D01+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ movl $0xf301f301, %eax # imm = 0xF301F301+ vmovd %eax, %xmm1+ vpbroadcastd %xmm1, %ymm1+ vmovdqa (%rsi), %ymm2+ vmovdqa 0x20(%rsi), %ymm3+ vmovdqa (%rdx), %ymm4+ vmovdqa 0x20(%rdx), %ymm5+ vmovdqa (%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, (%rdi)+ vmovdqa %ymm9, 0x20(%rdi)+ vmovdqa 0x40(%rsi), %ymm2+ vmovdqa 0x60(%rsi), %ymm3+ vmovdqa 0x40(%rdx), %ymm4+ vmovdqa 0x60(%rdx), %ymm5+ vmovdqa 0x20(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x40(%rdi)+ vmovdqa %ymm9, 0x60(%rdi)+ vmovdqa 0x80(%rsi), %ymm2+ vmovdqa 0xa0(%rsi), %ymm3+ vmovdqa 0x80(%rdx), %ymm4+ vmovdqa 0xa0(%rdx), %ymm5+ vmovdqa 0x40(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x80(%rdi)+ vmovdqa %ymm9, 0xa0(%rdi)+ vmovdqa 0xc0(%rsi), %ymm2+ vmovdqa 0xe0(%rsi), %ymm3+ vmovdqa 0xc0(%rdx), %ymm4+ vmovdqa 0xe0(%rdx), %ymm5+ vmovdqa 0x60(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0xc0(%rdi)+ vmovdqa %ymm9, 0xe0(%rdi)+ vmovdqa 0x100(%rsi), %ymm2+ vmovdqa 0x120(%rsi), %ymm3+ vmovdqa 0x100(%rdx), %ymm4+ vmovdqa 0x120(%rdx), %ymm5+ vmovdqa 0x80(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x100(%rdi)+ vmovdqa %ymm9, 0x120(%rdi)+ vmovdqa 0x140(%rsi), %ymm2+ vmovdqa 0x160(%rsi), %ymm3+ vmovdqa 0x140(%rdx), %ymm4+ vmovdqa 0x160(%rdx), %ymm5+ vmovdqa 0xa0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x140(%rdi)+ vmovdqa %ymm9, 0x160(%rdi)+ vmovdqa 0x180(%rsi), %ymm2+ vmovdqa 0x1a0(%rsi), %ymm3+ vmovdqa 0x180(%rdx), %ymm4+ vmovdqa 0x1a0(%rdx), %ymm5+ vmovdqa 0xc0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x180(%rdi)+ vmovdqa %ymm9, 0x1a0(%rdi)+ vmovdqa 0x1c0(%rsi), %ymm2+ vmovdqa 0x1e0(%rsi), %ymm3+ vmovdqa 0x1c0(%rdx), %ymm4+ vmovdqa 0x1e0(%rdx), %ymm5+ vmovdqa 0xe0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x1c0(%rdi)+ vmovdqa %ymm9, 0x1e0(%rdi)+ vmovdqa 0x200(%rsi), %ymm2+ vmovdqa 0x220(%rsi), %ymm3+ vmovdqa 0x200(%rdx), %ymm4+ vmovdqa 0x220(%rdx), %ymm5+ vmovdqa 0x100(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa (%rdi), %ymm8+ vmovdqa 0x20(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, (%rdi)+ vmovdqa %ymm9, 0x20(%rdi)+ vmovdqa 0x240(%rsi), %ymm2+ vmovdqa 0x260(%rsi), %ymm3+ vmovdqa 0x240(%rdx), %ymm4+ vmovdqa 0x260(%rdx), %ymm5+ vmovdqa 0x120(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x40(%rdi), %ymm8+ vmovdqa 0x60(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x40(%rdi)+ vmovdqa %ymm9, 0x60(%rdi)+ vmovdqa 0x280(%rsi), %ymm2+ vmovdqa 0x2a0(%rsi), %ymm3+ vmovdqa 0x280(%rdx), %ymm4+ vmovdqa 0x2a0(%rdx), %ymm5+ vmovdqa 0x140(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x80(%rdi)+ vmovdqa %ymm9, 0xa0(%rdi)+ vmovdqa 0x2c0(%rsi), %ymm2+ vmovdqa 0x2e0(%rsi), %ymm3+ vmovdqa 0x2c0(%rdx), %ymm4+ vmovdqa 0x2e0(%rdx), %ymm5+ vmovdqa 0x160(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0xc0(%rdi), %ymm8+ vmovdqa 0xe0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0xc0(%rdi)+ vmovdqa %ymm9, 0xe0(%rdi)+ vmovdqa 0x300(%rsi), %ymm2+ vmovdqa 0x320(%rsi), %ymm3+ vmovdqa 0x300(%rdx), %ymm4+ vmovdqa 0x320(%rdx), %ymm5+ vmovdqa 0x180(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x100(%rdi), %ymm8+ vmovdqa 0x120(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x100(%rdi)+ vmovdqa %ymm9, 0x120(%rdi)+ vmovdqa 0x340(%rsi), %ymm2+ vmovdqa 0x360(%rsi), %ymm3+ vmovdqa 0x340(%rdx), %ymm4+ vmovdqa 0x360(%rdx), %ymm5+ vmovdqa 0x1a0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x140(%rdi), %ymm8+ vmovdqa 0x160(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x140(%rdi)+ vmovdqa %ymm9, 0x160(%rdi)+ vmovdqa 0x380(%rsi), %ymm2+ vmovdqa 0x3a0(%rsi), %ymm3+ vmovdqa 0x380(%rdx), %ymm4+ vmovdqa 0x3a0(%rdx), %ymm5+ vmovdqa 0x1c0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1a0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x180(%rdi)+ vmovdqa %ymm9, 0x1a0(%rdi)+ vmovdqa 0x3c0(%rsi), %ymm2+ vmovdqa 0x3e0(%rsi), %ymm3+ vmovdqa 0x3c0(%rdx), %ymm4+ vmovdqa 0x3e0(%rdx), %ymm5+ vmovdqa 0x1e0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x1c0(%rdi), %ymm8+ vmovdqa 0x1e0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x1c0(%rdi)+ vmovdqa %ymm9, 0x1e0(%rdi)+ vmovdqa 0x400(%rsi), %ymm2+ vmovdqa 0x420(%rsi), %ymm3+ vmovdqa 0x400(%rdx), %ymm4+ vmovdqa 0x420(%rdx), %ymm5+ vmovdqa 0x200(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa (%rdi), %ymm8+ vmovdqa 0x20(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, (%rdi)+ vmovdqa %ymm9, 0x20(%rdi)+ vmovdqa 0x440(%rsi), %ymm2+ vmovdqa 0x460(%rsi), %ymm3+ vmovdqa 0x440(%rdx), %ymm4+ vmovdqa 0x460(%rdx), %ymm5+ vmovdqa 0x220(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x40(%rdi), %ymm8+ vmovdqa 0x60(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x40(%rdi)+ vmovdqa %ymm9, 0x60(%rdi)+ vmovdqa 0x480(%rsi), %ymm2+ vmovdqa 0x4a0(%rsi), %ymm3+ vmovdqa 0x480(%rdx), %ymm4+ vmovdqa 0x4a0(%rdx), %ymm5+ vmovdqa 0x240(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x80(%rdi)+ vmovdqa %ymm9, 0xa0(%rdi)+ vmovdqa 0x4c0(%rsi), %ymm2+ vmovdqa 0x4e0(%rsi), %ymm3+ vmovdqa 0x4c0(%rdx), %ymm4+ vmovdqa 0x4e0(%rdx), %ymm5+ vmovdqa 0x260(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0xc0(%rdi), %ymm8+ vmovdqa 0xe0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0xc0(%rdi)+ vmovdqa %ymm9, 0xe0(%rdi)+ vmovdqa 0x500(%rsi), %ymm2+ vmovdqa 0x520(%rsi), %ymm3+ vmovdqa 0x500(%rdx), %ymm4+ vmovdqa 0x520(%rdx), %ymm5+ vmovdqa 0x280(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x100(%rdi), %ymm8+ vmovdqa 0x120(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x100(%rdi)+ vmovdqa %ymm9, 0x120(%rdi)+ vmovdqa 0x540(%rsi), %ymm2+ vmovdqa 0x560(%rsi), %ymm3+ vmovdqa 0x540(%rdx), %ymm4+ vmovdqa 0x560(%rdx), %ymm5+ vmovdqa 0x2a0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x140(%rdi), %ymm8+ vmovdqa 0x160(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x140(%rdi)+ vmovdqa %ymm9, 0x160(%rdi)+ vmovdqa 0x580(%rsi), %ymm2+ vmovdqa 0x5a0(%rsi), %ymm3+ vmovdqa 0x580(%rdx), %ymm4+ vmovdqa 0x5a0(%rdx), %ymm5+ vmovdqa 0x2c0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1a0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x180(%rdi)+ vmovdqa %ymm9, 0x1a0(%rdi)+ vmovdqa 0x5c0(%rsi), %ymm2+ vmovdqa 0x5e0(%rsi), %ymm3+ vmovdqa 0x5c0(%rdx), %ymm4+ vmovdqa 0x5e0(%rdx), %ymm5+ vmovdqa 0x2e0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x1c0(%rdi), %ymm8+ vmovdqa 0x1e0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x1c0(%rdi)+ vmovdqa %ymm9, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(polyvec_basemul_acc_montgomery_cached_k3_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 3) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S view
@@ -0,0 +1,1032 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4)+/*yaml+ Name: polyvec_basemul_acc_montgomery_cached_k4_avx2_asm+ Description: x86_64 AVX2 base multiplication with accumulation (k=4)+ Signature: void mlk_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm(int16_t *r, const int16_t *a, const int16_t *b, const int16_t *b_cache)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t *r+ description: Output polynomial (256 x int16_t)+ rsi:+ type: buffer+ size_bytes: 2048+ permissions: read-only+ c_parameter: const int16_t *a+ description: Input polyvec a (4 x 256 x int16_t)+ rdx:+ type: buffer+ size_bytes: 2048+ permissions: read-only+ c_parameter: const int16_t *b+ description: Input polyvec b (4 x 256 x int16_t)+ rcx:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int16_t *b_cache+ description: Mulcache for b (4 x 128 x int16_t)+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k4_avx2_asm)+MLK_ASM_FN_SYMBOL(polyvec_basemul_acc_montgomery_cached_k4_avx2_asm)++ .cfi_startproc+ movl $0xd010d01, %eax # imm = 0xD010D01+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ movl $0xf301f301, %eax # imm = 0xF301F301+ vmovd %eax, %xmm1+ vpbroadcastd %xmm1, %ymm1+ vmovdqa (%rsi), %ymm2+ vmovdqa 0x20(%rsi), %ymm3+ vmovdqa (%rdx), %ymm4+ vmovdqa 0x20(%rdx), %ymm5+ vmovdqa (%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, (%rdi)+ vmovdqa %ymm9, 0x20(%rdi)+ vmovdqa 0x40(%rsi), %ymm2+ vmovdqa 0x60(%rsi), %ymm3+ vmovdqa 0x40(%rdx), %ymm4+ vmovdqa 0x60(%rdx), %ymm5+ vmovdqa 0x20(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x40(%rdi)+ vmovdqa %ymm9, 0x60(%rdi)+ vmovdqa 0x80(%rsi), %ymm2+ vmovdqa 0xa0(%rsi), %ymm3+ vmovdqa 0x80(%rdx), %ymm4+ vmovdqa 0xa0(%rdx), %ymm5+ vmovdqa 0x40(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x80(%rdi)+ vmovdqa %ymm9, 0xa0(%rdi)+ vmovdqa 0xc0(%rsi), %ymm2+ vmovdqa 0xe0(%rsi), %ymm3+ vmovdqa 0xc0(%rdx), %ymm4+ vmovdqa 0xe0(%rdx), %ymm5+ vmovdqa 0x60(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0xc0(%rdi)+ vmovdqa %ymm9, 0xe0(%rdi)+ vmovdqa 0x100(%rsi), %ymm2+ vmovdqa 0x120(%rsi), %ymm3+ vmovdqa 0x100(%rdx), %ymm4+ vmovdqa 0x120(%rdx), %ymm5+ vmovdqa 0x80(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x100(%rdi)+ vmovdqa %ymm9, 0x120(%rdi)+ vmovdqa 0x140(%rsi), %ymm2+ vmovdqa 0x160(%rsi), %ymm3+ vmovdqa 0x140(%rdx), %ymm4+ vmovdqa 0x160(%rdx), %ymm5+ vmovdqa 0xa0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x140(%rdi)+ vmovdqa %ymm9, 0x160(%rdi)+ vmovdqa 0x180(%rsi), %ymm2+ vmovdqa 0x1a0(%rsi), %ymm3+ vmovdqa 0x180(%rdx), %ymm4+ vmovdqa 0x1a0(%rdx), %ymm5+ vmovdqa 0xc0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x180(%rdi)+ vmovdqa %ymm9, 0x1a0(%rdi)+ vmovdqa 0x1c0(%rsi), %ymm2+ vmovdqa 0x1e0(%rsi), %ymm3+ vmovdqa 0x1c0(%rdx), %ymm4+ vmovdqa 0x1e0(%rdx), %ymm5+ vmovdqa 0xe0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x1c0(%rdi)+ vmovdqa %ymm9, 0x1e0(%rdi)+ vmovdqa 0x200(%rsi), %ymm2+ vmovdqa 0x220(%rsi), %ymm3+ vmovdqa 0x200(%rdx), %ymm4+ vmovdqa 0x220(%rdx), %ymm5+ vmovdqa 0x100(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa (%rdi), %ymm8+ vmovdqa 0x20(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, (%rdi)+ vmovdqa %ymm9, 0x20(%rdi)+ vmovdqa 0x240(%rsi), %ymm2+ vmovdqa 0x260(%rsi), %ymm3+ vmovdqa 0x240(%rdx), %ymm4+ vmovdqa 0x260(%rdx), %ymm5+ vmovdqa 0x120(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x40(%rdi), %ymm8+ vmovdqa 0x60(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x40(%rdi)+ vmovdqa %ymm9, 0x60(%rdi)+ vmovdqa 0x280(%rsi), %ymm2+ vmovdqa 0x2a0(%rsi), %ymm3+ vmovdqa 0x280(%rdx), %ymm4+ vmovdqa 0x2a0(%rdx), %ymm5+ vmovdqa 0x140(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x80(%rdi)+ vmovdqa %ymm9, 0xa0(%rdi)+ vmovdqa 0x2c0(%rsi), %ymm2+ vmovdqa 0x2e0(%rsi), %ymm3+ vmovdqa 0x2c0(%rdx), %ymm4+ vmovdqa 0x2e0(%rdx), %ymm5+ vmovdqa 0x160(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0xc0(%rdi), %ymm8+ vmovdqa 0xe0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0xc0(%rdi)+ vmovdqa %ymm9, 0xe0(%rdi)+ vmovdqa 0x300(%rsi), %ymm2+ vmovdqa 0x320(%rsi), %ymm3+ vmovdqa 0x300(%rdx), %ymm4+ vmovdqa 0x320(%rdx), %ymm5+ vmovdqa 0x180(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x100(%rdi), %ymm8+ vmovdqa 0x120(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x100(%rdi)+ vmovdqa %ymm9, 0x120(%rdi)+ vmovdqa 0x340(%rsi), %ymm2+ vmovdqa 0x360(%rsi), %ymm3+ vmovdqa 0x340(%rdx), %ymm4+ vmovdqa 0x360(%rdx), %ymm5+ vmovdqa 0x1a0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x140(%rdi), %ymm8+ vmovdqa 0x160(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x140(%rdi)+ vmovdqa %ymm9, 0x160(%rdi)+ vmovdqa 0x380(%rsi), %ymm2+ vmovdqa 0x3a0(%rsi), %ymm3+ vmovdqa 0x380(%rdx), %ymm4+ vmovdqa 0x3a0(%rdx), %ymm5+ vmovdqa 0x1c0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1a0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x180(%rdi)+ vmovdqa %ymm9, 0x1a0(%rdi)+ vmovdqa 0x3c0(%rsi), %ymm2+ vmovdqa 0x3e0(%rsi), %ymm3+ vmovdqa 0x3c0(%rdx), %ymm4+ vmovdqa 0x3e0(%rdx), %ymm5+ vmovdqa 0x1e0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x1c0(%rdi), %ymm8+ vmovdqa 0x1e0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x1c0(%rdi)+ vmovdqa %ymm9, 0x1e0(%rdi)+ vmovdqa 0x400(%rsi), %ymm2+ vmovdqa 0x420(%rsi), %ymm3+ vmovdqa 0x400(%rdx), %ymm4+ vmovdqa 0x420(%rdx), %ymm5+ vmovdqa 0x200(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa (%rdi), %ymm8+ vmovdqa 0x20(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, (%rdi)+ vmovdqa %ymm9, 0x20(%rdi)+ vmovdqa 0x440(%rsi), %ymm2+ vmovdqa 0x460(%rsi), %ymm3+ vmovdqa 0x440(%rdx), %ymm4+ vmovdqa 0x460(%rdx), %ymm5+ vmovdqa 0x220(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x40(%rdi), %ymm8+ vmovdqa 0x60(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x40(%rdi)+ vmovdqa %ymm9, 0x60(%rdi)+ vmovdqa 0x480(%rsi), %ymm2+ vmovdqa 0x4a0(%rsi), %ymm3+ vmovdqa 0x480(%rdx), %ymm4+ vmovdqa 0x4a0(%rdx), %ymm5+ vmovdqa 0x240(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x80(%rdi)+ vmovdqa %ymm9, 0xa0(%rdi)+ vmovdqa 0x4c0(%rsi), %ymm2+ vmovdqa 0x4e0(%rsi), %ymm3+ vmovdqa 0x4c0(%rdx), %ymm4+ vmovdqa 0x4e0(%rdx), %ymm5+ vmovdqa 0x260(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0xc0(%rdi), %ymm8+ vmovdqa 0xe0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0xc0(%rdi)+ vmovdqa %ymm9, 0xe0(%rdi)+ vmovdqa 0x500(%rsi), %ymm2+ vmovdqa 0x520(%rsi), %ymm3+ vmovdqa 0x500(%rdx), %ymm4+ vmovdqa 0x520(%rdx), %ymm5+ vmovdqa 0x280(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x100(%rdi), %ymm8+ vmovdqa 0x120(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x100(%rdi)+ vmovdqa %ymm9, 0x120(%rdi)+ vmovdqa 0x540(%rsi), %ymm2+ vmovdqa 0x560(%rsi), %ymm3+ vmovdqa 0x540(%rdx), %ymm4+ vmovdqa 0x560(%rdx), %ymm5+ vmovdqa 0x2a0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x140(%rdi), %ymm8+ vmovdqa 0x160(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x140(%rdi)+ vmovdqa %ymm9, 0x160(%rdi)+ vmovdqa 0x580(%rsi), %ymm2+ vmovdqa 0x5a0(%rsi), %ymm3+ vmovdqa 0x580(%rdx), %ymm4+ vmovdqa 0x5a0(%rdx), %ymm5+ vmovdqa 0x2c0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1a0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x180(%rdi)+ vmovdqa %ymm9, 0x1a0(%rdi)+ vmovdqa 0x5c0(%rsi), %ymm2+ vmovdqa 0x5e0(%rsi), %ymm3+ vmovdqa 0x5c0(%rdx), %ymm4+ vmovdqa 0x5e0(%rdx), %ymm5+ vmovdqa 0x2e0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x1c0(%rdi), %ymm8+ vmovdqa 0x1e0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x1c0(%rdi)+ vmovdqa %ymm9, 0x1e0(%rdi)+ vmovdqa 0x600(%rsi), %ymm2+ vmovdqa 0x620(%rsi), %ymm3+ vmovdqa 0x600(%rdx), %ymm4+ vmovdqa 0x620(%rdx), %ymm5+ vmovdqa 0x300(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa (%rdi), %ymm8+ vmovdqa 0x20(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, (%rdi)+ vmovdqa %ymm9, 0x20(%rdi)+ vmovdqa 0x640(%rsi), %ymm2+ vmovdqa 0x660(%rsi), %ymm3+ vmovdqa 0x640(%rdx), %ymm4+ vmovdqa 0x660(%rdx), %ymm5+ vmovdqa 0x320(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x40(%rdi), %ymm8+ vmovdqa 0x60(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x40(%rdi)+ vmovdqa %ymm9, 0x60(%rdi)+ vmovdqa 0x680(%rsi), %ymm2+ vmovdqa 0x6a0(%rsi), %ymm3+ vmovdqa 0x680(%rdx), %ymm4+ vmovdqa 0x6a0(%rdx), %ymm5+ vmovdqa 0x340(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x80(%rdi)+ vmovdqa %ymm9, 0xa0(%rdi)+ vmovdqa 0x6c0(%rsi), %ymm2+ vmovdqa 0x6e0(%rsi), %ymm3+ vmovdqa 0x6c0(%rdx), %ymm4+ vmovdqa 0x6e0(%rdx), %ymm5+ vmovdqa 0x360(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0xc0(%rdi), %ymm8+ vmovdqa 0xe0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0xc0(%rdi)+ vmovdqa %ymm9, 0xe0(%rdi)+ vmovdqa 0x700(%rsi), %ymm2+ vmovdqa 0x720(%rsi), %ymm3+ vmovdqa 0x700(%rdx), %ymm4+ vmovdqa 0x720(%rdx), %ymm5+ vmovdqa 0x380(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x100(%rdi), %ymm8+ vmovdqa 0x120(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x100(%rdi)+ vmovdqa %ymm9, 0x120(%rdi)+ vmovdqa 0x740(%rsi), %ymm2+ vmovdqa 0x760(%rsi), %ymm3+ vmovdqa 0x740(%rdx), %ymm4+ vmovdqa 0x760(%rdx), %ymm5+ vmovdqa 0x3a0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x140(%rdi), %ymm8+ vmovdqa 0x160(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x140(%rdi)+ vmovdqa %ymm9, 0x160(%rdi)+ vmovdqa 0x780(%rsi), %ymm2+ vmovdqa 0x7a0(%rsi), %ymm3+ vmovdqa 0x780(%rdx), %ymm4+ vmovdqa 0x7a0(%rdx), %ymm5+ vmovdqa 0x3c0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1a0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x180(%rdi)+ vmovdqa %ymm9, 0x1a0(%rdi)+ vmovdqa 0x7c0(%rsi), %ymm2+ vmovdqa 0x7e0(%rsi), %ymm3+ vmovdqa 0x7c0(%rdx), %ymm4+ vmovdqa 0x7e0(%rdx), %ymm5+ vmovdqa 0x3e0(%rcx), %ymm6+ vpmullw %ymm2, %ymm1, %ymm13+ vpmullw %ymm3, %ymm1, %ymm14+ vpmullw %ymm13, %ymm4, %ymm7+ vpmullw %ymm13, %ymm5, %ymm9+ vpmullw %ymm14, %ymm6, %ymm8+ vpmullw %ymm14, %ymm4, %ymm10+ vpmulhw %ymm7, %ymm0, %ymm7+ vpmulhw %ymm9, %ymm0, %ymm9+ vpmulhw %ymm8, %ymm0, %ymm8+ vpmulhw %ymm10, %ymm0, %ymm10+ vpmulhw %ymm2, %ymm4, %ymm11+ vpmulhw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm13+ vpmulhw %ymm3, %ymm4, %ymm14+ vpsubw %ymm7, %ymm11, %ymm7+ vpsubw %ymm9, %ymm12, %ymm9+ vpsubw %ymm13, %ymm8, %ymm8+ vpsubw %ymm10, %ymm14, %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa 0x1c0(%rdi), %ymm8+ vmovdqa 0x1e0(%rdi), %ymm10+ vpaddw %ymm7, %ymm8, %ymm7+ vpaddw %ymm9, %ymm10, %ymm9+ vmovdqa %ymm7, 0x1c0(%rdi)+ vmovdqa %ymm9, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(polyvec_basemul_acc_montgomery_cached_k4_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_reduce_avx2_asm.S view
@@ -0,0 +1,234 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Kyber implementation @[REF_AVX2].+ *+ * Changes:+ * - Add call to csub in reduce128_avx to produce outputs+ * in [0,1,...,q-1] rather than [0,1,...,q], matching the+ * semantics of mlk_poly_reduce(),+ * - Use a macro instead of a local function call.+ */++#include "../../../common.h"++#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)+/*yaml+ Name: reduce_avx2_asm+ Description: x86_64 AVX2 modular reduction+ Signature: void mlk_reduce_avx2_asm(int16_t *r)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t *r+ description: Input/output polynomial (256 x int16_t)+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_reduce_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(reduce_avx2_asm)+MLK_ASM_FN_SYMBOL(reduce_avx2_asm)++ .cfi_startproc+ movl $0xd010d01, %eax # imm = 0xD010D01+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ movl $0x4ebf4ebf, %eax # imm = 0x4EBF4EBF+ vmovd %eax, %xmm1+ vpbroadcastd %xmm1, %ymm1+ vmovdqa (%rdi), %ymm2+ vmovdqa 0x20(%rdi), %ymm3+ vmovdqa 0x40(%rdi), %ymm4+ vmovdqa 0x60(%rdi), %ymm5+ vmovdqa 0x80(%rdi), %ymm6+ vmovdqa 0xa0(%rdi), %ymm7+ vmovdqa 0xc0(%rdi), %ymm8+ vmovdqa 0xe0(%rdi), %ymm9+ vpmulhw %ymm1, %ymm2, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm2, %ymm2+ vpmulhw %ymm1, %ymm3, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm3, %ymm3+ vpmulhw %ymm1, %ymm4, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm4, %ymm4+ vpmulhw %ymm1, %ymm5, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm5, %ymm5+ vpmulhw %ymm1, %ymm6, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm6, %ymm6+ vpmulhw %ymm1, %ymm7, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm7, %ymm7+ vpmulhw %ymm1, %ymm8, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm8, %ymm8+ vpmulhw %ymm1, %ymm9, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm9, %ymm9+ vpsubw %ymm0, %ymm2, %ymm2+ vpsraw $0xf, %ymm2, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm2, %ymm2+ vpsubw %ymm0, %ymm3, %ymm3+ vpsraw $0xf, %ymm3, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm3, %ymm3+ vpsubw %ymm0, %ymm4, %ymm4+ vpsraw $0xf, %ymm4, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm4, %ymm4+ vpsubw %ymm0, %ymm5, %ymm5+ vpsraw $0xf, %ymm5, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm5, %ymm5+ vpsubw %ymm0, %ymm6, %ymm6+ vpsraw $0xf, %ymm6, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm6, %ymm6+ vpsubw %ymm0, %ymm7, %ymm7+ vpsraw $0xf, %ymm7, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm7, %ymm7+ vpsubw %ymm0, %ymm8, %ymm8+ vpsraw $0xf, %ymm8, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm8, %ymm8+ vpsubw %ymm0, %ymm9, %ymm9+ vpsraw $0xf, %ymm9, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm9, %ymm9+ vmovdqa %ymm2, (%rdi)+ vmovdqa %ymm3, 0x20(%rdi)+ vmovdqa %ymm4, 0x40(%rdi)+ vmovdqa %ymm5, 0x60(%rdi)+ vmovdqa %ymm6, 0x80(%rdi)+ vmovdqa %ymm7, 0xa0(%rdi)+ vmovdqa %ymm8, 0xc0(%rdi)+ vmovdqa %ymm9, 0xe0(%rdi)+ vmovdqa 0x100(%rdi), %ymm2+ vmovdqa 0x120(%rdi), %ymm3+ vmovdqa 0x140(%rdi), %ymm4+ vmovdqa 0x160(%rdi), %ymm5+ vmovdqa 0x180(%rdi), %ymm6+ vmovdqa 0x1a0(%rdi), %ymm7+ vmovdqa 0x1c0(%rdi), %ymm8+ vmovdqa 0x1e0(%rdi), %ymm9+ vpmulhw %ymm1, %ymm2, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm2, %ymm2+ vpmulhw %ymm1, %ymm3, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm3, %ymm3+ vpmulhw %ymm1, %ymm4, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm4, %ymm4+ vpmulhw %ymm1, %ymm5, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm5, %ymm5+ vpmulhw %ymm1, %ymm6, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm6, %ymm6+ vpmulhw %ymm1, %ymm7, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm7, %ymm7+ vpmulhw %ymm1, %ymm8, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm8, %ymm8+ vpmulhw %ymm1, %ymm9, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm9, %ymm9+ vpsubw %ymm0, %ymm2, %ymm2+ vpsraw $0xf, %ymm2, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm2, %ymm2+ vpsubw %ymm0, %ymm3, %ymm3+ vpsraw $0xf, %ymm3, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm3, %ymm3+ vpsubw %ymm0, %ymm4, %ymm4+ vpsraw $0xf, %ymm4, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm4, %ymm4+ vpsubw %ymm0, %ymm5, %ymm5+ vpsraw $0xf, %ymm5, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm5, %ymm5+ vpsubw %ymm0, %ymm6, %ymm6+ vpsraw $0xf, %ymm6, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm6, %ymm6+ vpsubw %ymm0, %ymm7, %ymm7+ vpsraw $0xf, %ymm7, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm7, %ymm7+ vpsubw %ymm0, %ymm8, %ymm8+ vpsraw $0xf, %ymm8, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm8, %ymm8+ vpsubw %ymm0, %ymm9, %ymm9+ vpsraw $0xf, %ymm9, %ymm12+ vpand %ymm0, %ymm12, %ymm12+ vpaddw %ymm12, %ymm9, %ymm9+ vmovdqa %ymm2, 0x100(%rdi)+ vmovdqa %ymm3, 0x120(%rdi)+ vmovdqa %ymm4, 0x140(%rdi)+ vmovdqa %ymm5, 0x160(%rdi)+ vmovdqa %ymm6, 0x180(%rdi)+ vmovdqa %ymm7, 0x1a0(%rdi)+ vmovdqa %ymm8, 0x1c0(%rdi)+ vmovdqa %ymm9, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(reduce_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_rej_uniform_avx2_asm.S view
@@ -0,0 +1,136 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*************************************************+ * Name: mlk_rej_uniform_avx2_asm+ *+ * Description: Run rejection sampling on uniform random bytes to generate+ * uniform random integers mod q+ *+ * Arguments: - int16_t *r: pointer to output buffer of MLKEM_N+ * 16-bit coefficients.+ * - const uint8_t *buf: pointer to input buffer+ * (assumed to be uniform random bytes)+ * - unsigned buflen: length of input buffer in bytes.+ * Must be a multiple of 12.+ *+ * Returns number of sampled 16-bit integers (at most MLKEM_N).+ **************************************************/+#include "../../../common.h"++#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)+/*yaml+ Name: rej_uniform_avx2_asm+ Description: x86_64 AVX2 rejection sampling+ Signature: uint64_t mlk_rej_uniform_avx2_asm(int16_t *r, const uint8_t *buf, unsigned buflen, const uint8_t *table)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 512+ permissions: write-only+ c_parameter: int16_t *r+ description: Output buffer (256 x int16_t)+ rsi:+ type: buffer+ size_bytes: rdx+ permissions: read-only+ c_parameter: const uint8_t *buf+ description: Input buffer+ rdx:+ type: scalar+ c_parameter: unsigned buflen+ description: Length of input buffer (must be multiple of 12)+ test_with: 504 # MLKEM_GEN_MATRIX_NBLOCKS * MLK_XOF_RATE+ rcx:+ type: buffer+ size_bytes: 4096+ permissions: read-only+ c_parameter: const uint8_t *table+ description: Lookup table+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_rej_uniform_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(rej_uniform_avx2_asm)+MLK_ASM_FN_SYMBOL(rej_uniform_avx2_asm)++ .cfi_startproc+ subq $0x210, %rsp # imm = 0x210+ .cfi_adjust_cfa_offset 0x210+ xorl %eax, %eax+ testq %rdx, %rdx+ je Lmlk_rej_uniform_asm_end+ movabsq $0xd010d010d010d01, %rax # imm = 0xD010D010D010D01+ movq %rax, %xmm0+ pinsrq $0x1, %rax, %xmm0+ movabsq $0xfff0fff0fff0fff, %rax # imm = 0xFFF0FFF0FFF0FFF+ movq %rax, %xmm5+ pinsrq $0x1, %rax, %xmm5+ movabsq $0x504040302010100, %rax # imm = 0x504040302010100+ movq %rax, %xmm4+ movabsq $0xb0a0a0908070706, %rax # imm = 0xB0A0A0908070706+ pinsrq $0x1, %rax, %xmm4+ movq $0x0, %rax+ movq $0x0, %r8+ movq $0x5555, %r9 # imm = 0x5555++Lmlk_rej_uniform_asm_loop_start:+ movq (%rsi,%r8), %xmm2+ pinsrd $0x2, 0x8(%rsi,%r8), %xmm2+ pshufb %xmm4, %xmm2+ movdqa %xmm2, %xmm3+ psrlw $0x4, %xmm3+ pblendw $0xaa, %xmm3, %xmm2 # xmm2 = xmm2[0],xmm3[1],xmm2[2],xmm3[3],xmm2[4],xmm3[5],xmm2[6],xmm3[7]+ pand %xmm5, %xmm2+ movdqa %xmm0, %xmm1+ pcmpgtw %xmm2, %xmm1+ pmovmskb %xmm1, %r11d+ pextq %r9, %r11, %r11+ movq %r11, %r10+ shlq $0x4, %r10+ movdqu (%rcx,%r10), %xmm3+ pshufb %xmm3, %xmm2+ movdqu %xmm2, (%rsp,%rax,2)+ popcntq %r11, %r11+ addq %r11, %rax+ cmpq $0x100, %rax # imm = 0x100+ jae Lmlk_rej_uniform_asm_final_copy+ addq $0xc, %r8+ cmpq %r8, %rdx+ ja Lmlk_rej_uniform_asm_loop_start++Lmlk_rej_uniform_asm_final_copy:+ movq $0x100, %rcx # imm = 0x100+ cmpq $0x100, %rax # imm = 0x100+ cmovaq %rcx, %rax+ movq %rsp, %rsi+ movq %rax, %rcx+ shlq %rcx+ rep movsb (%rsi), %es:(%rdi)++Lmlk_rej_uniform_asm_end:+ addq $0x210, %rsp # imm = 0x210+ .cfi_adjust_cfa_offset -0x210+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(rej_uniform_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/mlkem_tomont_avx2_asm.S view
@@ -0,0 +1,172 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * Implementation from Kyber reference repository @[REF_AVX2]+ *+ * Changes:+ * - Add call to csub in reduce128_avx to produce outputs+ * in [0,1,...,q-1] rather than [0,1,...,q], matching the+ * semantics of mlk_poly_reduce(),+ * - Use a macro instead of a local function call.+ */++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ !defined(MLK_CONFIG_NO_KEYPAIR_API)+/*yaml+ Name: tomont_avx2_asm+ Description: x86_64 AVX2 Montgomery conversion+ Signature: void mlk_tomont_avx2_asm(int16_t *r)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t *r+ description: Input/output polynomial (256 x int16_t)+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_tomont_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(tomont_avx2_asm)+MLK_ASM_FN_SYMBOL(tomont_avx2_asm)++ .cfi_startproc+ movl $0xd010d01, %eax # imm = 0xD010D01+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ movl $0x50495049, %eax # imm = 0x50495049+ vmovd %eax, %xmm1+ vpbroadcastd %xmm1, %ymm1+ movl $0x5490549, %eax # imm = 0x5490549+ vmovd %eax, %xmm2+ vpbroadcastd %xmm2, %ymm2+ vmovdqa (%rdi), %ymm3+ vmovdqa 0x20(%rdi), %ymm4+ vmovdqa 0x40(%rdi), %ymm5+ vmovdqa 0x60(%rdi), %ymm6+ vmovdqa 0x80(%rdi), %ymm7+ vmovdqa 0xa0(%rdi), %ymm8+ vmovdqa 0xc0(%rdi), %ymm9+ vmovdqa 0xe0(%rdi), %ymm10+ vpmullw %ymm1, %ymm3, %ymm11+ vpmulhw %ymm2, %ymm3, %ymm3+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm11, %ymm3, %ymm3+ vpmullw %ymm1, %ymm4, %ymm12+ vpmulhw %ymm2, %ymm4, %ymm4+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm4, %ymm4+ vpmullw %ymm1, %ymm5, %ymm13+ vpmulhw %ymm2, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm13, %ymm13+ vpsubw %ymm13, %ymm5, %ymm5+ vpmullw %ymm1, %ymm6, %ymm14+ vpmulhw %ymm2, %ymm6, %ymm6+ vpmulhw %ymm0, %ymm14, %ymm14+ vpsubw %ymm14, %ymm6, %ymm6+ vpmullw %ymm1, %ymm7, %ymm15+ vpmulhw %ymm2, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm15, %ymm15+ vpsubw %ymm15, %ymm7, %ymm7+ vpmullw %ymm1, %ymm8, %ymm11+ vpmulhw %ymm2, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm11, %ymm8, %ymm8+ vpmullw %ymm1, %ymm9, %ymm12+ vpmulhw %ymm2, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm9, %ymm9+ vpmullw %ymm1, %ymm10, %ymm13+ vpmulhw %ymm2, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm13, %ymm13+ vpsubw %ymm13, %ymm10, %ymm10+ vmovdqa %ymm3, (%rdi)+ vmovdqa %ymm4, 0x20(%rdi)+ vmovdqa %ymm5, 0x40(%rdi)+ vmovdqa %ymm6, 0x60(%rdi)+ vmovdqa %ymm7, 0x80(%rdi)+ vmovdqa %ymm8, 0xa0(%rdi)+ vmovdqa %ymm9, 0xc0(%rdi)+ vmovdqa %ymm10, 0xe0(%rdi)+ vmovdqa 0x100(%rdi), %ymm3+ vmovdqa 0x120(%rdi), %ymm4+ vmovdqa 0x140(%rdi), %ymm5+ vmovdqa 0x160(%rdi), %ymm6+ vmovdqa 0x180(%rdi), %ymm7+ vmovdqa 0x1a0(%rdi), %ymm8+ vmovdqa 0x1c0(%rdi), %ymm9+ vmovdqa 0x1e0(%rdi), %ymm10+ vpmullw %ymm1, %ymm3, %ymm11+ vpmulhw %ymm2, %ymm3, %ymm3+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm11, %ymm3, %ymm3+ vpmullw %ymm1, %ymm4, %ymm12+ vpmulhw %ymm2, %ymm4, %ymm4+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm4, %ymm4+ vpmullw %ymm1, %ymm5, %ymm13+ vpmulhw %ymm2, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm13, %ymm13+ vpsubw %ymm13, %ymm5, %ymm5+ vpmullw %ymm1, %ymm6, %ymm14+ vpmulhw %ymm2, %ymm6, %ymm6+ vpmulhw %ymm0, %ymm14, %ymm14+ vpsubw %ymm14, %ymm6, %ymm6+ vpmullw %ymm1, %ymm7, %ymm15+ vpmulhw %ymm2, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm15, %ymm15+ vpsubw %ymm15, %ymm7, %ymm7+ vpmullw %ymm1, %ymm8, %ymm11+ vpmulhw %ymm2, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm11, %ymm8, %ymm8+ vpmullw %ymm1, %ymm9, %ymm12+ vpmulhw %ymm2, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm9, %ymm9+ vpmullw %ymm1, %ymm10, %ymm13+ vpmulhw %ymm2, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm13, %ymm13+ vpsubw %ymm13, %ymm10, %ymm10+ vmovdqa %ymm3, 0x100(%rdi)+ vmovdqa %ymm4, 0x120(%rdi)+ vmovdqa %ymm5, 0x140(%rdi)+ vmovdqa %ymm6, 0x160(%rdi)+ vmovdqa %ymm7, 0x180(%rdi)+ vmovdqa %ymm8, 0x1a0(%rdi)+ vmovdqa %ymm9, 0x1c0(%rdi)+ vmovdqa %ymm10, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(tomont_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && !MLK_CONFIG_NO_KEYPAIR_API */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
+ cbits/mlkem/src/native/x86_64/src/rej_uniform_table.c view
@@ -0,0 +1,545 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#include "arith_native_x86_64.h"++/*+ * Lookup table used by rejection sampling of the public matrix.+ * See autogen for details.+ */+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_rej_uniform_table[4096] = {+ 255, 255, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 0 */,+ 0, 1, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 1 */,+ 2, 3, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 2 */,+ 0, 1, 2, 3, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 3 */,+ 4, 5, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 4 */,+ 0, 1, 4, 5, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 5 */,+ 2, 3, 4, 5, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 6 */,+ 0, 1, 2, 3, 4, 5, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 7 */,+ 6, 7, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 8 */,+ 0, 1, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 9 */,+ 2, 3, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 10 */,+ 0, 1, 2, 3, 6, 7, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 11 */,+ 4, 5, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 12 */,+ 0, 1, 4, 5, 6, 7, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 13 */,+ 2, 3, 4, 5, 6, 7, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 14 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 15 */,+ 8, 9, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 16 */,+ 0, 1, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 17 */,+ 2, 3, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 18 */,+ 0, 1, 2, 3, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 19 */,+ 4, 5, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 20 */,+ 0, 1, 4, 5, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 21 */,+ 2, 3, 4, 5, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 22 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 23 */,+ 6, 7, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 24 */,+ 0, 1, 6, 7, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 25 */,+ 2, 3, 6, 7, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 26 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 27 */,+ 4, 5, 6, 7, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 28 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 29 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 30 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 255, 255, 255, 255, 255, 255 /* 31 */,+ 10, 11, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 32 */,+ 0, 1, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 33 */,+ 2, 3, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 34 */,+ 0, 1, 2, 3, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 35 */,+ 4, 5, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 36 */,+ 0, 1, 4, 5, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 37 */,+ 2, 3, 4, 5, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 38 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 39 */,+ 6, 7, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 40 */,+ 0, 1, 6, 7, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 41 */,+ 2, 3, 6, 7, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 42 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 43 */,+ 4, 5, 6, 7, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 44 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 45 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 46 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 47 */,+ 8, 9, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 48 */,+ 0, 1, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 49 */,+ 2, 3, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 50 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 51 */,+ 4, 5, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 52 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 53 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 54 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 55 */,+ 6, 7, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 56 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 57 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 58 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 59 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 60 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 61 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 62 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 255, 255, 255, 255 /* 63 */,+ 12, 13, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 64 */,+ 0, 1, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 65 */,+ 2, 3, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 66 */,+ 0, 1, 2, 3, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 67 */,+ 4, 5, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 68 */,+ 0, 1, 4, 5, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 69 */,+ 2, 3, 4, 5, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 70 */,+ 0, 1, 2, 3, 4, 5, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 71 */,+ 6, 7, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 72 */,+ 0, 1, 6, 7, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 73 */,+ 2, 3, 6, 7, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 74 */,+ 0, 1, 2, 3, 6, 7, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 75 */,+ 4, 5, 6, 7, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 76 */,+ 0, 1, 4, 5, 6, 7, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 77 */,+ 2, 3, 4, 5, 6, 7, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 78 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 79 */,+ 8, 9, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 80 */,+ 0, 1, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 81 */,+ 2, 3, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 82 */,+ 0, 1, 2, 3, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 83 */,+ 4, 5, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 84 */,+ 0, 1, 4, 5, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 85 */,+ 2, 3, 4, 5, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 86 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 87 */,+ 6, 7, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 88 */,+ 0, 1, 6, 7, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 89 */,+ 2, 3, 6, 7, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 90 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 91 */,+ 4, 5, 6, 7, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 92 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 93 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 94 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 12, 13, 255, 255, 255, 255 /* 95 */,+ 10, 11, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 96 */,+ 0, 1, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 97 */,+ 2, 3, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 98 */,+ 0, 1, 2, 3, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 99 */,+ 4, 5, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 100 */,+ 0, 1, 4, 5, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 101 */,+ 2, 3, 4, 5, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 102 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 103 */,+ 6, 7, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 104 */,+ 0, 1, 6, 7, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 105 */,+ 2, 3, 6, 7, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 106 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 107 */,+ 4, 5, 6, 7, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 108 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 109 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 110 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 111 */,+ 8, 9, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 112 */,+ 0, 1, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 113 */,+ 2, 3, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 114 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 115 */,+ 4, 5, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 116 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 117 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 118 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 119 */,+ 6, 7, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 120 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 121 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 122 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 123 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 124 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 125 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 126 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 12, 13, 255, 255 /* 127 */,+ 14, 15, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 128 */,+ 0, 1, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 129 */,+ 2, 3, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 130 */,+ 0, 1, 2, 3, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 131 */,+ 4, 5, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 132 */,+ 0, 1, 4, 5, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 133 */,+ 2, 3, 4, 5, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 134 */,+ 0, 1, 2, 3, 4, 5, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 135 */,+ 6, 7, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 136 */,+ 0, 1, 6, 7, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 137 */,+ 2, 3, 6, 7, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 138 */,+ 0, 1, 2, 3, 6, 7, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 139 */,+ 4, 5, 6, 7, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 140 */,+ 0, 1, 4, 5, 6, 7, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 141 */,+ 2, 3, 4, 5, 6, 7, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 142 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 143 */,+ 8, 9, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 144 */,+ 0, 1, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 145 */,+ 2, 3, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 146 */,+ 0, 1, 2, 3, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 147 */,+ 4, 5, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 148 */,+ 0, 1, 4, 5, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 149 */,+ 2, 3, 4, 5, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 150 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 151 */,+ 6, 7, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 152 */,+ 0, 1, 6, 7, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 153 */,+ 2, 3, 6, 7, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 154 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 155 */,+ 4, 5, 6, 7, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 156 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 157 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 158 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 14, 15, 255, 255, 255, 255 /* 159 */,+ 10, 11, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 160 */,+ 0, 1, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 161 */,+ 2, 3, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 162 */,+ 0, 1, 2, 3, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 163 */,+ 4, 5, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 164 */,+ 0, 1, 4, 5, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 165 */,+ 2, 3, 4, 5, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 166 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 167 */,+ 6, 7, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 168 */,+ 0, 1, 6, 7, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 169 */,+ 2, 3, 6, 7, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 170 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 171 */,+ 4, 5, 6, 7, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 172 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 173 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 174 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 175 */,+ 8, 9, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 176 */,+ 0, 1, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 177 */,+ 2, 3, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 178 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 179 */,+ 4, 5, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 180 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 181 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 182 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 183 */,+ 6, 7, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 184 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 185 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 186 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 187 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 188 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 189 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 190 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 14, 15, 255, 255 /* 191 */,+ 12, 13, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 192 */,+ 0, 1, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 193 */,+ 2, 3, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 194 */,+ 0, 1, 2, 3, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 195 */,+ 4, 5, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 196 */,+ 0, 1, 4, 5, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 197 */,+ 2, 3, 4, 5, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 198 */,+ 0, 1, 2, 3, 4, 5, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 199 */,+ 6, 7, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 200 */,+ 0, 1, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 201 */,+ 2, 3, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 202 */,+ 0, 1, 2, 3, 6, 7, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 203 */,+ 4, 5, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 204 */,+ 0, 1, 4, 5, 6, 7, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 205 */,+ 2, 3, 4, 5, 6, 7, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 206 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 207 */,+ 8, 9, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 208 */,+ 0, 1, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 209 */,+ 2, 3, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 210 */,+ 0, 1, 2, 3, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 211 */,+ 4, 5, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 212 */,+ 0, 1, 4, 5, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 213 */,+ 2, 3, 4, 5, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 214 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 215 */,+ 6, 7, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 216 */,+ 0, 1, 6, 7, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 217 */,+ 2, 3, 6, 7, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 218 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 219 */,+ 4, 5, 6, 7, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 220 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 221 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 222 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 12, 13, 14, 15, 255, 255 /* 223 */,+ 10, 11, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 224 */,+ 0, 1, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 225 */,+ 2, 3, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 226 */,+ 0, 1, 2, 3, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 227 */,+ 4, 5, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 228 */,+ 0, 1, 4, 5, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 229 */,+ 2, 3, 4, 5, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 230 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 231 */,+ 6, 7, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 232 */,+ 0, 1, 6, 7, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 233 */,+ 2, 3, 6, 7, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 234 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 235 */,+ 4, 5, 6, 7, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 236 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 237 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 238 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 239 */,+ 8, 9, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 240 */,+ 0, 1, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 241 */,+ 2, 3, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 242 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 243 */,+ 4, 5, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 244 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 245 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 246 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 247 */,+ 6, 7, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 248 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 249 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 250 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 251 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 252 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 253 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 254 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 12, 13, 14, 15 /* 255 */,+};++#else /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ */++MLK_EMPTY_CU(avx2_rej_uniform_table)++#endif /* !(MLK_ARITH_BACKEND_X86_64_DEFAULT && \+ !MLK_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mlkem/src/params.h view
@@ -0,0 +1,76 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_PARAMS_H+#define MLK_PARAMS_H++#if !defined(MLK_CONFIG_PARAMETER_SET)+#error MLK_CONFIG_PARAMETER_SET is not defined+#endif++#if MLK_CONFIG_PARAMETER_SET == 512+#define MLKEM_K 2+#elif MLK_CONFIG_PARAMETER_SET == 768+#define MLKEM_K 3+#elif MLK_CONFIG_PARAMETER_SET == 1024+#define MLKEM_K 4+#else+#error Invalid value for MLK_CONFIG_PARAMETER_SET. Must be 512, 768, or 1024.+#endif++#define MLKEM_N 256+#define MLKEM_Q 3329+#define MLKEM_Q_HALF ((MLKEM_Q + 1) / 2) /* 1665 */+#define MLKEM_UINT12_LIMIT 4096++#define MLKEM_SYMBYTES 32 /* size in bytes of hashes, and seeds */+#define MLKEM_SSBYTES 32 /* size in bytes of shared key */++#define MLKEM_POLYBYTES 384+#define MLKEM_POLYVECBYTES (MLKEM_K * MLKEM_POLYBYTES)++#define MLKEM_POLYCOMPRESSEDBYTES_D4 128+#define MLKEM_POLYCOMPRESSEDBYTES_D5 160+#define MLKEM_POLYCOMPRESSEDBYTES_D10 320+#define MLKEM_POLYCOMPRESSEDBYTES_D11 352++#if MLKEM_K == 2+#define MLKEM_ETA1 3+#define MLKEM_DU 10+#define MLKEM_DV 4+#define MLKEM_POLYCOMPRESSEDBYTES_DV MLKEM_POLYCOMPRESSEDBYTES_D4+#define MLKEM_POLYCOMPRESSEDBYTES_DU MLKEM_POLYCOMPRESSEDBYTES_D10+#define MLKEM_POLYVECCOMPRESSEDBYTES_DU (MLKEM_K * MLKEM_POLYCOMPRESSEDBYTES_DU)+#elif MLKEM_K == 3+#define MLKEM_ETA1 2+#define MLKEM_DU 10+#define MLKEM_DV 4+#define MLKEM_POLYCOMPRESSEDBYTES_DV MLKEM_POLYCOMPRESSEDBYTES_D4+#define MLKEM_POLYCOMPRESSEDBYTES_DU MLKEM_POLYCOMPRESSEDBYTES_D10+#define MLKEM_POLYVECCOMPRESSEDBYTES_DU (MLKEM_K * MLKEM_POLYCOMPRESSEDBYTES_DU)+#elif MLKEM_K == 4+#define MLKEM_ETA1 2+#define MLKEM_DU 11+#define MLKEM_DV 5+#define MLKEM_POLYCOMPRESSEDBYTES_DV MLKEM_POLYCOMPRESSEDBYTES_D5+#define MLKEM_POLYCOMPRESSEDBYTES_DU MLKEM_POLYCOMPRESSEDBYTES_D11+#define MLKEM_POLYVECCOMPRESSEDBYTES_DU (MLKEM_K * MLKEM_POLYCOMPRESSEDBYTES_DU)+#endif /* MLKEM_K == 4 */++#define MLKEM_ETA2 2++#define MLKEM_INDCPA_MSGBYTES (MLKEM_SYMBYTES)+#define MLKEM_INDCPA_PUBLICKEYBYTES (MLKEM_POLYVECBYTES + MLKEM_SYMBYTES)+#define MLKEM_INDCPA_SECRETKEYBYTES (MLKEM_POLYVECBYTES)+#define MLKEM_INDCPA_BYTES \+ (MLKEM_POLYVECCOMPRESSEDBYTES_DU + MLKEM_POLYCOMPRESSEDBYTES_DV)++#define MLKEM_INDCCA_PUBLICKEYBYTES (MLKEM_INDCPA_PUBLICKEYBYTES)+/* 32 bytes of additional space to save H(pk) */+#define MLKEM_INDCCA_SECRETKEYBYTES \+ (MLKEM_INDCPA_SECRETKEYBYTES + MLKEM_INDCPA_PUBLICKEYBYTES + \+ 2 * MLKEM_SYMBYTES)+#define MLKEM_INDCCA_CIPHERTEXTBYTES (MLKEM_INDCPA_BYTES)++#endif /* !MLK_PARAMS_H */
+ cbits/mlkem/src/poly.c view
@@ -0,0 +1,581 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ *+ * - [REF]+ * CRYSTALS-Kyber C reference implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/ref+ */++#include "common.h"+#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)+++#include "cbmc.h"+#include "debug.h"+#include "poly.h"+#include "sampling.h"+#include "symmetric.h"+#include "verify.h"++/**+ * Montgomery multiplication modulo MLKEM_Q.+ *+ * @reference{`fqmul()` in the reference implementation @[REF].}+ *+ * @param a First factor. Can be any int16_t.+ * @param b Second factor. Must be signed canonical+ * (abs value < (MLKEM_Q+1)/2).+ *+ * @return 16-bit integer congruent to a*b*R^{-1} mod MLKEM_Q, and+ * smaller than MLKEM_Q in absolute value.+ */+static MLK_INLINE int16_t mlk_fqmul(int16_t a, int16_t b)+__contract__(+ requires(b > -MLKEM_Q_HALF && b < MLKEM_Q_HALF)+ ensures(return_value > -MLKEM_Q && return_value < MLKEM_Q)+)+{+ int16_t res;+ mlk_assert_abs_bound(&b, 1, MLKEM_Q_HALF);++ res = mlk_montgomery_reduce((int32_t)a * (int32_t)b);+ /* Bounds:+ * |res| <= ceil(|a| * |b| / 2^16) + (MLKEM_Q + 1) / 2+ * <= ceil(2^15 * ((MLKEM_Q - 1)/2) / 2^16) + (MLKEM_Q + 1) / 2+ * <= ceil((MLKEM_Q - 1) / 4) + (MLKEM_Q + 1) / 2+ * < MLKEM_Q+ */++ mlk_assert_abs_bound(&res, 1, MLKEM_Q);+ return res;+}++/**+ * Barrett reduction; given a 16-bit integer a, computes the centered+ * representative congruent to a mod MLKEM_Q in [-(MLKEM_Q-1)/2, (MLKEM_Q-1)/2].+ *+ * @reference{`barrett_reduce()` in the reference implementation @[REF].}+ *+ * @param a Input integer to be reduced.+ *+ * @return Integer in [-(MLKEM_Q-1)/2, (MLKEM_Q-1)/2] congruent to @p a modulo+ * MLKEM_Q.+ */+static MLK_INLINE int16_t mlk_barrett_reduce(int16_t a)+__contract__(+ ensures(return_value > -MLKEM_Q_HALF && return_value < MLKEM_Q_HALF)+)+{+ /* Barrett reduction approximates+ * ```+ * round(a/MLKEM_Q)+ * = round(a*(2^N/MLKEM_Q))/2^N)+ * ~= round(a*round(2^N/MLKEM_Q)/2^N)+ * ```+ * Here, we pick N=26.+ */+ const int32_t magic = 20159; /* check-magic: 20159 == round(2^26 / MLKEM_Q) */++ /*+ * PORTABILITY: Right-shift on a signed integer is+ * implementation-defined for negative left argument.+ * Here, we assume it's sign-preserving "arithmetic" shift right.+ * See (C99 6.5.7 (5))+ */+ const int32_t t = (magic * a + ((int32_t)1 << 25)) >> 26;++ /*+ * t is in -10 .. +10, so we need 32-bit math to+ * evaluate t * MLKEM_Q and the subsequent subtraction+ */+ int16_t res = (int16_t)(a - t * MLKEM_Q);++ mlk_assert_abs_bound(&res, 1, MLKEM_Q_HALF);+ return res;+}++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+/* Reference: `poly_tomont()` in the reference implementation @[REF]. */+MLK_STATIC_TESTABLE void mlk_poly_tomont_c(mlk_poly *r)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_abs_bound(r->coeffs, 0, MLKEM_N, MLKEM_Q))+)+{+ unsigned i;+ const int16_t f = 1353; /* check-magic: 1353 == signed_mod(2^32, MLKEM_Q) */+ for (i = 0; i < MLKEM_N; i++)+ __loop__(+ invariant(i <= MLKEM_N)+ invariant(array_abs_bound(r->coeffs, 0, i, MLKEM_Q))+ decreases(MLKEM_N - i))+ {+ r->coeffs[i] = mlk_fqmul(r->coeffs[i], f);+ }++ mlk_assert_abs_bound(r, MLKEM_N, MLKEM_Q);+}++MLK_INTERNAL_API+void mlk_poly_tomont(mlk_poly *r)+{+#if defined(MLK_USE_NATIVE_POLY_TOMONT)+ int ret;+ ret = mlk_poly_tomont_native(r->coeffs);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ mlk_assert_abs_bound(r, MLKEM_N, MLKEM_Q);+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_TOMONT */++ mlk_poly_tomont_c(r);+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++/**+ * Constant-time conversion of signed representatives modulo MLKEM_Q within+ * range [-(MLKEM_Q-1), MLKEM_Q-1] into unsigned representatives within+ * range [0, MLKEM_Q-1].+ *+ * @reference{Not present in the reference implementation @[REF]. Used here+ * to implement different semantics of `poly_reduce()`; see below. In the+ * reference implementation @[REF] this logic is part of all compression+ * functions (see `compress.c`).}+ *+ * @param c Signed coefficient to be converted.+ *+ * @return Unsigned representative in [0, MLKEM_Q).+ */+static MLK_INLINE int16_t mlk_scalar_signed_to_unsigned_q(int16_t c)+__contract__(+ requires(c > -MLKEM_Q && c < MLKEM_Q)+ ensures(return_value >= 0 && return_value < MLKEM_Q)+ ensures(return_value == (int32_t)c + (((int32_t)c < 0) * MLKEM_Q)))+{+ mlk_assert_abs_bound(&c, 1, MLKEM_Q);++ /* Add MLKEM_Q if c is negative, but in constant time.+ *+ * Note that c + MLKEM_Q does not overflow in int16_t,+ * so the cast to uint16_t is safe. */+ c = mlk_ct_sel_int16((int16_t)(c + MLKEM_Q), c, mlk_ct_cmask_neg_i16(c));++ mlk_assert_bound(&c, 1, 0, MLKEM_Q);+ return c;+}++/* Reference: `poly_reduce()` in the reference implementation @[REF]+ * - We use _unsigned_ canonical outputs, while the reference+ * implementation uses _signed_ canonical outputs.+ * Accordingly, we need a conditional addition of MLKEM_Q+ * here to go from signed to unsigned representatives.+ * This conditional addition is then dropped from all+ * polynomial compression functions instead (see `compress.c`). */+MLK_STATIC_TESTABLE void mlk_poly_reduce_c(mlk_poly *r)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+ unsigned i;++ for (i = 0; i < MLKEM_N; i++)+ __loop__(+ invariant(i <= MLKEM_N)+ invariant(array_bound(r->coeffs, 0, i, 0, MLKEM_Q))+ decreases(MLKEM_N - i))+ {+ /* Barrett reduction, giving signed canonical representative */+ int16_t t = mlk_barrett_reduce(r->coeffs[i]);+ /* Conditional addition to get unsigned canonical representative */+ r->coeffs[i] = mlk_scalar_signed_to_unsigned_q(t);+ }++ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+}++MLK_INTERNAL_API+void mlk_poly_reduce(mlk_poly *r)+{+#if defined(MLK_USE_NATIVE_POLY_REDUCE)+ int ret;+ ret = mlk_poly_reduce_native(r->coeffs);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_REDUCE */++ mlk_poly_reduce_c(r);+}++/* Reference: `poly_add()` in the reference implementation @[REF].+ * - We use destructive version (output=first input) to avoid+ * reasoning about aliasing in the CBMC specification */+MLK_INTERNAL_API+void mlk_poly_add(mlk_poly *r, const mlk_poly *b)+{+ unsigned i;+ for (i = 0; i < MLKEM_N; i++)+ __loop__(+ invariant(i <= MLKEM_N)+ invariant(forall(k0, i, MLKEM_N, r->coeffs[k0] == loop_entry(*r).coeffs[k0]))+ invariant(forall(k1, 0, i, r->coeffs[k1] == loop_entry(*r).coeffs[k1] + b->coeffs[k1]))+ decreases(MLKEM_N - i))+ {+ /* The preconditions imply that the addition stays within int16_t. */+ r->coeffs[i] = (int16_t)(r->coeffs[i] + b->coeffs[i]);+ }+}++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `poly_sub()` in the reference implementation @[REF].+ * - We use destructive version (output=first input) to avoid+ * reasoning about aliasing in the CBMC specification */+MLK_INTERNAL_API+void mlk_poly_sub(mlk_poly *r, const mlk_poly *b)+{+ unsigned i;+ for (i = 0; i < MLKEM_N; i++)+ __loop__(+ invariant(i <= MLKEM_N)+ invariant(forall(k0, i, MLKEM_N, r->coeffs[k0] == loop_entry(*r).coeffs[k0]))+ invariant(forall(k1, 0, i, r->coeffs[k1] == loop_entry(*r).coeffs[k1] - b->coeffs[k1]))+ decreases(MLKEM_N - i))+ {+ /* The preconditions imply that the subtraction stays within int16_t. */+ r->coeffs[i] = (int16_t)(r->coeffs[i] - b->coeffs[i]);+ }+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#include "zetas.inc"++/* Reference: Does not exist in the reference implementation @[REF].+ * - The reference implementation does not use a+ * multiplication cache ('mulcache'). This idea originates+ * from @[NeonNTT] and is used at the C level here. */+MLK_STATIC_TESTABLE void mlk_poly_mulcache_compute_c(mlk_poly_mulcache *x,+ const mlk_poly *a)+__contract__(+ requires(memory_no_alias(x, sizeof(mlk_poly_mulcache)))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ assigns(memory_slice(x, sizeof(mlk_poly_mulcache)))+)+{+ unsigned i;+ for (i = 0; i < MLKEM_N / 4; i++)+ __loop__(+ invariant(i <= MLKEM_N / 4)+ invariant(array_abs_bound(x->coeffs, 0, 2 * i, MLKEM_Q))+ decreases(MLKEM_N / 4 - i))+ {+ x->coeffs[2 * i + 0] = mlk_fqmul(a->coeffs[4 * i + 1], mlk_zetas[64 + i]);+ /* The values in zeta table are <= MLKEM_Q in absolute value,+ * so the negation in int16_t is safe. */+ x->coeffs[2 * i + 1] =+ mlk_fqmul(a->coeffs[4 * i + 3], (int16_t)(-mlk_zetas[64 + i]));+ }++ /*+ * This bound is true for the C implementation, but not needed+ * in the higher level bounds reasoning. It is thus omitted+ * from the spec to not unnecessarily constrain native+ * implementations, but checked here nonetheless.+ */+ mlk_assert_abs_bound(x, MLKEM_N / 2, MLKEM_Q);+}++MLK_INTERNAL_API+void mlk_poly_mulcache_compute(mlk_poly_mulcache *x, const mlk_poly *a)+{+#if defined(MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE)+ int ret;+ ret = mlk_poly_mulcache_compute_native(x->coeffs, a->coeffs);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE */++ mlk_poly_mulcache_compute_c(x, a);+}++/*+ * Computes a block CT butterflies with a fixed twiddle factor,+ * using Montgomery multiplication.+ * Parameters:+ * - r: Pointer to base of polynomial (_not_ the base of butterfly block)+ * - root: Twiddle factor to use for the butterfly. This must be in+ * Montgomery form and signed canonical.+ * - start: Offset to the beginning of the butterfly block+ * - len: Index difference between coefficients subject to a butterfly+ * - bound: Ghost variable describing coefficient bound: Prior to `start`,+ * coefficients must be bound by `bound + MLKEM_Q`. Post `start`,+ * they must be bound by `bound`.+ * When this function returns, output coefficients in the index range+ * [start, start+2*len) have bound bumped to `bound + MLKEM_Q`.+ * Example:+ * - start=8, len=4+ * This would compute the following four butterflies+ * 8 -- 12+ * 9 -- 13+ * 10 -- 14+ * 11 -- 15+ * - start=4, len=2+ * This would compute the following two butterflies+ * 4 -- 6+ * 5 -- 7+ */++/* Reference: Embedded in `ntt()` in the reference implementation @[REF]. */+static void mlk_ntt_butterfly_block(int16_t r[MLKEM_N], int16_t zeta,+ unsigned start, unsigned len,+ unsigned bound)+__contract__(+ requires(start < MLKEM_N)+ requires(1 <= len && len <= MLKEM_N / 2 && start + 2 * len <= MLKEM_N)+ requires(0 <= bound && bound < INT16_MAX - MLKEM_Q)+ requires(-MLKEM_Q_HALF < zeta && zeta < MLKEM_Q_HALF)+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(array_abs_bound(r, 0, start, bound + MLKEM_Q))+ requires(array_abs_bound(r, start, MLKEM_N, bound))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_abs_bound(r, 0, start + 2*len, bound + MLKEM_Q))+ ensures(array_abs_bound(r, start + 2 * len, MLKEM_N, bound)))+{+ /* `bound` is a ghost variable only needed in the CBMC specification */+ unsigned j;+ ((void)bound);+ for (j = start; j < start + len; j++)+ __loop__(+ invariant(start <= j && j <= start + len)+ /*+ * Coefficients are updated in strided pairs, so the bounds for the+ * intermediate states alternate twice between the old and new bound+ */+ invariant(array_abs_bound(r, 0, j, bound + MLKEM_Q))+ invariant(array_abs_bound(r, j, start + len, bound))+ invariant(array_abs_bound(r, start + len, j + len, bound + MLKEM_Q))+ invariant(array_abs_bound(r, j + len, MLKEM_N, bound))+ decreases(start + len - j))+ {+ int16_t t;+ t = mlk_fqmul(r[j + len], zeta);+ /* The precondition implies that the arithmetic does not overflow. */+ r[j + len] = (int16_t)(r[j] - t);+ r[j] = (int16_t)(r[j] + t);+ }+}++/*+ * Compute one layer of forward NTT+ * Parameters:+ * - r: Pointer to base of polynomial+ * - layer: Variable indicating which layer is being applied.+ */++/* Reference: Embedded in `ntt()` in the reference implementation @[REF]. */+static void mlk_ntt_layer(int16_t r[MLKEM_N], unsigned layer)+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(1 <= layer && layer <= 7)+ requires(array_abs_bound(r, 0, MLKEM_N, layer * MLKEM_Q))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_abs_bound(r, 0, MLKEM_N, (layer + 1) * MLKEM_Q)))+{+ unsigned start, k, len;+ /* Twiddle factors for layer n are at indices 2^(n-1)..2^n-1. */+ k = 1u << (layer - 1);+ len = (unsigned)MLKEM_N >> layer;+ for (start = 0; start < MLKEM_N; start += 2 * len)+ __loop__(+ invariant(start < MLKEM_N + 2 * len)+ invariant(k <= MLKEM_N / 2 && 2 * len * k == start + MLKEM_N)+ invariant(array_abs_bound(r, 0, start, layer * MLKEM_Q + MLKEM_Q))+ invariant(array_abs_bound(r, start, MLKEM_N, layer * MLKEM_Q))+ decreases(MLKEM_N - start))+ {+ int16_t zeta = mlk_zetas[k++];+ mlk_ntt_butterfly_block(r, zeta, start, len, layer * MLKEM_Q);+ }+}++/*+ * Compute full forward NTT+ * NOTE: This particular implementation satisfies a much tighter+ * bound on the output coefficients (5*q) than the contractual one (8*q),+ * but this is not needed in the calling code. Should we change the+ * base multiplication strategy to require smaller NTT output bounds,+ * the proof may need strengthening.+ */++/* Reference: `ntt()` in the reference implementation @[REF].+ * - Iterate over `layer` instead of `len` in the outer loop+ * to simplify computation of zeta index. */+MLK_STATIC_TESTABLE void mlk_poly_ntt_c(mlk_poly *p)+__contract__(+ requires(memory_no_alias(p, sizeof(mlk_poly)))+ requires(array_abs_bound(p->coeffs, 0, MLKEM_N, MLKEM_Q))+ assigns(memory_slice(p, sizeof(mlk_poly)))+ ensures(array_abs_bound(p->coeffs, 0, MLKEM_N, MLK_NTT_BOUND))+)+{+ unsigned layer;+ int16_t *r;++ mlk_assert_abs_bound(p, MLKEM_N, MLKEM_Q);++ r = p->coeffs;++ for (layer = 1; layer <= 7; layer++)+ __loop__(+ invariant(1 <= layer && layer <= 8)+ invariant(array_abs_bound(r, 0, MLKEM_N, layer * MLKEM_Q))+ decreases(8 - layer))+ {+ mlk_ntt_layer(r, layer);+ }++ /* Check the stronger bound */+ mlk_assert_abs_bound(p, MLKEM_N, MLK_NTT_BOUND);+}++MLK_INTERNAL_API+void mlk_poly_ntt(mlk_poly *r)+{+#if defined(MLK_USE_NATIVE_NTT)+ int ret;+ mlk_assert_abs_bound(r, MLKEM_N, MLKEM_Q);+ ret = mlk_ntt_native(r->coeffs);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ mlk_assert_abs_bound(r, MLKEM_N, MLK_NTT_BOUND);+ return;+ }+#endif /* MLK_USE_NATIVE_NTT */++ mlk_poly_ntt_c(r);+}+++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+/* Compute one layer of inverse NTT */++/* Reference: Embedded into `invntt()` in the reference implementation @[REF] */+static void mlk_invntt_layer(int16_t *r, unsigned layer)+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(1 <= layer && layer <= 7)+ requires(array_abs_bound(r, 0, MLKEM_N, MLKEM_Q))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_abs_bound(r, 0, MLKEM_N, MLKEM_Q)))+{+ unsigned start, k, len;+ len = (unsigned)MLKEM_N >> layer;+ k = (1u << layer) - 1;+ for (start = 0; start < MLKEM_N; start += 2 * len)+ __loop__(+ invariant(array_abs_bound(r, 0, MLKEM_N, MLKEM_Q))+ invariant(start <= MLKEM_N && k <= 127)+ /* Normalised form of k == MLKEM_N / len - 1 - start / (2 * len) */+ invariant(2 * len * k + start == 2 * MLKEM_N - 2 * len)+ decreases(MLKEM_N - start))+ {+ unsigned j;+ int16_t zeta = mlk_zetas[k--];+ for (j = start; j < start + len; j++)+ __loop__(+ invariant(start <= j && j <= start + len)+ invariant(start <= MLKEM_N && k <= 127)+ invariant(array_abs_bound(r, 0, MLKEM_N, MLKEM_Q))+ decreases(start + len - j))+ {+ int16_t t = r[j];+ /* The preconditions imply that the arithmetic does not overflow. */+ r[j] = mlk_barrett_reduce((int16_t)(t + r[j + len]));+ r[j + len] = (int16_t)(r[j + len] - t);+ r[j + len] = mlk_fqmul(r[j + len], zeta);+ }+ }+}++/* Reference: `invntt()` in the reference implementation @[REF]+ * - We normalize at the beginning of the inverse NTT,+ * while the reference implementation normalizes at+ * the end. This allows us to drop a call to `poly_reduce()`+ * from the base multiplication. */+MLK_STATIC_TESTABLE void mlk_poly_invntt_tomont_c(mlk_poly *p)+__contract__(+ requires(memory_no_alias(p, sizeof(mlk_poly)))+ assigns(memory_slice(p, sizeof(mlk_poly)))+ ensures(array_abs_bound(p->coeffs, 0, MLKEM_N, MLK_INVNTT_BOUND))+)+{+ unsigned j, layer;+ const int16_t f = 1441; /* check-magic: 1441 == pow(2,32 - 7,MLKEM_Q) */+ int16_t *r = p->coeffs;++ /*+ * Scale input polynomial to account for Montgomery factor+ * and NTT twist. This also brings coefficients down to+ * absolute value < MLKEM_Q.+ */+ for (j = 0; j < MLKEM_N; j++)+ __loop__(+ invariant(j <= MLKEM_N)+ invariant(array_abs_bound(r, 0, j, MLKEM_Q))+ decreases(MLKEM_N - j))+ {+ r[j] = mlk_fqmul(r[j], f);+ }++ /* Run the invNTT layers */+ for (layer = 7; layer > 0; layer--)+ __loop__(+ invariant(0 <= layer && layer < 8)+ invariant(array_abs_bound(r, 0, MLKEM_N, MLKEM_Q))+ decreases(layer))+ {+ mlk_invntt_layer(r, layer);+ }++ mlk_assert_abs_bound(p, MLKEM_N, MLK_INVNTT_BOUND);+}++MLK_INTERNAL_API+void mlk_poly_invntt_tomont(mlk_poly *r)+{+#if defined(MLK_USE_NATIVE_INTT)+ int ret;+ ret = mlk_intt_native(r->coeffs);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ mlk_assert_abs_bound(r, MLKEM_N, MLK_INVNTT_BOUND);+ return;+ }+#endif /* MLK_USE_NATIVE_INTT */++ mlk_poly_invntt_tomont_c(r);+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */+#else /* !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(mlk_poly)++#endif /* MLK_CONFIG_MULTILEVEL_NO_SHARED */
+ cbits/mlkem/src/poly.h view
@@ -0,0 +1,296 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#ifndef MLK_POLY_H+#define MLK_POLY_H+++#include "cbmc.h"+#include "common.h"+#include "debug.h"+#include "verify.h"++/* Absolute exclusive upper bound for the output of the inverse NTT */+#define MLK_INVNTT_BOUND (8 * MLKEM_Q)++/* Absolute exclusive upper bound for the output of the forward NTT */+#define MLK_NTT_BOUND (8 * MLKEM_Q)++/**+ * Element of R_q = Z_q[X]/(X^n + 1). Represents polynomial+ * coeffs[0] + X*coeffs[1] + X^2*coeffs[2] + ... + X^{n-1}*coeffs[n-1].+ */+typedef struct+{+ int16_t coeffs[MLKEM_N]; /**< Polynomial coefficients. */+} MLK_ALIGN mlk_poly;++/**+ * INTERNAL representation of precomputed data speeding up+ * the base multiplication of two polynomials in NTT domain.+ */+typedef struct+{+ int16_t coeffs[MLKEM_N >> 1]; /**< Cached coefficients. */+} MLK_ALIGN mlk_poly_mulcache;++/**+ * Generic Montgomery reduction; given a 32-bit integer a, computes a 16-bit+ * integer congruent to a * R^-1 mod MLKEM_Q, where R=2^16.+ *+ * @param a Input integer to be reduced, of absolute value smaller or equal+ * to INT32_MAX - 2^15 * MLKEM_Q.+ *+ * @return Integer congruent to a * R^-1 modulo MLKEM_Q, with absolute value+ * <= ceil(|a| / 2^16) + (MLKEM_Q + 1)/2.+ */+static MLK_ALWAYS_INLINE int16_t mlk_montgomery_reduce(int32_t a)+__contract__(+ requires(a < +(INT32_MAX - (((int32_t)1 << 15) * MLKEM_Q)) &&+ a > -(INT32_MAX - (((int32_t)1 << 15) * MLKEM_Q)))+ /* We don't attempt to express an input-dependent output bound+ * as the post-condition here. There are two call-sites for this+ * function:+ * - The base multiplication: Here, we need no output bound.+ * - mlk_fqmul: Here, we inline this function and prove another spec+ * for mlk_fqmul which does have a post-condition bound. */+)+{+ /* check-magic: 62209 == unsigned_mod(pow(MLKEM_Q, -1, 2^16), 2^16) */+ const uint32_t QINV = 62209;++ /* Compute a*q^{-1} mod 2^16 in unsigned representatives. */+ const uint16_t a_reduced = mlk_cast_int32_to_uint16(a);+ const uint16_t a_inverted = (a_reduced * QINV) & UINT16_MAX;++ /* Lift to signed canonical representative mod 2^16. */+ const int16_t t = mlk_cast_uint16_to_int16(a_inverted);++ int32_t r;++ mlk_assert(a < +(INT32_MAX - (((int32_t)1 << 15) * MLKEM_Q)) &&+ a > -(INT32_MAX - (((int32_t)1 << 15) * MLKEM_Q)));++ r = a - ((int32_t)t * MLKEM_Q);++ /*+ * PORTABILITY: Right-shift on a signed integer is, strictly-speaking,+ * implementation-defined for negative left argument. Here,+ * we assume it's sign-preserving "arithmetic" shift right. (C99 6.5.7 (5))+ */+ r = r >> 16;+ /* Bounds: |r >> 16| <= ceil(|r| / 2^16)+ * <= ceil(|a| / 2^16 + MLKEM_Q / 2)+ * <= ceil(|a| / 2^16) + (MLKEM_Q + 1) / 2+ *+ * (Note that |a >> n| = ceil(|a| / 2^16) for negative a)+ */+ return (int16_t)r;+}++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+#define mlk_poly_tomont MLK_NAMESPACE(poly_tomont)+/**+ * In-place conversion of all coefficients of a polynomial from the normal+ * domain to the Montgomery domain.+ *+ * Bounds: output < MLKEM_Q in absolute value.+ *+ * @spec{Internal normalization required in `mlk_indcpa_keypair_derand` as+ * part of matrix-vector multiplication @[FIPS203, Algorithm 13, K-PKE.KeyGen,+ * L18].}+ *+ * @param[in,out] r Input/output polynomial.+ */+MLK_INTERNAL_API+void mlk_poly_tomont(mlk_poly *r)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_abs_bound(r->coeffs, 0, MLKEM_N, MLKEM_Q))+);+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++#define mlk_poly_mulcache_compute MLK_NAMESPACE(poly_mulcache_compute)+/**+ * Compute the mulcache for a polynomial in NTT domain.+ *+ * The mulcache of a degree-2 polynomial b := b0 + b1*X in Fq[X]/(X^2-zeta)+ * is the value b1*zeta, needed when computing products of b in+ * Fq[X]/(X^2-zeta).+ *+ * The mulcache of a polynomial in NTT domain -- which is a 128-tuple of+ * degree-2 polynomials in Fq[X]/(X^2-zeta), for varying zeta, is the+ * 128-tuple of mulcaches of those polynomials.+ *+ * @spec{Caches `b_1 * \gamma` in @[FIPS203, Algorithm 12, BaseCaseMultiply,+ * L1].}+ *+ * @param[out] x Mulcache to be populated.+ * @param[in] a Input polynomial.+ */+/*+ * NOTE: The default C implementation of this function populates+ * the mulcache with values in (-q,q), but this is not needed for the+ * higher level safety proofs, and thus not part of the spec.+ */+MLK_INTERNAL_API+void mlk_poly_mulcache_compute(mlk_poly_mulcache *x, const mlk_poly *a)+__contract__(+ requires(memory_no_alias(x, sizeof(mlk_poly_mulcache)))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ assigns(memory_slice(x, sizeof(mlk_poly_mulcache)))+);++#define mlk_poly_reduce MLK_NAMESPACE(poly_reduce)+/**+ * Convert a polynomial to unsigned canonical representatives.+ *+ * The input coefficients can be arbitrary integers in int16_t. The output+ * coefficients are in [0,1,..,MLKEM_Q-1].+ *+ * @spec{Normalizes on unsigned canonical representatives ahead of calling+ * @[FIPS203, Compress_d, Eq (4.7)]. This is not made explicit in FIPS 203.}+ *+ * @param[in,out] r Input/output polynomial.+ */+/*+ * NOTE: The semantics of mlk_poly_reduce() is different in+ * the reference implementation, which requires+ * signed canonical output data. Unsigned canonical+ * outputs are better suited to the only remaining+ * use of mlk_poly_reduce() in the context of (de)serialization.+ */+MLK_INTERNAL_API+void mlk_poly_reduce(mlk_poly *r)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+);++#define mlk_poly_add MLK_NAMESPACE(poly_add)+/**+ * Add two polynomials in place.+ *+ * The coefficients of @p r and @p b must be such that the addition does not+ * overflow. Otherwise, the behaviour of this function is undefined.+ *+ * @spec{@[FIPS203, 2.4.5, Arithmetic With Polynomials and NTT+ * Representations]. Used in @[FIPS203, Algorithm 14 (K-PKE.Encrypt), L21].}+ *+ * @param[in,out] r Input-output polynomial to be added to.+ * @param[in] b Input polynomial that should be added to @p r. Must be+ * disjoint from @p r.+ */+/*+ * NOTE: The reference implementation uses a 3-argument mlk_poly_add.+ * We specialize to the accumulator form to avoid reasoning about aliasing.+ */+MLK_INTERNAL_API+void mlk_poly_add(mlk_poly *r, const mlk_poly *b)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(b, sizeof(mlk_poly)))+ requires(forall(k0, 0, MLKEM_N, (int32_t) r->coeffs[k0] + b->coeffs[k0] <= INT16_MAX))+ requires(forall(k1, 0, MLKEM_N, (int32_t) r->coeffs[k1] + b->coeffs[k1] >= INT16_MIN))+ ensures(forall(k, 0, MLKEM_N, r->coeffs[k] == old(*r).coeffs[k] + b->coeffs[k]))+ assigns(memory_slice(r, sizeof(mlk_poly)))+);++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_sub MLK_NAMESPACE(poly_sub)+/**+ * Subtract two polynomials; no modular reduction is performed.+ *+ * @spec{@[FIPS203, 2.4.5, Arithmetic With Polynomials and NTT+ * Representations]. Used in @[FIPS203, Algorithm 15, K-PKE.Decrypt, L6].}+ *+ * @param[in,out] r Input-output polynomial to be subtracted from.+ * @param[in] b Second input polynomial.+ */+/*+ * NOTE: The reference implementation uses a 3-argument mlk_poly_sub.+ * We specialize to the accumulator form to avoid reasoning about aliasing.+ */+MLK_INTERNAL_API+void mlk_poly_sub(mlk_poly *r, const mlk_poly *b)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(b, sizeof(mlk_poly)))+ requires(forall(k0, 0, MLKEM_N, (int32_t) r->coeffs[k0] - b->coeffs[k0] <= INT16_MAX))+ requires(forall(k1, 0, MLKEM_N, (int32_t) r->coeffs[k1] - b->coeffs[k1] >= INT16_MIN))+ ensures(forall(k, 0, MLKEM_N, r->coeffs[k] == old(*r).coeffs[k] - b->coeffs[k]))+ assigns(memory_slice(r, sizeof(mlk_poly)))+);+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#define mlk_poly_ntt MLK_NAMESPACE(poly_ntt)+/**+ * Compute the negacyclic number-theoretic transform (NTT) of a polynomial+ * in place.+ *+ * The input is assumed to be in normal order and coefficient-wise bound by+ * MLKEM_Q in absolute value.+ *+ * The output polynomial is in bitreversed order, or of a custom order if+ * MLK_USE_NATIVE_NTT_CUSTOM_ORDER is set, and coefficient-wise bound+ * by MLK_NTT_BOUND in absolute value.+ *+ * (NOTE: Sometimes the input to the NTT is actually smaller, which gives+ * better bounds.)+ *+ * @spec{Implements @[FIPS203, Algorithm 9, NTT].}+ *+ * @param[in,out] r Input/output polynomial.+ */+MLK_INTERNAL_API+void mlk_poly_ntt(mlk_poly *r)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(array_abs_bound(r->coeffs, 0, MLKEM_N, MLKEM_Q))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_abs_bound(r->coeffs, 0, MLKEM_N, MLK_NTT_BOUND))+);++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_invntt_tomont MLK_NAMESPACE(poly_invntt_tomont)+/**+ * Compute the inverse negacyclic number-theoretic transform (NTT) of a+ * polynomial in place; input assumed to be in bitreversed order, output in+ * normal order.+ *+ * The input is assumed to be in bitreversed order, or of a custom order if+ * MLK_USE_NATIVE_NTT_CUSTOM_ORDER is set, and can have arbitrary+ * coefficients in int16_t.+ *+ * The output polynomial is in normal order, and coefficient-wise bound by+ * MLK_INVNTT_BOUND in absolute value.+ *+ * @spec{Implements composition of @[FIPS203, Algorithm 10, NTT^{-1}] and+ * elementwise modular multiplication with a suitable Montgomery factor+ * introduced during the base multiplication.}+ *+ * @param[in,out] r Input/output polynomial.+ */+MLK_INTERNAL_API+void mlk_poly_invntt_tomont(mlk_poly *r)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_abs_bound(r->coeffs, 0, MLKEM_N, MLK_INVNTT_BOUND))+);+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#endif /* !MLK_POLY_H */
+ cbits/mlkem/src/poly_k.c view
@@ -0,0 +1,522 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ *+ * - [REF]+ * CRYSTALS-Kyber C reference implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/ref+ */++#include "poly_k.h"++#include "debug.h"+#include "sampling.h"+#include "symmetric.h"+#include "verify.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mlkem-native (e.g. with varying parameter sets)+ * within a single compilation unit. */+#define mlk_poly_cbd_eta1 MLK_ADD_PARAM_SET(mlk_poly_cbd_eta1)+#define mlk_poly_cbd_eta2 MLK_ADD_PARAM_SET(mlk_poly_cbd_eta2)+#define mlk_polyvec_basemul_acc_montgomery_cached_c \+ MLK_ADD_PARAM_SET(mlk_polyvec_basemul_acc_montgomery_cached_c)+/* End of parameter set namespacing */++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `polyvec_compress()` in the reference implementation @[REF]+ * - In contrast to the reference implementation, we assume+ * unsigned canonical coefficients here.+ * The reference implementation works with coefficients+ * in the range [-(MLKEM_Q-1), MLKEM_Q-1]. */+MLK_INTERNAL_API+void mlk_polyvec_compress_du(uint8_t r[MLKEM_POLYVECCOMPRESSEDBYTES_DU],+ const mlk_polyvec *a)+{+ unsigned i;+ mlk_assert_bound_2d(a->vec, MLKEM_K, MLKEM_N, 0, MLKEM_Q);++ for (i = 0; i < MLKEM_K; i++)+ {+ mlk_poly_compress_du(r + i * MLKEM_POLYCOMPRESSEDBYTES_DU, &a->vec[i]);+ }+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `polyvec_decompress()` in the reference implementation @[REF]. */+MLK_INTERNAL_API+void mlk_polyvec_decompress_du(mlk_polyvec *r,+ const uint8_t a[MLKEM_POLYVECCOMPRESSEDBYTES_DU])+{+ unsigned i;+ for (i = 0; i < MLKEM_K; i++)+ {+ mlk_poly_decompress_du(&r->vec[i], a + i * MLKEM_POLYCOMPRESSEDBYTES_DU);+ }++ mlk_assert_bound_2d(r->vec, MLKEM_K, MLKEM_N, 0, MLKEM_Q);+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API) || !defined(MLK_CONFIG_NO_ENCAPS_API)+/* Reference: `polyvec_tobytes()` in the reference implementation @[REF].+ * - In contrast to the reference implementation, we assume+ * unsigned canonical coefficients here.+ * The reference implementation works with coefficients+ * in the range [-(MLKEM_Q-1), MLKEM_Q-1]. */+MLK_INTERNAL_API+void mlk_polyvec_tobytes(uint8_t r[MLKEM_POLYVECBYTES], const mlk_polyvec *a)+{+ unsigned i;+ mlk_assert_bound_2d(a->vec, MLKEM_K, MLKEM_N, 0, MLKEM_Q);++ for (i = 0; i < MLKEM_K; i++)+ __loop__(+ assigns(i, memory_slice(r, MLKEM_POLYVECBYTES))+ invariant(i <= MLKEM_K)+ decreases(MLKEM_K - i)+ )+ {+ mlk_poly_tobytes(&r[i * MLKEM_POLYBYTES], &a->vec[i]);+ }+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `polyvec_frombytes()` in the reference implementation @[REF]. */+MLK_INTERNAL_API+void mlk_polyvec_frombytes(mlk_polyvec *r, const uint8_t a[MLKEM_POLYVECBYTES])+{+ unsigned i;+ for (i = 0; i < MLKEM_K; i++)+ {+ mlk_poly_frombytes(&r->vec[i], a + i * MLKEM_POLYBYTES);+ }++ mlk_assert_bound_2d(r->vec, MLKEM_K, MLKEM_N, 0, MLKEM_UINT12_LIMIT);+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++/* Reference: `polyvec_ntt()` in the reference implementation @[REF]. */+MLK_INTERNAL_API+void mlk_polyvec_ntt(mlk_polyvec *r)+{+ unsigned i;+ for (i = 0; i < MLKEM_K; i++)+ {+ mlk_poly_ntt(&r->vec[i]);+ }++ mlk_assert_abs_bound_2d(r->vec, MLKEM_K, MLKEM_N, MLK_NTT_BOUND);+}++/* Reference: `polyvec_invntt_tomont()` in the reference implementation @[REF].+ * - We normalize at the beginning of the inverse NTT,+ * while the reference implementation normalizes at+ * the end. This allows us to drop a call to `poly_reduce()`+ * from the base multiplication. */+#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_INTERNAL_API+void mlk_polyvec_invntt_tomont(mlk_polyvec *r)+{+ unsigned i;+ for (i = 0; i < MLKEM_K; i++)+ {+ mlk_poly_invntt_tomont(&r->vec[i]);+ }++ mlk_assert_abs_bound_2d(r->vec, MLKEM_K, MLKEM_N, MLK_INVNTT_BOUND);+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++/* Reference: `polyvec_basemul_acc_montgomery()` in the+ * reference implementation @[REF].+ * - We use a multiplication cache ('mulcache') here+ * which is not present in the reference implementation @[REF].+ * This idea originates from @[NeonNTT] and is used+ * at the C level here.+ * - We compute the coefficients of the scalar product in 32-bit+ * coefficients and perform only a single modular reduction+ * at the end. The reference implementation uses 2 * MLKEM_K+ * more modular reductions since it reduces after every modular+ * multiplication. */+MLK_STATIC_TESTABLE void mlk_polyvec_basemul_acc_montgomery_cached_c(+ mlk_poly *r, const mlk_polyvec *a, const mlk_polyvec *b,+ const mlk_polyvec_mulcache *b_cache)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, sizeof(mlk_polyvec)))+ requires(memory_no_alias(b, sizeof(mlk_polyvec)))+ requires(memory_no_alias(b_cache, sizeof(mlk_polyvec_mulcache)))+ requires(forall(k1, 0, MLKEM_K,+ array_bound(a->vec[k1].coeffs, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT)))+ assigns(memory_slice(r, sizeof(mlk_poly)))+)+{+ unsigned i;+ mlk_assert_bound_2d(a->vec, MLKEM_K, MLKEM_N, 0, MLKEM_UINT12_LIMIT);++ for (i = 0; i < MLKEM_N / 2; i++)+ __loop__(invariant(i <= MLKEM_N / 2)+ decreases(MLKEM_N / 2 - i))+ {+ unsigned k;+ int32_t t[2] = {0};+ for (k = 0; k < MLKEM_K; k++)+ __loop__(+ invariant(k <= MLKEM_K &&+ t[0] <= (int32_t) k * 2 * MLKEM_UINT12_LIMIT * 32768 &&+ t[0] >= - ((int32_t) k * 2 * MLKEM_UINT12_LIMIT * 32768) &&+ t[1] <= ((int32_t) k * 2 * MLKEM_UINT12_LIMIT * 32768) &&+ t[1] >= - ((int32_t) k * 2 * MLKEM_UINT12_LIMIT * 32768))+ decreases(MLKEM_K - k))+ {+ t[0] += (int32_t)a->vec[k].coeffs[2 * i + 1] * b_cache->vec[k].coeffs[i];+ t[0] += (int32_t)a->vec[k].coeffs[2 * i] * b->vec[k].coeffs[2 * i];+ t[1] += (int32_t)a->vec[k].coeffs[2 * i] * b->vec[k].coeffs[2 * i + 1];+ t[1] += (int32_t)a->vec[k].coeffs[2 * i + 1] * b->vec[k].coeffs[2 * i];+ }+ r->coeffs[2 * i + 0] = mlk_montgomery_reduce(t[0]);+ r->coeffs[2 * i + 1] = mlk_montgomery_reduce(t[1]);+ }+}++MLK_INTERNAL_API+void mlk_polyvec_basemul_acc_montgomery_cached(+ mlk_poly *r, const mlk_polyvec *a, const mlk_polyvec *b,+ const mlk_polyvec_mulcache *b_cache)+{+#if defined(MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED)+ {+ int ret;+ mlk_assert_bound_2d(a->vec, MLKEM_K, MLKEM_N, 0, MLKEM_UINT12_LIMIT);+#if MLKEM_K == 2+ ret = mlk_polyvec_basemul_acc_montgomery_cached_k2_native(+ r->coeffs, (const int16_t *)a, (const int16_t *)b,+ (const int16_t *)b_cache);+#elif MLKEM_K == 3+ ret = mlk_polyvec_basemul_acc_montgomery_cached_k3_native(+ r->coeffs, (const int16_t *)a, (const int16_t *)b,+ (const int16_t *)b_cache);+#elif MLKEM_K == 4+ ret = mlk_polyvec_basemul_acc_montgomery_cached_k4_native(+ r->coeffs, (const int16_t *)a, (const int16_t *)b,+ (const int16_t *)b_cache);+#endif+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+ }+#endif /* MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED */++ mlk_polyvec_basemul_acc_montgomery_cached_c(r, a, b, b_cache);+}++/* Reference: Does not exist in the reference implementation @[REF].+ * - The reference implementation does not use a+ * multiplication cache ('mulcache'). This idea originates+ * from @[NeonNTT] and is used at the C level here. */+MLK_INTERNAL_API+void mlk_polyvec_mulcache_compute(mlk_polyvec_mulcache *x, const mlk_polyvec *a)+{+ unsigned i;+ for (i = 0; i < MLKEM_K; i++)+ {+ mlk_poly_mulcache_compute(&x->vec[i], &a->vec[i]);+ }+}++/* Reference: `polyvec_reduce()` in the reference implementation @[REF].+ * - We use _unsigned_ canonical outputs, while the reference+ * implementation uses _signed_ canonical outputs.+ * Accordingly, we need a conditional addition of MLKEM_Q+ * here to go from signed to unsigned representatives.+ * This conditional addition is then dropped from all+ * polynomial compression functions instead (see `compress.c`). */+MLK_INTERNAL_API+void mlk_polyvec_reduce(mlk_polyvec *r)+{+ unsigned i;+ for (i = 0; i < MLKEM_K; i++)+ {+ mlk_poly_reduce(&r->vec[i]);+ }++ mlk_assert_bound_2d(r->vec, MLKEM_K, MLKEM_N, 0, MLKEM_Q);+}++/* Reference: `polyvec_add()` in the reference implementation @[REF].+ * - We use destructive version (output=first input) to avoid+ * reasoning about aliasing in the CBMC specification */+MLK_INTERNAL_API+void mlk_polyvec_add(mlk_polyvec *r, const mlk_polyvec *b)+{+ unsigned i;+ for (i = 0; i < MLKEM_K; i++)+ __loop__(+ assigns(i, memory_slice(r, sizeof(mlk_polyvec)))+ invariant(i <= MLKEM_K)+ invariant(forall(j0, i, MLKEM_K,+ forall(k0, 0, MLKEM_N,+ ((int32_t)r->vec[j0].coeffs[k0] + b->vec[j0].coeffs[k0] <= INT16_MAX) &&+ ((int32_t)r->vec[j0].coeffs[k0] + b->vec[j0].coeffs[k0] >= INT16_MIN))))+ invariant(forall(j2, 0, i,+ forall(k2, 0, MLKEM_N,+ (r->vec[j2].coeffs[k2] <= INT16_MAX) &&+ (r->vec[j2].coeffs[k2] >= INT16_MIN))))+ decreases(MLKEM_K - i)+ )+ {+ mlk_poly_add(&r->vec[i], &b->vec[i]);+ }+}++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+/* Reference: `polyvec_tomont()` in the reference implementation @[REF]. */+MLK_INTERNAL_API+void mlk_polyvec_tomont(mlk_polyvec *r)+{+ unsigned i;+ for (i = 0; i < MLKEM_K; i++)+ {+ mlk_poly_tomont(&r->vec[i]);+ }++ mlk_assert_abs_bound_2d(r->vec, MLKEM_K, MLKEM_N, MLKEM_Q);+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */+++/**+ * Given an array of uniformly random bytes, compute a polynomial with+ * coefficients distributed according to a centered binomial distribution+ * with parameter MLKEM_ETA1.+ *+ * @spec{Implements @[FIPS203, Algorithm 8, SamplePolyCBD_eta1], where eta1+ * is specified per parameter set in @[FIPS203, Table 2] and represented as+ * MLKEM_ETA1 here.}+ *+ * @reference{`poly_cbd_eta1` in the reference implementation @[REF].}+ *+ * @param[out] r Output polynomial.+ * @param[in] buf Input byte array.+ */+static MLK_INLINE void mlk_poly_cbd_eta1(+ mlk_poly *r, const uint8_t buf[MLKEM_ETA1 * MLKEM_N / 4])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(buf, MLKEM_ETA1 * MLKEM_N / 4))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_abs_bound(r->coeffs, 0, MLKEM_N, MLKEM_ETA1 + 1))+)+{+#if MLKEM_ETA1 == 2+ mlk_poly_cbd2(r, buf);+#elif MLKEM_ETA1 == 3+ mlk_poly_cbd3(r, buf);+#else+#error "Invalid value of MLKEM_ETA1"+#endif+}++#if !defined(MLK_CONFIG_NO_KEYPAIR_API) || \+ (MLKEM_ETA1 == MLKEM_ETA2 && (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API)))+/* Reference: Does not exist in the reference implementation @[REF].+ * - This implements a x4-batched version of `poly_getnoise_eta1()`+ * from the reference implementation, to leverage+ * batched Keccak-f1600.*/+MLK_INTERNAL_API+void mlk_poly_getnoise_eta1_4x(mlk_poly *r0, mlk_poly *r1, mlk_poly *r2,+ mlk_poly *r3, const uint8_t seed[MLKEM_SYMBYTES],+ uint8_t nonce0, uint8_t nonce1, uint8_t nonce2,+ uint8_t nonce3)+{+ MLK_ALIGN uint8_t buf[4][MLK_ALIGN_UP(MLKEM_ETA1 * MLKEM_N / 4)];+ MLK_ALIGN uint8_t extkey[4][MLK_ALIGN_UP(MLKEM_SYMBYTES + 1)];+ mlk_memcpy(extkey[0], seed, MLKEM_SYMBYTES);+ mlk_memcpy(extkey[1], seed, MLKEM_SYMBYTES);+ mlk_memcpy(extkey[2], seed, MLKEM_SYMBYTES);+ mlk_memcpy(extkey[3], seed, MLKEM_SYMBYTES);+ extkey[0][MLKEM_SYMBYTES] = nonce0;+ extkey[1][MLKEM_SYMBYTES] = nonce1;+ extkey[2][MLKEM_SYMBYTES] = nonce2;+ extkey[3][MLKEM_SYMBYTES] = nonce3;++#if !defined(FIPS202_X4_DEFAULT_IMPLEMENTATION) && \+ !defined(MLK_CONFIG_SERIAL_FIPS202_ONLY)+ mlk_prf_eta1_x4(buf, extkey);+#else+ mlk_prf_eta1(buf[0], extkey[0]);+ mlk_prf_eta1(buf[1], extkey[1]);+ mlk_prf_eta1(buf[2], extkey[2]);+ if (r3 != NULL)+ {+ mlk_prf_eta1(buf[3], extkey[3]);+ }+#endif /* !(!FIPS202_X4_DEFAULT_IMPLEMENTATION && \+ !MLK_CONFIG_SERIAL_FIPS202_ONLY) */++ mlk_poly_cbd_eta1(r0, buf[0]);+ mlk_poly_cbd_eta1(r1, buf[1]);+ mlk_poly_cbd_eta1(r2, buf[2]);+ if (r3 != NULL)+ {+ mlk_poly_cbd_eta1(r3, buf[3]);+ mlk_assert_abs_bound(r3, MLKEM_N, MLKEM_ETA1 + 1);+ }++ mlk_assert_abs_bound(r0, MLKEM_N, MLKEM_ETA1 + 1);+ mlk_assert_abs_bound(r1, MLKEM_N, MLKEM_ETA1 + 1);+ mlk_assert_abs_bound(r2, MLKEM_N, MLKEM_ETA1 + 1);++ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(buf, sizeof(buf));+ mlk_zeroize(extkey, sizeof(extkey));+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API || (MLKEM_ETA1 == MLKEM_ETA2 && \+ (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API)) */++#if (MLKEM_K == 2 || MLKEM_K == 4) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API))+/**+ * Given an array of uniformly random bytes, compute a polynomial with+ * coefficients distributed according to a centered binomial distribution+ * with parameter MLKEM_ETA2.+ *+ * @spec{Implements @[FIPS203, Algorithm 8, SamplePolyCBD_eta2], where eta2+ * is specified per parameter set in @[FIPS203, Table 2] and represented as+ * MLKEM_ETA2 here.}+ *+ * @reference{`poly_cbd_eta2` in the reference implementation @[REF].}+ *+ * @param[out] r Output polynomial.+ * @param[in] buf Input byte array.+ */+static MLK_INLINE void mlk_poly_cbd_eta2(+ mlk_poly *r, const uint8_t buf[MLKEM_ETA2 * MLKEM_N / 4])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(buf, MLKEM_ETA2 * MLKEM_N / 4))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_abs_bound(r->coeffs, 0, MLKEM_N, MLKEM_ETA2 + 1)))+{+#if MLKEM_ETA2 == 2+ mlk_poly_cbd2(r, buf);+#else+#error "Invalid value of MLKEM_ETA2"+#endif+}++/* Reference: `poly_getnoise_eta2()` in the reference implementation @[REF].+ * - We include buffer zeroization. */+MLK_INTERNAL_API+void mlk_poly_getnoise_eta2(mlk_poly *r, const uint8_t seed[MLKEM_SYMBYTES],+ uint8_t nonce)+{+ MLK_ALIGN uint8_t buf[MLKEM_ETA2 * MLKEM_N / 4];+ MLK_ALIGN uint8_t extkey[MLKEM_SYMBYTES + 1];++ mlk_memcpy(extkey, seed, MLKEM_SYMBYTES);+ extkey[MLKEM_SYMBYTES] = nonce;+ mlk_prf_eta2(buf, extkey);++ mlk_poly_cbd_eta2(r, buf);++ mlk_assert_abs_bound(r, MLKEM_N, MLKEM_ETA2 + 1);++ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(buf, sizeof(buf));+ mlk_zeroize(extkey, sizeof(extkey));+}+#endif /* (MLKEM_K == 2 || MLKEM_K == 4) && (!MLK_CONFIG_NO_ENCAPS_API || \+ !MLK_CONFIG_NO_DECAPS_API) */++#if MLKEM_K == 2 && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API))+/* Reference: Does not exist in the reference implementation @[REF].+ * - This implements a x4-batched version of `poly_getnoise_eta1()`+ * and `poly_getnoise_eta2()` from the reference implementation,+ * leveraging batched Keccak-f1600.+ * - If a x4-batched Keccak-f1600 is available, we squeeze+ * more random data than needed for the eta2 calls, to be+ * be able to use a x4-batched Keccak-f1600. */+MLK_INTERNAL_API+void mlk_poly_getnoise_eta1122_4x(mlk_poly *r0, mlk_poly *r1, mlk_poly *r2,+ mlk_poly *r3,+ const uint8_t seed[MLKEM_SYMBYTES],+ uint8_t nonce0, uint8_t nonce1,+ uint8_t nonce2, uint8_t nonce3)+{+#if MLKEM_ETA2 >= MLKEM_ETA1+#error mlk_poly_getnoise_eta1122_4x assumes MLKEM_ETA1 > MLKEM_ETA2+#endif+ MLK_ALIGN uint8_t buf[4][MLK_ALIGN_UP(MLKEM_ETA1 * MLKEM_N / 4)];+ MLK_ALIGN uint8_t extkey[4][MLK_ALIGN_UP(MLKEM_SYMBYTES + 1)];++ mlk_memcpy(extkey[0], seed, MLKEM_SYMBYTES);+ mlk_memcpy(extkey[1], seed, MLKEM_SYMBYTES);+ mlk_memcpy(extkey[2], seed, MLKEM_SYMBYTES);+ mlk_memcpy(extkey[3], seed, MLKEM_SYMBYTES);+ extkey[0][MLKEM_SYMBYTES] = nonce0;+ extkey[1][MLKEM_SYMBYTES] = nonce1;+ extkey[2][MLKEM_SYMBYTES] = nonce2;+ extkey[3][MLKEM_SYMBYTES] = nonce3;++ /* On systems with fast batched Keccak, we use 4-fold batched PRF,+ * even though that means generating more random data in buf[2] and buf[3]+ * than necessary. */+#if !defined(FIPS202_X4_DEFAULT_IMPLEMENTATION) && \+ !defined(MLK_CONFIG_SERIAL_FIPS202_ONLY)+ mlk_prf_eta1_x4(buf, extkey);+#else+ mlk_prf_eta1(buf[0], extkey[0]);+ mlk_prf_eta1(buf[1], extkey[1]);+ mlk_prf_eta2(buf[2], extkey[2]);+ mlk_prf_eta2(buf[3], extkey[3]);+#endif /* !(!FIPS202_X4_DEFAULT_IMPLEMENTATION && \+ !MLK_CONFIG_SERIAL_FIPS202_ONLY) */++ mlk_poly_cbd_eta1(r0, buf[0]);+ mlk_poly_cbd_eta1(r1, buf[1]);+ mlk_poly_cbd_eta2(r2, buf[2]);+ mlk_poly_cbd_eta2(r3, buf[3]);++ mlk_assert_abs_bound(r0, MLKEM_N, MLKEM_ETA1 + 1);+ mlk_assert_abs_bound(r1, MLKEM_N, MLKEM_ETA1 + 1);+ mlk_assert_abs_bound(r2, MLKEM_N, MLKEM_ETA2 + 1);+ mlk_assert_abs_bound(r3, MLKEM_N, MLKEM_ETA2 + 1);++ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(buf, sizeof(buf));+ mlk_zeroize(extkey, sizeof(extkey));+}+#endif /* MLKEM_K == 2 && (!MLK_CONFIG_NO_ENCAPS_API || \+ !MLK_CONFIG_NO_DECAPS_API) */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef mlk_poly_cbd_eta1+#undef mlk_poly_cbd_eta2+#undef mlk_polyvec_basemul_acc_montgomery_cached_c
+ cbits/mlkem/src/poly_k.h view
@@ -0,0 +1,622 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#ifndef MLK_POLY_K_H+#define MLK_POLY_K_H++#include "common.h"+#include "compress.h"+#include "poly.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mlkem-native (e.g. with varying parameter sets)+ * within a single compilation unit. */+#define mlk_polyvec MLK_ADD_PARAM_SET(mlk_polyvec)+#define mlk_polymat MLK_ADD_PARAM_SET(mlk_polymat)+#define mlk_polyvec_mulcache MLK_ADD_PARAM_SET(mlk_polyvec_mulcache)+/* End of parameter set namespacing */++/** Vector of MLKEM_K polynomials. */+typedef struct+{+ mlk_poly vec[MLKEM_K]; /**< Component polynomials. */+} MLK_ALIGN mlk_polyvec;++/** MLKEM_K x MLKEM_K matrix of polynomials. */+typedef struct+{+ mlk_polyvec vec[MLKEM_K]; /**< Rows of the matrix. */+} MLK_ALIGN mlk_polymat;++/** Vector of MLKEM_K mlk_poly_mulcache entries. */+typedef struct+{+ mlk_poly_mulcache vec[MLKEM_K]; /**< Per-component caches. */+} MLK_ALIGN mlk_polyvec_mulcache;++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_compress_du MLK_NAMESPACE_K(poly_compress_du)+/**+ * Compression (du bits) and subsequent serialization of a polynomial.+ *+ * @spec{Implements `ByteEncode_{d_u} (Compress_{d_u} (u))` in @[FIPS203,+ * Algorithm 14 (K-PKE.Encrypt), L22], with level-specific d_u defined in+ * @[FIPS203, Table 2], and given by MLKEM_DU here.}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_DU+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ */+static MLK_INLINE void mlk_poly_compress_du(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_DU], const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_DU))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_DU)))+{+#if MLKEM_DU == 10+ mlk_poly_compress_d10(r, a);+#elif MLKEM_DU == 11+ mlk_poly_compress_d11(r, a);+#else+#error "Invalid value of MLKEM_DU"+#endif+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_decompress_du MLK_NAMESPACE_K(poly_decompress_du)+/**+ * De-serialization and subsequent decompression (du bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_du.+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @spec{Implements `Decompress_{d_u} (ByteDecode_{d_u} (u))` in @[FIPS203,+ * Algorithm 15 (K-PKE.Decrypt), L3], with level-specific d_u defined in+ * @[FIPS203, Table 2], and given by MLKEM_DU here.}+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_DU+ * bytes).+ */+static MLK_INLINE void mlk_poly_decompress_du(+ mlk_poly *r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_DU])+__contract__(+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_DU))+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q)))+{+#if MLKEM_DU == 10+ mlk_poly_decompress_d10(r, a);+#elif MLKEM_DU == 11+ mlk_poly_decompress_d11(r, a);+#else+#error "Invalid value of MLKEM_DU"+#endif+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_compress_dv MLK_NAMESPACE_K(poly_compress_dv)+/**+ * Compression (dv bits) and subsequent serialization of a polynomial.+ *+ * @spec{Implements `ByteEncode_{d_v} (Compress_{d_v} (v))` in @[FIPS203,+ * Algorithm 14 (K-PKE.Encrypt), L23], with level-specific d_v defined in+ * @[FIPS203, Table 2], and given by MLKEM_DV here.}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_DV+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ */+static MLK_INLINE void mlk_poly_compress_dv(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_DV], const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_DV))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_DV)))+{+#if MLKEM_DV == 4+ mlk_poly_compress_d4(r, a);+#elif MLKEM_DV == 5+ mlk_poly_compress_d5(r, a);+#else+#error "Invalid value of MLKEM_DV"+#endif+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */+++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_decompress_dv MLK_NAMESPACE_K(poly_decompress_dv)+/**+ * De-serialization and subsequent decompression (dv bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_dv.+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @spec{Implements `Decompress_{d_v} (ByteDecode_{d_v} (v))` in @[FIPS203,+ * Algorithm 15 (K-PKE.Decrypt), L4], with level-specific d_v defined in+ * @[FIPS203, Table 2], and given by MLKEM_DV here.}+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_DV+ * bytes).+ */+static MLK_INLINE void mlk_poly_decompress_dv(+ mlk_poly *r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_DV])+__contract__(+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_DV))+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q)))+{+#if MLKEM_DV == 4+ mlk_poly_decompress_d4(r, a);+#elif MLKEM_DV == 5+ mlk_poly_decompress_d5(r, a);+#else+#error "Invalid value of MLKEM_DV"+#endif+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_polyvec_compress_du MLK_NAMESPACE_K(polyvec_compress_du)+/**+ * Compress and serialize a vector of polynomials.+ *+ * @spec{Implements `ByteEncode_{d_u} (Compress_{d_u} (u))` in @[FIPS203,+ * Algorithm 14 (K-PKE.Encrypt), L22], with level-specific d_u defined in+ * @[FIPS203, Table 2], and given by MLKEM_DU here.}+ *+ * @param[out] r Output byte array (needs space for+ * MLKEM_POLYVECCOMPRESSEDBYTES_DU bytes).+ * @param[in] a Input vector of polynomials. Coefficients must be unsigned+ * canonical, i.e. in [0,1,..,MLKEM_Q-1].+ */+MLK_INTERNAL_API+void mlk_polyvec_compress_du(uint8_t r[MLKEM_POLYVECCOMPRESSEDBYTES_DU],+ const mlk_polyvec *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYVECCOMPRESSEDBYTES_DU))+ requires(memory_no_alias(a, sizeof(mlk_polyvec)))+ requires(forall(k0, 0, MLKEM_K,+ array_bound(a->vec[k0].coeffs, 0, MLKEM_N, 0, MLKEM_Q)))+ assigns(memory_slice(r, MLKEM_POLYVECCOMPRESSEDBYTES_DU))+);+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_polyvec_decompress_du MLK_NAMESPACE_K(polyvec_decompress_du)+/**+ * De-serialize and decompress a vector of polynomials; approximate inverse+ * of mlk_polyvec_compress_du.+ *+ * @spec{Implements `Decompress_{d_u} (ByteDecode_{d_u} (u))` in @[FIPS203,+ * Algorithm 15 (K-PKE.Decrypt), L3], with level-specific d_u defined in+ * @[FIPS203, Table 2], and given by MLKEM_DU here.}+ *+ * @param[out] r Output vector of polynomials. Coefficients are normalized+ * to [0,1,..,MLKEM_Q-1].+ * @param[in] a Input byte array (of length MLKEM_POLYVECCOMPRESSEDBYTES_DU+ * bytes).+ */+MLK_INTERNAL_API+void mlk_polyvec_decompress_du(mlk_polyvec *r,+ const uint8_t a[MLKEM_POLYVECCOMPRESSEDBYTES_DU])+__contract__(+ requires(memory_no_alias(a, MLKEM_POLYVECCOMPRESSEDBYTES_DU))+ requires(memory_no_alias(r, sizeof(mlk_polyvec)))+ assigns(memory_slice(r, sizeof(mlk_polyvec)))+ ensures(forall(k0, 0, MLKEM_K,+ array_bound(r->vec[k0].coeffs, 0, MLKEM_N, 0, MLKEM_Q)))+);+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API) || !defined(MLK_CONFIG_NO_ENCAPS_API)+#define mlk_polyvec_tobytes MLK_NAMESPACE_K(polyvec_tobytes)+/**+ * Serialize a vector of polynomials.+ *+ * @spec{Implements ByteEncode_12 @[FIPS203, Algorithm 5]. Extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays] and+ * @[FIPS203, 2.4.6, Matrices and Vectors].}+ *+ * @param[out] r Output byte array (needs space for MLKEM_POLYVECBYTES bytes).+ * @param[in] a Input vector of polynomials. Each polynomial must have+ * coefficients in [0,1,..,MLKEM_Q-1].+ */+MLK_INTERNAL_API+void mlk_polyvec_tobytes(uint8_t r[MLKEM_POLYVECBYTES], const mlk_polyvec *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mlk_polyvec)))+ requires(memory_no_alias(r, MLKEM_POLYVECBYTES))+ requires(forall(k0, 0, MLKEM_K,+ array_bound(a->vec[k0].coeffs, 0, MLKEM_N, 0, MLKEM_Q)))+ assigns(memory_slice(r, MLKEM_POLYVECBYTES))+);+#endif /* !MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_polyvec_frombytes MLK_NAMESPACE_K(polyvec_frombytes)+/**+ * De-serialize a vector of polynomials; inverse of mlk_polyvec_tobytes.+ *+ * @spec{Implements ByteDecode_12 @[FIPS203, Algorithm 6]. Extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays] and+ * @[FIPS203, 2.4.6, Matrices and Vectors].}+ *+ * @param[out] r Output vector of polynomials. Coefficients will be+ * normalized in [0,1,..,4095].+ * @param[in] a Input byte array (of length MLKEM_POLYVECBYTES bytes).+ */+MLK_INTERNAL_API+void mlk_polyvec_frombytes(mlk_polyvec *r, const uint8_t a[MLKEM_POLYVECBYTES])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_polyvec)))+ requires(memory_no_alias(a, MLKEM_POLYVECBYTES))+ assigns(memory_slice(r, sizeof(mlk_polyvec)))+ ensures(forall(k0, 0, MLKEM_K,+ array_bound(r->vec[k0].coeffs, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT)))+);+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#define mlk_polyvec_ntt MLK_NAMESPACE_K(polyvec_ntt)+/**+ * Apply forward NTT to all elements of a vector of polynomials.+ *+ * The input is assumed to be in normal order and coefficient-wise bound by+ * MLKEM_Q in absolute value.+ *+ * The output polynomial is in bitreversed order, and coefficient-wise bound+ * by MLK_NTT_BOUND in absolute value.+ *+ * @spec{Implements @[FIPS203, Algorithm 9, NTT]. Extended to vectors as per+ * @[FIPS203, 2.4.6, Matrices and Vectors].}+ *+ * @param[in,out] r Input/output vector of polynomials.+ */+MLK_INTERNAL_API+void mlk_polyvec_ntt(mlk_polyvec *r)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_polyvec)))+ requires(forall(j, 0, MLKEM_K,+ array_abs_bound(r->vec[j].coeffs, 0, MLKEM_N, MLKEM_Q)))+ assigns(memory_slice(r, sizeof(mlk_polyvec)))+ ensures(forall(j, 0, MLKEM_K,+ array_abs_bound(r->vec[j].coeffs, 0, MLKEM_N, MLK_NTT_BOUND)))+);++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_polyvec_invntt_tomont MLK_NAMESPACE_K(polyvec_invntt_tomont)+/**+ * Apply inverse NTT to all elements of a vector of polynomials and multiply+ * by Montgomery factor 2^16.+ *+ * The input is assumed to be in bitreversed order, and can have arbitrary+ * coefficients in int16_t.+ *+ * The output polynomial is in normal order, and coefficient-wise bound by+ * MLK_INVNTT_BOUND in absolute value.+ *+ * @spec{Implements @[FIPS203, Algorithm 10, NTT^{-1}]. Extended to vectors+ * as per @[FIPS203, 2.4.6, Matrices and Vectors].}+ *+ * @param[in,out] r Input/output vector of polynomials.+ */+MLK_INTERNAL_API+void mlk_polyvec_invntt_tomont(mlk_polyvec *r)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_polyvec)))+ assigns(memory_slice(r, sizeof(mlk_polyvec)))+ ensures(forall(j, 0, MLKEM_K,+ array_abs_bound(r->vec[j].coeffs, 0, MLKEM_N, MLK_INVNTT_BOUND)))+);+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#define mlk_polyvec_basemul_acc_montgomery_cached \+ MLK_NAMESPACE_K(polyvec_basemul_acc_montgomery_cached)+/**+ * Scalar product of two vectors of polynomials in NTT domain, using+ * mulcache for the second operand.+ *+ * Bounds: every coefficient of @p a is assumed to be in [0,1,..,4095]. No+ * bounds guarantees for the coefficients in the result.+ *+ * @spec{Implements @[FIPS203, Section 2.4.7, Eq (2.14)], @[FIPS203,+ * Algorithm 11, MultiplyNTTs], and @[FIPS203, Algorithm 12,+ * BaseCaseMultiply].}+ *+ * @param[out] r Output polynomial.+ * @param[in] a First input polynomial vector.+ * @param[in] b Second input polynomial vector.+ * @param[in] b_cache Mulcache for the second input polynomial vector. Can+ * be computed via mlk_polyvec_mulcache_compute().+ */+MLK_INTERNAL_API+void mlk_polyvec_basemul_acc_montgomery_cached(+ mlk_poly *r, const mlk_polyvec *a, const mlk_polyvec *b,+ const mlk_polyvec_mulcache *b_cache)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, sizeof(mlk_polyvec)))+ requires(memory_no_alias(b, sizeof(mlk_polyvec)))+ requires(memory_no_alias(b_cache, sizeof(mlk_polyvec_mulcache)))+ requires(forall(k1, 0, MLKEM_K,+ array_bound(a->vec[k1].coeffs, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT)))+ assigns(memory_slice(r, sizeof(mlk_poly)))+);++#define mlk_polyvec_mulcache_compute MLK_NAMESPACE_K(polyvec_mulcache_compute)+/**+ * Compute the mulcache for a vector of polynomials in NTT domain.+ *+ * The mulcache of a degree-2 polynomial b := b0 + b1*X in Fq[X]/(X^2-zeta)+ * is the value b1*zeta, needed when computing products of b in+ * Fq[X]/(X^2-zeta).+ *+ * The mulcache of a polynomial in NTT domain -- which is a 128-tuple of+ * degree-2 polynomials in Fq[X]/(X^2-zeta), for varying zeta, is the+ * 128-tuple of mulcaches of those polynomials.+ *+ * The mulcache of a vector of polynomials is the vector of mulcaches of+ * its entries.+ *+ * @spec{Caches `b_1 * \gamma` in @[FIPS203, Algorithm 12, BaseCaseMultiply,+ * L1].}+ *+ * @param[out] x Mulcache to be populated.+ * @param[in] a Input polynomial vector.+ */+/*+ * NOTE: The default C implementation of this function populates+ * the mulcache with values in (-q,q), but this is not needed for the+ * higher level safety proofs, and thus not part of the spec.+ */+MLK_INTERNAL_API+void mlk_polyvec_mulcache_compute(mlk_polyvec_mulcache *x, const mlk_polyvec *a)+__contract__(+ requires(memory_no_alias(x, sizeof(mlk_polyvec_mulcache)))+ requires(memory_no_alias(a, sizeof(mlk_polyvec)))+ assigns(memory_slice(x, sizeof(mlk_polyvec_mulcache)))+);++#define mlk_polyvec_reduce MLK_NAMESPACE_K(polyvec_reduce)+/**+ * Apply Barrett reduction to each coefficient of each element of a vector+ * of polynomials. For details of the Barrett reduction see comments in+ * poly.c.+ *+ * @spec{Normalizes on unsigned canonical representatives ahead of calling+ * @[FIPS203, Compress_d, Eq (4.7)]. This is not made explicit in FIPS 203.}+ *+ * @param[in,out] r Input/output polynomial vector.+ */+/*+ * NOTE: The semantics of mlk_polyvec_reduce() is different in+ * the reference implementation, which requires+ * signed canonical output data. Unsigned canonical+ * outputs are better suited to the only remaining+ * use of mlk_poly_reduce() in the context of (de)serialization.+ */+MLK_INTERNAL_API+void mlk_polyvec_reduce(mlk_polyvec *r)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_polyvec)))+ assigns(memory_slice(r, sizeof(mlk_polyvec)))+ ensures(forall(k0, 0, MLKEM_K,+ array_bound(r->vec[k0].coeffs, 0, MLKEM_N, 0, MLKEM_Q)))+);++#define mlk_polyvec_add MLK_NAMESPACE_K(polyvec_add)+/**+ * Add vectors of polynomials.+ *+ * The coefficients of @p r and @p b must be such that the addition does+ * not overflow. Otherwise, the behaviour of this function is undefined.+ *+ * The coefficients returned in @p *r are in int16_t which is sufficient to+ * prove type-safety of calling units. Therefore, no stronger ensures clause+ * is required on this function.+ *+ * @spec{@[FIPS203, 2.4.5, Arithmetic With Polynomials and NTT+ * Representations]. Used in @[FIPS203, Algorithm 14 (K-PKE.Encrypt), L19].}+ *+ * @param[in,out] r Input-output vector of polynomials to be added to.+ * @param[in] b Second input vector of polynomials.+ */+MLK_INTERNAL_API+void mlk_polyvec_add(mlk_polyvec *r, const mlk_polyvec *b)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_polyvec)))+ requires(memory_no_alias(b, sizeof(mlk_polyvec)))+ requires(forall(j0, 0, MLKEM_K,+ forall(k0, 0, MLKEM_N,+ (int32_t)r->vec[j0].coeffs[k0] + b->vec[j0].coeffs[k0] <= INT16_MAX)))+ requires(forall(j1, 0, MLKEM_K,+ forall(k1, 0, MLKEM_N,+ (int32_t)r->vec[j1].coeffs[k1] + b->vec[j1].coeffs[k1] >= INT16_MIN)))+ assigns(memory_slice(r, sizeof(mlk_polyvec)))+);++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+#define mlk_polyvec_tomont MLK_NAMESPACE_K(polyvec_tomont)+/**+ * In-place conversion of all coefficients of a polynomial vector from the+ * normal domain to the Montgomery domain.+ *+ * Bounds: output < MLKEM_Q in absolute value.+ *+ * @spec{Internal normalization required in `mlk_indcpa_keypair_derand` as+ * part of matrix-vector multiplication @[FIPS203, Algorithm 13, K-PKE.KeyGen,+ * L18].}+ *+ * @param[in,out] r Input/output polynomial vector.+ */+MLK_INTERNAL_API+void mlk_polyvec_tomont(mlk_polyvec *r)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_polyvec)))+ assigns(memory_slice(r, sizeof(mlk_polyvec)))+ ensures(forall(j, 0, MLKEM_K,+ array_abs_bound(r->vec[j].coeffs, 0, MLKEM_N, MLKEM_Q)))+);+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API) || \+ (MLKEM_ETA1 == MLKEM_ETA2 && (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API)))+#define mlk_poly_getnoise_eta1_4x MLK_NAMESPACE_K(poly_getnoise_eta1_4x)+/**+ * Batch sample four polynomials deterministically from a seed and nonces,+ * with output polynomials close to centered binomial distribution with+ * parameter MLKEM_ETA1.+ *+ * @spec{Implements 4x `SamplePolyCBD_{eta1} (PRF_{eta1} (sigma, N))`:+ * @[FIPS203, Algorithm 8, SamplePolyCBD_eta] and @[FIPS203, Eq (4.3),+ * PRF_eta]. `SamplePolyCBD_{eta1} (PRF_{eta1} (sigma, N))` appears in+ * @[FIPS203, Algorithm 13, K-PKE.KeyGen, L{9, 13}] and @[FIPS203,+ * Algorithm 14, K-PKE.Encrypt, L10].}+ *+ * @param[out] r0 Output polynomial.+ * @param[out] r1 Output polynomial.+ * @param[out] r2 Output polynomial.+ * @param[out] r3 Output polynomial. May be NULL.+ * @param[in] seed Input seed (of length MLKEM_SYMBYTES bytes).+ * @param nonce0 One-byte input nonce.+ * @param nonce1 One-byte input nonce.+ * @param nonce2 One-byte input nonce.+ * @param nonce3 One-byte input nonce.+ */+MLK_INTERNAL_API+void mlk_poly_getnoise_eta1_4x(mlk_poly *r0, mlk_poly *r1, mlk_poly *r2,+ mlk_poly *r3, const uint8_t seed[MLKEM_SYMBYTES],+ uint8_t nonce0, uint8_t nonce1, uint8_t nonce2,+ uint8_t nonce3)+__contract__(+ requires(memory_no_alias(seed, MLKEM_SYMBYTES))+ requires(memory_no_alias(r0, sizeof(mlk_poly)))+ requires(memory_no_alias(r1, sizeof(mlk_poly)))+ requires(memory_no_alias(r2, sizeof(mlk_poly)))+ requires(r3 == NULL || memory_no_alias(r3, sizeof(mlk_poly)))+ assigns(memory_slice(r0, sizeof(mlk_poly)))+ assigns(memory_slice(r1, sizeof(mlk_poly)))+ assigns(memory_slice(r2, sizeof(mlk_poly)))+ assigns(r3 != NULL: memory_slice(r3, sizeof(mlk_poly)))+ ensures(array_abs_bound(r0->coeffs,0, MLKEM_N, MLKEM_ETA1 + 1))+ ensures(array_abs_bound(r1->coeffs,0, MLKEM_N, MLKEM_ETA1 + 1))+ ensures(array_abs_bound(r2->coeffs,0, MLKEM_N, MLKEM_ETA1 + 1))+ ensures(r3 != NULL ==> array_abs_bound(r3->coeffs,0, MLKEM_N, MLKEM_ETA1 + 1))+);++#if MLKEM_ETA1 == MLKEM_ETA2+/*+ * We only require mlk_poly_getnoise_eta2_4x for ml-kem-768 and ml-kem-1024+ * where MLKEM_ETA2 = MLKEM_ETA1 = 2.+ * For ml-kem-512, mlk_poly_getnoise_eta1122_4x is used instead.+ */+#define mlk_poly_getnoise_eta2_4x mlk_poly_getnoise_eta1_4x+#endif /* MLKEM_ETA1 == MLKEM_ETA2 */+#endif /* !MLK_CONFIG_NO_KEYPAIR_API || (MLKEM_ETA1 == MLKEM_ETA2 && \+ (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API)) */++#if (MLKEM_K == 2 || MLKEM_K == 4) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API))+#define mlk_poly_getnoise_eta2 MLK_NAMESPACE_K(poly_getnoise_eta2)+/**+ * Sample a polynomial deterministically from a seed and a nonce, with+ * output polynomial close to centered binomial distribution with parameter+ * MLKEM_ETA2.+ *+ * @spec{Implements `SamplePolyCBD_{eta2} (PRF_{eta2} (sigma, N))`:+ * @[FIPS203, Algorithm 8, SamplePolyCBD_eta] and @[FIPS203, Eq (4.3),+ * PRF_eta]. `SamplePolyCBD_{eta2} (PRF_{eta2} (sigma, N))` appears in+ * @[FIPS203, Algorithm 14, K-PKE.Encrypt, L14].}+ *+ * @param[out] r Output polynomial.+ * @param[in] seed Input seed (of length MLKEM_SYMBYTES bytes).+ * @param nonce One-byte input nonce.+ */+MLK_INTERNAL_API+void mlk_poly_getnoise_eta2(mlk_poly *r, const uint8_t seed[MLKEM_SYMBYTES],+ uint8_t nonce)+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(seed, MLKEM_SYMBYTES))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_abs_bound(r->coeffs, 0, MLKEM_N, MLKEM_ETA2 + 1))+);+#endif /* (MLKEM_K == 2 || MLKEM_K == 4) && (!MLK_CONFIG_NO_ENCAPS_API || \+ !MLK_CONFIG_NO_DECAPS_API) */++#if MLKEM_K == 2 && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API))+#define mlk_poly_getnoise_eta1122_4x MLK_NAMESPACE_K(poly_getnoise_eta1122_4x)+/**+ * Batch sample four polynomials deterministically from a seed and nonces,+ * with output polynomials close to centered binomial distribution with+ * parameter MLKEM_ETA1 and MLKEM_ETA2.+ *+ * @spec{Implements two instances each of+ * `SamplePolyCBD_{eta1} (PRF_{eta1} (sigma, N))` and+ * `SamplePolyCBD_{eta2} (PRF_{eta2} (sigma, N))`:+ * @[FIPS203, Algorithm 8, SamplePolyCBD_eta] and @[FIPS203, Eq (4.3),+ * PRF_eta]. `SamplePolyCBD_{eta2} (PRF_{eta2} (sigma, N))` appears in+ * @[FIPS203, Algorithm 14, K-PKE.Encrypt, L14].}+ *+ * @param[out] r0 Output polynomial.+ * @param[out] r1 Output polynomial.+ * @param[out] r2 Output polynomial.+ * @param[out] r3 Output polynomial.+ * @param[in] seed Input seed (of length MLKEM_SYMBYTES bytes).+ * @param nonce0 One-byte input nonce.+ * @param nonce1 One-byte input nonce.+ * @param nonce2 One-byte input nonce.+ * @param nonce3 One-byte input nonce.+ */+MLK_INTERNAL_API+void mlk_poly_getnoise_eta1122_4x(mlk_poly *r0, mlk_poly *r1, mlk_poly *r2,+ mlk_poly *r3,+ const uint8_t seed[MLKEM_SYMBYTES],+ uint8_t nonce0, uint8_t nonce1,+ uint8_t nonce2, uint8_t nonce3)+__contract__(+ requires(memory_no_alias(r0, sizeof(mlk_poly)))+ requires(memory_no_alias(r1, sizeof(mlk_poly)))+ requires(memory_no_alias(r2, sizeof(mlk_poly)))+ requires(memory_no_alias(r3, sizeof(mlk_poly)))+ requires(memory_no_alias(seed, MLKEM_SYMBYTES))+ assigns(memory_slice(r0, sizeof(mlk_poly)))+ assigns(memory_slice(r1, sizeof(mlk_poly)))+ assigns(memory_slice(r2, sizeof(mlk_poly)))+ assigns(memory_slice(r3, sizeof(mlk_poly)))+ ensures(array_abs_bound(r0->coeffs,0, MLKEM_N, MLKEM_ETA1 + 1)+ && array_abs_bound(r1->coeffs,0, MLKEM_N, MLKEM_ETA1 + 1)+ && array_abs_bound(r2->coeffs,0, MLKEM_N, MLKEM_ETA2 + 1)+ && array_abs_bound(r3->coeffs,0, MLKEM_N, MLKEM_ETA2 + 1))+);+#endif /* MLKEM_K == 2 && (!MLK_CONFIG_NO_ENCAPS_API || \+ !MLK_CONFIG_NO_DECAPS_API) */++#endif /* !MLK_POLY_K_H */
+ cbits/mlkem/src/randombytes.h view
@@ -0,0 +1,52 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_RANDOMBYTES_H+#define MLK_RANDOMBYTES_H+++#include "cbmc.h"+#include "common.h"++#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+#if !defined(MLK_CONFIG_CUSTOM_RANDOMBYTES)+/**+ * Fill a buffer with cryptographically secure random bytes.+ *+ * mlkem-native does not provide an implementation of this function.+ * It must be provided by the consumer.+ *+ * To use a custom random byte source with a different name or signature,+ * set MLK_CONFIG_CUSTOM_RANDOMBYTES and define mlk_randombytes directly.+ *+ * @param[out] out Output buffer.+ * @param outlen Number of random bytes to write.+ *+ * @retval 0 Success.+ * @retval other Failure; top-level APIs propagate this as MLK_ERR_RNG_FAIL.+ */+int randombytes(uint8_t *out, size_t outlen);++/**+ * Internal wrapper around randombytes().+ *+ * Fills a buffer with cryptographically secure random bytes.+ *+ * This function can be replaced by setting MLK_CONFIG_CUSTOM_RANDOMBYTES+ * and defining mlk_randombytes directly.+ *+ * @param[out] out Output buffer.+ * @param outlen Number of random bytes to write.+ *+ * @retval 0 Success.+ * @retval other Failure; top-level APIs propagate this as MLK_ERR_RNG_FAIL.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_randombytes(uint8_t *out, size_t outlen)+__contract__(+ requires(memory_no_alias(out, outlen))+ assigns(memory_slice(out, outlen))) { return randombytes(out, outlen); }+#endif /* !MLK_CONFIG_CUSTOM_RANDOMBYTES */+#endif /* !MLK_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLK_RANDOMBYTES_H */
+ cbits/mlkem/src/sampling.c view
@@ -0,0 +1,371 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ *+ * - [REF]+ * CRYSTALS-Kyber C reference implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/ref+ */++#include "common.h"+#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#include "debug.h"+#include "sampling.h"+#include "symmetric.h"+#include "verify.h"++/* Reference: `rej_uniform()` in the reference implementation @[REF].+ * - Our signature differs from the reference implementation+ * in that it adds the offset and always expects the base of the+ * target buffer. This avoids shifting the buffer base in the+ * caller, which appears tricky to reason about. */+MLK_STATIC_TESTABLE unsigned mlk_rej_uniform_c(int16_t *r, unsigned target,+ unsigned offset,+ const uint8_t *buf,+ unsigned buflen)+__contract__(+ requires(offset <= target && target <= 4096 && buflen <= 4096 && buflen % 3 == 0)+ requires(memory_no_alias(r, sizeof(int16_t) * target))+ requires(memory_no_alias(buf, buflen))+ requires(array_bound(r, 0, offset, 0, MLKEM_Q))+ assigns(memory_slice(r, sizeof(int16_t) * target))+ ensures(offset <= return_value && return_value <= target)+ ensures(array_bound(r, 0, return_value, 0, MLKEM_Q)))+{+ unsigned ctr, pos;+ int16_t val0, val1;++ mlk_assert_bound(r, offset, 0, MLKEM_Q);++ ctr = offset;+ pos = 0;+ /* pos + 3 cannot overflow due to the assumption buflen <= 4096 */+ while (ctr < target && pos + 3 <= buflen)+ __loop__(+ invariant(offset <= ctr && ctr <= target && pos <= buflen)+ invariant(array_bound(r, 0, ctr, 0, MLKEM_Q))+ decreases(buflen - pos))+ {+ /* Safety:+ * - The explicit cast to uint16_t ensures that << 8 does+ * not signed-overflow even on a 16-bit system.+ * - The conversion to int16_t is safe due to the explicit 0xFFF+ * truncation.+ */+ val0 = (int16_t)(((buf[pos + 0] >> 0) | ((uint16_t)buf[pos + 1] << 8)) &+ 0xFFF);+ val1 = (int16_t)(((buf[pos + 1] >> 4) | (buf[pos + 2] << 4)) & 0xFFF);+ pos += 3;++ if (val0 < MLKEM_Q)+ {+ r[ctr++] = val0;+ }+ if (ctr < target && val1 < MLKEM_Q)+ {+ r[ctr++] = val1;+ }+ }++ mlk_assert_bound(r, ctr, 0, MLKEM_Q);+ return ctr;+}++/**+ * Run rejection sampling on uniform random bytes to generate uniform random+ * integers mod MLKEM_Q.+ *+ * @reference{`rej_uniform()` in the reference implementation @[REF]. Our+ * signature differs from the reference in that it adds the offset and always+ * expects the base of the target buffer; this avoids shifting the buffer+ * base in the caller, which is tricky to reason about. Has an optional+ * fallback to a native implementation.}+ *+ * @param[out] r Output buffer.+ * @param target Requested number of 16-bit integers (uniform mod MLKEM_Q).+ * Must be <= 4096.+ * @param offset Number of 16-bit integers that have already been+ * sampled. Must be <= @p target.+ * @param[in] buf Input buffer (assumed to be uniform random bytes).+ * @param buflen Length of input buffer in bytes. Must be <= 4096 and a+ * multiple of 3.+ *+ * @note Strictly speaking, only a few values of @p buflen near UINT_MAX need+ * excluding. The limit of 4096 is somewhat arbitrary but sufficient+ * for all uses of this function. Similarly, the actual limit for+ * @p target is UINT_MAX/2.+ *+ * @return New offset of sampled 16-bit integers, at most @p target and at+ * least the initial @p offset. If the new offset is strictly less+ * than @p target, the entire input buffer is guaranteed to have been+ * consumed; otherwise no information is provided on how many bytes+ * of the input buffer have been consumed.+ */+static unsigned mlk_rej_uniform(int16_t *r, unsigned target, unsigned offset,+ const uint8_t *buf, unsigned buflen)+__contract__(+ requires(offset <= target && target <= 4096 && buflen <= 4096 && buflen % 3 == 0)+ requires(memory_no_alias(r, sizeof(int16_t) * target))+ requires(memory_no_alias(buf, buflen))+ requires(array_bound(r, 0, offset, 0, MLKEM_Q))+ assigns(memory_slice(r, sizeof(int16_t) * target))+ ensures(offset <= return_value && return_value <= target)+ ensures(array_bound(r, 0, return_value, 0, MLKEM_Q))+)+{+#if defined(MLK_USE_NATIVE_REJ_UNIFORM)+ if (offset == 0)+ {+ int ret;+ ret = mlk_rej_uniform_native(r, target, buf, buflen);+ if (ret != MLK_NATIVE_FUNC_FALLBACK)+ {+ unsigned res = (unsigned)ret;+ mlk_assert_bound(r, res, 0, MLKEM_Q);+ return res;+ }+ }+#endif /* MLK_USE_NATIVE_REJ_UNIFORM */++ return mlk_rej_uniform_c(r, target, offset, buf, buflen);+}++#ifndef MLKEM_GEN_MATRIX_NBLOCKS+#define MLKEM_GEN_MATRIX_NBLOCKS \+ ((12 * MLKEM_N / 8 * ((uint32_t)1 << 12) / MLKEM_Q + MLK_XOF_RATE) / \+ MLK_XOF_RATE)+#endif++#if !defined(MLK_CONFIG_SERIAL_FIPS202_ONLY)+/* Reference: Does not exist in the reference implementation @[REF].+ * - x4-batched version of `rej_uniform()` from the+ * reference implementation, leveraging x4-batched Keccak-f1600. */+MLK_INTERNAL_API+void mlk_poly_rej_uniform_x4(mlk_poly *vec0, mlk_poly *vec1, mlk_poly *vec2,+ mlk_poly *vec3,+ uint8_t seed[4][MLK_ALIGN_UP(MLKEM_SYMBYTES + 2)])+{+ /* Temporary buffers for XOF output before rejection sampling */+ MLK_ALIGN uint8_t+ buf[4][MLK_ALIGN_UP(MLKEM_GEN_MATRIX_NBLOCKS * MLK_XOF_RATE)];++ /* Tracks the number of coefficients we have already sampled */+ unsigned ctr[4];+ mlk_xof_x4_ctx statex;+ unsigned buflen;++ mlk_xof_x4_init(&statex);+ mlk_xof_x4_absorb(&statex, seed, MLKEM_SYMBYTES + 2);++ /*+ * Initially, squeeze heuristic number of MLKEM_GEN_MATRIX_NBLOCKS.+ * This should generate the matrix entries with high probability.+ */+ mlk_xof_x4_squeezeblocks(buf, MLKEM_GEN_MATRIX_NBLOCKS, &statex);+ buflen = MLKEM_GEN_MATRIX_NBLOCKS * MLK_XOF_RATE;+ ctr[0] = mlk_rej_uniform(vec0->coeffs, MLKEM_N, 0, buf[0], buflen);+ ctr[1] = mlk_rej_uniform(vec1->coeffs, MLKEM_N, 0, buf[1], buflen);+ ctr[2] = mlk_rej_uniform(vec2->coeffs, MLKEM_N, 0, buf[2], buflen);+ ctr[3] = mlk_rej_uniform(vec3->coeffs, MLKEM_N, 0, buf[3], buflen);++ /*+ * So long as not all matrix entries have been generated, squeeze+ * one more block a time until we're done.+ */+ buflen = MLK_XOF_RATE;+ while (ctr[0] < MLKEM_N || ctr[1] < MLKEM_N || ctr[2] < MLKEM_N ||+ ctr[3] < MLKEM_N)+ __loop__(+ assigns(ctr, statex,+ memory_slice(vec0, sizeof(mlk_poly)),+ memory_slice(vec1, sizeof(mlk_poly)),+ memory_slice(vec2, sizeof(mlk_poly)),+ memory_slice(vec3, sizeof(mlk_poly)),+ object_whole(buf))+ invariant(ctr[0] <= MLKEM_N && ctr[1] <= MLKEM_N)+ invariant(ctr[2] <= MLKEM_N && ctr[3] <= MLKEM_N)+ invariant(array_bound(vec0->coeffs, 0, ctr[0], 0, MLKEM_Q))+ invariant(array_bound(vec1->coeffs, 0, ctr[1], 0, MLKEM_Q))+ invariant(array_bound(vec2->coeffs, 0, ctr[2], 0, MLKEM_Q))+ invariant(array_bound(vec3->coeffs, 0, ctr[3], 0, MLKEM_Q)))+ {+ mlk_xof_x4_squeezeblocks(buf, 1, &statex);+ ctr[0] = mlk_rej_uniform(vec0->coeffs, MLKEM_N, ctr[0], buf[0], buflen);+ ctr[1] = mlk_rej_uniform(vec1->coeffs, MLKEM_N, ctr[1], buf[1], buflen);+ ctr[2] = mlk_rej_uniform(vec2->coeffs, MLKEM_N, ctr[2], buf[2], buflen);+ ctr[3] = mlk_rej_uniform(vec3->coeffs, MLKEM_N, ctr[3], buf[3], buflen);+ }++ mlk_xof_x4_release(&statex);++ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(buf, sizeof(buf));+}+#endif /* !MLK_CONFIG_SERIAL_FIPS202_ONLY */++MLK_INTERNAL_API+void mlk_poly_rej_uniform(mlk_poly *entry, uint8_t seed[MLKEM_SYMBYTES + 2])+{+ mlk_xof_ctx state;+ MLK_ALIGN uint8_t buf[MLKEM_GEN_MATRIX_NBLOCKS * MLK_XOF_RATE];+ unsigned ctr, buflen;++ mlk_xof_init(&state);+ mlk_xof_absorb(&state, seed, MLKEM_SYMBYTES + 2);++ /* Initially, squeeze + sample heuristic number of MLKEM_GEN_MATRIX_NBLOCKS.+ */+ /* This should generate the matrix entry with high probability. */+ mlk_xof_squeezeblocks(buf, MLKEM_GEN_MATRIX_NBLOCKS, &state);+ buflen = MLKEM_GEN_MATRIX_NBLOCKS * MLK_XOF_RATE;+ ctr = mlk_rej_uniform(entry->coeffs, MLKEM_N, 0, buf, buflen);++ /* Squeeze + sample one more block a time until we're done */+ buflen = MLK_XOF_RATE;+ while (ctr < MLKEM_N)+ __loop__(+ assigns(ctr, state, memory_slice(entry, sizeof(mlk_poly)), object_whole(buf))+ invariant(ctr <= MLKEM_N)+ invariant(array_bound(entry->coeffs, 0, ctr, 0, MLKEM_Q)))+ {+ mlk_xof_squeezeblocks(buf, 1, &state);+ ctr = mlk_rej_uniform(entry->coeffs, MLKEM_N, ctr, buf, buflen);+ }++ mlk_xof_release(&state);++ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(buf, sizeof(buf));+}++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_ETA1 == 2 || \+ !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+/**+ * Load 4 bytes into a 32-bit integer in little-endian order.+ *+ * @reference{`load32_littleendian()` in the reference implementation @[REF].}+ *+ * @param[in] x Input byte array.+ *+ * @return 32-bit unsigned integer loaded from @p x.+ */+static uint32_t mlk_load32_littleendian(const uint8_t x[4])+{+ uint32_t r;+ r = (uint32_t)x[0];+ r |= (uint32_t)x[1] << 8;+ r |= (uint32_t)x[2] << 16;+ r |= (uint32_t)x[3] << 24;+ return r;+}++/* Reference: `cbd2()` in the reference implementation @[REF]. */+MLK_INTERNAL_API+void mlk_poly_cbd2(mlk_poly *r, const uint8_t buf[2 * MLKEM_N / 4])+{+ unsigned i;+ for (i = 0; i < MLKEM_N / 8; i++)+ __loop__(+ invariant(i <= MLKEM_N / 8)+ invariant(array_abs_bound(r->coeffs, 0, 8 * i, 3))+ decreases(MLKEM_N / 8 - i))+ {+ unsigned j;+ uint32_t t = mlk_load32_littleendian(buf + 4 * i);+ uint32_t d = t & 0x55555555;+ d += (t >> 1) & 0x55555555;++ for (j = 0; j < 8; j++)+ __loop__(+ invariant(i <= MLKEM_N / 8 && j <= 8)+ invariant(array_abs_bound(r->coeffs, 0, 8 * i + j, 3))+ decreases(8 - j))+ {+ /* Safety: The & 0x3 masks each value to 2 bits (range [0, 3]), so the+ * truncation and subsequent subtraction in int16_t is lossless. */+ const int16_t a = (int16_t)((d >> (4 * j + 0)) & 0x3);+ const int16_t b = (int16_t)((d >> (4 * j + 2)) & 0x3);+ r->coeffs[8 * i + j] = (int16_t)(a - b);+ }+ }+}+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_ETA1 == 2 || \+ !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_ETA1 == 3+/**+ * Load 3 bytes into a 32-bit integer in little-endian order.+ *+ * This function is only needed for ML-KEM-512.+ *+ * @reference{`load24_littleendian()` in the reference implementation @[REF].}+ *+ * @param[in] x Input byte array.+ *+ * @return 32-bit unsigned integer loaded from @p x (most significant byte+ * is zero).+ */+static uint32_t mlk_load24_littleendian(const uint8_t x[3])+{+ uint32_t r;+ r = (uint32_t)x[0];+ r |= (uint32_t)x[1] << 8;+ r |= (uint32_t)x[2] << 16;+ return r;+}++/* Reference: `cbd3()` in the reference implementation @[REF]. */+MLK_INTERNAL_API+void mlk_poly_cbd3(mlk_poly *r, const uint8_t buf[3 * MLKEM_N / 4])+{+ unsigned i;+ for (i = 0; i < MLKEM_N / 4; i++)+ __loop__(+ invariant(i <= MLKEM_N / 4)+ invariant(array_abs_bound(r->coeffs, 0, 4 * i, 4))+ decreases(MLKEM_N / 4 - i))+ {+ unsigned j;+ const uint32_t t = mlk_load24_littleendian(buf + 3 * i);+ uint32_t d = t & 0x00249249;+ d += (t >> 1) & 0x00249249;+ d += (t >> 2) & 0x00249249;++ for (j = 0; j < 4; j++)+ __loop__(+ invariant(i <= MLKEM_N / 4 && j <= 4)+ invariant(array_abs_bound(r->coeffs, 0, 4 * i + j, 4))+ decreases(4 - j))+ {+ /* Safety: The & 0x7 masks each value to 3 bits (range [0, 7]), so the+ * truncation and subsequent subtraction in int16_t is lossless. */+ const int16_t a = (int16_t)((d >> (6 * j + 0)) & 0x7);+ const int16_t b = (int16_t)((d >> (6 * j + 3)) & 0x7);+ r->coeffs[4 * i + j] = (int16_t)(a - b);+ }+ }+}+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_ETA1 == 3 */++#else /* !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(sampling)++#endif /* MLK_CONFIG_MULTILEVEL_NO_SHARED */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef MLKEM_GEN_MATRIX_NBLOCKS
+ cbits/mlkem/src/sampling.h view
@@ -0,0 +1,111 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#ifndef MLK_SAMPLING_H+#define MLK_SAMPLING_H++#include "cbmc.h"+#include "common.h"+#include "poly.h"++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_ETA1 == 2 || \+ !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_cbd2 MLK_NAMESPACE(poly_cbd2)+/**+ * Given an array of uniformly random bytes, compute a polynomial with+ * coefficients distributed according to a centered binomial distribution+ * with parameter eta=2.+ *+ * @spec{Implements @[FIPS203, Algorithm 8, SamplePolyCBD_2].}+ *+ * @param[out] r Output polynomial.+ * @param[in] buf Input byte array.+ */+MLK_INTERNAL_API+void mlk_poly_cbd2(mlk_poly *r, const uint8_t buf[2 * MLKEM_N / 4]);+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_ETA1 == 2 || \+ !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_ETA1 == 3+#define mlk_poly_cbd3 MLK_NAMESPACE(poly_cbd3)+/**+ * Given an array of uniformly random bytes, compute a polynomial with+ * coefficients distributed according to a centered binomial distribution+ * with parameter eta=3.+ *+ * This function is only needed for ML-KEM-512.+ *+ * @spec{Implements @[FIPS203, Algorithm 8, SamplePolyCBD_3].}+ *+ * @param[out] r Output polynomial.+ * @param[in] buf Input byte array.+ */+MLK_INTERNAL_API+void mlk_poly_cbd3(mlk_poly *r, const uint8_t buf[3 * MLKEM_N / 4]);+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_ETA1 == 3 */++#if !defined(MLK_CONFIG_SERIAL_FIPS202_ONLY)+#define mlk_poly_rej_uniform_x4 MLK_NAMESPACE(poly_rej_uniform_x4)+/**+ * Generate four polynomials using rejection sampling on (pseudo-)uniformly+ * random bytes sampled from a seed.+ *+ * @spec{Implements @[FIPS203, Algorithm 7, SampleNTT].}+ *+ * @param[out] vec0 Polynomial to be sampled.+ * @param[out] vec1 Polynomial to be sampled.+ * @param[out] vec2 Polynomial to be sampled.+ * @param[out] vec3 Polynomial to be sampled.+ * @param[in] seed Consecutive array of 4 seed buffers of size+ * MLKEM_SYMBYTES + 2 each, plus padding for alignment.+ */+MLK_INTERNAL_API+void mlk_poly_rej_uniform_x4(mlk_poly *vec0, mlk_poly *vec1, mlk_poly *vec2,+ mlk_poly *vec3,+ uint8_t seed[4][MLK_ALIGN_UP(MLKEM_SYMBYTES + 2)])+__contract__(+ requires(memory_no_alias(vec0, sizeof(mlk_poly)))+ requires(memory_no_alias(vec1, sizeof(mlk_poly)))+ requires(memory_no_alias(vec2, sizeof(mlk_poly)))+ requires(memory_no_alias(vec3, sizeof(mlk_poly)))+ requires(memory_no_alias(seed, 4 * MLK_ALIGN_UP(MLKEM_SYMBYTES + 2)))+ assigns(memory_slice(vec0, sizeof(mlk_poly)))+ assigns(memory_slice(vec1, sizeof(mlk_poly)))+ assigns(memory_slice(vec2, sizeof(mlk_poly)))+ assigns(memory_slice(vec3, sizeof(mlk_poly)))+ ensures(array_bound(vec0->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ ensures(array_bound(vec1->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ ensures(array_bound(vec2->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ ensures(array_bound(vec3->coeffs, 0, MLKEM_N, 0, MLKEM_Q)));+#endif /* !MLK_CONFIG_SERIAL_FIPS202_ONLY */++#define mlk_poly_rej_uniform MLK_NAMESPACE(poly_rej_uniform)+/**+ * Generate a polynomial using rejection sampling on (pseudo-)uniformly+ * random bytes sampled from a seed.+ *+ * @spec{Implements @[FIPS203, Algorithm 7, SampleNTT].}+ *+ * @param[out] entry Polynomial to be sampled.+ * @param[in] seed Seed buffer of size MLKEM_SYMBYTES + 2.+ */+MLK_INTERNAL_API+void mlk_poly_rej_uniform(mlk_poly *entry, uint8_t seed[MLKEM_SYMBYTES + 2])+__contract__(+ requires(memory_no_alias(entry, sizeof(mlk_poly)))+ requires(memory_no_alias(seed, MLKEM_SYMBYTES + 2))+ assigns(memory_slice(entry, sizeof(mlk_poly)))+ ensures(array_bound(entry->coeffs, 0, MLKEM_N, 0, MLKEM_Q)));++#endif /* !MLK_SAMPLING_H */
+ cbits/mlkem/src/symmetric.h view
@@ -0,0 +1,70 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#ifndef MLK_SYMMETRIC_H+#define MLK_SYMMETRIC_H+++#include "cbmc.h"+#include "common.h"+#include MLK_FIPS202_HEADER_FILE+#if !defined(MLK_CONFIG_SERIAL_FIPS202_ONLY)+#include MLK_FIPS202X4_HEADER_FILE+#endif++/* Macros denoting FIPS 203 specific Hash functions */++/* Hash function H, @[FIPS203, Section 4.1, Eq (4.4)] */+#define mlk_hash_h(OUT, IN, INBYTES) mlk_sha3_256(OUT, IN, INBYTES)++/* Hash function G, @[FIPS203, Section 4.1, Eq (4.5)] */+#define mlk_hash_g(OUT, IN, INBYTES) mlk_sha3_512(OUT, IN, INBYTES)++/* Hash function J, @[FIPS203, Section 4.1, Eq (4.4)] */+#define mlk_hash_j(OUT, IN, INBYTES) \+ mlk_shake256(OUT, MLKEM_SYMBYTES, IN, INBYTES)++/* PRF function, @[FIPS203, Section 4.1, Eq (4.3)]+ * Referring to (eq 4.3), `OUT` is assumed to contain `s || b`. */+#define mlk_prf_eta(ETA, OUT, IN) \+ mlk_shake256(OUT, (ETA) * MLKEM_N / 4, IN, MLKEM_SYMBYTES + 1)+#define mlk_prf_eta1(OUT, IN) mlk_prf_eta(MLKEM_ETA1, OUT, IN)+#define mlk_prf_eta2(OUT, IN) mlk_prf_eta(MLKEM_ETA2, OUT, IN)+#define mlk_prf_eta1_x4(OUT, IN) \+ mlk_shake256x4((OUT)[0], (OUT)[1], (OUT)[2], (OUT)[3], \+ (MLKEM_ETA1 * MLKEM_N / 4), (IN)[0], (IN)[1], (IN)[2], \+ (IN)[3], MLKEM_SYMBYTES + 1)++/* XOF function, FIPS 203 4.1 */+#define mlk_xof_ctx mlk_shake128ctx+#define mlk_xof_x4_ctx mlk_shake128x4ctx+#define mlk_xof_init(CTX) mlk_shake128_init((CTX))+#define mlk_xof_absorb(CTX, IN, INBYTES) \+ mlk_shake128_absorb_once((CTX), (IN), (INBYTES))+#define mlk_xof_squeezeblocks(BUF, NBLOCKS, CTX) \+ mlk_shake128_squeezeblocks((BUF), (NBLOCKS), (CTX))+#define mlk_xof_release(CTX) mlk_shake128_release((CTX))++#define mlk_xof_x4_init(CTX) mlk_shake128x4_init((CTX))+#define mlk_xof_x4_absorb(CTX, IN, INBYTES) \+ mlk_shake128x4_absorb_once((CTX), (IN)[0], (IN)[1], (IN)[2], (IN)[3], \+ (INBYTES))+#define mlk_xof_x4_squeezeblocks(BUF, NBLOCKS, CTX) \+ mlk_shake128x4_squeezeblocks((BUF)[0], (BUF)[1], (BUF)[2], (BUF)[3], \+ (NBLOCKS), (CTX))+#define mlk_xof_x4_release(CTX) mlk_shake128x4_release((CTX))++#define MLK_XOF_RATE SHAKE128_RATE++#endif /* !MLK_SYMMETRIC_H */
+ cbits/mlkem/src/sys.h view
@@ -0,0 +1,320 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_SYS_H+#define MLK_SYS_H++#if !defined(MLK_CONFIG_NO_ASM) && (defined(__GNUC__) || defined(__clang__))+#define MLK_HAVE_INLINE_ASM+#endif++/* Try to find endianness, if not forced through CFLAGS already */+#if !defined(MLK_SYS_LITTLE_ENDIAN) && !defined(MLK_SYS_BIG_ENDIAN)+#if defined(__BYTE_ORDER__)+#if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__+#define MLK_SYS_LITTLE_ENDIAN+#elif __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__+#define MLK_SYS_BIG_ENDIAN+#else+#error "__BYTE_ORDER__ defined, but don't recognize value."+#endif+#endif /* __BYTE_ORDER__ */++/* MSVC does not define __BYTE_ORDER__. However, MSVC only supports+ * little endian x86, x86_64, and AArch64. It is, hence, safe to assume+ * little endian. */+#if defined(_MSC_VER) && (defined(_M_X64) || defined(_M_AMD64) || \+ defined(_M_IX86) || defined(_M_ARM64))+#define MLK_SYS_LITTLE_ENDIAN+#endif++#endif /* !MLK_SYS_LITTLE_ENDIAN && !MLK_SYS_BIG_ENDIAN */++/* Check if we're running on an AArch64 little endian system. _M_ARM64 is set by+ * MSVC. */+#if defined(__AARCH64EL__) || defined(_M_ARM64)+#define MLK_SYS_AARCH64+#endif++/* Check if the AArch64 compilation target supports NEON (Advanced SIMD).+ *+ * Some compilers also define __ARM_NEON__, but __ARM_NEON is the most reliable+ * signal. Specifically, clang on Apple appears to keep __ARM_NEON__ set even if+ * -march=armv8-a+nosimd is set.+ *+ * gcc 4.8 -- the first gcc version introducing Neon support -- sets neither+ * __ARM_NEON nor __ARM_NEON__; in fact, there is no preprocessor signal that+ * Neon is enabled. If you use gcc 4.8, you should set __ARM_NEON manually.+ * gcc 4.9 onwards do set __ARM_NEON.+ */+#if defined(MLK_SYS_AARCH64) && defined(__ARM_NEON)+#define MLK_SYS_AARCH64_NEON+#endif++/* Check if we're running on an AArch64 big endian system. */+#if defined(__AARCH64EB__)+#define MLK_SYS_AARCH64_EB+#endif++/* Check if we're running on an Armv8.1-M system with MVE */+#if defined(__ARM_ARCH_8_1M_MAIN__) || defined(__ARM_FEATURE_MVE)+#define MLK_SYS_ARMV81M_MVE+#endif++/* Check if we're running on an x86_64 system. */+#if defined(__x86_64__) || defined(_M_X64) || defined(_M_AMD64)+#define MLK_SYS_X86_64+#if defined(__AVX2__)+#define MLK_SYS_X86_64_AVX2+#endif+#endif /* __x86_64__ || _M_X64 || _M_AMD64 */++#if defined(MLK_SYS_LITTLE_ENDIAN) && defined(__powerpc64__)+#define MLK_SYS_PPC64LE+#endif++#if defined(__riscv) && defined(__riscv_xlen) && __riscv_xlen == 64+#define MLK_SYS_RISCV64+#endif++#if defined(MLK_SYS_RISCV64) && defined(__riscv_vector) && \+ defined(__riscv_v_intrinsic)+#define MLK_SYS_RISCV64_RVV+#endif++#if defined(__riscv) && defined(__riscv_xlen) && __riscv_xlen == 32+#define MLK_SYS_RISCV32+#endif++#if defined(_WIN32)+#define MLK_SYS_WINDOWS+#endif++#if defined(__linux__)+#define MLK_SYS_LINUX+#endif++#if defined(__APPLE__)+#define MLK_SYS_APPLE+#endif++#if defined(MLK_FORCE_AARCH64) && !defined(MLK_SYS_AARCH64)+#error "MLK_FORCE_AARCH64 is set, but we don't seem to be on an AArch64 system."+#endif++#if defined(MLK_FORCE_AARCH64_EB) && !defined(MLK_SYS_AARCH64_EB)+#error \+ "MLK_FORCE_AARCH64_EB is set, but we don't seem to be on an AArch64 system."+#endif++#if defined(MLK_FORCE_X86_64) && !defined(MLK_SYS_X86_64)+#error "MLK_FORCE_X86_64 is set, but we don't seem to be on an X86_64 system."+#endif++#if defined(MLK_FORCE_PPC64LE) && !defined(MLK_SYS_PPC64LE)+#error "MLK_FORCE_PPC64LE is set, but we don't seem to be on a PPC64LE system."+#endif++#if defined(MLK_FORCE_RISCV64) && !defined(MLK_SYS_RISCV64)+#error "MLK_FORCE_RISCV64 is set, but we don't seem to be on a RISCV64 system."+#endif++#if defined(MLK_FORCE_RISCV32) && !defined(MLK_SYS_RISCV32)+#error "MLK_FORCE_RISCV32 is set, but we don't seem to be on a RISCV32 system."+#endif++/*+ * MLK_INLINE: Hint for inlining.+ * - MSVC: __inline+ * - C99+: inline+ * - GCC/Clang C90: __attribute__((unused)) to silence warnings+ * - Other C90: empty+ */+#if !defined(MLK_INLINE)+#if defined(_MSC_VER)+#define MLK_INLINE __inline+#elif defined(inline) || \+ (defined(__STDC_VERSION__) && __STDC_VERSION__ >= 199901L)+#define MLK_INLINE inline+#elif defined(__GNUC__) || defined(__clang__)+#define MLK_INLINE __attribute__((unused))+#else+#define MLK_INLINE+#endif+#endif /* !MLK_INLINE */++/*+ * MLK_ALWAYS_INLINE: Force inlining.+ * - MSVC: __forceinline+ * - GCC/Clang C99+: MLK_INLINE __attribute__((always_inline))+ * - Other: MLK_INLINE (no forced inlining)+ */+#if !defined(MLK_ALWAYS_INLINE)+#if defined(_MSC_VER)+#define MLK_ALWAYS_INLINE __forceinline+#elif (defined(__GNUC__) || defined(__clang__)) && \+ (defined(inline) || \+ (defined(__STDC_VERSION__) && __STDC_VERSION__ >= 199901L))+#define MLK_ALWAYS_INLINE MLK_INLINE __attribute__((always_inline))+#else+#define MLK_ALWAYS_INLINE MLK_INLINE+#endif+#endif /* !MLK_ALWAYS_INLINE */++/*+ * MLK_NOINLINE: Prevent inlining.+ * - MSVC: __declspec(noinline)+ * - GCC/Clang: __attribute__((noinline))+ * - Other: empty+ */+#if !defined(MLK_NOINLINE)+#if defined(_MSC_VER)+#define MLK_NOINLINE __declspec(noinline)+#elif defined(__GNUC__) || defined(__clang__)+#define MLK_NOINLINE __attribute__((noinline))+#else+#define MLK_NOINLINE+#endif+#endif /* !MLK_NOINLINE */++#ifndef MLK_STATIC_TESTABLE+#define MLK_STATIC_TESTABLE static+#endif++/*+ * C90 does not have the restrict compiler directive yet.+ * We don't use it in C90 builds.+ */+#if !defined(restrict)+#if defined(__STDC_VERSION__) && __STDC_VERSION__ >= 199901L+#define MLK_RESTRICT restrict+#else+#define MLK_RESTRICT+#endif++#else /* !restrict */++#define MLK_RESTRICT restrict+#endif /* restrict */++#define MLK_DEFAULT_ALIGN 32+#define MLK_ALIGN_UP(N) \+ ((((N) + (MLK_DEFAULT_ALIGN - 1)) / MLK_DEFAULT_ALIGN) * MLK_DEFAULT_ALIGN)+#if defined(__GNUC__)+#define MLK_ALIGN __attribute__((aligned(MLK_DEFAULT_ALIGN)))+#elif defined(_MSC_VER)+#define MLK_ALIGN __declspec(align(MLK_DEFAULT_ALIGN))+#else+#define MLK_ALIGN /* No known support for alignment constraints */+#endif+++/* New X86_64 CPUs support control-flow protection using the CET instructions.+ * When enabled (through -fcf-protection=), all compilation units (including+ * empty ones) need to support CET for this to work.+ * For assembly, this means that source files need to signal support for+ * CET by setting the appropriate note.gnu.property section.+ * This can be achieved by including the <cet.h> header in all assembly file.+ * This file also provides the _CET_ENDBR macro which needs to be placed at+ * every potential target of an indirect branch.+ * If CET is enabled _CET_ENDBR maps to the endbr64 instruction, otherwise+ * it is empty.+ * In case the compiler does not support CET (e.g., <gcc8, <clang11),+ * the __CET__ macro is not set and we default to nothing.+ * Note that we only issue _CET_ENDBR instructions through the MLK_ASM_FN_SYMBOL+ * macro as the global symbols are the only possible targets of indirect+ * branches in our code.+ */+#if defined(MLK_SYS_X86_64)+#if defined(__CET__)+#include <cet.h>+#define MLK_CET_ENDBR _CET_ENDBR+#else+#define MLK_CET_ENDBR+#endif+#endif /* MLK_SYS_X86_64 */++#if defined(MLK_CONFIG_CT_TESTING_ENABLED) && !defined(__ASSEMBLER__)+#include <valgrind/memcheck.h>+#define MLK_CT_TESTING_SECRET(ptr, len) \+ VALGRIND_MAKE_MEM_UNDEFINED((ptr), (len))+#define MLK_CT_TESTING_DECLASSIFY(ptr, len) \+ VALGRIND_MAKE_MEM_DEFINED((ptr), (len))+#else /* MLK_CONFIG_CT_TESTING_ENABLED && !__ASSEMBLER__ */+#define MLK_CT_TESTING_SECRET(ptr, len) \+ do \+ { \+ } while (0)+#define MLK_CT_TESTING_DECLASSIFY(ptr, len) \+ do \+ { \+ } while (0)+#endif /* !(MLK_CONFIG_CT_TESTING_ENABLED && !__ASSEMBLER__) */++#if defined(__GNUC__) || defined(__clang__)+#define MLK_MUST_CHECK_RETURN_VALUE __attribute__((warn_unused_result))+#else+#define MLK_MUST_CHECK_RETURN_VALUE+#endif++/* The x86_64 assembly backend uses the SysV calling convention. On Windows,+ * where the Microsoft x64 calling convention is the default, it can still be+ * used with compilers that allow choosing the calling convention per+ * function: GCC and Clang support __attribute__((sysv_abi)), which makes+ * calls to the annotated function follow the SysV calling convention.+ *+ * MLK_SYSV_ABI_SUPPORTED signals that the toolchain can call SysV assembly+ * routines; the x86_64 assembly backend is only enabled if it is defined.+ * MLK_SYSV_ABI is the attribute carried by declarations of x86_64 assembly+ * routines. Both macros can be set externally for toolchains offering an+ * equivalent mechanism that is not recognized here. */+#if defined(MLK_SYS_X86_64) && !defined(MLK_SYSV_ABI_SUPPORTED)+#if !defined(MLK_SYS_WINDOWS) || defined(__GNUC__) || defined(__clang__)+#define MLK_SYSV_ABI_SUPPORTED+#endif+#endif++#if !defined(MLK_SYSV_ABI)+#if defined(MLK_SYS_WINDOWS) && defined(MLK_SYSV_ABI_SUPPORTED)+#define MLK_SYSV_ABI __attribute__((sysv_abi))+#else+#define MLK_SYSV_ABI+#endif+#endif /* !MLK_SYSV_ABI */++#if !defined(__ASSEMBLER__)+/* System capability enumeration */+typedef enum+{+ /* x86_64 */+ MLK_SYS_CAP_X86_64_AVX2,+ /* AArch64 */+ MLK_SYS_CAP_AARCH64_NEON,+ MLK_SYS_CAP_AARCH64_SHA3,+ /* Armv8.1-M */+ MLK_SYS_CAP_ARMV81M_MVE+} mlk_sys_cap;++#if !defined(MLK_CONFIG_CUSTOM_CAPABILITY_FUNC)+#include "cbmc.h"++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_sys_check_capability(mlk_sys_cap cap)+__contract__(+ ensures(return_value == 0 || return_value == 1)+)+{+ /* By default, we rely on compile-time feature detection/specification:+ * If a feature is enabled at compile-time, we assume it is supported by+ * the host that the resulting library/binary will be built on.+ * If this assumption is not true, you MUST overwrite this function.+ * See the documentation of MLK_CONFIG_CUSTOM_CAPABILITY_FUNC in+ * mlkem_native_config.h for more information. */+ (void)cap;+ return 1;+}+#endif /* !MLK_CONFIG_CUSTOM_CAPABILITY_FUNC */+#endif /* !__ASSEMBLER__ */++#endif /* !MLK_SYS_H */
+ cbits/mlkem/src/verify.c view
@@ -0,0 +1,20 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#include "verify.h"++#if !defined(MLK_USE_ASM_VALUE_BARRIER) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)+/*+ * Masking value used in constant-time functions from+ * verify.h to block the compiler's range analysis and+ * thereby reduce the risk of compiler-introduced branches.+ */+volatile uint64_t mlk_ct_opt_blocker_u64 = 0;++#else /* !MLK_USE_ASM_VALUE_BARRIER && !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(verify)++#endif /* !(!MLK_USE_ASM_VALUE_BARRIER && !MLK_CONFIG_MULTILEVEL_NO_SHARED) */
+ cbits/mlkem/src/verify.h view
@@ -0,0 +1,447 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ *+ * - [REF]+ * CRYSTALS-Kyber C reference implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/ref+ *+ * - [libmceliece]+ * libmceliece implementation of Classic McEliece+ * Bernstein, Chou+ * https://lib.mceliece.org/+ *+ * - [optblocker]+ * PQC forum post on opt-blockers using volatile globals+ * Daniel J. Bernstein+ * https://groups.google.com/a/list.nist.gov/g/pqc-forum/c/hqbtIGFKIpU/m/H14H0wOlBgAJ+ */++#ifndef MLK_VERIFY_H+#define MLK_VERIFY_H+++#include "cbmc.h"+#include "common.h"++/* Constant-time comparisons and conditional operations++ We reduce the risk for compilation into variable-time code+ through the use of 'value barriers'.++ Functionally, a value barrier is a no-op. To the compiler, however,+ it constitutes an arbitrary modification of its input, and therefore+ harden's value propagation and range analysis.++ We consider two approaches to implement a value barrier:+ - An empty inline asm block which marks the target value as clobbered.+ - XOR'ing with the value of a volatile global that's set to 0;+ see @[optblocker] for a discussion of this idea, and+ @[libmceliece, inttypes/crypto_intN.h] for an implementation.++ The first approach is cheap because it only prevents the compiler+ from reasoning about the value of the variable past the barrier,+ but does not directly generate additional instructions.++ The second approach generates redundant loads and XOR operations+ and therefore comes at a higher runtime cost. However, it appears+ more robust towards optimization, as compilers should never drop+ a volatile load.++ We use the empty-ASM value barrier for GCC and clang, and fall+ back to the global volatile barrier otherwise.++ The global value barrier can be forced by setting+ MLK_CONFIG_NO_ASM_VALUE_BARRIER.++*/++#if defined(MLK_HAVE_INLINE_ASM) && !defined(MLK_CONFIG_NO_ASM_VALUE_BARRIER)+#define MLK_USE_ASM_VALUE_BARRIER+#endif++#if !defined(MLK_USE_ASM_VALUE_BARRIER)++/*+ * Declaration of global volatile that the global value barrier+ * is loading from and masking with.+ */+#define mlk_ct_opt_blocker_u64 MLK_NAMESPACE(ct_opt_blocker_u64)+extern volatile uint64_t mlk_ct_opt_blocker_u64;++/* Helper functions for obtaining global masks of various sizes */++/* This contract is not proved but treated as an axiom.+ *+ * Its validity relies on the assumption that the global opt-blocker+ * constant mlk_ct_opt_blocker_u64 is not modified.+ */+static MLK_INLINE uint64_t mlk_ct_get_optblocker_u64(void)+__contract__(ensures(return_value == 0)) { return mlk_ct_opt_blocker_u64; }++static MLK_INLINE uint8_t mlk_ct_get_optblocker_u8(void)+__contract__(ensures(return_value == 0)) { return (uint8_t)mlk_ct_get_optblocker_u64(); }++static MLK_INLINE uint32_t mlk_ct_get_optblocker_u32(void)+__contract__(ensures(return_value == 0)) { return (uint32_t)mlk_ct_get_optblocker_u64(); }++static MLK_INLINE int32_t mlk_ct_get_optblocker_i32(void)+__contract__(ensures(return_value == 0)) { return (int32_t)mlk_ct_get_optblocker_u64(); }++/* Opt-blocker based implementation of value barriers */+static MLK_INLINE uint32_t mlk_value_barrier_u32(uint32_t b)+__contract__(ensures(return_value == b)) { return (b ^ mlk_ct_get_optblocker_u32()); }++static MLK_INLINE int32_t mlk_value_barrier_i32(int32_t b)+__contract__(ensures(return_value == b)) { return (b ^ mlk_ct_get_optblocker_i32()); }++static MLK_INLINE uint8_t mlk_value_barrier_u8(uint8_t b)+__contract__(ensures(return_value == b)) { return (b ^ mlk_ct_get_optblocker_u8()); }++#else /* !MLK_USE_ASM_VALUE_BARRIER */++static MLK_INLINE uint32_t mlk_value_barrier_u32(uint32_t b)+__contract__(ensures(return_value == b))+{+ __asm__ volatile("" : "+r"(b));+ return b;+}++static MLK_INLINE int32_t mlk_value_barrier_i32(int32_t b)+__contract__(ensures(return_value == b))+{+ __asm__ volatile("" : "+r"(b));+ return b;+}++static MLK_INLINE uint8_t mlk_value_barrier_u8(uint8_t b)+__contract__(ensures(return_value == b))+{+ __asm__ volatile("" : "+r"(b));+ return b;+}++#endif /* MLK_USE_ASM_VALUE_BARRIER */++#ifdef CBMC+#pragma CPROVER check push+#pragma CPROVER check disable "conversion"+#endif+/**+ * Cast uint16 value to int16.+ *+ * @param x Input value.+ *+ * @return For uint16_t x, the unique y in int16_t so that x == y mod 2^16.+ * Concretely:+ * - x < 32768: returns x+ * - x >= 32768: returns x - 65536+ */+static MLK_ALWAYS_INLINE int16_t mlk_cast_uint16_to_int16(uint16_t x)+{+ /*+ * PORTABILITY: This relies on uint16_t -> int16_t+ * being implemented as the inverse of int16_t -> uint16_t,+ * which is implementation-defined (C99 6.3.1.3 (3))+ * CBMC (correctly) fails to prove this conversion is OK,+ * so we have to suppress that check here+ */+ return (int16_t)x;+}+#ifdef CBMC+#pragma CPROVER check pop+#endif++/**+ * Cast int32 value to uint16 as per C standard.+ *+ * @param x Input value.+ *+ * @return For int32_t x, the unique y in uint16_t so that x == y mod 2^16.+ */+static MLK_ALWAYS_INLINE uint16_t mlk_cast_int32_to_uint16(int32_t x)+{+ return (uint16_t)(x & (int32_t)UINT16_MAX);+}++/**+ * Cast int16 value to uint16 as per C standard.+ *+ * @param x Input value.+ *+ * @return For int16_t x, the unique y in uint16_t so that x == y mod 2^16.+ */+static MLK_ALWAYS_INLINE uint16_t mlk_cast_int16_to_uint16(int32_t x)+{+ return mlk_cast_int32_to_uint16(x);+}++/**+ * Return 0 if input is non-negative, and -1 otherwise.+ *+ * @reference{Embedded in the polynomial compression function in the+ * reference implementation @[REF]. Used as part of signed->unsigned+ * conversion for modular representatives to detect whether the input is+ * negative. This happens in `mlk_poly_reduce()` here, and as part of+ * polynomial compression functions in the reference implementation. See+ * `mlk_poly_reduce()`. We use value barriers to reduce the risk of+ * compiler-introduced branches.}+ *+ * @param x Value to be converted into a mask.+ *+ * @return Mask value (0 or 0xFFFF).+ */+static MLK_INLINE uint16_t mlk_ct_cmask_neg_i16(int16_t x)+__contract__(ensures(return_value == ((x < 0) ? 0xFFFF : 0)))+{+ int32_t tmp = mlk_value_barrier_i32((int32_t)x);+ /*+ * PORTABILITY: Right-shift on a signed integer is+ * implementation-defined for negative left argument.+ * Here, we assume it's sign-preserving "arithmetic" shift right.+ * See (C99 6.5.7 (5))+ */+ tmp >>= 16;+ return mlk_cast_int32_to_uint16(tmp);+}++/**+ * Return 0 if input is zero, and -1 otherwise.+ *+ * @reference{Embedded in `cmov_int16()` in the reference implementation+ * @[REF]. Uses a value barrier and shift instead of `b = -b` to convert+ * condition into mask.}+ *+ * @param x Value to be converted into a mask.+ *+ * @return Mask value (0 or 0xFFFF).+ */+static MLK_INLINE uint16_t mlk_ct_cmask_nonzero_u16(uint16_t x)+__contract__(ensures(return_value == ((x == 0) ? 0 : 0xFFFF)))+{+ int32_t tmp = mlk_value_barrier_i32(-((int32_t)x));+ /*+ * PORTABILITY: Right-shift on a signed integer is+ * implementation-defined for negative left argument.+ * Here, we assume it's sign-preserving "arithmetic" shift right.+ * See (C99 6.5.7 (5))+ */+ tmp >>= 16;+ return mlk_cast_int32_to_uint16(tmp);+}++/**+ * Return 0 if input is zero, and -1 otherwise.+ *+ * @reference{Embedded in `verify()` and `cmov()` in the reference+ * implementation @[REF]. We include a value barrier not present in the+ * reference implementation, to prevent the compiler from realizing that+ * this function returns a mask.}+ *+ * @param x Value to be converted into a mask.+ *+ * @return Mask value (0 or 0xFF).+ */+static MLK_INLINE uint8_t mlk_ct_cmask_nonzero_u8(uint8_t x)+__contract__(ensures(return_value == ((x == 0) ? 0 : 0xFF)))+{+ uint16_t mask = mlk_ct_cmask_nonzero_u16((uint16_t)x);+ return (uint8_t)(mask & 0xFF);+}++/**+ * Functionally equivalent to cond ? a : b, but implemented with guards+ * against compiler-introduced branches.+ *+ * @spec{With `a = MLKEM_Q_HALF` and `b=0`, this essentially implements+ * `Decompress_1` @[FIPS203, Eq (4.8)] in `mlk_poly_frommsg()`. With+ * `a = x + MLKEM_Q`, `b = x`, and `cond` indicating whether `x` is negative,+ * implements signed->unsigned conversion of modular representatives.+ * Questions of representation are not considered in the specification+ * @[FIPS203, Section 2.4.1, "The pseudocode is agnostic regarding how an+ * integer modulo 𝑚 is represented in actual implementations"].}+ *+ * @reference{Embedded in the polynomial compression function in the+ * reference implementation @[REF]. Used as part of signed->unsigned+ * conversion for modular representatives. This happens in `mlk_poly_reduce()`+ * here, and as part of polynomial compression functions in @[REF]. See+ * `mlk_poly_reduce()`. Barrier to reduce the risk of compiler-introduced+ * branches. For `a = MLKEM_Q_HALF` and `b=0`, also embedded in+ * `poly_frommsg()` from the reference implementation, which uses+ * `cmov_int16()` instead.}+ *+ * @param a First alternative.+ * @param b Second alternative.+ * @param cond Condition variable.+ *+ * @return @p a if @p cond != 0, else @p b.+ */+static MLK_INLINE int16_t mlk_ct_sel_int16(int16_t a, int16_t b, uint16_t cond)+__contract__(ensures(return_value == (cond ? a : b)))+{+ uint16_t au = mlk_cast_int16_to_uint16(a);+ uint16_t bu = mlk_cast_int16_to_uint16(b);+ uint16_t res = bu ^ (mlk_ct_cmask_nonzero_u16(cond) & (au ^ bu));+ return mlk_cast_uint16_to_int16(res);+}++/**+ * Functionally equivalent to cond ? a : b, but implemented with guards+ * against compiler-introduced branches.+ *+ * @reference{Embedded into `cmov()` in the reference implementation @[REF].+ * Uses a value barrier to get mask from condition value.}+ *+ * @param a First alternative.+ * @param b Second alternative.+ * @param cond Condition variable.+ *+ * @return @p a if @p cond != 0, else @p b.+ */+static MLK_INLINE uint8_t mlk_ct_sel_uint8(uint8_t a, uint8_t b, uint8_t cond)+__contract__(ensures(return_value == (cond ? a : b)))+{+ return b ^ (mlk_ct_cmask_nonzero_u8(cond) & (a ^ b));+}++/**+ * Compare two arrays for equality in constant time.+ *+ * @spec{Used to securely compute conditional move in @[FIPS203, Algorithm+ * 18 (ML-KEM.Decaps_Internal, L9-11].}+ *+ * @reference{`cmov()` in the reference implementation @[REF]. We return+ * `uint8_t`, not `int`. We use an additional XOR-accumulator in the+ * comparison loop which prevents early abort if the OR-accumulator is 0xFF.+ * We use a value barrier to convert the OR-accumulator into a mask; the+ * reference implementation uses a shift which the compiler can argue to+ * result in either 0 or 0xFF..FF.}+ *+ * @param[in] a First byte array.+ * @param[in] b Second byte array.+ * @param len Length of the byte arrays, upper-bounded to UINT16_MAX to+ * control proof complexity only.+ *+ * @retval 0 The byte arrays are equal.+ * @retval 0xFF The byte arrays are not equal.+ */+static MLK_INLINE uint8_t mlk_ct_memcmp(const uint8_t *a, const uint8_t *b,+ const size_t len)+__contract__(+ requires(len <= UINT16_MAX)+ requires(memory_no_alias(a, len))+ requires(memory_no_alias(b, len))+ ensures((return_value == 0) || (return_value == 0xFF))+ ensures((return_value == 0) == forall(i, 0, len, (a[i] == b[i]))))+{+ uint8_t r = 0, s = 0;+ unsigned i;++ for (i = 0; i < len; i++)+ __loop__(+ invariant(i <= len)+ invariant((r == 0) == (forall(k, 0, i, (a[k] == b[k]))))+ decreases(len - i))+ {+ r |= a[i] ^ b[i];+ /* s is useless, but prevents the loop from being aborted once r=0xff. */+ s ^= a[i] ^ b[i];+ }++ /*+ * - Convert r into a mask; this may not be necessary, but is an additional+ * safeguard+ * towards leaking information about a and b.+ * - XOR twice with s, separated by a value barrier, to prevent the compile+ * from dropping the s computation in the loop.+ */+ return (mlk_value_barrier_u8(mlk_ct_cmask_nonzero_u8(r) ^ s) ^ s);+}++/**+ * Copy len bytes from x to r if b is zero; don't modify x if b is non-zero.+ * Assumes two's complement representation of negative integers. Runs in+ * constant time.+ *+ * @spec{Used to securely compute conditional move in @[FIPS203, Algorithm+ * 18 (ML-KEM.Decaps_Internal, L9-11].}+ *+ * @reference{`cmov()` in the reference implementation @[REF]. We move if+ * condition value is `0`, not `1`. We use `mlk_ct_sel_uint8` for+ * constant-time selection.}+ *+ * @param[out] r Output byte array.+ * @param[in] x Input byte array.+ * @param len Number of bytes to be copied.+ * @param b Condition value.+ */+static MLK_INLINE void mlk_ct_cmov_zero(uint8_t *r, const uint8_t *x,+ size_t len, uint8_t b)+__contract__(+ requires(len <= UINT32_MAX)+ requires(memory_no_alias(r, len))+ requires(memory_no_alias(x, len))+ assigns(memory_slice(r, len))+ ensures(forall(i, 0, len, (r[i] == (b == 0 ? x[i] : old(r)[i])))))+{+ size_t i;+ for (i = 0; i < len; i++)+ __loop__(+ invariant(i <= len)+ invariant(forall(k, 0, i, r[k] == (b == 0 ? x[k] : loop_entry(r)[k])))+ decreases(len - i))+ {+ r[i] = mlk_ct_sel_uint8(r[i], x[i], b);+ }+}++/**+ * Force-zeroize a buffer.+ *+ * @spec{Used to implement @[FIPS203, Section 3.3, Destruction of+ * intermediate values].}+ *+ * @reference{Not present in the reference implementation @[REF].}+ *+ * @param[out] ptr Buffer to be zeroed.+ * @param len Number of bytes to be zeroed.+ */+#if !defined(MLK_CONFIG_CUSTOM_ZEROIZE)+#if defined(MLK_SYS_WINDOWS)+#include <windows.h>+#elif !defined(MLK_HAVE_INLINE_ASM)+#error No plausibly-secure implementation of mlk_zeroize available. Please provide your own using MLK_CONFIG_CUSTOM_ZEROIZE.+#endif++static MLK_INLINE void mlk_zeroize(void *ptr, size_t len)+__contract__(+ requires(len <= UINT32_MAX)+ requires(memory_no_alias(ptr, len))+ assigns(memory_slice(ptr, len))+ ensures(array_zeroized_u8((uint8_t *)ptr, len)))+{+#if defined(MLK_SYS_WINDOWS)+ SecureZeroMemory(ptr, len);+#else+ mlk_memset(ptr, 0, len);+ /* This follows OpenSSL and seems sufficient to prevent the compiler+ * from optimizing away the memset.+ *+ * If there was a reliable way to detect availability of memset_s(),+ * that would be preferred. */+ __asm__ volatile("" : : "r"(ptr) : "memory");+#endif /* !MLK_SYS_WINDOWS */+}+#endif /* !MLK_CONFIG_CUSTOM_ZEROIZE */++#endif /* !MLK_VERIFY_H */
+ cbits/mlkem/src/zetas.inc view
@@ -0,0 +1,30 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */+++/*+ * Table of zeta values used in the reference NTT and inverse NTT.+ * See autogen for details.+ */+static MLK_ALIGN const int16_t mlk_zetas[128] = {+ -1044, -758, -359, -1517, 1493, 1422, 287, 202, -171, 622, 1577,+ 182, 962, -1202, -1474, 1468, 573, -1325, 264, 383, -829, 1458,+ -1602, -130, -681, 1017, 732, 608, -1542, 411, -205, -1571, 1223,+ 652, -552, 1015, -1293, 1491, -282, -1544, 516, -8, -320, -666,+ -1618, -1162, 126, 1469, -853, -90, -271, 830, 107, -1421, -247,+ -951, -398, 961, -1508, -725, 448, -1065, 677, -1275, -1103, 430,+ 555, 843, -1251, 871, 1550, 105, 422, 587, 177, -235, -291,+ -460, 1574, 1653, -246, 778, 1159, -147, -777, 1483, -602, 1119,+ -1590, 644, -872, 349, 418, 329, -156, -75, 817, 1097, 603,+ 610, 1322, -1285, -1465, 384, -1215, -136, 1218, -1335, -874, 220,+ -1187, -1659, -1185, -1530, -1278, 794, -1510, -854, -870, 478, -108,+ -308, 996, 991, 958, -1460, 1522, 1628,+};
+ cbits/tests/ct/ct_mlkem.c view
@@ -0,0 +1,45 @@+/* ML-KEM-768, which is the parameter set TLS uses. The seed and the+ * decapsulation key are secret; the encapsulation key and the ciphertext are+ * published, and the shared secret is the answer the caller asked for.+ *+ * mlkem-native's AArch64 and x86-64 assembly carries a HOL-Light proof of+ * secret-independent timing, so a report from the backend build means the+ * proof does not cover what crypton built, or that crypton reached it+ * wrongly. The portable C has no such proof, which is the reason to ask it+ * the question at all. */+#include "tests/ct/ct.h"++int crypton_mlkem768_keypair_derand(uint8_t *pk, uint8_t *sk,+ const uint8_t *coins);+int crypton_mlkem768_enc_derand(uint8_t *ct, uint8_t *ss, const uint8_t *pk,+ const uint8_t *coins);+int crypton_mlkem768_dec(uint8_t *ss, const uint8_t *ct, const uint8_t *sk);++int main(void) {+ uint8_t pk[1184], sk[2400], ct[1088], ss[32], ss2[32];+ uint8_t kcoins[64], ecoins[32];++ ct_fill(kcoins, sizeof kcoins);+ ct_fill(ecoins, sizeof ecoins);++ CT_SECRET(kcoins, sizeof kcoins);+ if (crypton_mlkem768_keypair_derand(pk, sk, kcoins) != 0) return 1;+ CT_PUBLIC(pk, sizeof pk);+ ct_sink(pk, sizeof pk);++ /* The message encapsulation draws is as secret as the key it derives. */+ CT_SECRET(ecoins, sizeof ecoins);+ if (crypton_mlkem768_enc_derand(ct, ss, pk, ecoins) != 0) return 1;+ CT_PUBLIC(ct, sizeof ct);+ CT_PUBLIC(ss, sizeof ss);+ ct_sink(ct, sizeof ct);+ ct_sink(ss, sizeof ss);++ /* sk is still undefined here: decapsulation is the half that runs on the+ * long-lived secret, and the half an attacker can drive by sending+ * ciphertexts of their choosing. */+ if (crypton_mlkem768_dec(ss2, ct, sk) != 0) return 1;+ CT_PUBLIC(ss2, sizeof ss2);+ ct_sink(ss2, sizeof ss2);+ return 0;+}
cbits/tests/ct/run.sh view
@@ -46,11 +46,37 @@ ct_define=-DCRYPTON_CT_VALGRIND fi +# MLK_CONFIG_CT_TESTING_ENABLED turns on mlkem-native's own secret and+# declassify annotations, which are in the source at the points the algorithm+# says a value becomes public. The important one is the public seed: it is+# derived from the key generation seed, so it starts out secret, and the+# matrix it then seeds is generated by rejection sampling, which branches on+# the bytes it draws. Upstream declassifies the seed first -- rho travels in+# the encapsulation key and is not secret -- and without that the driver+# reports a leak that is not one. Those points are maintained by the people+# whose proofs cover this code, so they are a better answer than marking by+# hand from outside.+#+# Only with valgrind, since the macros include valgrind/memcheck.h.+mlkem_src="cbits/mlkem/crypton_mlkem.c"+mlkem_inc="-Icbits/mlkem${ct_define:+ -DMLK_CONFIG_CT_TESTING_ENABLED}"+# And the hand-written backend, where the architecture has one. Both builds+# have to be silent: unlike the AES pair below there is no variable-time+# implementation here for one of them to expose.+mlkem_native_src="$mlkem_src cbits/mlkem/crypton_mlkem_asm.S"+mlkem_native_inc="$mlkem_inc -DCRYPTON_MLKEM_NATIVE_BACKEND"+++# run_one <name> <sources> <includes> [driver]+#+# The driver defaults to ct_<name>.c. It is given separately where one+# driver is built twice -- the same question asked of the portable C and of+# the hand-written backend -- rather than copying the file to a second name. run_one() {- name=$1; srcs=$2; inc=$3+ name=$1; srcs=$2; inc=$3; drv=${4:-$1} # shellcheck disable=SC2086 $cc -O2 -g $ct_define -Icbits -Icbits/include64 $inc \- -o "$out/$name" "cbits/tests/ct/ct_$name.c" $srcs 2> "$out/$name.cc" || {+ -o "$out/$name" "cbits/tests/ct/ct_$drv.c" $srcs 2> "$out/$name.cc" || { echo "FAIL $name did not build"; sed -n '1,12p' "$out/$name.cc"; status=1; return } if [ "$have_valgrind" = no ]; then@@ -150,6 +176,7 @@ run_one ed25519 "cbits/ed25519/ed25519.c cbits/crypton_sha512.c" "-Icbits/ed25519" run_one decaf "$decaf_src" "$decaf_inc" run_one chapoly "cbits/crypton_chacha.c cbits/crypton_poly1305.c" ""+run_one mlkem "$mlkem_src" "$mlkem_inc" run_one aes "$aes_src" "" # Only where the instructions exist. Elsewhere there is nothing to measure@@ -157,6 +184,17 @@ case $(uname -m) in aarch64 | arm64) run_one aes_armv8 "$armv8_src" "$armv8_inc"+ ;;+esac++# The ML-KEM backend, on both architectures that have one. x86-64 wants the+# flags its own build system passes, or the AVX2 sources do not assemble.+case $(uname -m) in+aarch64 | arm64)+ run_one mlkem_native "$mlkem_native_src" "$mlkem_native_inc" mlkem+ ;;+x86_64 | amd64)+ run_one mlkem_native "$mlkem_native_src" "$mlkem_native_inc -mavx2 -mbmi2" mlkem ;; esac
crypton.cabal view
@@ -1,6 +1,6 @@ cabal-version: 3.0 name: crypton-version: 2.1.7+version: 2.1.8 -- crypton's own code is BSD-3-Clause. The parts of -- cbits/aes/gcm_fused_x86.c that follow picotls's fusion are MIT, and the -- vendored s2n-bignum assembly in cbits/s2n is taken under ISC; each has@@ -15,6 +15,8 @@ cbits/aes/LICENSE.fusion cbits/asm/LICENSE.cryptogams cbits/s2n/LICENSE+ cbits/mlkem/LICENSE+ cbits/mldsa/LICENSE copyright: 2006-2022 Vincent Hanquez <vincent@snarc.org> and contributors, 2023-2026 Kazu Yamamoto <kazu@iij.ad.jp>@@ -62,6 +64,68 @@ cbits/blake2/ref/*.h cbits/blake2/sse/*.h cbits/crypton_hash_prefix.c+ cbits/mlkem/*.S+ cbits/mlkem/*.c+ cbits/mlkem/*.h+ cbits/mlkem/*.md+ cbits/mlkem/*.sh+ cbits/mlkem/COMMIT+ cbits/mlkem/LICENSE+ cbits/mlkem/src/*.c+ cbits/mlkem/src/*.h+ cbits/mlkem/src/*.inc+ cbits/mlkem/src/fips202/*.c+ cbits/mlkem/src/fips202/*.h+ cbits/mlkem/src/fips202/native/*.h+ cbits/mlkem/src/fips202/native/aarch64/*.h+ cbits/mlkem/src/fips202/native/aarch64/src/*.S+ cbits/mlkem/src/fips202/native/aarch64/src/*.c+ cbits/mlkem/src/fips202/native/aarch64/src/*.h+ cbits/mlkem/src/fips202/native/x86_64/*.h+ cbits/mlkem/src/fips202/native/x86_64/src/*.S+ cbits/mlkem/src/fips202/native/x86_64/src/*.c+ cbits/mlkem/src/fips202/native/x86_64/src/*.h+ cbits/mlkem/src/native/*.h+ cbits/mlkem/src/native/aarch64/*.h+ cbits/mlkem/src/native/aarch64/*.md+ cbits/mlkem/src/native/aarch64/src/*.S+ cbits/mlkem/src/native/aarch64/src/*.c+ cbits/mlkem/src/native/aarch64/src/*.h+ cbits/mlkem/src/native/x86_64/*.h+ cbits/mlkem/src/native/x86_64/*.md+ cbits/mlkem/src/native/x86_64/src/*.S+ cbits/mlkem/src/native/x86_64/src/*.c+ cbits/mlkem/src/native/x86_64/src/*.h+ cbits/mldsa/*.S+ cbits/mldsa/*.c+ cbits/mldsa/*.h+ cbits/mldsa/*.md+ cbits/mldsa/*.sh+ cbits/mldsa/COMMIT+ cbits/mldsa/LICENSE+ cbits/mldsa/src/*.c+ cbits/mldsa/src/*.h+ cbits/mldsa/src/*.inc+ cbits/mldsa/src/fips202/*.c+ cbits/mldsa/src/fips202/*.h+ cbits/mldsa/src/fips202/native/*.h+ cbits/mldsa/src/fips202/native/aarch64/*.h+ cbits/mldsa/src/fips202/native/aarch64/src/*.S+ cbits/mldsa/src/fips202/native/aarch64/src/*.c+ cbits/mldsa/src/fips202/native/aarch64/src/*.h+ cbits/mldsa/src/fips202/native/x86_64/*.h+ cbits/mldsa/src/fips202/native/x86_64/src/*.S+ cbits/mldsa/src/fips202/native/x86_64/src/*.c+ cbits/mldsa/src/fips202/native/x86_64/src/*.h+ cbits/mldsa/src/native/*.h+ cbits/mldsa/src/native/aarch64/*.h+ cbits/mldsa/src/native/aarch64/src/*.S+ cbits/mldsa/src/native/aarch64/src/*.c+ cbits/mldsa/src/native/aarch64/src/*.h+ cbits/mldsa/src/native/x86_64/*.h+ cbits/mldsa/src/native/x86_64/src/*.S+ cbits/mldsa/src/native/x86_64/src/*.c+ cbits/mldsa/src/native/x86_64/src/*.h cbits/decaf/ed448goldilocks/decaf.c cbits/decaf/ed448goldilocks/decaf_tables.c cbits/decaf/include/*.h@@ -207,6 +271,7 @@ Crypto.Hash Crypto.Hash.Algorithms Crypto.Hash.IO+ Crypto.KEM Crypto.KDF.Argon2 Crypto.KDF.BCrypt Crypto.KDF.BCryptPBKDF@@ -245,6 +310,8 @@ Crypto.PubKey.Ed25519 Crypto.PubKey.Ed448 Crypto.PubKey.EdDSA+ Crypto.PubKey.MLKEM+ Crypto.PubKey.MLDSA Crypto.PubKey.MaskGenFunction Crypto.PubKey.Rabin.Basic Crypto.PubKey.Rabin.Modified@@ -618,6 +685,44 @@ cbits/s2n/x86_att/p521_jscalarmul.S cbits/s2n/x86_att/p521_jscalarmul_alt.S + -- The PQ Code Package's mlkem-native and mldsa-native, vendored under+ -- cbits/mlkem and cbits/mldsa. Apache-2.0 OR ISC OR MIT, the same+ -- three-way form as s2n-bignum above, and by the same authors, so+ -- crypton takes them on the same terms. The portable C is built+ -- everywhere; the hand-written backends are added below where the+ -- architecture has them.+ --+ -- Each wrapper includes its amalgamation once per parameter set, which+ -- is how one build offers all three. See cbits/mlkem/crypton_mlkem.c.+ c-sources:+ cbits/mlkem/crypton_mlkem.c+ cbits/mldsa/crypton_mldsa.c+ include-dirs:+ cbits/mlkem+ cbits/mldsa++ -- Windows is left out of the backends for the reason s2n-bignum is:+ -- the x86-64 assembly is written for the SysV ABI -- upstream guards+ -- its own backend header on that -- and the AArch64 files carry ELF and+ -- Mach-O directives and nothing for COFF. Windows gets the portable C,+ -- which is the same code every other architecture gets.+ if ((arch(x86_64) || arch(aarch64)) && !os(windows))+ cc-options:+ -DCRYPTON_MLKEM_NATIVE_BACKEND -DCRYPTON_MLDSA_NATIVE_BACKEND+ asm-options:+ -DCRYPTON_MLKEM_NATIVE_BACKEND -DCRYPTON_MLDSA_NATIVE_BACKEND+ asm-sources:+ cbits/mlkem/crypton_mlkem_asm.S+ cbits/mldsa/crypton_mldsa_asm.S++ -- What the two packages' own build systems pass for the x86-64+ -- backend; without them the AVX2 sources do not assemble. AArch64+ -- needs nothing beyond the baseline for the arithmetic, and takes+ -- the armv8.4 Keccak from the -march the block below already sets.+ if arch(x86_64)+ cc-options: -mavx2 -mbmi2+ asm-options: -mavx2 -mbmi2+ if ((flag(support_rdrand) && (arch(i386) || arch(x86_64))) && !os(windows)) cpp-options: -DSUPPORT_RDRAND c-sources: cbits/crypton_rdrand.c@@ -858,6 +963,10 @@ PubKey.DSASpec PubKey.ECCSpec PubKey.ECDSASpec+ PubKey.MLKEMSpec+ PubKey.MLKEMVectors+ PubKey.MLDSASpec+ PubKey.MLDSAVectors PubKey.ElGamalSpec PubKey.MGF1Spec PubKey.OAEPSpec
+ tests/PubKey/MLDSASpec.hs view
@@ -0,0 +1,251 @@+{-# LANGUAGE RankNTypes #-}+{-# LANGUAGE ScopedTypeVariables #-}++-- | ML-DSA against NIST's ACVP vectors, and against itself.+--+-- The vectors are the point. A signature this module makes and then verifies+-- says only that its two halves agree; the vectors say the signature is the+-- one FIPS 204 asks for, which is what a peer will check it against.+module PubKey.MLDSASpec (spec) where++import qualified Data.ByteArray as B+import Data.ByteArray.Encoding (Base (Base16), convertFromBase)+import qualified Data.ByteString as BS+import Data.Proxy (Proxy (..))+import Control.Monad (forM_, when)+import Data.List (nub)+import Test.Hspec hiding (context)++import Crypto.Error+import Crypto.PubKey.MLDSA++import Imports ()+import PubKey.MLDSAVectors++hex :: String -> BS.ByteString+hex s = case convertFromBase Base16 (BS.pack (map (fromIntegral . fromEnum) s)) of+ Left e -> error ("bad hex in a test vector: " ++ e)+ Right b -> b++withSet :: String -> (forall p. MLDSA p => Proxy p -> r) -> r+withSet "ML-DSA-44" k = k (Proxy :: Proxy MLDSA44)+withSet "ML-DSA-65" k = k (Proxy :: Proxy MLDSA65)+withSet "ML-DSA-87" k = k (Proxy :: Proxy MLDSA87)+withSet s _ = error ("unknown parameter set in a test vector: " ++ s)++ctxOf :: String -> Context+ctxOf "" = emptyContext+ctxOf s = case context (hex s) of+ CryptoPassed c -> c+ CryptoFailed e -> error (show e)++spec :: Spec+spec = do+ describe "ACVP keyGen" $+ mapM_ keyGenCase keyGenVectors+ describe "ACVP sigGen" $+ mapM_ sigGenCase sigGenVectors+ describe "what a signature is bound to" $+ mapM_ bindingCase sigGenVectors+ describe "ACVP sigGen, the external-mu interface" $+ mapM_ extMuCase extMuVectors+ describe "the message representative" $+ mapM_ muCase sigGenVectors+ describe "the message representative, a piece at a time" $+ mapM_ muStreamCase sigGenVectors+ describe "the seed a key pair came from" $ do+ seedKeeps "ML-DSA-44" (Proxy :: Proxy MLDSA44)+ seedKeeps "ML-DSA-65" (Proxy :: Proxy MLDSA65)+ seedKeeps "ML-DSA-87" (Proxy :: Proxy MLDSA87)+ describe "round trip" $ do+ roundTrip "ML-DSA-44" (Proxy :: Proxy MLDSA44)+ roundTrip "ML-DSA-65" (Proxy :: Proxy MLDSA65)+ roundTrip "ML-DSA-87" (Proxy :: Proxy MLDSA87)++keyGenCase :: KeyGenVector -> Spec+keyGenCase v =+ it (kgSet v ++ " tcId " ++ show (kgId v)) $+ withSet (kgSet v) $ \p ->+ case keyPairFromSeed p (hex (kgSeed v)) of+ CryptoFailed e -> expectationFailure (show e)+ CryptoPassed (vk, sk) -> do+ B.convert vk `shouldBe` hex (kgPk v)+ B.convert sk `shouldBe` hex (kgSk v)++sigGenCase :: SigGenVector -> Spec+sigGenCase v =+ it (label v) $+ withSet (sgSet v) $ \(_ :: Proxy p) ->+ case signingKey (hex (sgSk v)) :: CryptoFailable (SigningKey p) of+ CryptoFailed e -> expectationFailure (show e)+ CryptoPassed sk ->+ let ctx = ctxOf (sgContext v)+ msg = hex (sgMessage v)+ got+ | sgDeterministic v =+ CryptoPassed (signDeterministic sk ctx msg)+ | otherwise = signWith sk ctx msg (hex (sgRnd v))+ in case got of+ CryptoFailed e -> expectationFailure (show e)+ CryptoPassed s ->+ B.convert s `shouldBe` hex (sgSignature v)++-- | The vector's own signature, verified -- and then the three ways it should+-- stop verifying. Signing and verifying with this module alone could agree+-- on a wrong domain prefix; this pins the prefix to the vector's signature+-- and then shows the context is really part of it.+bindingCase :: SigGenVector -> Spec+bindingCase v =+ it (label v) $+ withSet (sgSet v) $ \(_ :: Proxy p) ->+ case ( signingKey (hex (sgSk v)) :: CryptoFailable (SigningKey p)+ , signature (hex (sgSignature v)) :: CryptoFailable (Signature p)+ ) of+ (CryptoPassed sk, CryptoPassed sig) -> do+ let vk = toPublic sk+ ctx = ctxOf (sgContext v)+ msg = hex (sgMessage v)+ verify vk ctx msg sig `shouldBe` True+ verify vk ctx (flipFirst msg) sig `shouldBe` False+ verify vk (otherContext (sgContext v)) msg sig `shouldBe` False+ case signature (flipFirst (hex (sgSignature v))) of+ CryptoPassed bad -> verify vk ctx msg bad `shouldBe` False+ CryptoFailed e -> expectationFailure (show e)+ (CryptoFailed e, _) -> expectationFailure (show e)+ (_, CryptoFailed e) -> expectationFailure (show e)+ where+ flipFirst b+ | BS.null b = BS.singleton 1+ | otherwise = BS.cons (BS.head b `seq` BS.head b + 1) (BS.tail b)+ -- any context other than the one it was signed under+ otherContext "" = ctxOf "00"+ otherContext _ = emptyContext++-- Signing a representative the vector supplies.+extMuCase :: ExtMuVector -> Spec+extMuCase v =+ it (xmSet v ++ " tcId " ++ show (xmId v) ++ det) $+ withSet (xmSet v) $ \(_ :: Proxy p) ->+ case ( signingKey (hex (xmSk v)) :: CryptoFailable (SigningKey p)+ , mu (hex (xmMu v))+ ) of+ (CryptoPassed sk, CryptoPassed m) -> do+ let got+ | xmDeterministic v =+ CryptoPassed (signExternalMuDeterministic sk m)+ | otherwise = signExternalMuWith sk m (hex (xmRnd v))+ case got of+ CryptoFailed e -> expectationFailure (show e)+ CryptoPassed sig -> do+ B.convert sig `shouldBe` hex (xmSignature v)+ verifyExternalMu (toPublic sk) m sig `shouldBe` True+ (CryptoFailed e, _) -> expectationFailure (show e)+ (_, CryptoFailed e) -> expectationFailure (show e)+ where+ det = if xmDeterministic v then ", deterministic" else ", hedged"++-- messageRepresentative, against a vector that never mentions mu.+--+-- The vectors for the external-mu interface supply the representative, so+-- using them would only say that signing it works, not that this computes+-- the right one. Taking a vector from the ordinary interface and computing+-- the representative from its key, context and message does say that: the+-- signature has to come out the same as the one the vector gives for+-- signing that message directly.+muCase :: SigGenVector -> Spec+muCase v =+ it (label v) $+ withSet (sgSet v) $ \(_ :: Proxy p) ->+ case signingKey (hex (sgSk v)) :: CryptoFailable (SigningKey p) of+ CryptoFailed e -> expectationFailure (show e)+ CryptoPassed sk -> do+ let vk = toPublic sk+ ctx = ctxOf (sgContext v)+ msg = hex (sgMessage v)+ m = messageRepresentative vk ctx msg+ B.length m `shouldBe` muSize+ let viaMu = B.convert (signExternalMuDeterministic sk m)+ direct = B.convert (signDeterministic sk ctx msg)+ (viaMu :: BS.ByteString) `shouldBe` direct+ -- and for the deterministic vectors it is the+ -- signature the vector itself gives+ when (sgDeterministic v) $+ viaMu `shouldBe` hex (sgSignature v)++-- The streaming form, against the same vectors.+--+-- Where the message is cut must not matter, so every cut is tried: none,+-- at the front, at the back, at thirds, and one byte at a time. All of+-- them have to give what 'messageRepresentative' gives for the whole+-- message, and -- for the deterministic vectors -- signing that has to+-- give the signature the vector itself holds. Without that last step the+-- test would only say two of this module's paths agree with each other.+muStreamCase :: SigGenVector -> Spec+muStreamCase v =+ it (label v) $+ withSet (sgSet v) $ \(_ :: Proxy p) ->+ case signingKey (hex (sgSk v)) :: CryptoFailable (SigningKey p) of+ CryptoFailed e -> expectationFailure (show e)+ CryptoPassed sk -> do+ let vk = toPublic sk+ ctx = ctxOf (sgContext v)+ msg = hex (sgMessage v)+ whole = B.convert (messageRepresentative vk ctx msg)+ n = BS.length msg+ cuts = nub [0, 1, n `div` 3, n `div` 2, n - 1, n]+ chunked :: [BS.ByteString] -> BS.ByteString+ chunked cs =+ B.convert (muFinalize (muUpdates (muInit vk ctx) cs))+ forM_ (filter (\i -> i >= 0 && i <= n) cuts) $ \i ->+ let (a, b) = BS.splitAt i msg+ in chunked [a, b] `shouldBe` (whole :: BS.ByteString)+ chunked (map BS.singleton (BS.unpack msg))+ `shouldBe` (whole :: BS.ByteString)+ -- an empty piece is not a piece+ chunked [BS.empty, msg, BS.empty]+ `shouldBe` (whole :: BS.ByteString)+ let streamed =+ muFinalize $+ muUpdate (muUpdate (muInit vk ctx) (BS.take 1 msg)) $+ BS.drop 1 msg+ when (sgDeterministic v) $+ (B.convert (signExternalMuDeterministic sk streamed) :: BS.ByteString)+ `shouldBe` hex (sgSignature v)++-- The seed generateKeyPairAndSeed hands back has to be the one the pair+-- was derived from: expanding it again has to give that very pair, not+-- merely some pair. Two generated pairs also have to differ.+seedKeeps :: MLDSA p => String -> Proxy p -> Spec+seedKeeps name p =+ it (name ++ ": the seed comes back and rebuilds the pair") $ do+ (vk, sk, seed) <- generateKeyPairAndSeed p+ B.length seed `shouldBe` seedSize+ case keyPairFromSeed p seed of+ CryptoPassed (vk', sk') -> do+ (B.convert vk' :: BS.ByteString) `shouldBe` B.convert vk+ (B.convert sk' :: BS.ByteString) `shouldBe` B.convert sk+ CryptoFailed e -> expectationFailure (show e)+ (vk2, _, _) <- generateKeyPairAndSeed p+ (B.convert vk2 :: BS.ByteString) `shouldNotBe` B.convert vk++roundTrip :: MLDSA p => String -> Proxy p -> Spec+roundTrip name p =+ it (name ++ ": a signature this module makes, it verifies") $+ case keyPairFromSeed p (BS.replicate 32 5) of+ CryptoFailed e -> expectationFailure (show e)+ CryptoPassed (vk, sk) -> do+ let msg = BS.pack [1 .. 40]+ ctx = ctxOf "aabb"+ sig = signDeterministic sk ctx msg+ verify vk ctx msg sig `shouldBe` True+ toPublic sk `shouldBe` vk+ -- the same key and message twice give the same signature+ signDeterministic sk ctx msg `shouldBe` sig++label :: SigGenVector -> String+label v =+ sgSet v+ ++ " tcId "+ ++ show (sgId v)+ ++ (if sgDeterministic v then ", deterministic" else ", hedged")+ ++ (if null (sgContext v) then ", no context" else ", with a context")
+ tests/PubKey/MLDSAVectors.hs view
@@ -0,0 +1,207 @@+-- | ACVP test vectors for ML-DSA, from NIST's ACVP-Server.+--+-- Generated from gen-val/json-files/ML-DSA-{keyGen,sigGen}-FIPS204. The+-- signature vectors are the pure, external-interface groups, which is what+-- this module implements.+--+-- Every combination -- deterministic or hedged, with a context string or+-- without -- is taken at ML-DSA-44, which is the smallest, and the ones with+-- a context at the other two. That exercises each path through the Haskell+-- at 44 and would still catch a size constant that is wrong at 65 or 87,+-- without carrying four times the vectors for it.+--+-- They are here rather than downloaded when the suite runs, so that it needs+-- no network and a release tarball carries what it was tested against.+module PubKey.MLDSAVectors (+ KeyGenVector (..),+ SigGenVector (..),+ ExtMuVector (..),+ keyGenVectors,+ sigGenVectors,+ extMuVectors,+) where++-- | A seed in, the key pair out.+data KeyGenVector = KeyGenVector+ { kgSet :: String+ , kgId :: Int+ , kgSeed :: String+ , kgPk :: String+ , kgSk :: String+ }++-- | A signing key, a message and a context in, the signature out. When+-- 'sgDeterministic' the randomness is zeroes and 'sgRnd' is empty; otherwise+-- 'sgRnd' is the randomness the vector was produced with.+data SigGenVector = SigGenVector+ { sgSet :: String+ , sgId :: Int+ , sgDeterministic :: Bool+ , sgSk :: String+ , sgMessage :: String+ , sgContext :: String+ , sgRnd :: String+ , sgSignature :: String+ }++keyGenVectors :: [KeyGenVector]+keyGenVectors =+ [ KeyGenVector+ "ML-DSA-44"+ 1+ "7194B13C95231010AFD2C909992BD2003BA6F437C3886BDBE3F6B867A14BA161"+ "0B89806F0EEC39F2891116152ED4319D4260DFB8AC0710765BD497E6E1DE17783CF81E435A412EABEF5DB3AF5D15867BBB4C60F8CF98BA31BAD6D41A5F8EB0C11B632C3F19D844A223C353BD182883DCF13B5C97823D0C0E6902DB25AD8D344A37F59F4AFACA5BC8874792DA1E6A3EAE742AB7034B20A4AB75A93BCA4B68002DD242CED348920B7E5ABF645A0E2E79617BCB3EE7BA972B3E718D3EFFC59B1869814BA3F526927477B12BF25CBAD8B04B09905FDAD3820715A8B9A905DE1CD65EFF6B0B0886305EFB6CFEEC9E90B5EF9A5AAEC45C753298E8DF9B017CE0FEC9B7431B20775CE8CB11F1F42D1D9FE936D0803196E71ADDC26CC430CC3B69760C7CCAFAB7651E21BAA28F92BBFF1C4A6EEF156D6F08F80B5E3B6FC943E6E984378B90888D09A6EA38B0BA86A3446211452E076DC9F65620014205D5271C7A44FEC3CC5375EB246AFFC11B26CAFB8B96CEE3A68E31642E3D69B9130795F25ED818EBB211CD8BE648ADB5C8A120C8186017727FBCAB31C7425C08FE9195DE6BDBADA5778D727EE5CDE0674FACB7AB81786357B529C71DDB24DF770E8E95E5F3112BD297B352CB91B08ED1097A98E87BD7CE4235B8DD42292CD4C59D87C1F0FF00734AA22D7CAE4361ADC47742C897601048526702538828BA3C3A959990C0E99463FD22417E147FF2DAA74C0C8D3A06E9703A2E160590086DB8011A3D9CEC5AE6348706F87CB2379632CE56E660A0BA1B30E3846C5B5C6C0339DD993E543A5322AF5A11FC7040A2DF23A0B43E882D7A0FF4431A723BBB918AFF7F14BC045CBE94BCAB27AE3109147B588665EF486006562B1297016EFDE787B46237060EE431E0F011166F916AA0789A7647103B7400A1CBCF0E22BD7B6DD2BB3EC51EC98F0EC6A5BAA4CDC83F993D302F8FA849F2046B78AA32F0B3751885ECB941799E250E6546DCA5C20C24845190F239EDC20DDA77353D555DE61509CA6D3C6DC3195BBC6F1703CB03EAD5E7FCBCF5D196E9AB71522408E11D6337C74F9A31EB22AD084A19132BF72E7076A9743ED070ABA78789791824E050CD27694C2648263D1200811FA1B81A00B8FC09CB7A338795E54F6598D7753395F05C60E6EBA9630912B7AA8CAAB3017565DEF72C7929F4E7736C2B8043FEB448801E2DED704E834294B69F6A109C0968214FDC5C3FF0D1B1555D617E16DF61829231962C59B22A10FE400F8B8CB2A3F19FB4B2E8D087F22687506E7F0D061857D1C1789C7F55B899FF4B322982D64BD0AA751D5BEE320B135C7F5DDCD5E6245B57DD22F44042F2BA6DE942365A59FD0C6B0F20C07B71277C6EE7DD9D225032605AED1D3CF8242EB85C33A0AFC3AB42764088D8F4A80FAF804CD84360B2055181E58A0B5AD4C367ABC667982045AD0FD7E048AF8C326D5DB60233302B107E515B15B0F90E5F348C54192B559B4C0A86CDF0719387EA3FF6B1D60B324A98963C56927E2B8DD5A39AC792AEB85EBDBD8DC34B395C2B4DEF4D853AC21A7660348EA8C96C943DE0BAFF3AA6849179E5EF2BAA1731C81C605BEC3860FC4A6A08CC9F75BDE9533511780FF1E0B01D34C0DC3EB80A7E2F52A7A4B815DDA98EA775DFE0C5B3D419B05934DDA05A9616C0978CC99CC8D7B68227BD846419D765956C3D7AA811CE60AF22DF322FEF0DCE38C4278E0237F1D29EF139E201C8ECB4D36E79910D06C5CA4CAA8C2886B96DE6EDD40D2499E30EB942F22BEBF6ED5C8E37DF9557E74D67DC467BAAFA68F1CE37C8BD9B3A4F9DE71670128125AA16ACA7232239575E1C6819C820AD16832F23647DD53C5740A8552F86901AA4F883EFD5A3EFD7C3BF458C5122712D44BE43306C9B8264"+ "0B89806F0EEC39F2891116152ED4319D4260DFB8AC0710765BD497E6E1DE17786CDEC899F1C6534284585DDA4DF03E45E4D39B4526015A7B3D65F8BF875452560DB3223594A1FCE8DB48C8F1793611A17FCC0006FFEA26CF7094D8325037288F8AF7D062833B71B8C0F06108442786E3DE59D649162273EF179AEADBBD48CC981CA86463082A0020220CC10003B160C0B82C20A70409C521649820E3A84409C689832621E20832C1B04C1C8911429420193789C12462E4C661643600D8484A0998601902621B104CC2A4495A1600444690A3800CD3889098A49024C3901A344D594052614411248101C416261308881C818083A86142200D23262A1B2501D0422464C4911315094092641B071208956C513830A20831CBB84153240E99C41111A1241A297003A74011B744A2442DDAC0041C21611222684B126821C3104082600A2352C43890CC967063206553486D84124C0239328A0640112689E49841114871D22652E3C2240A98889802891B8031E4089140B404534286411826D102929C946180C84801134118348509B76984180D0A404AE2340E0181809C04662481299AA424A2B8445C38491AC20DD9B045A286491C864011036C13B78044B2618498700A1989A00024E1C2488CB09063B8100A408C18171009B76C1108249136404BC29104A96901B525044802C300319C380E0A26229CC4841A04306184201A1300A1940460982912924D12144AE1068D80B62D1AA5855C240849A20401B56880B449D9B21163240A98424414B40401264282008C0B04651B21409A488494866D18B380C4A801020326CC44491922208A364264048D1205814AB20401426A5886201034110039295A98654138080CC964D3849021A788E0088901386012222859488290C20C538430823626818051A4C871819431E0386ED3A22121186E8B027151240ED13689903408A3480A0B34894C040608314C81280400894503920D010208E0C011D1041288482E01280E08C161C11820C4C84800242A24C545D80020C8368498C885D4B40CC3062942B00D98487002A7899B303152284E10860C23B124901829C2C84462B269431824098621D2062989020A1CA76023C00448426CA4902CE09681C2C66D23286C40206A4404210C3405C980814B0892A0162151B001C1148D413849DC968C212244C4A00D18049240083209126A00177109118114962402C22D109591A1084D0B316C01466ACA8291CA1252D3B23018094D44C4845490411FB45680A1B2685FF6F4075D76E422EE9B1DD39048A2C6B0C1C441689316AEE550C179A8E55B87627071ED299FCBFACFD13CFBE61A9F87BE0579A19357C4E7B0F125D354DCEC0CA0EAEB5E116FBD87EFB3049ADE6F28921D56ED84487CAD51BD84F7C07FF460B09B4201E7B9DF1801DA2771C2DA20D2B687A44D4D1F49DA38BCF9AE30713ACD86E32F52FA8735DCDEDF1CDA4A5C2C8880784A5B7C3AC395FF8C4BD1A0303868CD3E6D1F7EF258F38F5CF560AB9E8357272222123821D25C141258A6B132F9F99D014584BFE23AD4957F6692CD7E8327CF66E581A598C4FF3C7CBF5316B3AB028FF82E8BC3250F250E6A451996BFAA00A1458A86F8304F2A839DD1B929EAE4E53E919AEA13BA09569B14148ECE44CB27650EAE5FB352061C7301D8DD9CB5156BC15DE1F578AD95D6505CCFD485D99867D48BC6910A491856D30B17FD8952B774A70F57A24458CD9B18D0A222BC5B307A34EE347106A9B76609505E7C81495F88D8DA05D742188F01820EBB8AC559FF3417A33E6CD4FDA1D60C1A6D37C3D27F51717645ADEE9020F10748F7CDB0D5B142F465C54FE5D70CB787EB47B8741A162BF373AFA1C8DB3985900C6A8B9035006CEDB7EB9854D3C50E1C30C8A6B34269D85D7C683EDB1BE1455ECE1C9768EA9C9A140036E8EAD9A19D9F167E52CA10DA5FA7FE2BED7C0ADB0A45C642AC02ECB5B1C5199AD5D6227CB4F506A973D696908C15791513783736FFFE5A47395CE2E7CE1C42E7F6541825A2BDE5617F53B5155AC30E3EC43BB4EF5AD727ACE5A5ADCC7F036A1BB606F7C943C112B372DB92832639DB2F488A3ED64E43609DD93B43F38DB939F6BAE61E3E44772929E65F43D739A061AE3272021A220387A43BE3A985AD713999F75E040DF53DA81801BF165052A68179E6BB1FD4F624C31EDAB74F6E2E7EFC31EDA78103BFCB32B837BA07C5D37E922440DE741BE0029BE98DC86739324D73E62B3FD09B9EE00EF8DBCFD0ED687C5269BF3A4F84AFE2B8FB52BBD118E718DF5972038CDAB018CF8AC7D6785C958AC5B9B23785DE1A90B9BE64279015FCE8C36C87453E392DA20CC72C533D115BC0F53A385A0DF6127B3A81592552B7CF0E8AF3797869683FF0C42D2A189C04442966B37CA321A4DBF02067447D50D09F92E4E63A272A97E0460FF1BFBF82F61412BBBDEAE83D0843F5B38E10A53DC5CA86A4C6AEB17601FB560A8852BC60C25767C2489F95143FCF75B648E814374BA98DFA753D08A53F02377F551C9AA374301CE7C84B1A48E6F750794E5386361815155053175DB12E29EF9D49920D7704CC343DA6021249479E5E5405F3BEDE4DBD612009C34E659C3C9D7C59DA97D104F47D4B314D1C0E2414F8746CA658A9731C18A0C89F61E11F617B197093ADDEFA42CA0DF723F93A12ED83C352B05F8F5039B2C8D321C4992D0DF249BC5148E0C27519430A9E70A5CC24B8E0217D6F9BD04737A4FCF7351C7670269DAD9DA97858A1FA9DA23DFAB215172BFF72962F62406D2C5747CA0F273EF8F31761F99CAF2F417685F971C3415FD1C79D7A4E75EC50A6B7B795A35BC45EDAA824EE71E651830C96AC2905C1EAF817B8E833C9BE77242A6B43FFFBC108DAC12631EBC86BC06BD7E506F827B142F03476357C9AE5B98B447DBE0F408C75535FD9341FA3693F2887FACE3B72CECFD62CBF908319A22336AC43E57E2A46C42B28B6E55D7075B5CFED7D8C0A8A1A4ADAAE79B4A09B7CCFC04FD99FEBE3A0DE1F8A9B43F97A2F93CD758AD546C0A1C3801E3BCFE1ADC246E7A34044CF56BFBDF5A5035C2B8D1E19A3A7243C225BE23EBA7FFF8E052C3845310F4B2B7393BDC15B766BA0B3CB45BD0A1CC693C947A5A964DD39287828EE86EB6C2DB9D8967EDF4B75C78EF4B34561EA1A9D93BB8E1209381E9D1F2C0E61EDBAB542E07E2C3C71F4841A3F2117F26B608CA244C46663C3FBD6A6A20F55C8C778C585BAC5DBEFDE74A7EFA8658D95B12EE9412BC8CC24333BB2A3E994E887A8140FE482EDFECA89E887531A536BC13FC44AF7B595B06E6B122E59A324992C553D6278AF277E5C545B126105D1A180D2CF769ABBCD9B8DB72330E6548521FF4569C674E60D35923B86F0166CD24D8AC7FA4F49743E7E2C90BCF3E66955C6F5CA430024902C536D0E0F5D3A637C033A3BA6F9778475C455E440A5E03B485F7C8263F5D007A8A1B3DEF7AE943DD38633715B50A2E76228521EB1D0CAAEAB48951C1E395DF94F9A63313DBCCF1A6BE8C0954388AEBB0AE472E741A8006DC0299F5E4073E89A3D097512321BA8037C391CBEE65977354B3739CB04FAE7663D86E9CE04BEF14D3615B9DF81AEA3E4"+ , KeyGenVector+ "ML-DSA-65"+ 26+ "A991FD42B071D49C48AE3E75C647459E0DAAD1E1BA356A04801912D3294BCFF8"+ "36DB0B5DCE98BD190CB139E80B71B49C7D7040B71C5A1F3412C46BDE939192B1B57CCB88AC2714C1240CB0EB62C689E031AEA3D9F3EB3ED7BFA45931D288DCAE3413199B31A7032560DCE8A61E195D13A1440615C2F3AA7DD28C5B1B742BFA400052186721F13D3DF9DCFAEE348B10D66913C7148913E085E1A4A03C659398DADC6A8E0E0C1A7F9F44D30436DB90FD65A6AB8F36137338255653BAAE8DA21526A333426DBD9F76CCE0F43212643E854D772018B35CE726BCAAA5AB0651BAF8C122E13929BB35B6E4963DF2595FDC7237CDAA7234BF776B07F353CCDBA12AD3E025138E3492D7F8E929DB55DC23E23075F66D57A10492E6A10AE7B758ACC2291CA18BA1CA07A5B574AB6D8AAC18B9524990AD2F110225B7D82F696300A660A166AD35B3C57ECBAB77117C79656FA8AE2A19A7DEFB2AFD2AF54683D043BE0F933B8EAE0D591448ED55D00068CD9FE10B067FCFAAC53AEDB1E9B667E36C4E30231F85C7AA0A474AF2FA4776226F4479555E155528D78B98183CBDF7FAE4E7301140F163EB71E991D15FAD4A0D2F25A5A62FA2E9BCC823CC2927662E40C538213DED9E2DF508E911E4924E507A50861FBB050EDBBF56D937206F8FBC6F4CEAD4CD10D06B73AADCD4AA39703A7A2BFFAE68B7BAA47341B699DA9F3B167D4D90EFEE0A07EE3529A3B5E8648B9CB07EE973E1D8DCCF1D16E95092C4A0184CCB4902D6086D9F444ACA5FA45F43CA91B351E82585989FFCBD6D2C3471D6B8593AA46F29D0DD9B44E8AA4D8F9A0BC886BB7982C56AAB11E23BFDBD8BC674732FADACACAE25FA416B2D0CC7743827293336507DA4B14C1F0AA2E929AF975466DADC89A016F33A0CCA2D5C08114CF04B02358805A772536432C44DBE9886130D2D3A0FFA0E175875A2207686F5E562B879EB2957573AB706B942468C20CC69BC566D29D9F151F3CFAE71CC97CE4A30722D4679FC1C089B5009935931EE60AAC5496B0FD5F24E514C0E20FA1DCC7729184A50FC85FAD1D2F32F715FBF55666E49F5A19761F2DD1AA5D1A33C7916EB6A794981C0334176ABC493EF30D9EAEAAD42E705989DCFCDEB578529A700BD14076A348A2062D6483CC63CD7F55136587AAC0E531A06EB2DE74E61CFCBBCE18F2ADA5A741F683BF101F71432EE659DD1508E0C8FB2400E0CCBE435DA3466D543D3EF5BA369E125C0B84D855EFE6D4A22FF929A7A7A984E448D23871E09B88A0BC3F3B7DA55DD2EDFB5A6DAA102819FF50CD4DCCD0A95D2F27354668065D4A56C31FB18B92B2A8DB2C6453BAA9333AEA6EABB1BD6411D584DD5900262057A707F81CC5137DBDD9AC1079BB98DA78A8E4BD1B2E0546C2B3D956FCC280D37D855E31F1E4315B387A742280F057F3219EAE512884AE7EC4D2E3A72265B1D0163FBCBF616B2E289B0EAF9C63437D50B7B50CE408F5B4562F2ABF510C19F5E8A0ACE264DB6E0F2A69A7D0B4A5E62A2B964F08C8FFE9C295F5773BDD7FAB054A13822D428FAE28AE5E4ADC9D9F6E4DFFC457A3E49F0BCF62B32961C4667B60960452AFD917FDD00D954FA30C8533E5629F90AF85948DC1BAD889F91832DDF9B738254C9E7939726C37AB4557C2CE363C1391816C467537B471E5985E8084C277B62BE514922D352E20689EABB3EB91C343F36E77B152D5E85AFD088F4D02E7024B248A7420F58C7EBBEB480CAE39B56164F5ACD37A4F56B3DB6E1CC6B7C8C96CD3C44A69D9AC99175257BAB7FD83C5B574B5C9702C0FD13A5B176C60F82D2DFFF50C2AF25D96E0F8D27EC818D499E479B9642AED4A4A0E6AF5F14CC5E1299EABAE055EF3C763D1E350E2D76E92CEE47A4233368466A298AFB4CA108A325D2A4F8B79F21EE7349C1C186ECD7897F9886CF27EC01B05388870484867F84BAE2C016D04A3762241907C4DE207798DD125A2CEBB6C2982F779E04117BDD65CD7FF0361A59D3EC05F6D903B6D15554BEEFB6D40D96A0D4B37AE76C69C1B9592088B7DB878F95ABEDCB5FD5423ED93DF1B27D01A4DC9F4438E7C55F35B0AEB7395B08E1ECBF15CB2B61D043C0454AEAEF2D487093FA0D7DE3FC6CAF084B6A0F15A5CB05D9340D4E6763983DC45B7828539C77A60D5E081A03FE29949D916392B6D989B4C8C047E3635A76BA88AC18A7A18CCFF5C7E06B02A43D2DFF169FA449739E382BD020E0963C14A9ACAE6B6561C722D2BDA183F33EE6A904DF0207F5B098E56335CC063F9640C4997F593218D502F6B382354C73979E93C4B1B2471965FFA5A0DC8EE8ECA9F5697E7EF08DC0EEFD9AD75CD4122194B450201DFD73CDA46A7B2478DF66129FBB9C75B774213F9615BD990F8D07501FED440FD25D6CB912B8ECFA678D887A4EE28677E6E0491D49EFC7A3B34B9815C5C22983DA280D0AFDE2324F5281BC8B796DDACFD82723BBD9AA34B0C96075B36848591E47B80086897846FA76D092BC8BC6200837BFB5545039F8602B7EA49F63C0C3B8317EEEB7612F8E818DDE09E43C7DA76FD2FF6847906A45DA3D993E8EAED9FB3B1E579D8CC6900C89522AAEB0B4A80DA1E66AB8DFD62DFE4E4D77A3A77E5BD669207C70AA8537DD6D80A647B0420D79531A7456052C3C989F0F08DE3D343C40067680B39ECE95A17AAC8A622D1D5D95B38CA0F11D94E5B0A7634EEF4055517ACE79F0DF1D7C172E0246ABB2AB6B135EE1A38A3B84F86FD7C3CAF178CD4446D0B554256AD45C657E1192070ABA7DF480F489EBDF9753A79CCBC6AA893913C5F1271F1C6035"+ "36DB0B5DCE98BD190CB139E80B71B49C7D7040B71C5A1F3412C46BDE939192B133824E8FA472BEAD7A4D3BF6DDC323BB37E0D8D50854522DBBFF2890D5FD9736EC39B00C9031E14E972D23B06640342D1A78FA5DF0EF12AA3236A50AC6E7BD8FB80C402EC98F22A4E2BB6E9E26E516E7E01BAAD986920061E5DAD927E4E0061A3866606015637231056063365487762585526887675878487477638236280364817056370582625626040150855601281021415375658360081814003238344132863410087754120308825448505583635861483488522781450115530724037288452180645532851376425811578032786032168836177052271783100201556873405285855840687261141784114410655808751130742173183267801325883830046480283005852511571077126111413273338305581727825323153811440104652051222210226383420042340322308056166352828043247278611463585302027847233562568428033166345443003065880210047365367315160748730010865734738305232222523752820711625615767042435315314503041375164660366121405180624802071164334230625718652833518622521854441353358223121352540300388484252536762043855528882400160748523278371774111874135436141440273310385340855586032187104381475057571182780222731565348534387571046724673888254604354385508308043028205873204570401180432163575734240531768606715214426833325634520763763723518740465145177515257128367252225710368810875341411301258348454232761048362867773260383180381738162335055515482041865381233334185878354587451337064781024554443887677673632652073866018425757626437708228375720344830144464713214160163503461453087771277385166532247065330167588150054362558464618363012667170148045255155505852527457605055186031538645654218602787417881724640247712454302818075713048757672721263432274814710043321053707037323076550400607427725460304754370164874174660112813541057331681152601054103866853711065347808675867280645076661507844584617431350801506735705655588557763038770722137306354434225726148356748145448405337674501367155743323121430777237601375131842781822541637737684866136033562874806077454652343344523376688865207837368261460663174085715881486277771105030663307516560865370605141151427575626557305834181601010676611140375585358407327605130777628845428075077042880047212574387827502731387663076364011400711625371477843100456442114008255062481181022270487235380661013734700364144346574117201004664624034383102461407855152244654114385024337324032207628818203135837764028525504622646060546321503604141638311777672736251736116485482155586726005155741501257214560667436087333810604210000768248328716753662713101822523612840613402468265615133652047073250607757568023357276561280717402160118308041601228257128106430611343883016085704472043883082027544828877406838744175438740321103025587868268600854465810776568573488081068834626204671561152001437524050818730116108867426350618732585768271733354605757872840002565184766480704586166333368630416538633768348004707041106457232184732732723056625067668604641302261520414588670616402767446637810213634724784500051171174854851110503637815161314655084024111051400267461562547284356765843660727856543172543618510316827312736388341513864280454700853333375821431615420043800422886563456470388125547022771880188550286833688023527412111250433820145308E4EE45B2C9CA523459C49BF1604ACBACCA7FEC3ED15FBD497EEA72B009E7EAD6EEFF7DAD65C246DC26DCBB2973BD2D305993E843581C4B2A4FF45ACC882CBAE751E63E925F84EA3FDF722FBBD9AE92F18CC4B26F188BB6B4254497704531041139AC623381B1117F51DEB204938C82B278E01AE4969EB587E51E5CCACD12F60E4AA2B5367A2CB34D2D442C4E124A3BA2CBA083FAE8CE72F39C7513D80574A06C02383451E653DA50C2FE71DCBE1609CAB5F6F0D1B7AAF21844466FA1AF79C671216792FCC2018E633EF173FB4CF7E767E1705DB9939EA2C185DC3AEC7BED66EE13CD8814D790287CCBF859EB646E58EBEBA9D081F811B653A25C8FAD64DD6FCBD6290E8F55320C51474E18882BCA350A33768B9DB737FFD1B7CA5BAF2579526438CB086A6C9ABC3FD387907A6D35CCA79F6F1C9801709EA1D02DCE59676E8C002E464A660F9E410E81E75A62F59A80E95C058EE950BF0804C5E8A0ADA205D71EC2E6E4F929AB17B47708BA00E4BF64452BEE8483FBB68627D57475D86922547B7CCEB7F7A7EFF98F46FD6451346E1314F8F694BA16C27234EEC848FB7EDB442D845E55825AB01B829DE26F76C275D5609BE4C36212F9837CCEB4628CA7899F214C9B06E7DB5EB9BC24522475DE460994A89C583A9D3C885C9B74C7D6271B8702778E84DF19F13015655C9D06804BB1237A17A93E19BD46232FAFF39D6E1A6ACE3495022C265638FC97B59B86A38EAA8F8A03BBCAFA35EED6576D109396203E19DCF1D151836CC7BF46C818B9959A53F0502464BA199F400E412B30902264F1E4909F52F2031D9EA8396C2F735A2DA81A389A792031418F2BF73A767F42DB0EE14FE1CBEA5A99B5B95C6137FA4621DEF9B4F6069CA4709242D15C797CDBB71EE507B22FBE95825DA5AA4D9BEAADC3C898AB122629EA717C6934E5A4170DE7EEB5D608684DA36A318F18E559F1DC9D18104271913EFB0EE8CD2379FF1B31C575C71D90A54E61185E1399E1AD710B4E960A7B41C88CE2F4889B0B2C7B3C208A1F14D750DB138FCE86FADD7072296155022AB24665C95A7719E92A1CF9FAEF424F90E945C905D7908013A0E0A5B9C7819BEC35B06DAD1A146E75D345DE5D18565A1D30C862C122BA4E87CF21CC63B87347B3ED90015EDC1B5D3125956BFAD93335495A74107B8AF62B50B4C56B48E2A3293F56DC7CBC09F2F6FBDE9A96CAAA2787892E3D04238D9F988125D8BEE2EB6299EC5C678AFE42D3441BF6E4B22EB6975E6DEC06AA507AA7C787D70EB97B60DB056D658F7404FA409939B8F4A321B7722F0FCA1404E79FFCD87D2A6EE13D508B6A193AF891E587E4F0E502D68784C55E9662B0905D912BDA6C678D14E5424781E0E86FB2BE43D7618DA355CDE8617F7254870F84E4641E7E16FDF3B6DB86B60BD873E8D432B95F53C4119DEC7A711CC8F2BF901E38B692A54E7ED9D8A2DBC4CF5F271F4BB1A4678E6A19F370486984045A1B4F8521C9A6F811D49BF7C182E8175F75CC1421486ABD75130329EC1C87594055A2F0773EBB3CE3DFE073DE8955F15EA3C10D75B6F67AC2D639C3575E7C0BCEC031F902E90235BA405DA6B2570EC9A0635AF9A3EB05CAFF00AB4767DF1C5B0C01DF45569D2F11C58061C848FA8D94CA4E9671FF94A440FBA63956EBD924E2EAA2A2CA5E21FA24ECCE4A1B82CEC6F55E5E2E6B1C2416E4CD1C2FA0DEC1B07C6E415A75075662CB6FFAE761AB7FE771EA54BC5F4A53ACC7B64A545A25627950057F336B9751FCCE479820BCC2CB25A92781AF13CE3572A0CF0D946798D44BFE0BFB5BAB4786E6E024971A1FF320B8C931329F8FD12E17E794058A809D849B333CD98D66FBBC18C216D0F01ACDA3E50F712691099336897D449AC15E77BA6FD29C6042D50ECDDEA4E9D578E2054B56B2645C97990B87251AD09B76F029DAB7498404C9294ACF6535E65B74D0BF345DFAC95FE020308A931BCE8635C78A2FC0F67910C63633E5FE9004846ED9EFC411235028EABDBD4FF10788FBB081AB51187F30020091486241900DA3BADC323AFF56DB7BE0DA078C69E535422E04FFD8EA8661C559BC311BD8D26EA8486832D839536F59B79E1D0443B938DD621758CB2EBBF41F87290590042913C761650481AFF1ED1F5A3E10B4C9D2B198DF8F2EF9FA7EB53FBAB3E04450678770FCEC45ED3B78591E9F643A53702C404A1EA13638455E43427A807607F7091E4FD7E3FD29901026CA4C724C1CBD2CC4C9F50AB48D02C451ECDB5E8E9020D214F3DAD92A6321F0C2B25C4D20FE09E10A4BC26B578C09E288A0796F97C04C45370AD42E1B920671355A6DFF3A10C3C9EA96F569A90E9380F5EE4A91EED1CBB644D890F0D755425AB16A92687D6E194B17374C4CCBBF924043BCF84CC6956835DE8A2A42E91E139169EC1B54AD1B7C0FB1488C8E98AE3EB2D5ACA80E1790C49EFCB98E50FE03EF3ED2040D10DAB1B3B69DA0A903EFB56686199AC2863CD04032A78290AAC30E600FB8E1DAFBC8F54F0E485BB47212A46EB81505890C67905D6900572F1693CF75E2143872B420C0348AD08CE8BA2F3FDAC84926B9E8FA40231E6EE7BE6CE6F115A62A57825ECDD19670FE5C236477245701D30208099C4430F053C46911EB51E94819F59404C90F7282E149E60321921D0D197F8A73E1D7BAD4F63659ADD3BBFFB5C2B71A4C3A9A0CE5ADD6DC9181C31A040B1D143891663D0361255BDCEB5A6EF9A414ABCE158030F8664B1C42242ADBAB2112EAF420699D1EC1FA2BFC9ECCA599F4A1795F7CF8593057C835127700A6703C6786BB75F1002DADB09CD7097E5FF978000115FA6278B912258E945E37461D00D9D5827C231A76B47FA58F68A363705E3B8109D364C89E510D84367D2092C501399250F482719F9C5F9D68C7896650AA91DDD17D0FA75160A2586C5F6846E411D7CE9D040114360FD948705FB60FF93C6C1320240DFDD132360C5B8F99384DB7E9147FD93D0D745A763A48A76EA7E9E06AA0BAC02CCF341FA505354606750D3D4F60B28FFF171F5B656C231640D82498BDFB874773274CA568CE2B2B824CE5793B86EF02AA7CEE6A027F27C89BE27C519C0BB356359A98A1C011551CC106A78E6D675CE68FCA86AF5F6F28B9DBF9B464A26836A328F6AAD6B01243E7BE9EE2BB6DB1FCC2E8B52CD64AC5D62000976A78BE4948F36513C48C8BCC0205DCBFA3F7F1018986A2734205690E26485E24188F6EA5BE3EB5E13BCE8BB2E748943D7BB0E30125011E074ADF4ABA8F1C24962372BCDA4622794D3C04FD99F5428EB2F45C829048E117CF5E3E53C881A8B3EB740B5BE99C6AA7CAFD7B5190A19FA4117EE8B0C9D1E501DB54C1C65C30E62E7787987722AC90708ADFE64784FE5D89C07FC59E01EBC02EF6E3FB86F44403A25AC4A276AB8A639867E9C13A3E7CB127743E5DE2695AC25AD2BCD8036BC601D663EEA17FBC26958671A116EC5055C53EA1AD7CD5F0685B20BA2BFDD699B83B9866190FC9364D34E3EC342AE3881B934047D577"+ , KeyGenVector+ "ML-DSA-87"+ 51+ "A16F5B0796703E2D1A0140A35CBF36EFABE70E752BA59B6A9A0E9C4B05302F73"+ "A5787E8044248F3F85AAC54E9469FC98F1B1138CC127B120F9946C80B96E3D89CCFE38C995645D4B6A559EACB2AFB81621D765C6E42E73031D44CBE74D322C7B16249576EB4C500253538D1A2C6B408E681B93B9014E3147DFBECF9D9858F7E8635F8598BA6847127D216A888FFB1636CC761616A0389C39A4245695DDE0C86CCF8A3BC5A50EA6FDBCC0A34457D4DEFE35F775C5993685AEF2237C31912A619FF804AFE8AC3418C13502820AEE5249D6E577EE0B2A5E3E8DA2DD30F514A076B50556D7581BA1F9B2E4671756A63065C20EBDA6EE2C33C9D97AB14C5F2204FC5359ADB2BDAA3AAB7AE1DACFF18A67D801BCB8F054BBC444F0FD0001A7908EBDCDF2B84F3C026EEC1282498D31AC33AE6A309ACAB17B70DC9F0EFBE52648D1AD2A4CD5964DC619B66CDC9D35EDA7A3DC21C729B9929024DB8B852DFDF102A086845702FD249EE43B54D2D033ABA95110C4F0A66EEAEE78F3C41D5D792A37D1D2299252A5498A44CA354F6F37FDC2F3B72B1A8378A5BA8B4997556E5A6F4125AD946BD4FC402C4320B27111BC204B8B5448F43F7E77A8166A48137D85584BEFE2D9C85CCCD9BBF3F8A4E05930180AFDD697DA82ED9F1069150FFC76578C941CDBCC5BD2ECB7D6ABF9E68327DF51C26C8B42CB1DD8BDA98A82C4C6BA6A991E651BDFF68F0A62D5DBFEA020D4303E0C53D474CB11D553C5BEE1156917D72AC2A6CCC4EC1C775D41EE660E2485A45AF5A7CA083DDDCC3EE4FFFC5E77BA97CC4473303C77D8B6FDB35E2A20627BEBBE327DD2D1AD1F880CA8EAE1E0067A9093929E5AB18D406A518EDD8C1E0F0AB07736F55FCECF4A3B2ADCE1CE3E080B58DDF85A2262D0803A7E5B4E485E642BA533E2EC7A43F9E8DB20F75292ADAC704395469408C15641A9C28B83E8C1C799FAE0652F51369978F4C089FE15DD78C5F560CD28F5F75FE0A39A60A61AEF6C7D802141E9809E7ACA68A38BE9BDF5312258704F8B11AF4262220CB55641FE95DF83EC9F786D6E69202C91EC4CAA4E38A21C3CC609C28B8F65552FE8334850858BBB30B837D874867EAD330A1F5B4D7F6BEC4748BE54A782D5B21A192BC00A7F2240FBD900460785D1971A94C587493D21CDA249676EDAC6C865147E269488B9CA78FAEFDC778FF60E79735DD5539879182424B054FAE8A9E153BFFD6957C533AEDC105E43AD7626312C8D229327D3E72AA9644BF3E2C9A08ABF807CF3A472C7DAF0AC4290A9E8F88D07AE8FEDB8A4B218C3EBBDBE53882781F2DE034B17FEFE69302B4975CF43E03645DC53BD355AF988B3A3D2431BF9D9A865750051EED7BAEFD1BAD4939C7EC293509A5851A145F79DCBEEFD195571AC2172AC6036812B4DC8040D186B8984CF9DC24F8F765166C6B2E8389DD24ED63CEE951442861669BDA0622BD90B041DC477C01A0D95D547E07A892CE1F26275ABC6F97702E03B776E2CA71E3D0EBB88D1ADF591122E6F0EC95A6CFBB976AE64BD0C7F074CB6E78E644DEE8200E2D626435907B9134000DBC3B50271F5A8F254D2CF02DA039D458E80FA13A33567AA9B374B47B799D0FF1A14CA92A50EFA192EA1641FE29A380E11386528B7248D6DE6AA90E1C9744713B768264B72A13CCAA94F3E19ADA5129E0D8155EC1B267E2CDC7F0E7A08F203D9A18F8C8D18529709EF746A21D5FEC36EB547A2FD4490092AD0F06C55EE0A1CB104E2C3C3CF2BCBCBFFEAEC744A13E37310962CE40E072E7ECAE223943A03002077D3D96C7B4C3FB1E2C9EE9C5222A63252F26372BC2BD94507CE5728CEB7A0AA33527E0662B4561D1BD255806450772EABB7ED40F787A2E3C664FA6BC3BE9BCD84FC7B42F16F94CC56CFB67297E475177E19E51010CFB74BA996BA1C3D793EA010C99901E2223B375E364BA193772B329D5D05AED959EBB924B698F0657AF43EEE8E0D54D15C93511209AAA9E10251BF81AD8B467D8FAEE2A440782C372F55A62CFEF408803E4B3BD8EFB94061EAB525BD31957779B89EB1C75AB97F278B3EAF99A05686E04873D7A685868D1C0F4510ADAB0B267FE4D3CB70C35B295AFC671B60C6575B60EEB99756E7B204A24D7095DF277BE18668E0F1AA5621D16F204575F76C3B13385BAF61AFF7F36D56111FD88FF093BEC4FFC26BA63720660B5BD209A9C14C0AC6B4E08B38FF580C18E39A22A9AC36912892537B0FD42800445DBEF1A03D2D8624C1B125519927C282EEFAC4AFC16128B07456FBC99D34C783AE08EBB0C46915971429FE64E448F1678BF76C7C20F91AD80E213E39AD89962AEA44E06EAB351C9B3A5859D5A884DE0CD767260CB60A96F508C62B731047CB1F3CA6AE28742771F4EC3BE45DF500F0132B6E947D2468AAB9410BC1F581328471B3E53E8E29A0794F4EA8BDF9FDB3922CECBD2AAAF75A2AF4C7CEBA03382332D39DFB770559789708930E4C77966C7364E2A4D762C122BE3FCBD37276EA6194071BB7C18E2627524F3AB2BE7C0C6E59F62D1F075E1951DD3352F6CA9762F243F6691FE6A9DCE4278B178F688E433B990272000F23C92F74A1CFF1429BDF1C3FE1B9AA1C0E7A58ABFFBB54D3F38ED93B11FB12233CFF48C812A227747849012F7BB85529D87459E11DCA9F7A9C3242054DB6BB93B48A47D69BA791B2A5D694A8776B22B8881BB2D192AB0A45AEC71A7171F844E3B1C4149D118EFA899E88D96A296D5EB811AF923AFD1A92E0C71464B90C8E5EC24954FE8C9BFB518A308B3D301D8D6C62E5AD40F63BCE24699AF0BBD2E48FD9AA7C3A1C597731700E3D86E16AD36A67BC031BF381B82CF40B1B235F51727913BBC311A186FDC23EE267739740CC57C36BE05F9331EA40279C34C44FB8FDFCC7E9CDA54C394CEF9C17F8529A56E3BCCBA327CABD07F5F9C567B70DFBD9D7139E39B7E1A493482321B11C881BA9D44CBABEA64F1AF18C18B3B5F98FF4A95FD907E112AED857B033BB7E0FF9466CF5FC691CCC1142BFB8CE81BAD7DB554409B79E23D0A41063857E4AA05583525E69EDAA017F6A7FA96C25A38E7A7D01E22F96ADBB683B4EB8A487202426D68D297B30BE0A803D4FCA1E037DB3BD63A2F06EFD68BD4D7BEAD8338D36AC1425163B3C739B86AC9D967ED54DE24CA58BC129CFF73585B667DC33F32BC4A4D9E106076849BC2574F0E5AB2C0B5CBDBCA59BB644BC86F3339B23B3B12959FC7887F7291136343E54F4A56C344DC5BE641DBCE62AED4D4BB958C4D54D7196DB4ACB8A1F43F88CDBB758546F44C59293501E0569F00B97FC244507C248031D07EE4DBB6A97270B772791E758E582989D93124AB82F15A9DCA469C1A0E98CC75FF683430036F16B4E94E94DECA06EBE7FE2B8E9C0A690AAD7A91877799FCF24CE275844A63E68F1A5C799BC83C7F5384CD9322E20B619D39D0031482FFED8224F6C522C8533CB29C04AF154D920D747D81616EA219E358385AA9FDB8E94A7EE5B53F2CC31B3A7BAC787E54AB9536FC42A3E369043C6F5C11D0F7D452852C3FB3F1845942186044385FB9E482962B1DAEBD2B3DF125D1A61843F71A272E3A1D97AAB97DE5831C56D16A6A6620F8C6CD0F4F1E41AFE6895BA3664D23A03EAB3531126E87335F4BCBBF2E39C47B5A58BA15066F79717D8296667553ECD0991F14D42F8934D753929F146E4B58D00A3E0139D66"+ "A5787E8044248F3F85AAC54E9469FC98F1B1138CC127B120F9946C80B96E3D898CFB9219F5616156C30E9ED4C2C060256A1248DC7B0BA16B8A37C9412493D4DD2674A8FF7151B493FB452DAA920A5FEB5CE5E8D70677735C0543AEC2BB161C07E2C1065FC5CCA8EA34EA53C7D879A57F60BEF6D318280489CD568229DEA833EA03326C82482218104DCA1070C3280111834900B2691B3544E408061935600398441BB509C19285C3200C8A488AD4842D4AC08881B24811C851589845A1C80521424C58248E0C298899A46584A220114606D4C04018B704E2C085C91668A330621AA090D9183154A40421086E189408D34604DC24264204125B0221541401D4A240242206822892889890643884C11671C2340423978D4408449008450184510B31690B30900217641388001885711C89650AC770D3C870D0A021E124015C202A94864090425063142890C211044528002552642046982841E3346D220202D4006A2319500A40814884204C02529A468C1AA028C32002DA14041B208C01286CD4106493108CD3966CD32652E148224842005042889B4470DAB468E2805101262052404108328254A0601247255A928C000632819849D9386AC846848AB28121A16D88A085092530C2908D9C4040D14222D9A440E1045043165219016524354C1108608B4689100551E4008D03B161D306060A2049491850E1C888580691480621E486704042918B46108A128123189214474562281193C64904B66DCC2008A002208C284E90A84103B9640A058980881014957160A621591864C2029019154A13278D21904804800902440901A29001C750D3262913C9448C88082433629994518400118AB0700882255924404288099A407210B26CA12011912485A000520840600946899388911B46111B906C84002D1B1642122012CC3042CB806988126C8010884C9064C9125262886889B48190922123B711C1228C12B88908069210044513094502068C01B6508B0800148869004028620245D42204A0C68088B0004B128000388ED040844CB65103169221A249D4824101408D02923058B4241104650B0044134586C8A40C238610C3208EC9404A2393050CC08553164C11170ADBB64C593804D13040C8442601326019268A434465E03460E4308D22416403484D09384890900954800818104D82C27021C55123377043340408818C88C661D0404A4224880C26490481314A94211C1025040391D93089E0063181348D24388E84A82584888C4A40669A220A48120D1B012699266814C74CD24291244332228920980668A006260849890BB340A4086AD4040D58268DC1C07182484ECCA84CE0380D01166208C8409C046DC2A2898B0629C106312386300A9365E3347280A4616486711AA50121A581CC006ACCC02CC108644B304D931231104090914246024048DB182E611691D34800CC34890C888898182114018611A3601906211A053208044D4A40105B444914429111384413110442462223366D44B22559B0650B161163C62DD22621188408D2A66C0B120E92904510955053186919B16920A06524436ED9C6480126600CA02CA3062110C62C94C68118128141262408439009310D0A276A09040909482883B480609800208824199730A1B461D4382AD08680182026248984A4B661C0462D4AC60D09C97019C50408438224888091C8851AB50D44B08DD416289A8224C0864C03246C200546182912A046654224281110501A1961C9B48082A2810B0708CAA06011A920E216655A103094266A0BC52522046DC08091A232724C4621DBC068181701633085D44089890086CB2204224650D4168000432688844C2280911845722400054BB42C81B02908256AD8402D00A43088341010153009323100404E13C31123A420114921432860C1206D51C62DD1108A19339108312D5B346984022644A064C11681944465189500DBC04C9A2850C2183218B869D2A28100B131648061D0126D14A3242283711C0880C80050A1402E0A838C9B9028D296510C144DA1940880420810320A09198891B609D9B67123B44C903468D286891C906C00248A52362418418EE0C64580426C1CC77118124882184D502272D1384C0C14881B187100299121934544842D9B8210DC28285B2052C1346CE4802500448164044CC9222E14435010474293206519B83080122D1C006163108924C34CF692DDF651A0F08DDE43395656847184F33E137C04F3426909179E9DEF4A914BCBEB9A4530ADAED1ADBEC4D3AFFBB57A70BCC7FD6BE902F7B19A13B6255BA27AF9CA8028D67403BE3E5DC168244A989165899D0E759104B072114F7B6CC107D7514235468DD88420616EEDEA618B8B6CF692981087ECD26EB7C049BA29E1918456154B4B62D7F2054B2DFA8D56BDA46605726C47EDC9A3F82EC1AA41235888846C3C428815D4607EADEBD81AB7B72E6A5D92ABE542E2C9F1C64E3B54FDBECAC29DF0C45E0BAE8AEB2427073355A2C4DFDB8B633C77E9D438DC8C4786AD5FDABD817AA011C18ED90F4B6A0542AD5EE5B42224283392AF775B1122B8CF27B126AD2D806CBA48DD6D6D0FC97A3F123E4A3964F579A6345C2468D9F8C0A5388BEEEB15657A150D7C73D4EC6395CD3F24DDCCCE2FDE77F6335A22A79463E071050FCA1DA34E04E224B4C6759C54E08359A7B228F7F7936D45452B9A8E793BBD87EDA4F6D81189AA5501B8957CD54B9ED4690814811AAD2AFB3DD8E9464A829FD82CF4D941BC86A749C72F82AF36AFE766812F0367DCFA395B9A5D2A38EEC569EDAA793164080188540B070E1F9E3C2921C2115E456ED7B3C7EB8E4063517133CEA544E4B81A3AB24CBC4B58D2BEBC6B7094ACF1A8E3DC22198D1E85B42E80A09A138023D1FB2C9D0C4650CF2D3BE3E5D43F57DA4CA25F23D0E58373161810BFEDAB0DDECF0F1BAFD5229E07C5BF15E52AF391876BD2751C57F8D8BEA0AC50E110178DB04456CFB1FB7B174F5BD3471C5CB572B5B052F0525225E694DE9588A8372831139A5EB41B6A58AB0E590186472C151209F2D2BF99D6D9E583BBFEE0B2E2A5E26265884CF6D2B0F498442E6CFF79A7333FB0835391145F4075758B04DFF4B3F86B9AD8BE10FF01E4DDF067B6E66EA96CC9D335B36A4C9ECDCB89A591CA4834B5ABEF15560CCD78396568851B4BA0ECDB7831C5B27AC459ACA7D307D97868A8613636A3D7845BA2110781D200C2AF170983C5D773379D350F70F48DDD3B37D02E750FD0BEAC3951EE52C831BBED1B26EC3B086CC5E901EFB1B7392B2643FF7328210CAD3CC0529D9909569B7A77E7B459425E1E6E717155B3A7CBF2A62F8C5436F4668F7EDC652189FB54FE49885CDE2B35F7B7BAB41E460F99B3621D73E1C07DEE392661A0066BC40002422F2E3C0CCE1ABA02B1ACC0B4792F3BB9950DB6638E89CF9467CC716FD31BF08C70EF7B0695197656F1F07EC01797D84D332DBC440E8C8B48F3B4399EA8DE593FCD36A8C4AE69C275E69CB7C2F80F284B0A05B25F8BA4597CB56B18C1150F4127BA1054CAD65CEEBEAC54ED8210A61CAF049AD212995FF924157D8B6827F73AA46FF5534C7187BA24705B68F7C6C7998E60F54DD77038D4902CD923D137ED447C157E647AE91C85096A8A3A13B4C7829CC4BB5B05C1BDC2673ADA69ED599DD9F379182C7B7DE21DA504C7C2283B1E10AC2AE9EC092B95027F27F1364401B3CF9E6E2270DF1B607CBD7DA8F9AB36D340C9057D7A38DE7F7197DD75A4E24BA5525ACCFF7B6E0BE21136E08D6AED502864B8B6DEFB690AEA7ADBF36D75D37437693F279E57AB7E8048E031AB029EA799A45C46FF8491087372C787E36C9093A0926B15D383B4F8CA26201272F97F91AEA95B05BFF56D1D35E047E4D19B234082FB13B754BF87F9A998D94D81AB5EC9B7DB5F182DF41F9D64264BFCACC3810919F9DF780FE00EAFDE884A70F17E8913DDD14BDE92D915544DFC830690A980BECA539DCE75CBFAADC80291D6DAAD7C871FEDEE7DBE20DC7ABCC56E0D4F317EA57791EF8DA9569C9C4DB52DC5135E7806A2821788D7E3BB43C44C827AFFF07302D21D14AB0F47B0E74A0973E05471259BFD650787B7CAADE4CD7AE155F01867C624DF65A0F47F65F675B48AF6671C39AEAA0F1660D5EFCADE3212BD70FAC0AC342EA12E686E79CFDBC4C0068FAAF6B801C5E9B6C687C0BA9E883CEFC3D59263458736C6FD358ECE9EADBEB5DA31D03A87F3A5EF581947E80116696097F916808CE634FD5455296BD8D9DD899CE19C34ADE6CFCEE97F3D8930AB5D24F33706AF6A315C6B4021756F8D1CA3C5ADBDBED71D0CCF3AC15B75AE7BEDE62F08760D4EF03D584674B7E4B98CAAF9B05904CE4A30896D42C78D31C5D3E548D1AAC12BA1B4AF6D5DEEE2BEC4B9242E9CE340D2B5C94197C736BA15FA2D40FB9ADCB840CE8B905E05EA928FC1CEF280C68A2BD3FE1FBD4ED47443DA3F42F482DB6BD467724EBA128FAE13E28F325FFAC53061B9BD009CB00100B2CF0998F8CFC498AB723FB093FA5F68B573595E50518EF184FB04DB336FB9335781BF7A9CCB5BEF71AC131F61A87FF2A07A2131BC40003FA7170F7F43D10BB53368EF7CBAED8208F5ED1DB135E5232E06DB1B394E88D99D8E9E4997326DB9F5367F0DA6F26B24F8E7F772E4D23E0F4BE11C7F91E7F643DD932772887F50A82DC5C2D3B916E612B53AA402731E99A401BBFBCF5D3182B3D85126288C160EC5A442A8F95D4929756CFA5DE4B98CCF93BB9EDF0C5E6717D52651EAA58633ABABC4B10C0A0EA14EF796C3724312E97E4D2A7265E006A325D27517F11547B91BAF56AD11268BFFD0E864962830200F1D85F9571AFFADBBF24FBF69F1E0A658E4B3F97726ADA13D7DEF371855A535826BE77AAB5F03D520E3C3F3AAA83DF603019B635B047D58210BADCF9252857A49AF7909942D793069F441B1133C855D71F9C5702E0DEB6EC2E406A617219F928F9ADD14848B73AA14AF4C22F9607814A53BD832147060C797345D9017653F166723F484A4CB271DAA3CBF64D70FDF7110C1C4864BA12414F23C06C79AE70C0F3973716C114C7C64C8537D76188DDC0B5E5F99CDAC1B93A6D67444C3238F586322083EAACA34B5A791E74957FD3EED8F1A0A297B4F72880F9F08618F0BB92A599797C206190973E9DAF2FDF0B9C06F6BF70603B9D2F9233A14B14F263F0F9E0B200342D4099EFF7C405D18DE9FF10D3A7CED4867BB12EE110984CB51F46ACC1A63FD4A8577EB6B6E0A35805390882538F203F8C96E6E9E44F9FC0CF891BCA0D3D49ABDAE3E0CF6EBD1C5CB231EDEC40BAF1D8C63828B41033D5DE389ABAD6FF95E497AF0753E798172FB638B93D49D96766A1EC78EDFB13A8C6862DAA245A59D3C36C408349C62AAB197D9308420CDE0A26F77D9DB8BC64CED2349FA0199A977520442CC787F9432CD9F07279D19AC77D0D3B35B6E04508108EDE9842EF22AE5DD3D8B96D8DFBD775B2EB150FD70ED80F3F3259B289203265F7A4E26BC7855562481A7705CEB34F1B3F917A28CAC6EFE47ED3ECC170C7FF29961EEF0299ACCCB6696606AAC9699D7992099C6BB2B288EB17B5907D78319A15D0CAFB8CE96B654BE2F1200B722164A4A5A3289D314B46BAD462EC95E9D069F82652C3C4095314B3504DF63463D0CF2F2D746CE2F883B8C6A7805DBCC94004CDE0B5A3A79C3549B3C312A426CBF1FE3E8D46CCF15375959D8DE784C725653152DB3B97917A1951F6FDF313FC593CD8425D6A4EE55F321D27CCE899C4AF8C83EC25A95669267080F6CDC4B16DCB0E2D9BAD8FFD1F31590BFC5C82C80BEE12143C210B6C104F3E173438D6CBE0F42E6F2EB6A2750AF3A1380D2125D8EDE383E6162065745400B90AF9682B2BCE9FEB2CA3D187B148A3A592772D4BF135105C938FDC7904A799A5DB2D7503515A54EF84F5DA5FAD4D83369652A32EADDD555ADEB6B3218BF2932C54AC95B552D12E27BA99BBC9119645594EAFB5BA859AA77683E294D3CC12E8A8A88BC8892675F3AA1E166C027A0686B33323EA45A9274D51DB772B15C5D06A66E179FF5AB8952B2D3CF78627B47F9A8911847F59F99387970FFC27DDA6A24DAEF8AD873665C9B7D5C85FA2DEC1FCCE52E680D3622B930B42597F09B72DCAA7C9661306243531AC1A20A3963DADA81974DD2C22AB535FF2D87870C9B73CD95D64C41F4F40C0884D980FDDE9445F3ACCCF66D9D5AA44DDF823448F1162A99CA017FF44B309550D1793CF3BE8DF7E6FCEF5F440127799D6E6411C4C8FDD31E4D8CA550F0226817309E233B7EF9AFF3D1E65132740C79263B62BCDFA974682D3C4314D9033E295D0D712C2881F6A0E1EDEBEAF922B6E410CC9A1D2DD9A0CAE142914C4845B3AC60CB9D71852EC35AD7FE74CDFB1D1F4E5560A77FAE212BCDE1D6AE651AFF3C6B18185F8F6D32933DC524A1C5BE33146C8C6FA37B01F722354B9DFAFC8A4F769CD7B304A3C088366C4C7BC6D8ED6283E1555B4A2F7F6CF9C462AB5B4D52D8CFE937FA855F008ECE0C2C02EE3787CFDF8C16200619FA7F956F41696CA78CC0D691C55B018870D26F236FD30785D7AEA9F90F16491488E3F615ED6AC390DC14E5FC8032D23579C24AFA29241830C4D2628D790179B792CAACB5EF865997DA83A60AC8305B09C133003121F6B3F0CBABCB74A62A7DD2AFDD1501E2AB594C6E3FBA51D83D68A68A835C726CE42D966215C0BA4BC7774A439A1D6DD12B4578F5E713CF500EC7D456095EAF9BBFF8A6779DE5136BD8980D90301A1584F221989794CD651057B021A9D6E15C030FDF04EBE6BCC85353EA23A4C1EF70BAD07846D81B948761DD83D18B09CFCB09E487A518BE97733E799E8A55083BA3409B9139AB44144F6E283D7532284FFA52A1CE1EAD335F5AFE051FC50A"+ ]++sigGenVectors :: [SigGenVector]+sigGenVectors =+ [ SigGenVector+ "ML-DSA-44"+ 5+ True+ "9CE7AE44A9A0ACC05723F4A0DEC292C44AC1F951E1463B748B65C453632142D7115B92334F0D082E57F5E6E1CD28A2D83B6260EEC4929DFF296019A16E64285E884C5E92340A7D60B459ACF8868409B0B57E70FBF78BCD8FF6D437C5491ECC07D19CCE70FAFF8AFF70B7F88CF3EC5622A6775DF8D549759D979A458453E15A77534421C1108210496149382059866D4106511C9209DB820823130661226249304621444C22076DDC902442020D8C804C21C00823248D2115651B9341032522D010844812491BC04901052614420CCC263014274DC1B66092C02594102E8B220C531672D0C640A2A0680031448B24288B92245A0282A100900CA4080C3672188905C28011CB30321A8900230408D004095C322CD1346C11935021C06C60A08003962D224132C9883114148D80A66192C2655140650449501CA26122492C63328608144C0191288B48629410682212901C3680A2328A5A188280C8659AA62D1CB1918C487042C685C9169113286E4B3201CCB4494B204220268C62C0281316864422505012088C062902B2609A862D08218902046C18936889026E0BB071040980E41012DC325010A8815C44612347018326469A8208C2462CCC462A1AC741C1082D09A56440A009621666223049591661D98804CA080219468404A640539045CA184C5CC081DCA64D42A62421A9285814865904310A411119C484A4406498842422B70543349002476D84188AA4446A23B3900849301CA6208AC68409A211D0C070CC2272824045219671611672D83629A4A42418C2218A16011425108C90299B4025C24892DA4001C30045E0044512256AA34068632410DA488AA306421C9808194022A0263111A609E18091C3140A4C2685E4B80C2280050A452861204D11396048B28C1418210C11651302690CA8099C822D4A2825619204C0149010455210932520A56CD144808C10805C365223452691929019178AA1420C8A384541164CC9802053B02D0C3484D1C4219498211B454991B08C1CB98844C0215844318B36800804015C980893044D41B40C9832501BB88108837052B22C50963198086C203624D342895B064A62A87199044A610808191601043531CAB428D3044D1A178CD4A86DCC120D19980C83466014A24083C86518008D991210C406649106065C3488CAB668943445A2C43012B020CC9665A1262E08872824488E40426C133824180110044521CC828D2444501B8085C4B224C44451C8B26DC13481DA304198B620BFB2957FBD886D99CED5B18DE21AD8438A531ADD8FAD11035ADBABB41390D6E7F6771026ED82AB331C70124EEE1D3CF44F319FCF489C2B5FBB87C4D8F68913C6D689BECCC38CBD54488FB340E1848FB96E4D5B18C21EE9E4778C04C57B809835DAAE90FB98DBE672195290E49A9CADCDDC7EAE1197DBDCAF287A382F6C4CD6F45034388053AA8B13B9BA082778F1F4C689FBD09BDB6497DD6F46AC0F73AE73A6A30E5F9FC363E34BE57E71C7D593DC73664633DC642D779AFD2D1E61722CE477BDBC79F8FDDB23252BB28F39D1218DF783A5F28A10224AA89F92A5212198EB2F252FE4EDBA12EB3BF384B871E457EBB06D30E1A3203994A65FD34E543EA06C55EFD90B11EFEA3211A93EE663A0D81BF41B0EE8A8A1E7B5BD51CB44CDAA4E7ED9BF537000CB06BEBE5C12B3AA254472897E07CB2C4EBEA64450E3DBF2DE5B6C21310D96ABBA04463C1B215BE1BB1893063498769975509D7A15B7FA2D9F28242436D68F538F492F55A6DE5F94E35AC6174E5E77CEEB379EBABA1B76A5E85ABF01E67E69AC54CF004445316AE3D95061A5C71548CC822A1888C3587E8FAC74668C876104D8B0BE84D97C7643BDAA884A724271589616A8C2FA683647785F20384BE9A66BB52DE5A0EE74FBBB2B35EA18DBB14E0F7BD71AF19745AADA829E78CC79E3C3DF1F5B96AA366223F3F31F87838D91A2ADE2770ACB9C7FDBC5D19CDBFD635B631F341F44AF5FD90994DB7572F2D4D71E24A2FB4C7EA198B5CC38C4D8614B71D9150C59810B851CB922900F4F262C9F43DF69D6316ADABE40EA37AD7C20163F66BE50B7F8C68D5B1987F2479A1A572F7EFDB34C2724F09CB44777ECF6FF1E20438139D74470ED6D40369140A7A22B03596A6B63B5EDF78EB42E084DEBD08B6B1C9279FAE6AC596176899879824F5E8C357CBECE77E8294DBFC4EDFB991D5753AE2F3756C2DC03FE34E314D95F68420E526B0AD52B96B1D2A212EB1133D327A07B7A26371AFE8A3681B12B999DF4716D652630170533CCDBD5846B72F745FB7EC661DC12900874474FB3EDC5A1BD78243E9CF45C997CDF9F3CCF0CE4A395ECB736A256D3EBD8BB296380C72B0207ED7B0BDE9BCEC76D02B8B4FA0944E9B2BCAED9BEC4C19176755BCE73776BF7FB8460ACFF7A67934FA76BB4F8A1C6900316073FBF3E2A81B5CF99A9512F9B0C3781C075445ECDB344758A339FE2484D45E80B1D024FC940BA6460E768A3E949E1C53F8EE0B3F79AEB895EE8DD046F98123475E846FF3412BE7DBACB37F3B1589ABFDADA96471BE56698534FBF0398381625E3A60A6A15C0B1333773ECC9C831525A1C3CE9C291361592978137FB33313D9DE592D8EC292EBBFDFC84CAFE790A992DD4A94426408857C2A5A3284654A8320F065AA4B1C9EF409CB14BD22ADBC54D9C34053B5D91A7C7B408B0A4551B3C67E8CC73D696AA9711336D62CDB93993D16AC5CD34AFAF60AC5707E26440CDEEFB92FC836F93F3C93F7E9B54D745C87236D85E4614A838793BB7360B747BF56B146C309DBBE90C84010F07052556FD9C9675AC0661E5302205EF46564B0795CCAC0F3713B8DB19B0124B10BF534215385C1BE4C08E02A13F06155BCFC64EAFB44BAA1ED3D1FE1AB0C3CE4967A4DD8EBE2EDA5F611AC71FDCAAA865E47CE1B62C199B487A5BF34FFA886FC4E3326B186E23BA519103CCB4C418CBF9A12D1D2373CCDE405380ED27EFD47902F786BB6BF6C407A87EC0FFA060B74029C326806AF28F71777101C64516A187B039F66C937B4B01245B0B284775A063F29199F581DFEF7519E2BD00DB1B23CB6427C40DE16D473B4EFDA1071C014B7E25D2D5411E6E8060AB2174498BFE159E9435E74ED44346EE38012A7A9010620D2244CA21AC03CFFFAED7195D9C8C4F7CC120F3150D25983C7C080B80207353492DC534CF0FFC8892F6FDA0EFED4A4932FD6B4FAC5A2198488DD128FF64296DE42AFE9C605E06C338A69460A39EAFE3A9768B2A9D8AAF812E9A82ACCAD1DE6469CFD24B16CF6D546018FB7CF56A0383D0F83C65CB53A26B7C4B75267F4BE18632D4494C2AB703E087F37634C822A5FB003D560F44EE5957114C7FFBD1BD133886FB5C47739E8CA7DB7C72EDCBC6D1377BF5E8D6BBE08467BB26BA359B6FE38EB4D0DC4DA29CFE68CD570B472E6E08B2991F28BE8406D470869C4F53BFA0E2302EC097B060D68057393EABF3B13D010BEB9865E0341E5EFF46EA25AF98627A04687BFE9346E1CF6C087F4F9DF6FCAAFB6B6CA9356F626FD7191836ACD57A4AD5FDA57F4E9E5BBC4A43F2501366B043352A3F80E2A6B3099190DDBE193C37D2AD9D952810D88A0EDFA9"+ "661880719B09183546633AFDCBDB764955F5080D3171C9B57F2C2BAF3B84EB4B3434307D97081F960C11853BF06BFD15061688E9749DC29796177ABAA02546CB9F2A68C8F07D60DDDF7DAAF30860CDF9E6CBD482508A759DFD20FBF8E6FF13B7048D9BBD64E4E1DB99D974EA3ABFB7911060B9B4DC4B593AE8CDADD159F9B65367E5B3BFF1DD5AA080B1CEAAC45E6AAAF155F19C527B9CB84D8D9BBF84618D67833949A569A9E601FD0C2F7904A6C8838C3232E760398CB08A1D69D8DE57ACAF2AD2518CE2F7DC243C52D6DF2B27D5B66579163110B83AD4C62B9C46CD2253D3EC5AA0DA5D2DFDA263429F9A0E52DC276336FB322A990ACBACAEBAE09A4067DBECE432074F39ED61AC40CB1A84E682800C378634FE40B71E62F0134B8EAAE32F5D238190D3E85FEA6BF973E2B86CBCC0F9498A7E6FC4FF6E7506EFAE23595A1F23838740A77CCA96B5F5F9FDAF2D25529E37A4E9C907B2E464280D9E7F15E064F7220FDDBE5AF79E33548521A9639783A46FC303BE52413261459B9E1BFEEB3B17A26146B0CAC1623B623C0B39132AFFC7299C6A50F7075968BFA780BC3201705FBF35D3625EAC41D913FBC5ABAAF208C658D11C6E4AFA4D75C4A45C0B9B763628E8B52C1CF05798450AB44A457C1D0D1CB31B3369295815EB52C4B9F7F14B8E5E4042C8B70EBCE18B3B46C65FE69F3A7D9C4E5729D4776AB06D3AD8D2E2D48BD11430E4B79AA96A0A4407E1940882262A979CCB7D1CA28F25F9636178C626921CDACFE20CCAD11A67CB74CC649D6314CFB39FF975F166A0F63DAF88B511367E116C7FF71D51DACA9111BCE3BD99CDD4811873AE4538FF9BD2EC34E694622B24AF31653172E52D4262E531748C388D2A2443EBE036E5B0769DF1DD4A943CAFC385491C4AC4F1186CF9A994F6AE9D8DB037C8234325B16ECB7C29238A5226EBF7DBE0111AE8C4F247871091A5B12CA0C07C30707F338AA39764040EB6399F7D8B4E46224F5D09E0A84857CEC948D2D08AE05C2AD0B13A3DFD2F6F24C96C3A9F55D9DA7781B7811DA70122BA7188D30688CD7BF1D8F55765C6C8DAD07AB1E1CA02C8569C4F5B9E06D2341F27170105D700EAD5AB373D67AB7629D320BD5351C72839BB7660D3F05E79B22F3FA1CD8F9AE89724CC780A04BE3FB67040BA17B90273C1F5A79BED6F24B64DB53CFCA662E76A8D377655C5881EE05764CFF8A7AC2FEE2BE603B5F9271F0E79910028E515A38669E76646A72D5AC4AAA0C20D7D9E474EDE21114AE85052F37D1E831F4DC98119714D96EFFA0D103D31EC064B1B9D2CFDB14DF1672F15D300D76054B86A8BB0B1ECDDABAA9BF99BD089405FBB512954D2D7EDABFD7A6EF71CF832E896DE9A7E90A857756F6F04EB5D4B6CD62631CB32C0F63252E81C5808624C1AAE8656E73C5F67918784E94A042AD6C85682D04437A7B93F7D1BF9E5222FDB5EB9742EC125A2EEA76B6972ECFF744E4CCE3F88E3C0068FDD487C892B07D2FBFA70E8BE5276A407DF65914BC78402BB074600B81D2A4B4778C7A4662D815439B631F6EEBA7B944D865DB22F087F5782AEA90B7F40892CF15FA33AE7C42B6492959ACA1E06B06513C641414A1BAAEAB3AF69919214E9D3814FFD9F1F8E5B8EC7A7CEDAF33206EB403C2069130A7CEEFEC2C84352F30C606F8DFD25F1F1A13330544BC60AA9AD7BF2BC859E7558617E74EE91C0C18BE3136DD883A0AEF9A7C5DB95C17589F0DD8F838ACF36067A39F22EAC1E98B77CC8F1CAD4B7E7383106C2F28B19AE4B8F2A9387483D8A7DADE825C0BCD2B5BB9951F4430C04B38465D7C1154CE3EF9FF9AE0161DA3208700CFECF883B4C10F21B0ED67BB8A8F9908C6DBA50AE967C7C71600CCCC4D0CA6FCCACEB069540F11F286FABC9606000067BC532A4A2D51DF80564C82D6B3961C7D836DBB8AE7556BE125D26CB8F81E13A8E78062698A2A6A8659A7CECFE047C4E3A3B4B419A8E7E81ECCA47497D82928151C731E8889C02237876472CF7438859A6A44E21033301BF608EC29AC61E1EACC67330875D22F3885D4E749B4EDF0FAF650D689C2F59F5814E8F0B7B245F72ED12950C07F2EE670E407D7BF9B4504F6C09E7FA380C3042D37A2375C5313305D84829EC9E4B1997604674CC712C70A5C4543DAF2A38BA16FE3358208A4EAD830DEEA8695914A4A9B81232F2222E752B9BADC372B441F380B1830042F5D2A9923897FBE6580B5287C7F1F01EB86B9AFE764DF333E1E61A09D50074AC9B6B9BA930EB5EBD2DEB788FE0A458CF4AC980262F9CAE907CD97E975429D2D09A322C20C13BE73991CA9B24DE6F0C5F7D6F32A7D00FDEC88C5CE51824F087C91056FFED04D63E7D7151350E643AEC61F1A288DCF2E327991F30F7FFB9588AC7D9FF5ECBD1FE02064C258BF8A9FEF8835C9E70AFB860C96F21D0EB06DD8AA2FC4ADF0D5CBD48006251403DBB52496A0B36D48A5B55D4A7E65C1C13D16154468AE0DD4214092E57E6FAC5835F47258C35A24C5A231483D98CF3168B152FFA28D5A084E1DE7F0D5891F89FDF2F619BFCE83243CC8929E70EC2BCC22161C05086D9F4E5FB430EBE796ED6840B2346964E15784B65B5D41DBA9CC8F09DF87B2E1503C325C5B0B0753498F8A0A6E483D7B25ABB9C90D55349FFBA95ACB94357720E7D698FFBF8F009B20921B9FED3D74F488233F56B8AAB7E92BF1EB5B9627E5652F7747E34CB59E64F26273C54E6081E704E0F3899A96A82FB9D5A55DF3F2FD945769F07F1479E57BACD9ED50DA71CED14FDF5E93CC4CCF1FF65C6D22214B3EEFB34E470D41C1B65384BFD669B69E8C2707D69932D2B0E2AE773240DADD6A2B7343A3EF56A8C6B7E9023154BEFB365237373A9C5CCE7283A2A4EFC3F9AAC894D96EC1A018D5C361943367A55663979E4B3BFBA362439A44E3BCC52A0A5AEB40B797B84689719E22EAACDFF703663067CEFA2ED4FB802C81B55B21A708AE0E742A6B1B96CB66886A10E7338EC55F900563CDC818DF15ABC33E73AD11EBAF1A1E3BA546A37B942597030DFAA7535B05626D1D4612BC4A19711E873AF0FD9B780B8E86E9CAB1FD22A9D08452F896CA5A6974AAAD63AA98A5157008A09E4E061B65F68161ADB47A0D0A70F0BB615F9979BF1578F718F16E8CE034AE44CF07939A8BE5A28210797FE26BC3AED23788E9B3B4C7B831D88201188891EB2DA6D5C0B94328344B035E41674751892A09F72ECB52452A8FB0E6B2C020B5D051E0C743F4EDA8C956544BF8DF5779189B5231CE68F3CBE4EACABE621C8E98870C2D98D06FFCDB5BD8501297473FE54E22F34158B7C20DFAC12740C049811978EB8338C4959C1BDCAAF99066EE8D0C2B9AF7E325F7741A34F23E6C7BD9864568C6F114435B8036062938DCC20700292DA31A468EF9E06A0FD68444CB92351577BB976707BABF45E186DD166183E0C946FA0EB0FEFE01003DF8DF02A161BCE33F9A50A2D047133C5409449FFA1E62DED935B6887564BFC86E67BE2AE7034F2616D724C452B9FEFD728415AEC67F67EC13B605F097C3BA546EA36CCF4495B079B80B0F620A8694EF264E4C3FC92E7BCB36F2B0A78D073126392F5AE799254BCEB3EECD324AAEEED45D92F7C8F9F383B99AF5F480647468754D5A10E95AF2721E0746334CE806C12B83251BE57C7E705A08A5DA1B335304E73BA5491BDA71F9798874376A4D9C6F750137E3796B942A8665EB5B546BFA824EA070D9478BD26376AE7789A704D635640A4615C591B56814FCC4EE3DE897E1021B3C4082F069BB26FD853EC3A9275B542933FC4D0CAD8AF103D9B146F960FBDD3B22CE503A6EA7F68312632CB66BEEBE0BE79ED4752C34E99C7AE198604E3960A1E52F97D5673BC81861693187177BE572218532C1ADB26ABAA3C76626C067FC43EA44225150EF911A60823796EF8D63E11868D83A7D5AE60770F8B4F85638AA98F011047B5CFDDAF9CFB02C93C9EBA368CFCC5F6B00FC64FECF7F4F3BE950CA9D15F17292E623AD77C3D936E249B586C0249E5920C5B09A72B3EAAF0A65AED2F692764950BBD7F718ED252BAB7F18E96E89B4EABC576D035F423E6AED3C9A9CFF74A60B231410851F5AA08DC91FECB12ED7EBA502AE57AC9DA95D6140E751C97AEE4B882047B14"+ ""+ ""+ "DD36EBCC94CD8834E17132B10458E43557357C09A5B855592ECB412BC27480F212C4105527CD9919AB9D2BCC2EB74B5BE49F964B7B302213CEF8DBF1617554F90BE7547809E1E061C0494430FA8B07972BFF6999DC645DC2ABE9059470DF537F6B312771688DEAA173FF211BE5E7D8F4819E3F45419BF79BB51600986351B62290C9C70157C0F34968FF50DDC771458EF8CBD9F61A70B821BD834E4ABA0ACB642C3A0B83904E3CCCB095807BD3CF1CBE20573DD9C3E044EA605614995FEB203B84720DDBEFBCD14E23564C4DD046E8C6F6FAD8638E0B2BC8967360E3FCF632715CF1C34DB2A7C183699BF44DECA5FCC427CB90819A1A1B233275D90C35C7889E35DB6F725D9C8812358E306B942127723A6A2E1E93ED1D2E2BCCFF996EB10F36D5B595FEB3A41AEDA88C489CDD4984E96079A29371CDA0C7CDB2FCD99F92BB4C8527E5A8471DA0187B507E7FC4830F93D2FCDBD1C07E6D542B6706EC7B1CBC5691FC1585558758014FA15BB47A44BE8E6E1FB8773E57137198150B86BD2D557A4FD08889EEDA4F0FF3F77056620786B306F843207942F6544688F949EDD984C8D2386795BB1E3B5A4F0016755EA9556277C8BF3A85F6E38D89F8473565722E8C80CB0CA1FE15C99A8C9D501841C316050EAEE5935A400899378C576A6826539EF7995AD73B37F7F6FC972E3663759D9C7DB2099747C3373969A1315EF3987BC8CAD92CAE546EE776BEA23C018B169F04D5D5A229CA19025837E20A0A948D147A9F4A68CD7D5805D7F65F0370DD9A7924A564921CDDF3E6F62111EE7EB7082B4A77AF80C5BDDBB496E3E1F8940620C2515B49F211EB3BD5D5E0B06F554E279F672C7812E82E8E1C80867A0F0762F9ED3DE4B22A24E1D24B4C315CE98F33F5754EAA1E98EDDC50A93BF5DD14A72AE454B61AC9C548062AAEF2E8996433DACF8E2238D31B434DE66688ED47069465DEDD6EE2C0E66DB2B24FF3C89A0FE8F012EBF6778C03BD9246B6354229D2ED78F4CA5E96B0A774CA9B78469373E5BE91BABA862B5A60F06889BF4346D93E08C27685CA5106A1D3F621EB7466C2CB4BA29F3ED03B06FE01F720146C5FCBB2A2F2943C9A48F965CC574DB9F36799783F728BD807FB78CBDE55595C14A714166598E3B5F78469A9E6BD38432C8B802FCF0A1287E75577EA1CCB9069A9E7C86AFA5B2126BBB17C1FF8AE4465A5A975F651FECD18C9012954AAFA9F5108310F2E05D401BC75E20D5E56305A9C2C663B805231FBA93B99D39E03B29EF65C91B253F0F80EA1930D5CDFA907797AE8120E100ED6F8D7A1831D95EAA28C9FAA3D9BD272D3BE1FB6C5AD393A515D842FDFBC78BF240435A2A477EEB2A9F03FEEB74CCC1D1E8C1D16ACB01017D64D9D6BDF43564D5B87F1F9C0850C90B745162C875F6542DC91E6B1BE2E2419A731033C8289AFEC90075C0DBB3482DDB3B3F45F04AFCF5062C3ABEAB65C63CE882353B3880C6CCC2705147CB9C1AC42D3783F06D1FE216BE45CFB87A67D230A5586CFE24DB1F682DF0F14105FE5015B664A898D9B14F517D1302A020ED92F887D3AC57D416572D44AC036509B13186C9317E0E9A97912FE7E2EC2DD62637CFE24390F51337776F6D62437BF832964C3D3697D411BDD3B1715423E5B71672980802CA2D00373AF55073687E65ADDD398AFA41A7F7423EA601839081D635AC5BE70036C2D0920978CE709560D4CA292DEE56D9F094C8153A9EB94D93300B3174A400BFDA080F66F39B06FC4DF062B58DF729FF3065B7D720C1AFB2D4ADAD20099505B6FC8F7DD3FDF1AECACE3E446A899D9D747F329D2FAB1BF9EBF73D83AB7D4E4BC70F5921539075BBEDFCC80598C50A76EA11BE3748A2558D85D34ED7B51C9153CD19E5C184A4515D1D1577AE078CF9E345C0DCF41C9773C6D5F25B7D8BA39C297FF2AF9D162C12E283A213FBB4FDD2D9A42E12254472E51DBD3B5E5D16DF16ABC48B9DA8D34D4784A62E9E7D57E9BB215DDE58B45EC65E2466BCB1F5210EF830EA1878713D2465E54C133486FA91C2F5C215AB029F1C1613FC85DE03F26F964A2C4636F55ABAD106D96B4634BB11AADF825D212400CF7059E189AE4E80B373AEA6C8D0CAEA1F545810A6D3C5B1012322BB2C730EED76F357AE2A03F3EDD4CB18DFA56643523B6F6528E83E31851579C2A214AB224A75DD98FA8383F028194E2BF4EEC96DC0EB9F2B31D9D66E66F50C84CB827744CC0D58EA770F2EE5ACA58B9594F42F43464C252102C3475FAC5E11A79E29F1365F5E72747B95FD99CF5FFA099A2A3FE6A43F8DC1ACA0D81361B4DE017C61F338F768D63CC24DCF7C4F334028A27CDA1265E7FC6703D63FFAA0C6DA10C623D8DC2AE72F4C134AB9EC1C8FC5C90CB6D63111594F164A40335FE3EFF9BD6E81969A8221D5FF3CFFDB54D260774E02A47A4A2708B639F4A1593267C5CEA8412F61A5780DAAFB89D04E5112B6C2C709B9279D01BEE877878EA15C14DB7DEFF1D64D74AC6F204730E9F24DE21194DDB88F0069B3F1E2014FF838A7B4BF6615E7C68D03638720651365EACECAE7A02B007E059935FA99F0C250E8FEC507E02EDEE75DB93601C5F217ED7626A5D43C92DF6AE8A51E1DA5F5368C897E996D0A986A0C63F0D20965EF56936714558F962F43293FA65F3673DCE9C13554919D8D698F91414BB85D0D47EB43160BDF26879092C6D3502219B5E499FA14D1ADBAF53D8B16A1823FC5CECE854E337BAFEDA725CA1C00E3C6F79EE2CFB8B87CC45EDD624162DA00CD3368597641A639219A77C62CA721ABAB7DBD28206ECB25E89A9C759DB2E181A5DC115602F3AF0453B1B04E6B449510FE42CA9E50234117858A0F3DA30221CF2850931A1118094820D77C5B4BE7013B723D477326F8B3172467A03BF2CB557ECE13DFCFFDB47FA6B16659336FB8873082E14E14D558BD4830337265E52A95F5A9A0210A737034B9C5061C1F9035EB0792444C5D21016E7CA779D7612DC05DA6660BC39192D87486D8D388083E0BFA40593ED8CE52D90F95B5E8D67CEED92B690860434B73D20244FC14A678B53F313F3806DA39F9128429F99997E51B1904CF64FEC3789E112AF100B77D7EC2495847721AB3E0E1BE7165E4C29F6B4267354B7BD95BA2448FEDB4AE4DCA5B43519E4C5356EB5CCA226148517847751204A764A8A8C4C073C788F98A2D1B57E9EF2946F7CDBB042AEB7E4782F9893140D865630F525C816FE74D98C22C124DC8ABF54F31068189B3D67458A9300BDAFF9A875484C6091BC4870702E38B1AE2CB4AE688EA324629CBB8006222F466283898AB8D3D6FC02051F212B405D8D8F91929597A0A4C4D2D5EA070A193E4550626674819DDCDFF3FCFF1121424F5C64666A80878C93E6EDEEEFF1F70000000000000000000000000000000C1F2F41"+ , SigGenVector+ "ML-DSA-44"+ 1+ True+ "D10A7DAADC675BFFE61DA485642DA00930CA6206C420755A687F6F9AFDF547D19DD48D7F417E56ED9A333584E14284D4B7B354A4A9162F6BD09664115F9BF3649A997870E0609AEC65F23FDD5884926E5914FC9A03A0D2081D9234C5A230F2A767A0BE128F439DEB23341D1A67FFED52E6146464C19E662C1C8E770744D8FFC612114D1C1222222664D3102AA23092DBC4305C202619A690DCB26C13040C0A2126C9382E1A0330012685229585CC226ED3844DD12012C224712242691A294922C428094988422268DC8065C4122E18A14060444C9946285C904DC3C2701403315AC64CD2148DA0064A9C3441DB288CE1A24588268448B42051440CC318051AB890193351A3C484609630999230E1246C01879058C82C14108C0BA44023282824880103117260C220A2C685D8442CC304324AB420800222DA8621C328261A054009480C42C60591182242962949482009266504A52980B42023A44823444D242081A292241017800B930418036AC0286910C9490C220414460512024958406099B610134004DA8045D13660A1266689022400054012A95084344A08B4294B222062802C1CB08419846504220059366501153161068619B02C1C3904C8A028D4340A8340400316860494000A164C20B1294A188C51A49023212E4B3644D2B80821C5640B194162026992802012B66011990D400201531472A290246030691B342264964800B62019080952008C538208020384C9B28460A86802263122B621A186898302066310250C132CA242728AB20C224671E1264451A82D230321A2202CCCA8611C498C514661E3900581C01159306E0244601117004BB2404430709348024444611A922D194781C4A0906346920A304E19280E244842E1308EC2C04C0C974808088283006D082631D090050841419B308EC9C4315416300B875082C2850BA909D9364D9918666146701928890B1391939089DA060823898D5C8444209980D88625002266232048E0960441926161002D8B986888C26011A1289192445A101064361018380921076580481212B68C084924583009232691C29450A1168419286C9C82491318240837842292811A304621B820114142E1C4109BA02450A4650129411319915C806D943870621021093411013429811084529830C3446413098A0CA30849262909B69184280C11158A08C90CCAA490D8B84D22C38521392EC4A86D1C1826C11070434208589201C4C2498044655A22685102924934040380013E561E982B8CA565AEA3EF00D4E294F5E7478BF127A82E36CF32813D6A2ABF054B56E3D0B1B52E45E11F2B1EAD362995C40675AE48304C1BB2534A1BA1949045FCB52C3E6A41FAF37FDF22104F68C09436A0CEE08A42FAEC891A417E0DCBF284172AC4F94F2890DC1D66FDF1CE8246B2C237153020B445F7F6B34651B3F380D1DCEDC4C7182962748C24E2280618CC7D98D45B57AF73241235A3C9492EB0B9F4293948FF40BAB03533CCF612023410FC19C3B33986E5C5B4892A657E32F79544AA1FA18271F4B6A92B888C968FA4B45196F0CCF15C4779E41573F2984CC5C34B57F9D3C758544A9163B9023AD4279DBA755B8092F311C4662F33DC7B14F903B2AB102E4F21D59637D192CB8DD60A650EDFE248D4452662CE148A5EAEA25664D5FD90B03A41DF219633E83C614D354DD509190A547E90E9C648902E64AAD4E086A9C7B29090658C1524BFF42AEFD64DD571DFDD2285FE62D9876FA77D885DB1A9D4D0622ABC73BA4B929D6E2FC98334EA4D4B2C190C7376F6AAC912A38E9241F7D08F4058161D8FFEA1970A53E2FA135AEBDD221470D36225A2A23C1E8922951CAB2A0E0A2628991EF72044A5EB715B3C7538655F34FD0F03002B9C15079CC9F8BF1E17531105E6E8387B70C14E3AD6C9F1D83354903F564E53E888DC152165C6C684868C150FDFF73BC39E65350B09D54DE51FCFA9D43E1A36FD0420642A78C628ED92D0734C30C2A544AEAFA81344AEE122B2D57987F244BA36C33890234BEAA43DFEAFFC8DE0EDB74393C458C775991D0C120CA1790DF28168E3F607C4CEE3359014800D8EB6DF570D591C9FE5D3E3B5EED9AC196BC61696926AB48B7F61A41D1D236517B01638508DB17ED4834F3EABAEDC5B0E31270119F6319670DE5715B99129EFCD92C83206C8086459452BCDB9808C114038510055494CFB652526B6763BD58B932E3B5A1FD24AE3F26E15F44B609CF97A204FEB07EF9AF3519795508424A1ECB061A3DFFB121BDAD5CD44B519FB7C49E9FB234AD76EF3CEB54EDFB13D1E7FF9DB99261BB39513EE6138FD5527CE67A56F2F688AA0437F1C4FC50670FB49268E48360538C0C21454F79666751887C5F98C495ADEFA4089E33FA1806019DCC25F6A20BEC3A95AC2525BC5A424DCD0D0DDCF88A59536889D836D2397E9B72A9DF58356567C87059B8FBB830BC490215F3B7FD7F01576F09A815DB55B527A9668ED2FF0AC4AA6D4FCEE920F953D292538014DBC57DA9B01F3E522F228E6A200E22A31E28DB9CCD5E672ABC99EDC5950909A02FDEA10B92C3A1D4383E492A731B1E9DEFE6E1DCA4625387E96BAFE38A02A3ED15B1637A3F72BEEF52AFD2AD74386D7797657C450D36842CCA8F3E300BA699C5C8AE0F067D584B9CA4AE44AF156B3F4DF7C409C6A1FF20008B95CB4D20D505A722299A5F96F0857B435602F166AE8C51ADB1B0B65C115193FD66D51B7B4E94FA659D78468FDE4C4BDA59DD5B7A45D9BD178D7BB3ACFF94CECFEACB3CBCC93BCD0C44E064AC885A604F6192030192BCC7CE7BBF8B221DC377F2548E4EF2F86E5762B92C619C63F575AD547CF9E32E8FB145AD80CDE6FA28473635681C97BDCCB85D6F15FF0C567DF09B5A2E2683CA3FC8096BD1FB6E36DF32B4CAA02270AB35D1BB2DBAA7970A6273E2BB4FAB09AEB66D40BAA51D1146B206BE11598053E379424A64DD7EF818DC7A4894B6B12DCCDA271A269748E062674B50F762B4401FD54288E46952ADEF25D30BEDFF1C8FB153A55A98C83E12FE0876C4F52D5E9B2CD986B71C67F46E05AB0201FE635582909730DEC0DD2A4B1FDA1BF6BD7FB118E5490BA54FD01D394E1E1B80B4287B785C9226190198D8BECB7296D85422632716C5E7384213DBC3C27A8305B80F5B3E91874020A24123253E3014F4333A296A2F61C859D34E980B0960E4628D2D1B1DA13F603EADAA5A6901BFDE1F8D1CB92DC0DF91A8D6FC7A65736516A3E2FB041B535B095DF6278A39DB66CCF0B365ECE9C9317CB05899B90F7E4696564A627CC9D025EDA8C54DF41CDCA82F95E247AD577DB90C50F579EA555C9A4F7F538526925B30FD876E85660C669FF4771A8748FD6B9AD434E19F10DC6E28C3902BA2CF5516C7870DD03F83D1098142AF43E29C6CA8C6E945D3AF983EE7C15C3C87044691BE225AA422EE600D06F18A6FD394317FF10C391CBA1789202E4D6853CC27518B849F5FF16AEBE26DF3F85A005F2A77001A925D1C67825CC64FE0092D90C2B28541F9E43871F77055BB2FDF0130C5254B03912BCB937E1E0BF2180D0F48A84A2CF093D5270EDFED0AB141A608C8A52BC6C4BA34128FF377A4BB046D90CF79B"+ "636C19A0652537F75E3931F27C7CF6025D78CC759B8AF4595563A320CEF0FD67FEA0883A4564A78D0C9401014BA1DC69124B77C5636EDDEBA65F050DB3438C603A7D9359E6094ED43D30E378759BB85A7156635A5137E492F96E439339E5BEF00D779C03FC4DC076A6C8FC4700D659EAF33EF2713399D7D6808FA6A680F49D3A77E44FCF85AB9964C797F4280E6421A014C388D61F36D45C4722AF95B1870D9AF10AAC22766EAE7B1512A708F61FF82F25582E8D8E8E442406C53E3426AF766F9FEA47079447F843511665645ADB228557D54428F6AD85391140696FB100927A7601E319A44F056A81253B14C756647035F035B829ED708204EC5699B17A688D2B2DBABE4B58768C8E6E156730892668DA1BE91A8EBDA730C02E711BDC94D8968D6B9EEEBE3E8B1F0AC376D11E31B17607332D8911F5F9C91B79FDD677804F053F3398E1002D9B07A4FA0F8786C2AABD05E89C2B9918CA254D9E1B5A46301A1A4AFD94C8EFB75EA35FC6A2DB1B466736D44407919B9A02D4FC87985ED1FBCE47D62BCDEEF97E020C7A37C69BECC7ADE6FDED561577F04517EB8E47F1C3D9DB25628FA2C82C64242A597E82E1B1ED6243795C38AD2031D679F3F9521EA85B8AFC1877884F0A2642359A76464D6A39113DACEF4D4ED5F49DEADEADEF6F99BFA0A0CBED1AC32B0F33C6AA0DA1BD1BB426E368ECDD49FB56E9333BB35F3DECED433B346202911C7A2A8A3948A4BF27D89E3DEADC5CDDFAD502CD7E373E55996C0F4FCDC896BAF211BFC5C9D81FFE24686A7909F8AAFB3C1004CA1CF4A3DB3FF4625847C9279252A73694D799EDFFA01FC222137E49FBC55619F45AE86128A6965EB7D7BF1DE78A7DF752F46F5256055FB37E1D4A729EAB3A1865B0B1104AFA2A0429EFF35E539DE06C861CF535DDBBC9A66A325B53ABC04DAB3612BBC7F60414472A4D0D1FACCAE5DA71F533EC56702A2078A8145E415452370E9F2839785161B9094CA608DC414F7F919DA8FEF75C4A5E4B341958BD1E94D056E961FB49777819F32B49A1CF76549100D27CD625A820709B71B0F16B55E552ED757F30D34F510E74CDA618CCE5D0FC2584ADA3708246D51A344A9A86ECE79A4415879F737916AE75ED586C7D3BA5E60C36DD52EF29687515F2AAE9FDC6FBCC39245471CCDC3302D69DCDD8F75CC7C51E486181B820BE9887537CA77CEC4B42C64FC84500C32E6DB89672AEC2F0C9929927D64809C11C02695435FA49EC99494B8FC8C1EE61ECF5916A43D42328EDBD4EC693153696D28F11B7A7ECB9C0D7FC6B95D6AFAD18582D3D0DDFDB502DB91141EFBB21E904DF223E76D9478F290271469402DC392EAC09E87D1A15B91105EAFE3616BC311E916D7210DD5A2141AB572399EC5618B74757ABEC407B7151331CD1FB883A81260A33EB1A94818658F628F88C67EC2ED7E6877097E23C517D87069164B31869A3E38A851B2EFE3FB8EF8E2891F8F957C4818413F1DD65B08F4E1918D33975E61EB9F185BFB1AF5CFA794EFD93CBA097A0D9CBD99A372395694936CA8AF8B806EC21F431909931905A1E4A1573CEFAC618A4118B40620813EB811EF0410A7B448EC1CB8E92B83BB1640B3CA24A0F16873A78D4053D46C8B8AF87EACC4A959B97C195248955353FEDAA4ED2081AB79AD64D6886DCA13CDAB24B21C3D1A6E6FF3713575AE5E564F0C055764218F914D9725991B249CCAFEE1B7B0AEE46F59463EC66CC1CF42AF46700B2538B905756BC3967B86509D36603304B84D94D64A313E1069A6EB64D412BBF9A85B0D0DF3AA0555752E909756987D5826A3556B3028563236BCAF8AB148C51C2DB735A3B859A81DA8C50EA06364CD0921ACBDA93ED9A0EE0EB557ABB7AE2A83CF040C13A19EF4D44C9434D11664E79E585728EBCC942029CAE742AD3DF4A1EDA625055B0325605A887A4D5F037BD551629590599BC0A8D45F0F3911F97D5E89E3384CEBB65A23F8B0C4AC65688F2904094C2A9C8E9525F310BBA5271E211F34EFA5DE2C8FDE9D9D222EFEC1C3B9900426574318A24E3FC164DCCA43E75627BF86702F08835F6B5C54DCDACE569ACFDB0F9C7D45D0A5A50EAE618BEB2AE26A0CFD455C8394AFA0A39D23BE575CEE22E0B7F536516314F4E303B3A824030870132437BC798B87D045FDD05AD588A8C256958EF3AB059E44A9CED81B64E59E83F80946F01F34B87FBBA04BFFA7CEC3BB1F189DC296F0796A2456F131FB337091ED88199943B7DAE45825F4246FC29B9B00D35EE5777C0C11929AB7FA290095AFC4ABB23359C49E6DFE153EA3121BF4EC411A4290E3453D9FBB736DE225D8F986C9063054AEDCA54374322AB51BC785007CEDD312975EB6050051100F0812A028CA4A6A971C71CA142CD9FF994A3DF0CB7265F4C601B3D123AD5CD668F0D21D73DB64F30C83076671A3D4D668825351D6CA00D125867B9DBD56FDAE82F36DE8F33281FCAC14CACC539B7B319F7128114BEFC9D6E9D55FF5C81107D75F6AF1FCE5644790A3AF4190A0EAB6757C33AEB53E4D0984E0535D4AC3259BEDA56BB70A00F7753E36566E5564AA2A3710231B53FEAD43067502E45AA27DF701F4A64ADB21C69887879D6271538E87D4ADFFC8CA278083F046E8D0A98EFCDB98D7C1D3FC7ED836D8ED42E4829D89FF45BC63577E16CEC3D892B9B3689F85DA6C20000B42D44437F57DDF88FB6926A9FA41603F5968FBA8D304A830B6DDB4B8664E4DA9478430B5EBCCB079CF91D880D370812D47FC18B9D9578F838A2004B5F7AF2AD3E9FA2A8B6EB82B7212C50062883250BB93BC57911237A3720FAF7E073ED34EC00460C3243C14D9AF8D3AE17CC99F2308274389466F2F34BD2E56A04867A8AD0441C9F1BB3A87FF1CF2C6B056E4D1C8BFE7D5BFEBD79099F61C0DAC3B31CC764D81D64E924F0AEDD72F60D7091550BADA81D204DDE63EDA906C30B6285F2E8DC9DFD5A79FAC5A81ECB3962AB2210636D2260B59BF9646133975FAECD0F7917D5A12F80F5C22D0E5EB919B140978508B5911ADB9DC78A0C1DF21DA7082462387C7AF9302BC5F487FEFE7D25058254513E68D32F47F691CED25B46E94A2B0CDCDF1F3394E8C37188C162D22F965C6764DC68D215004157B00FE1F49C4EB5EAD02D13826CE631B9117DD3719FA7A1AFC6F7949F34FC3AF0618B1AC6A1D4EE5C320279D045235EF87C9CCF50F1A3E3CAD582C98BAAE8EC4AFE59A5CBDFAD56E0CEC415837176328F833E872146394A0756F973F7B5BD204885EB4778B3E13703BE15893BAB3AC4497A7E48CA30723545D158AD16B114F657FFC0BD35BFAC0ECB09B8947C7E1B60030F250BAB14200843210624C1AD40BC529107560B3ED36E862A5CEB39B6564CF61A5A566D67875F56CCFB092D323928F4F24AD2CB127B5E669272337377D53F2B3457FAD634AE9C64D269869C3A01F85ED0620EF582814E15318B1C04EC32C05398432F2B97F8560040554C619090F1AF7620A53728D86E638392CF8BAFDDB91FCB172C2BB06F7E759937CDF1E4D2A70016CBE1A6518B6AC56FDF9B166EF0380D4631333790087B57C6F6C2939DDE431FA83E7BE7EAFF50FE4FBD342BC7456C5CCA6BCF28F14F3E92B3CBA6A18AB983A40B635BDB4C027915592587CF9B2E236D984C4D21834E830419C8EA6415AA24D78DD05AEBB205A9178D53777454585B9D30BC4BAE901075654F0EE1F61A0669D19ADB28F7CC5429E4FBDD32F8B4D212FABE046B7F7BBEB2B5C7263FDBE9A4F5BF4B2B4767531AA29B3BC76C43C63D8B8FEACB9A0FC6B5FDE1427B116EEF64AB6721F8354061FDC659143134789FF531D9942F40950C17B8816CA8641974B6CC6F3F0C39F81620BE2A60A7277E366989A7958624BCD3EB735BF04C601F2826003BB06AE1299DC760B8EB13808073A9CECCC88C4074F17CDA2135D7BF76D8A93EE1279A2705355134D6FB1539AD0CA933B0C338C9660727D5D86F177AD19D5BCE24C18D29A659920591F3489CC6554A38B0A85CD2A7CE006E504081973BE3271E98DC8EF84559DC0D95E098667678165906525EF8BCF3AAB3B9D282826727E52959BE15DDAE518B0EA119C55181BA40D3C3E8BE200C81DDADB93D49D0677D265B311920B64BD942C5736A7999A064F003A2EA008D8317437A7561B6F116FE781AB266B671BDE3BDC37642374FFB520E702E60CFE098D9C1E88BDE9A099DB807ABBA75ECAFD41E6372CB986644FFCF590F6A54A91EAA33909F158B902055895CDC86371E556279C4D77F79142878D6731CFEEF93F9958EE2D9CA4E2899F6E2EBDFC1DF1B72EB57FFFD4D3FAD8C0D5AEE3D24860241681DBBA1AA0AAC627712162C062183661497A20131FB1751308CE377B09FAB0D0FF10145452F0A8E5930F96590DE19F1D29FDDD0931233696301D1B49806927177A70B1F125605F167361364657A5202407C7D431022B41A11139B4B2072C46D955FDFE3E05A6D4FE2EEF2482C447DFF4CF2265D7327D0578AEB797FA5E8DF5A7F40A8E37A9FF467D9F06BE17A3F058A93413300CC44AD3D435A83D4C146404ECF6C8832CE104180F2F3A1206514B10EC03BC6E636AFC7087277F4BB1FE75A7A11D1F52B4DC6C559286AAF944C20B43715A3BC1CC1521FC598D6F8946A69EAA287EE411A4BC6B17C57873956745FDE6B0B5E240A89E26953EB41F1FF421CE161C124F42A8BC09639AE46FE2AA50B0689D53BB04BCBACC67C19CDDAE1005FC256B2DE617BE9B3A7BCAAEFB7505E8220C622F15E05729340229FDDDE337C402965975DACC6D1297B334123DF0BCE665AF70057212AC459B35FA533A465E7E8E089FB85B7B879211D441B225F26F2E5554442C61B5DC925EA4E1626E7EF6D9756A04AD1F124AA9CBE1CB4EB4FAFA8C9858C2212FC3D892A5BECAA6255383DD22F6BD107F0787278563F03A587241295B06B5E25C19D0227A83A4A649C01DCC1C13C0F61445835E4B1E4AFFF653164A551476332FAFF0E3FCC1660255C245C063E69D6ECCA63DC139D5824D0DCCF3D095FA788F8071B2B9828D9C4206CD33DD6B48CCEA33AC44CD9406A4D259737189037CDF89AFA10920D0C11B1163C1C7EEA7DBB08F56074D658418309992837AADAF0013DEABDBA92BC812E18101C8E5D4904755C4111099274B7B6548CCD143B2422CEA59A668260D9FD19FEF2E98D0E9610D7AE635BCCA8959BA36C1E948FD68021019E2B246D21B2D9B70EBA9E2E1724D4E91CA4945C25C1FAC907873439910BD49D2000B2E3802D7B42C0808EA25FA32BA157269BAB1D4101510EA49A9CF599389C5DFED142C62C10B84FB19E2713096EA2242AD3CAA8319BF243790F90E5CAF45FA8A1ABCE454FCC2AF044DCB93445CA1E358067EAF8E13D08CC1792925701FC2B6BD26CDE27A4228C14CBABE63A6809D50C6791E2F65F31157FD37268DBE1F69C2A17213E1AD685C7625E4A2AF6BDDC3F806E744E9D27438C1087B4AD6B85EA5EF6A1A316A78A4380B52F0A39EB429C3480DEEEC640ED3ECE2EC8C85E573AF76D98366D92F15D1D6E34B43BDBF67154D9E60CE862F2839463349574C4F87834F00B00AB401C82406941E049D5517264A28DC1D78B2B026E4376891532C65E4F7E898C4E5B14E29ABD97146560C70BED49C83B2348F7BD7D595CBF22B63FD77DA64111E03EBB2CDDB109CE48A85BCA039940540A4405918F69CA0988F293A48BE0BDC12BCAA52D9ED27F29D986274863E4B1B65BCAB83A676CDB856C59FF5DDAE25570F1BFAFB54F9B1FC27F6DFD6BEF3763874F6E86E050AE80DA80F5C9BD95FF96D7F2177FC76C0124E1AB08E8E8475C7B3517945C71B9478770861BB7747DCACB685545C86A6C7B1CACDF4028D71E06F6376947886067EC1CBA82EC4289AD0422D3E007EF3D94E57C9FF5C45DD602095CC8A46579A09EAB6F44835C50E91A11703C172AE07ACF9B9CE2E7260CD7309F16D2D55D9A1E4139FDC182BED930C864C6019B8E7618DB31253F83BADA9B6DF2035C8C184DE1130BCA8D43BCACF5DA1F9DC1C279C934FCA3EA223EAD290CC7407D2FB4B3506525887B64C272BCC55464EDACC6A1DCEA21B710DD7D736A8E6F8DFFFF810159C8CC63639E64EAEB0C806796A83C9999C740FF300D7C1DC98827DF4FAE7E3C2DEB0C294660681CE951C58CFA46FC916DBE3B7E3BAFEC66F213C4CBBB4E3CB1987BAE923AAA58D5457BAE1D736E308D54E32D8B5F17B7DC7D47D49F35AE98986F38FB2BD451759B4A9B96F38B6AADC6B2F97954E6ABE994A682E3AF45A370DB6F39872034E387CE37143749AB5E5F61F61942A68908D2F659300E5B1FACAF81E44151EACBA3DFF5A0035CB8A243C4B4DE035BF424551DC12452C414D828D293CCD98C70133D64520E67FCA7C9F7491E45E683F23F7C5B71B265282FF6EE5C9DAB26B0EB61099B62F59B7180C7CF264D82BA0334C57DB97CC5EE79AFA059ED7FE7C477F044CBD9481B4970F9F920602B14C889E36D44612216F4B63A51B7AC65956AA028FB3B160B9E707927A0D0717C29CB674CBF83E988EC732B5AD777B9957010C96A9C8AF77B15496757CD4D95424E8EB20A4C1D765ED43DB2F3B62B5405C846753612BF0ADCD5B36935E1BEC69E66E1862C7D81DBE646DB2AA666CD2F7F7E481693ECFC24D4BA2F8C2803BD3AFFE418C0E5ABD34EDED961323694C40C61DB6DF31CBDAAB726C731FFCB5FB4477912BE6B722E0BEC7165D33EA422EB06BAFBEBDC4B364CBE508BEFF09845B4B2170555219B7E074DB8B829697BFB4315B8C9458012CC4DCE345AEA7A7CC3A59BCF4F39E738A3D0FF7BFEA9980581183A066BB43DE1830151ED8DC56B7E834B62512FEA933549816C072703141D3A5CF9C568FEC825F6CBB9E36E31B910D7267F12C7D3627BDDA20534DA33AC3DE5BF1E47D643A5035E2924916571D84FB5473AD340F9DD59A3D0080D02274F1A5C1F5546806F5812FAC00BA52C054BE6B0075B294E095C3A069FB62E559FC36ACE5197B1588856117188DD9D33E4F85AFF5DADC5911D6CF4D0ADCFD780C99AD75D3F89255F4B445DCCEC47BAFD2DE7D57637F8EF7720A95777F7FC9CE623FDB942DBCA1BB96F91F74D24E63A75E3EBCD49B5FA7E82DDB3285A4395F1429282B227BD9DE9FA318A71B57ACD2A48049B0CE83E2E68D62576E0C4C54C44E046721B4F2E3F10EFD40C781ED0A3F8BD78EDA7877637B4ADD2E166CE8DBE06AAA84259738BA236AC238220F3DC3BB88B446D82CCF7ED239FE5A6F265EB3372F4FF5559882E3D2BFBB99E7599D79A54C4465102FFB206FF02CB87B1B69DEF4A981BF88D3000FEDC476D35D2F6EFEE27975C905436CC8EED8D8559870B8598318C956587FF31B0549B87CE8EA094855B5293364ECE7CE0FDA3F8BFE1559A6415FBD6667B4B41BBCDAB5FD116C932E54C67AE226F9B9407F311B0C8BBFE5BB46D9D98C086E0EC0A7503D1E2D66FF1AE6CB3BE3C090702AEBC5C476855B6BE9F8D354107789E990AA3B72A724950643A79067F799EE9E9AEA5105F1FEB90C19F751B333AAA6BB4EB2D5C26F2B5A7BF1B7C32F8175633BAC16FEB722C8AC667CAA9DF19096CE06D721DD5AD59AB5999183DE9E5525A6B7F8F249F1A18DDD1C1DAB5D14B4FBFFC0462855382D5C2DF618A0450FE4CA01C9BF52FD9057042F3C69FAB70380CE39E683B35866FF660AE682A105D0BA3716C4AEEDD2FED28383AE8F44AE3842B0971E810F36FB8EB0B79241D013EAB2BE710832F9C427ECCAE5CABEE727A7085CB6A68BB89F3B3F49CE7315CFB9C84E3285B6EBA685096722D7103A76454D23CD520B98621D48655CAEC7F32B5AB3DB2030F4345B646A585F18A900F497EE01B19E607384E88F36CA024000ED587FCBE814C254570EC22DD59070E1540277ACAFA55BAB169BEFE2955B331D824FE8CFF9FD137C349DB70D49771F74DB24820D2FC240C62879C8FB8106E21E1C0B94F5054C05E60F977EF07B7414DF8770F987694F292734FE1C0FD31303A38E22815E56A6F5C4BBFB5C48E985C444799C0E94BA70309D69FBAC0B10C1D16BE068763FB89E7B9605A407A9EC9312D0F92CC1A47EF40BF46239ECC24D0B78B39538DC0C66CD614B408D88C664A2D756CDE90689756694CC64DD0D29D9BC239E6A6C5CB3E9E0C9FFB6371B13B070C16F63C68E7099D32FE1FCC5ED20BD17725BA1493736123B1D5C59906C95588192520DBA415C17D4D7A1A26EFF073D239012DC73A542C2DF29A512A6457184C250EBDD1A545D3A5EDF468BFA054BE03FF2AF01E167A97D6F6C52177BAAE71B63ABB7AADF472FD2A204341D12A23F313120CD9D50D5207A0BBB144A26FF3E82FC593E39368D00FD800FE701D1A52873B68906AA3DD374A1D4EE27D034BFD2890615C24B25E5BA47CAFB70D576F568339B5909F9884498BEA4641EF0C4DBF8F0D56E0749D8C3DCB0828EEEC6921905E3EEFCAC3DB0A7F70D5FAB0EF2964CD6B9C202EE8CE947204786BAA17F51E869A366B53DAB95BB03C3FAE081090B536B240E40BF0BBC6CA91C9831E8C46BE1FDD0B40A560160D4FA710EAAF98263C8796F7F0D514FC91DBDBCC6AE70802C126EE462E4C184C2EA316F8078197A296FBB18C204E6DAC650C5758F18FB92448AF9281512E21ED14B7465E827909E934CD5B230C51B154A84074503A2C0474019972106BCB6FF21FBDDDD2A216EDDA5D9285F604BBBDBAC0FD4CC12041ABF28AE8747054667BB9C360DFC90B7252F5029592E51F8E5D7509C677E6853352FE87E945F40BEF6FEF78A5846230C68137E4D2F5734032DBCFDC7F60C456F35E48C20F908CA936C7F708D218FA5E3AF5698ACDDAE47B150E1776EE2AD3E777C290A0E39C25300C05BFB671884E4951BE0753473C1984FF2C5E35C099AFF57D689021624C1D4148B1C7E65FE3A87343D80F9D20514F08E4F2EC01DD6EE1D3EC1E9919FC9222615C28A49744FD77EB2D3CA3A26C9B1416818CB484AE7A6AED480BE4D6CFD5B0E77B7B393568D19599769B7B485B2ADED0F568EFAC7A9DDC955E0C68D996EE87A984FFCD4B6B097F70392F218B10C8F9B1E57BC4567F35E748C6A43FE742CEC898268B1055DBC44CB9BE6EE262E5A33A9DF3AAA85E86AD359554A90FB7C33DDF1B91072E4A66ED5B8DDA4B03B5CB3CF1D55C032521EA599F54D6774337512605C223AD9ED77997A119CC92B826020C0D97E060DEC9FF02C92FDFA7E20706281F927105C1D8D4D2363BCC2424B3887EFCB97E495F71B877C7E58672AAB0D26D3DF04540A5FEC58BE140D29640C7E74FA18957B575441865D317048764B61309D20CEC0C"+ "8C1F0F14834390D53E370F974037A24DCC7752B210A6387513EF685897E846E14ED2E6F0548224497CB32EC2CD4A6CD1FEF802E81768B7730F6396A88C0D886DF9DDE1AC09A146CE5F691BC71947D8AA97F565205540BF307EF0919A5084B298"+ ""+ "1951500245BAC3685E1C31EE20418604A0B693FC396F47EBA26D6A241493285F9D96A8A0972E031C07B4D56DF2CCBB688A1733E5882DCB8092EAB8055B8F0FCD8BECD771B1A95201FF9A8EEE6A14298E4165CC0523E9FE223022E92BDD480429A46D7E0D9E82BC200E8060C9260C05847F5B9E738193749053CA4CD71BBA1D11A2CCD3D3F857FB830836BCF0B09F2F8C4BE2E86C73F8353F65C2504A5A6F79E8503C148E17477BC45A3F688AD9BFD7CCB82A1576FD07B715304CE9B5091729BE74F6EC06646FE2E814933B3317D2E299B6B80E8FAFBD6CC710E3B4A2E913990B2C7E08C3D300B2C9F62F3AE322F274A85E3DC6F2307653A083B3B9134088AEAB5A97B5EEB565C12A11B19DBB299DC25A9CAF201231BC06492AAF381E49FC840A3FAAFA99E4F48CE06C5BCEA8D9C069349BB7ACB98B74AEF15B998994FFB80997B93672CBED2156D6B2DFF819D9B58DFA26D712D09A7DFF5006F25E5D1CE889FEA8D73B309680C6C473861E9A309A863127EDA77F3AB0B8B76AFD113555A2CF24124E1265CEA11E4A11E78F58255E3928B12F0223B88202A56AB2174CF0F95C43DCA78A99E8571FF088159CE6511278ADAEB03C7CCF5280099D21555E8D7D96C1CDC9F9CD8C3DF20CF5F0757E8A978BADA93D336955847F50BCD794CD784720196FE51EF4034F59D86B6D64EC977695EED60BEC329D236853922F0FB3216A1A5B1C65F27B2111D8DA01F2E7EF52F9325600F7F107B3A78CBCB7922B62681EA86632EC0E249F024D2A91F9C8A004BAB1D812ECD1037BED6122EE0989D7011D00398A4D9850C5C50F9FA75196DFCD0893CD6EB0C34377DBB83A5186046E803A9546E42785BDE21161D2DFA517BF3F6B539452848CA071C7AE57F15E7A6E1AF7BBF83AFD3131594C8829A870E738EC3AAB66A1652967B072C2B0917542D8B12FD6AD86714CB70EAA71F2BD70B3191A13E07EC8C04B8652E3420BFDAA30683EFE94D91390653866F4BA0215578FA737CFE07024AFF1C614A31A34398178A2D6D468871CFB0865591D6A919C9AF45D7C441651A754016EC3F137240237D0E856A2A883767F7907425F4B3FACCF3D102A728EFCEBEE43C3E995E54C40A2203E3FA0A4567DCBD8172B6BF03F5DEDB47658674AD339DB2249890648AAFDBB4F630D23DF6BC0AC13C7B8F206ACCFB40DBDDB8A350572402B62C81B1D7A01AE847AA9017828A64662C81A6B140B0390F2BE3E53592C051F268E842B2A467EE380F3E79951CF8EE23E54F40AA90CB2F18EB3B763F968A5453C65B85C07FAB395B002F16C785419732DDC8B00B720C776D3CDB890A1919ADF11ADBE680FCC1D8FD4F5308F8DFBF173FD44BD68052021C8CE8BED130F0132E1AE2AA6D9444C53FE676F85D4818F39707208300CC51343D0173BC100A388FA6233123D89D3DFE6D469110EED0BC3C9AF89E8FD59736A18F69743CC120E4471210E4B2F9D1E4A1B94BBE6D5FA9C3A4980739A22EBDD0F1838A3BA4137E1C344A887F70B44302A85AFA432BEDB6C6B24B9E6339AD83A0948A7625CE4BE6A0CA40685DC7EA673E404793F75FA440B8DEFE7C5784B21C814BE5CBD0E0CD5B84F5246133BDC18ED711E88FC7F18A7526A9BA62670C90FE9FEC6F293C6F62DEBA57F1AB6C6D07FE55334F0250B11624139567EF451C36A1C26691C8856B39387DC2AB4C8AF32DED57BEB84BFDA47B06503F9D34C9B36FA4521EAD4253A7BFCE0B7E96928F78F3F089C374B8735EEE30784EB1C47242C447802DE5C8D55EFA727787493641CD0AA169B18AAC14B2121595003AB3213C02903DFC36BE579AF3DD54E6F7A8651950F35034FC5C36941F5157508062C72AB95609478196F0CDD23C4364492000581EC91CCABEB2638E3D2B4877315B1C46D92BCFBD6C6C875708DA86A7C41311CCCE6A394F771F5E3A1486D5D0797B09EC838CCB0006BBDE6F3E0692CAB845F95E5CD341904DDA7F072C510C5C850ED1929594EED2B3427F67F4565EE030176C57371D4624FFE3096F82A83B7D8A5263DC697A6330061F4C56F99AAC21FA126846A24F09E50116AD3E622C1F36CF384CDBF49CF5EBCEC5C773CC6FC62C0CD65AC9320416319C1F718C3A67285D4AD40510C63E3F6C511A1B19647925C2D383EB4A774AFD80CD12C5698C6A1AA0A6158D888DA7FF26AE4E70743802232BA6D6477F952AA97F65B16C0F5B7FAC66EAB502DDDF4B4E0836F11A7521E3E3D8F273459B6AFC42E2835249B6FF6B7665A07120EC588114DECCC758C2B75321190309881C343600D0D9528B03E0044712A9F94E6AE6400373F6923238DC8038DC4A1D1C397D211361FE683C96DE856A8C9CD0F7135E8A217EF1CC498A0B3D24E946FC2F464E0E5352BB7F26347F10D72A2970DE8D77D99BE62AEEE801A10AF431CFE6A69FAF6BF7F0F400C598FF2865E22C2C3085D65C19C8DC1C09C2565F567E7B76EDA51AD53753E23483585153ABA59625BFF7E1425D9DC87CF7F6BBA1C7479249C187D855B4DEEBF106C6E00EF40B09BCC578EA9BA1ECC467E49CE6425973079941A380BB8BB3D1E3D734E7282E03B88A1FEBE0712CEA5625147FBAEB433E804BE741641A313E47FE64D07FEE0A7FF75E4E2DA7203B95E0D11F4F04CF2D1F13AFC248EC4E068F5D926148E624BD335C825ACEF5153D95865933EEEEB49C1B82681197F1A3DA24A6471543F0AB1E5C63AFE1244CC19222CEF14524CD982983391E8EE8CDD42C99E8D5CFB9E47447C5D048ED3E82AC3042D96ED7AA98C51EBF4507382CB111ACECB11D67D58D3F7ECBD1562EAE432D5D48EAEEF572391026506DB7CD46855B36A76018BCF970358FF392A0300FA38F543123112444FCAA445F17604F23DB8753BFCB2414E93390881527834FB303471E38BF6FE6AB773C4B2F60E2FF286F24D339A07373EF2A19E70C20FE5C305BCD167491BCA15FC934041263A62F4D4534D44EC7BAB6A79128556EC475EE99BEAE8D147B77640447142F59BBB53830B0229A22DAEDCAC05E29C9EF793BFFE6AED1B80340BE5B01E3F973E19A9309F68B1F30A7BE4364F5BA7E4AA21AF8DBECDB125FCAA2DCAD014A09A9087B6D0D728225E3CA08844B7DD62389BBC791D9412F3A520632326A9135DFD84A38491904430D0ECB147F1FF259CEB69263365A9D19F104C2D38CCB8B793F36E4AC9826BAA2B4E5DC604225E3C1F217BAB088DF18B043102965142CF2DF7B69D066774A91D628B0DCD7AE88AAE3A0F8143E77E2623C0ABD54A78DB5DB61E45B05E23F60ABFFDA34E6544D307A0C95773232A324D54656D8C90A7ACB5B9BBD0D3DADBE8F7FA2022393A3C45476C6E829DBDC5C6CDD1E0E4FAFE0C2A2D5D60849CACAFB2D4D6E3EF0517182B2C3F54577C839EA2A7DFE1E5EEF6FE0000000000001529374A"+ , SigGenVector+ "ML-DSA-65"+ 31+ True+ "AB5CABAB5C21B2C822F0B9B777C2368A89400F43F5E55F7C9263F985F3D21950787F6C89965BAC34ADCB36C46C81C45BB2A97A332E98D0FE9311A992A631F9D3B8B78F76F7E1BB4434A11362FD9D9D80C7DD6BBE6C06A5D80F635262DD8932FA1A7A5B5B2DF3592B3EB1FC7C483A4913FD98FEA25FDABC6488B6F06FD317283B14211703856180743131443863514707247734683531213288267203825188367650253033010245132773245512164055080418118558670651842813162354774387702523173555722023333318736338561136722838325612543303203125262618583380378123810785071284574673273362807333201836323335100351252046422406664246464022403363876337713142068301280017024462613008871685850447422654773856671440550247737614834867052541442350764640801302803557661822560380837144685726812260630284716130025674200750228877060667072317780080064180766854827147483667145776600118564604525704242004645600145066550307525132815561665870383704218473060657655835006403322475026884457376214661242802846787353147638277446300372376357756637004874784512753036872120436008171211632025840115482224775760013843117021585011451017044443783138306253462040241073763413520140002763605562252712755556287101007315167718673488502883773886888851604057350516157074717027315210787074625886757258833683007717740203166631557356472038447070286616653057341513426226176277575631386513371673510016324448782162250640772235047688201205574623017834676767243334308778062313104778541125612236261326266476227503446662864475738320575457568351716165601575281566112574746635002137326628503527078340641178863260660247611520743882658087061621017442722554722078018330314446115860453330145158321086187728214805555333230765568507856264835314005712501631506268325357220085347831614278188118248517083351760827377527874472153622727218732477062685630260500314826347214557282771682717140377380367525363238886707356080764521070026621863710300777041055245524147005231848572624241823462088050170583030268005663840457653542023818680383040780823236080748714750114520376272314006265656853766677727620843344012833126304433443670510022200847682711077578008241700385261578578537145855143540065185454253678525230742370061425207577423124878314522385765145820480122800164015447632441040423233602776058320154550062355066417517520613826187341803376572713184087136648210740244173030846412001103000873148355316781287028768507665120733888813770123264148332323841460123053610115763850577313176218431478238315358437425484146127588236132166765543574555712056371643767064181520207444313043233806463042308423442254811706817874661562274313577038060758242627064448431473447552260525704704724761188717331010717384238020076044771878138074082645284286572374337762466220263046614487234488816441183660488724608487607280008880481342078415731866733547824708507757707784406050033224018751004312507257574283816483017741538305710661710861226045787784510050332512313624508736526164783220226746820623370370264855282163738370715310310864006027575224138274567341783676248575650035115843414615403753587343467202080842758400841250154672384462040881335430770631831435868806246888106284601022478000835678170147030017458555567318068363537378183058005778280886837060626B884C0CD9D5CFD660BCA5015E9229EFE1917E75644FB75026A3B9AFE569A30DAE5E8A6286E624B994A4288578A58B542471A1A0583A3DEDD9CB058ABB1D8DCAB1E5AC4F11D7E0F8EBF2753CA7ABC63C29C63A278226C57004E462CEB2539B7299B2D51C4D588146214A91C3886FA73F6038502D95008D72F5DF8CBB1414E33700BC85862033B8B36DF80EFC28439397511F5E27645D0508B75491B2EE4B4101A189BCDB4783429FD0ACB888B35D4F5A6C9DE49E258D26C57763A78171566115F83685E3E7A456ABD6A25B57335ACB92C03ECC5276E3ED25D12CE50C447DA5E21B1A95EAC3D4DC9AB97354F64173398D699EF98B185C57DD71E9DA1AE359B300E2DFEC31D3B3AD390F3F994AA44F5B198A3E8007FDF627685561022F255E464520755AE436FFBCA77A22AFABBD57E9754891132B23494F5451CF6F97D4697FC58FA67C92C872ABD51ADA6342B58E5BAB451F08A440946173E154E81F6C5B62983BAFDE03BA39E50483374D7E0D8DF25B0B6CA35D78DC462DE4D0495D3C4F95C7750E5566684A3402364E1B0B5CBBC1DA8D7FCF99BBA296A46FCC7F1FA12AA663C838FAA2BC0D85245BB7B208E2463FF6D43EC7CC3E3230A5A658016E8A04AFF662BC9BE2EDD8F32C09A272E45709C91AD2475DBC0D128BE3AC06062B332C3131004B2F18414C11A3242AE1EB49C47160C7BDFF81510E6D31865FCBCC02DD6559CD6BFFEE54C44FCEBA57A77DA061636EE37138C7DD9F40900A41E814545B10FCE93FF6BAF5B8AF316D191216D9D460D4AA8517D5C4991A0A501D5E33DF909BA73E4A2E5FE3D06591B3FB45C51C88BDF128DB80623E031CFED0DA0DEFEC24F58F71FFD0300924079E0C9E40A60D50D3F271977A8EA269A6AA5BF26DBB02612F1755D65736381946B17F9529D27B11BC99FE6D010CB4849820F041928EF8929E897C290E9F3109850E276E026149437E8DEA6A3BFED7AE34C2F207CC2E480F4860D4BB0DF49B732D6B4B1BEF49D6196512789D06D2ABE40BB45EDFD3941BB8805B881591FD17178C24BEFB06F9F061F3A0DAAD19BAF4A1AB221DDD7D133B40B2840CD1A6BDFEBD6D1B681908E72AB3A817E29B58F062BF9F4F46F21EB71F97D6F055F74971F7AC4868AAA389B553972F5E09EE34A2CE319DA70C8111DF9751EE2C51BFB5B129DC1D23763FCB4961C4CBF1D67185170307D768942F0CE4051C556EE066E203B7914316A1064DA98660C6775CD5DEB9BC11DEF90A3D26DF0ECCED4A92AEDE273D2A5683653E460A808B7006C669A0F0ECA3534C27198FDF77B0C982881E5F158B2F5A0E5895FC03F508D0500ECB2259DF9A65783327D086F857D0DD815239994DDF65FDA2FC009535E5957E792C02D59EBFB0808D972F916A20CC8878C1F50A80122E2B6F1BD89BD92C28FA1518FA4C3DA8C6063D3E7968B677F2C00516142C4496C89208625856B1DF9182C50A07F64B66BB6BC75D945BA00F4F272E6DE2996A74C770892BC8084F11921B514B6166953B35CDA2F97BD49FD0F61F08F4B4ACDEBFE83C3BB5B31310977FC6A136FCB6374ED14E5101149464CE4AB5FAD47C81F23855BBA8B638E71B7861E13F828B228027C08CE4293E14592F5DA055C9D4F9383983B1FEECB4A5FA66760E9CDC9ECA0594775AC54E33E73BBD199F681B15EF27A9ECCEC9F312D1CCF6DC5CE3CC918345472160A73AEDDD5B3C83C53ECDF2F7E2A5A5BB9B0DC2E33D526C89FED05687894F3B3D7960B842C9212B3C209A221AC2F831FF34363BFEB965AE5F6E56871711C8FB6EA40523BB928079778031CD50C8465822EB1217DF12F8B604AA433F14BEEDE86A259C13B6D7ABA3DF45D018A77A1427D7683EC0CBAE6D8303D0FC7411ED3D5EAB3566CC088AC15F70412C4E3F43F500D08FFF7AAB7C477555EC7A4E566083FBADDC4638B56094B8FC9D9A5FC9F97815FD3FE79330ABAC930C93647324DB652689AAD9FAC5E8DB86DB191B23195A8FC25BEB8C7A2786C2C9F7B722828FAAE2ABD419932FDAEB0DC6B489129F5AB66166007C5DAC004D9D7CA1B8FF5E866AE738AE10513AD4DB6C9101A12D4663075363855A32E92C4A8C03167B21447F025424AF620C7F485BA5AA55DB0596FE4370E23A92BCF4101D6C6469EAB0FAEB9B1D0EED2754548F2892C3397416CDDBA9AA119BC635C1037EDB3276A202B1D51C0A2485922586FC27A02AD6F1A5CA28D98F8FC6BDAF164DC26C05247935F23BF8EF504798E30B154A665BEB837BAFEC6A19486CE2DF6DD544B0C20688A37DE9E5E0B3BF93A70A865AD0635040E358507B0E9D43907ADA991005496A1B887EEB5B8B03CB9D22159FF7AEA0A8901FEA51048A5C975DAF858D142A63FB5140A424ECDEBF2564DF4B56D9D4DE86820DF38246B20F65A880193560C484F459283F0783EFC9CF8969AB00A619718455F662EF800F12CFFCDB252E0C29F8ED425ADCB3FB73E255A0CF950868F1F4DA6ED9DA31B0398B4C98A31043E1BB88BCE90F877DD45E533A1C0AD1C62D40D1441D3034F6E7F97785C7C348E20550E6565D627012C05CA145B670997AA8FA0F68EA6FEED9CAA398D9248832309B0A5DEED6B220C0B1FC04AAC894B5FCE311DF67281FB3A645AFBBAF2114F7C8C9B241397CC5BCEE1BBE0A1B2FA9167D58A39534BA0BBBC828C40686EEDD9ACE35CEF2AEA5330E733CBCDA7E9D869615688BD284EDA1BE04EF7C5654E2E6439EDD572756219E4A5CA3846503FAA367AAB83003CD74F3BA470195B2FF538BA32E9BDA8C78AEF0881DE83ECDD95DBDE9BB95F5B4B81F4FA30B341B4EADADEEC8E33B772F58D554CA02DCAFECAA81A75330EAE5F59519E0C1524F60283C11CD2303FA233F22767F81DB1234DC5E590457E9A54ABC76970B0276183671E446A470A8E66BF3C672C32A3D1097F07B3897BBF2CCA4060BA7B80744DA7301D50F834A1179468AE326FF59625AB90F2DB491B92E8FDD82A278B415594FC09A3DB3E0F2B7BB001B400ED1CDDC81B3132AAFAE1FE662A1B3AC4ADD170E7158708A2E62DE526394DC73E210B28FB3F0B5C23FB2E316F6C385CE4219160C013354D702571082986B2D2F0CCB923F4098CEB6F332209B4DC3B2EE888239E90F88E101C543A3A3C501CA4772FE2D704098D5FB21BACD34B4F422A4FA5A7BF2EA3112D5459621EA1F028B107054DAA4846ABDFEE0D6862C7F445FCBA1B45BEDC8CFBDA29B5309F0FFE9D74172EC4CA52FEEF7FE4EAC3D5C632BCCBEAAEA53D63E372046EE2C6434597B15237B004342885112BA8DFCD75640B00C0DF3A4726532B9D7326414C5DE1B823D0A2DC14335935866B2AAAF2959DFAD4275D88452EBA5DB38466C289D316C7465EA29C63F89270FC66E5C7F487DFA9B840B483F8EE39F9A8F717B31FE3299C37F7BBA99F4BFE0FF8DBDFDC90AB0267582B3DB78B0B2F2611B1FF93A77187F02ECBC2C6AD260707F9F922B80FDDE0FB196FD919620A23822B4AE08F46B4C865A3548DD0CF90572836DF5881840D1812C5EEDB"+ "653983BA2B10B9B29DBEE1BBA6B98BE849D882CA19C29E0176C469498734AEEDA6A02C930D287649ECD2FA198B131084BE91B54110D97E91CC90E2A15AA58597D76DCEB73E4A79FE1A92A497546B925965C3EF0CCC865A75576ADD862690DC906A4E841D6D7648AD9E81C76B4ED5FCA91709323636F0E65C8B3B2E86F81780E2FEF26D5015A96397FA25FDB2E2E6B085F0D59EA672DA82D89E99C7D00E42D39B34AA51FF5BEC7FB5B0A411B9C884F6520C3CC8CBB2B48C4FF1C93C945F143C0BDEC537D838BCC2B9A1515BFF2A74A1DB2E26E44B4DB2606B838ABC1DA620060BA43B2D69602479F187167E131F0FD9ADE6E2B56620E1470C2F97FDB2AE49CF1F657A0F042288BEE48D052C95BE9002803C6A6775406B86B34050F21E0906F6B3AB3DA4219016184D58F6D2A694875C21B506B9A4FD7B5C099B9F1A3AD617DA5FABC01C1BB2B787F60E64F19D7521535A8AC835E366A174C85856D9180D26B2499347C8E8E2E7EAEBEA1F467243B56074B8286188616EA773A9FF039CA2721655D63453D5960B83EC5096510D151DC75CF37B8A237C887EE61D5FF365D33A813C2D5C8F0B1253F0424CDD10DF54C4D29C64D6D724AE43B1663F6529874DCFC96176A6D2E8BA2C32CAA038672DEEBE4C47E641359D9CF7E2FF444A37451E917BB30BFF9EA00CB9DD9A74731C9143F5B077AA4818C983D5611447DEA298C6A73DBCCAAA239F36F690FF0AB872FD0C2ADEE6E2FEEC5C4D455E2E9ED4979CB9E852E67028BB21DC927696A4660FE10300E8DE7B5B80198EAEE585CB0031AC8C3565199976FDF409594163187A0A32CB2A2ED08024D11E02EC5D6D1DC47B1EF58E10AF99DA0B32D168034D798306EB1C024A471E273A9DE4FFF4A4A9CE5D0CA7BBE61E66273515714DF8BAC2B61DF791073C7BD022DADCAC77D6CEF07C80C923C503977A97F2587D1B69F804E615BBCDE5212A9881D1AB5361B5587FB331DB46FDA1D772EBDCB231D4FCA2CD2D3E62CF56812157C6C7C59C7683564A242C7ABE37264249E5114F0679F05F466E21D7B8007CAC2DEF126C40EBBC843BE32B22334211EE04F6624C02AF116A3307B059FABF39764660BCCF262B39AC3214FF18E32466F6C71B96F276C30DE9BF39987997EE983AFB0F469A38D9099BE8D8101099D1A3CB80474792F935858CFBF5A830C62184422498E987F0E73D0FAD8F27FC5008DF68695E9B1C30BDD2E597378176E9B72A10D4601878390C2A1B637850CC1B1CC47D6B92E8F13259A23D1023941A02A188FD5B431914854BFA5150CD76D3FAE3082B51841771B58653ED749EC3335E7AE9D94C4AAF2348150283049D78864717DCD356F295EC7AD5480B67612A88CEFEEFC853BDEBD6B75F1EE7709246FE92BA4F9E2692324D433C973721668366FD73154B9C0A4197B60265893FD929120C6A71EAC827FBE4498BD39455A1D91DE85EC19AA4536803A517C622767CD816D1ECD9FA836CC4BE0DBA9E050EE6AF7D24DC0518F0FAFFE665F5568CE6089EE890E17A041FFEF9EE76B194A339CDF8FBA4AD2A156469F84CF1F8F97C5A3B70A15ED8D207E1532A4FF16A81E847F0DE5E72B0C856AAFDC063E85C574A4DEB58F899DB94182B6BAC6D04D41225CCE0A849424668B334097DF2F0FE728DC02F23048F4EC8F03921A4DA64FC6027EFB1615AEA187971EFE84662CCCD3AD045116964FC01D72E3B9361BCB7579C7A6B49C02BCFB79E2A932006677C04839BDD5C310177CB71AFDC301288430D91F54FC3664F69182D7712E12F6EB1B45F4E1CA5C9494CEC3DD1992943FC2FB10BC38EB71BDB529F40A7A57EEDB59D0BD9202D41394ED020AD6A9EB8EEAABFD2F6FF96CD798BD48C8F56F39969FA57DC8A88D302CF25B0A1900843EF67E2F534410F7944912793AD22FC66EAF06129F05A069BE4DE00E26F4427F50D38F789F640E4FB92EAB1E3266CB202B194F5C985167BE6F1D936183CA263F6744EFF5F25AEFADFB9E73F6C41F7741AF73C84D33B9C430D971C37F1209EE2D257F2DA9B86A93F36673065C923130328CA56B61F74BBE94FE9860864E461101FB0637F441F3F70AE5BF88E24EB2C93A0811B5330BD9335D737FADAF7579B2FF55451F6B7B384441E8774B2A9922049C47D600013576571A61A33C8645DF8220EA2204535F01100812FAA962E394F0AFF5E38EF7AA9D1169A86E6C9B40673C91DA3ECF99D20A8FC6DDAA9E46DFFF78DE4E8CED82716C6FA9BBE66DDED7FDB2B14189526C734F3CE365F0640A36CD40064D448903188E1B3CAB32D90EFDF5C741B4D5141005C612C78006008CA90AEC84E55BD49F3F6786659968329616CCE9768D0D052352356221D3C86AAEE3B4A4DDD8EFDBE76C445EB9F8764577E7F69FFAA4E4314DD8B574C13330D38E218EDBEED02A8CA6C78E5ABE5AEDD8C6F6A55"+ "32ACB4CD9FD5F046155769897EDB150CE0"+ ""+ "EABA6286EF3BA82B537455C41A33532C9CF71BCE65634BED72D74769D4049D6C3516FC0683DC654F0395BA0CF857C9D025C40B40DA0E32C5558BC1DAF06674B2D6779969BE02ACD6522A3222577E1997E2C7078A295C2DF27CDD942025B35FE65DE175902D62BE5141F22AB604C0B8A09F3BEFFCA336E9CF64D4747D8381BF8827C7D3CFCBBCFEF58516EF7E9F9B0B58E501B0B3C494C171B4FB9044C3D8F977CD00244C25D594721E7824406079B9502161DC25D1FDA0389BCDC2375F9974DFFB99CCAD95CE1B1099CA0B11B638AD8ABD3F1F708FB31AD29DD06C0594E4DDD14E1C934AFA772B3727A20583015DF38D09DDBDB71E53202669A68CA1D5199FC4384B76E98D8F731D8817FBBAD361D73EA4B0FA2204A2D0C573D4283CA41A2F1B7B2AB43E61C80B1D363ED35140A18E96EC857A0AA3D4FEC11823F385578CA9D90F11A59DC471F4D399B46DA5454C4216A6D319603C867DEAFA731EB81D8A45F67CD53719F7B289CF090AF5DB7069B6DBDB014B3448E7ABE999AC29AC7943E434E70D1D9BF0DDCA7DF95F407F0751A9BEDE275750F8927988347EA7550FE4753A7A85FAAC2C3A39A65FA20C01E53851806494E29EED7C0015AF6168CD5E15DFD1CB394E2DF6B46B496F454404D1DA659B09AB797BF48839F456C6F7BD594489811E341E7C03D074D8B84925A47BCF46044299D1FE7528E7E78642C55007D10B040E208AEB4E3A38C3B3CC757A17F160949D082073BE235D0817A51B68C2FD5442AD973C907A3C44750F827A6473ED0C3CB016D6D82F61D639AD015964EDDD9055CAACCA1C71B7D9AE5ED8FB3B3AD01EC6B08EEEE87D6759284E60917AD337C5F615478EF78B0A6EDB672F2BDFD95906598CC05B21D171CD1C35C198AA5CB931D7F769DE115CB6499218060CF8CF3C4695F17C3983061FC71EA223763A0D6544E149A112EC36ACDE5DB1E0F5429E72F417BF550AD743D77C6376D6FCDACD97DB708F4947FC322DB62B00F95B2BBD10C247A471409287D956916B2A79120ADB1A2C403C9FA979FDCC4D858B61B15FD589E8FC3EB7406D8BF67C7CDC8A22E110FA50E1EA1905196CC8AB9E3E434EED6647512EBBBC7F64F786E37E7A1AE27C49FF34ABC5164BF25ADAF66D5394F22483617EC2DE6DBF6A90D4C1085E92A8EB8E51BA1E03EAE41EF192154BE87D8607B997793A0D861F7E146844D989D3AD192470535EE576F0DA16BA8FE39786F7169CDD05AFEB8A0689C7F21D20D1CBDB572122DCA4AF4C6BF4F8CE11CFD632FA3778BEAC46A17F9DD1C27992CA7B8136E9C224F4376B8C4A042B2282079C60251FBF7B6C6E49E82F6DAAD47F917CC20A693F04E3F4F399F7BB1E3B489F6880AB90B379DCEC0162176962C77649C8B1DBC23B487CAE5B075EF611EE29A995A2F914234D973352095C4B41525546E239DE18DED572C64FE528A4AA6705DEE5046E6953D76FA4300B6286AC91947D3C890DF7D7FF57B33740771DCEC8DAFF878BA9DE76742AB7D07C91BEA969F6B584D15E5A322BF63077BCE4C40F1E93651D091CB92C0D2AAE1525D7FAE70B1A515B351D8229C3849582043F9C614E865A683B29B5D3C1EE426F052EF8868096408345C12C5EDC7253AC8F3F5CB9D70EC096F7ACED9868BD3ED2F521CFFD6B0D2B77FFACC261421CCDFD8B2768DC90F23DC94ED55E96EDE45015F4FA137A3661AEC4E35B2E591229DB8D19BD53619909233E6AB49E2B9A64D68314CA263CF59DF9CDE57F52366680BD45D5A0A4C9D41EEB72A774AA72983070AAB67DE710DA1A9C8804249207C62C425A40852A25402BAA69DC8500EEE992858705581A161FC2298B5BA04641D13AF1A7D279575B0A8167CC2ADA9FC41CAEEF5916564E372B0C175C530EEE60BBF0D86F24A855BB7CF9BDF4D8BC8144BA90E3D015C604751E358F46223EC2E5E26D3B63A9A178222B01669D30C4B6A0309556E6952FB61BD379E794F4DB7CC72B1449663AFA908E61915E5BA91F8926332E5807BD1B11683AC08CEE214B7AB80CB3DEEE143CFA61214C70A732C3EB68E67A2A7A6A651D75D6F0F0437783395EED4F32370A635F3BED9F399CBA16FE027B624E549ED6AB3179940DE0DA3F1D7EB9F9254264E8089EEF823854A49817185F75323B5305FC34DB8BBF8734869782AEA671D8DCC5606ADA903258E9DE147516A4B11C2F700E2A7D906895362DB97945BB64D6D81266975BE4C8CDBF0F78F316D925032D9C79FEC2B6ECE659402703209E3DAB3EDDBC4829F2BABB101DA2D53E55FF20E958B22EC4C626D139F9039A332C940E333BD4768377B448CAD8768552A32388E61AAA64D73057A917B511BEDB1CBABBA00EDDCD04079AE02F19946FB29034D523AB7A267FE9B0128F02EE7E703AE4553148A3B9996A06071C7FDEBAF15C663CA538961749171C5C1EE2CCEC3DE81433E0317EF8FC50A4FC215BDEB24AD9E7BBAB7F95D4651200C03F3633386896A7A5C6FD1E76D9CB92680791EA8E90DC4D53EAE013A4C1ECD06DFCA11A017E8EF908E940C6274765852B3CD14EB9814A4DB4CDA5C980B89B9F8134BDFE9C13E4C9275CAB41CB17AF1C4DB7502FAA4ADCBB773F89FA97E5A947A380F2DF4D246110EEF244DA451D2F2E172775FF9B67BC4215C5573D425B20638363F016C8DDB5FBBBC0874A3BFE87F6CC76C3034FC2CC408098881B06C709270939638FE2E395CCC33933AF5744E0CACDE1C8C5E2F31F0D9B9B5CA81D195A8C7437A57B77D465A49F4CBCD928B06F3FD1A013615B35514177C5D133399070FD407BD2FC30F380775A7F812B08E74E9F8C009C45711EB5F5024B6558D6E4DA6E2FB30ADD4784E556E482DFD0FE2401E50E21954655504336F211647696C87CB5950ED77BBA05FD9756A176E68D9564321BC929E018C0BB1567C90769344DE824FF54FE8663528E2B197F2CE4F80AF50146F11B4AC02B0806EC8EDBFFE92A3C913B771105D0055B6430FED4C9DC4AD27944E9D692DE1F01EAFCCB2E93CF4E6B459F6F961BED82C91B45A3D8AA0BFE11E0498D26011CF9AAC2FF306F5A7AE11AFDC419FF744BA1280B60946FC3AF38674CA30B72386D6E498ABD3704E3317840CFDDC1A63583BAB0E097CBFDADD701348646EA5F756C64D61B7A227613076E7A7D59A11B9C397520B864A2ED59570B5F5D7CFCFF50BF0703B4940FE60ABD134322F080EAD5FCAC94CAE41785C3FB8E99A6F5E7BADB261C0AFD7598C59D6B7DF2D308196EE9096274049E962553B102A2A8796A5D4CE739534867F1C9E9B1F5B0FCA613986D828642ECBD278C70391881C2E65D9E411DEC96BB11A61CE85A636981E6B005B65414912EF0168F625888F95A2425888B181B2876C0A1C5646F7D06B156E0ACF39785784F4078CECEAFE9AEFB6BED561CBAE07CEE010B1003AE4D0E5C1AC7C0D6E51C3A3B49B8E0667A3A0DD1A9594E7457E8CB884E42312EE1F87DE49D95011288AA4E6DD7FC018C87D24FE9411908430B6820351B78263FB42D5427AF700BE9FE705C9C6F5904614085291A47840E50BE77E504FE6571C0C1DA6C02DBB37DD6210B313D5A4082A03AB1812A11064196129BE8E703C51FD924DFC933FA64E5D3903E77B41024A7E5AE32A5DCFDA95242AB8B22F0AC5AEF1A235308F1A79FA546316524AE8C952731E839C1D9D6F2432DFCA06C91DEAAE862CC70B5BD04E8BDF26623C70126F7D223C8ABE59C2201BBE6111E573D3D718A920E32267338549A758C2705E9D0128DABD4E3B75A107DEA57F98ABD9876A9EFF78BF710B1641FA7B7A1AC7EF6A5258DD28DDFC4B6BF6B9F9C351F0A7400F6929BEDE58FE812BB3382B84F00087F9BDAEA9506B2E5748C2B6C81D91D320B3C238C184B48D75F79ED228B4B5071324FFD4F6727303AF60FD385C8A4D0432C7EE3223E5EB3CAC2F7ADCF9BD0A2BDFCAB32B9F12C18D3351F4D4F13155ECDCF81FE2C0FD200124542EA70988310589BBF2B6527D23F2AFC3B98742C16EB66B6FE3B3BAB26E951BAED4D026BD7D41B63B32533A67EA03054213E8C9B10BF240CE0988FA76E16625B5F19AB652F2C114612A6A3CC56B1B41F460B94F234515F9928E9FD7301F6D9DFA809389A8D8E8C455D088733B429E9AA27224487C116A789B548440C9C4D8119B114D7FE0D960FA03DA19BCB80D50F211A0EAB0DAC5305CCFE40878F85FEC8C1387617732C3A2BED6373B52B69956BB06EDFAD5702007F54F3815884B874AA8DA45303326E983ABCCD6D009ED63942010424A68AEAFF00807455C48D672305A0D6AAE491E2C4AA98E6350C5DEACD498A5AA8ECD0C6E727FD74DE88D65AC32D89416643D07529803CA0094DBF133E29262D2C6A5A9D2E2E7A8FFA7DC058E8EBB77957254470F55779BBF1F4276F4974F9BE5C12F7AEE1753B3371BA7EF5C6D0695C828BD8C0137104CA6882FEBCC5407189A27B1E42869F4C72198ABC22D2670F906C17595224992CED4141F037410052231457C1347042A4ED7BDC0E3678C0EBA062C4C33799C519CDD91D082149DEB9C679238EB20D3E224C0BEA4E022062F262DB9724DB86B6C6E35EB2453F3015AD53FF22E3C11E2C2C465CC63E52747FD5E10B8CBCBFCFDA102324577484A7A9C0D3D4DEE822233576ABD30000000000000000000000000000000000000001050B111E24"+ , SigGenVector+ "ML-DSA-87"+ 61+ True+ "99A25F4EFA0AABBE2D4FA7D9A415008E4C08F09965E4D0E6311005B5D77BBB90C3B9713A4D33EF04C82826481963E3DAEC7127AFB3627EF1B36F5EDB0C0900EEFB2C12FDE8F42C16442EAE18A59C1D4205E5F3A73C74AE510BF74C8AC96D691EBB8BADF50CCD253A975580BB60A065E7032A1B7548F570F62C75F3AC5EC5934A082969210832E232061A077081020808C964222511A0A884900000CC02680202444C3492108100A0109142906909277110272EE0268AC2322E182526509004D0C46153C20CC9220E5CA0648B948C44122501A38894B2101C3080A2804DA2022A61386E5B160CCA142684280801A74903C465E1401281826D83102189428464166C834632A38600C2846420875021A93053A470204522C8942C01C7411292058C90119C86041C041194C84524108C5B0472DA842C22355018394C149144E1000923B32CE33650E4408CA2328CC8104C0C81445A488541B821C034061BB66560846163A451C1360A0311052312918A9008244052883222228981DB924121196E14860C1CC54550C070CC0451D130101B4668D8A46190042D60B405A00030D0B66C081704000688580888C930800A232C5C40099B206AD9B4880405290A443050360642942911093284302918170D9384711C41849CC484A2A02463B011E108214316210C2386C1900D01917124A10518C84C2141481040720A116019B42418C0201B192849C6080CC1090B0449A1289021174D8A82459C046A03A81011432914A7699A3449C9486A89242C08800194203292264EDC402D81926900936522C46910198C60126C8AC26D2215866196804910924C2689E0366D089251A4103194C84D08034403156A60B885CB860814C05090826D1C4246DB4844C8420109B9112424059448519224048CC22914174AA1962894804DC0880510A909D89448C4A2281CA3885A0229228691A4346E52120502A7091C42802180510A8768C928908C04455A801194B6215C000A2435815B3044DB0412002140C3C4704CA665112820008700228149009601142909802869D3366413B069D3345054320224826C0421825B82690907401990881C43091C862104812460289151226814012484B0301C8808C3166550809020B781193322620408D2366C801865D31011121252C9960104044810A42950345214824951B60498A604D30860CA102221374DDC442C0CB6000A4348C24851DB4266939221D8A671C38884C032509920880903021C184810A189E2944114B2451217700B046442C86DD13850A1B2100A0460DB420D09A50D4B04929A084599822DC24481A136291C370640A48812488E044450588288D3A269004946A226494C866504220E13B70453120E59462C12B78800B890E1444924470A01082D1C844CA226891304901B88251339111A86054A042C13012902878C1B8045599885C2168913A465134288C1920014480903974C419224040230D3B63064905183C28CD3844D132111DA806504A6214C0228D330710938500A460614234582B86C4A08695A328C81902951262484086809414A20262993B2699C308C42260249C6719AA205923269D938306038618A1821603631DB9025E42265DAC86C11296E82107211468C12994043C28DD0A051CAA48DA09204D8260424002923C644CAB489138988E3208C5B221260C66D82284280049043922804300619025103376254A28898362111094E11496412096C42404D01317261967181244D11B5482145490BA56D0C124A11044894302024970009936D510462910006598881E3024214C1050111085C245143A80C82C66849B2611329629B3426A2164AE4300E1CC38083142DCBC48C03390109B78CD83831C1044A40462A0996651C87690AC568A2327089A40D4396309830618A4445582288611228A3C249C2B409DBA61113488A581091D23204011089214502544864C28605A44462048340DB46725C00848B408D20B025D3264A8AB44012144EE2200A42B86150228618B1216004481C4588214910140050C2248D994460DC0285521070D0B2900136528BB048E2826091226961326C0B485124878DE3348E1CC8840B825114988D09816958A621E33269913669232745D2A08512095158122C810082DA1604A0300A84C62D24200A5828291A4051198325528211C4B00044268E12414D0003289C0266024809589671D4262879E58A1AD5DA28F03197F51EA970E46DD935B0F548844DD2D11AB2E9900EAFF23F55A77AC1AC50568CC37CE42DA518EBD55CCFE9C598DECFC7F022DB948602C922EC0B47C8A894CAADE5641F8610EF5EA8FF11A184A2660B220DBEC802EBA96F0F3BD7916CDCFDBFA1C350D326D37E847E99BE53BA36C5E1B7489D8348F14DFB098D349A71A5C7BCEE4EAA725B5C919E2F969D3922421903FC16192F5C5B3435235B45E9A7EDFC1418AEC2EED9F772746684F7A326CC76C67BCD3EECE0A2B8A7A3D3184AD146B0771B73097A0EDD3B377287742E98EB3C98D226C490C30E5F35F7A3431DFA32BE938D5B307B2014DC2CC1FE25ACE88068F64E287904AE488F0EF5139B7F9BB258236415D2F280C483AAAD6C3514F898E3591F2D24148E360DF03CD11CF6D85812BB295493001D28832CCFAD5BA68E65F7A00362D2423FDBFF61D7A49C7334BDB896FBC3AFA019FD58C9D13D1F758763897F2120CDCF8573E87A690EC99F09F29AE282EE66EE9F402F363219EB3B3030A18E12DCACB71E1A766E7D68587EA02E34587EE559C495A5C9EBA24382061FB32223DD8EE519E6736624E58FE6340B2C884500420A11E420D67F277CC2953A12370523699744957BF08307448CC85B5ECD1F1AF01AB66AA272A6B859E87036362D353EB18B357D7C7F640BE9C24317736C8691EF7186A7AB3398414AFAA6D3B607D75FC20E0D88828EE6C6337C2E47740D0FC1DBFCE99E722786FD5C9D7B4EF54F86FD0809EA798D956E0663AD4CE52D606E1E6D91087AEECAC21CF158A9447675103AED30343DE6F901B1637D60D4E1F06DECBDB7FFF66CEDC29E4089D3BC9A5FAE5095788A53E114F4A92AE259179FB47C1149B536D37E8E2AF55638AA70E3F12891645559D7DC82A2DE76A95BB68393A2BADCF70B57430B9AD17A964D429A5C8A033913B7A9DA87C8582974A271851AF2E21BABB064E027FD0518E40F0E6A156A55A66BF318BD2E5DB7A29C64EF02EE0FCB00FB9215CC04234D100F071F441F56D654E64761912A2A9193D20D47E136D271B77E6410AC3D202CD27850A7B1360CF6A1C5CAABF393208F5DD0C0D70BCA90F9D67F760CC4E129A024DA880A91E0FC62B35AB31D6EAF4E1E8AF2410025B5A1266F17928D36D8808B611C8F8DCB4535199FCC41E4A5B2A6FCDCB3896616C0FA03BE4CFE30DFFC40AE721820A7033D32AF047C918DD804477670ECFE764D9DC3797E001DF50C0E11F230FD84D29740C8C11A3DB2DEEF40197E4029645884B6FFCD5BD0196DA6A8287519FA3AADC2F966772F2654508FB9CB94E378DF69492F2C892DC4267D2879826246D26F7EF35E2E2F32CCE7C724BC681A226F503E8666EB85C19DB49CAC3E62BC8CCA9284D9CE8EDB2EF78D64A2AFEF48D258BFD4AF1C955E0FADFD879079F4C91222DC4C614148F382E17575C98241816203C81FA1E7B123B3DBC7EC0AD1FFE5B70EDB61F185C3A7B640C868FE2A2BF465831E46F2DF91A580DAFAAE6D15E893C58EC7A0A5A56B13B5056AA6DDE734A1CF15DD5057B278ABDE46DEF562A26D71516E7C97DF9B43D9C7BBD089CCD86B03E2F1533C14B643B06EC8C7F7D97FEFEDE2FCCAB110952DA4354F6DCB4D2EE07CA50079B19E6BCBF124D570F3CB164738FB9F975D1FD163601D6B59E7C14ED0057616F99D25F17DE2C2A8B97EF8C39E92C032D0471C09F6AC132F47D84F51F9186BBB49D295CCE822045D81D1DFD2CAC83BE029BB239680462663A458FB63B2446216D5453CE3941A96D090C85C87218F3D3D90F4ABEDD0ABF67D57191FF51C36B2291B7F16A0FB3E63C69CDC64726D87513F072297721E29C06D1C8CF91F18F914BAFD55CB1F88F952B17678CBC371954CB21C425E8CD0F6746C7D3CA468AB646B26A2A35FBCC317A1D264D6E96D1B41A3A08DBCFC10BAF3D22D28F5F370E402F3BD5CC6D706BEE8D3DED5430F7DCDCB5FA970A4D25FEA422D9CB1779B0C5D577331437A077C17E659752C71C574B843C0FE09E549BA09D83CFA6497E5ECFB0E3F4A6166CFA58BB0088BA841841BC57308FB1BEE99110EEA8447D236622059FEF08FB2685CF9D5ABACBF66E15CBF1A4874732531AACC855358A3DDC62305104BD8EFC0904A7ECBF40B9C221E21CC562051E584AC013565FE28A3141B724F4473022C2258103FF4B5D869BF115C56B16226A178586217E3B06B456E42FA8D73E5CCD2E5D9C45F4E224DEC7203A57F0E7D34866CEEC856E1EDF421359A4A03EDD7817CEE7162EB8B6296B50B88267D9775CB32346CDA913B0F338215BCC478B315C90F4C87B37EAB0D8A94725F2BD2442F567A56A097340A08093D3515EAF1C3D3B4D886C41734928C1044B2A685488A7A1598438D2D16A9E0ABDA13E59E190C7568E32FC70E7D24DA8CC85A9E2DE53C63DE6DB66C877B54B8769D68C05309D4FD9BD5EB65543399502C43BC1572052722EF4619667A026C77F208511B9C2C7A2C01D5E3F0650DA404779D06897F9B18F0E131ABD3739D0AFC656B00561AC02BAF7DA05EE153ACB2CE1B4C269B386CC3C45444C54B797E702A87D959B1A231EA8BCBB2BB1500BE7CDA3637B1F71DB9CC56256983A7E3048D181DC5DA074DF32E5A23C9DF67D7D48AD0C35895F10599C30D039361AB8BBB47155F55476DC85B9C3D0E519C316065CD2D07459EC44AB3363F8F95D7A48DA7BD67B99E13A96D05F338F7DF89D1EBCA98D0A466138B775DCB4F64559731EFB320BD5698E05C7FA9741CB40A532FE0F4A3A3534591644409093A8B51336AF9F799304C53B41C617FD43075070A34D6FDBB547E9323C0E0A9B7C0E05B3322149859B7E840457681941274636AADEFB8127A4D4151F72BB14D549E68B7A8CDE3346424AB3AD24431B423CDCA7BB331E14228EF817516DE713A8FCF8505093411ED82500FC28C0B3BEFC7EE800D09358D82B5721CD375C537C8FE33A88730F70CC2718BAEEA7632CFA1192C1B76D2F47B3CF4ED70B964E3B5A043FAE86C5DAD02311A3D52D699BA99E2FFC4586C0B29ACA2D1615F53F62E66F339BD5B50F56D5865B17F4FE2BCB877959D2197FA62BD368F1289ADDDFE624190D1ED089E78589C35FFE21E2E655BD9E18B87E687384DA72A07E2A2E98E8011E3824F6973F2BB41333B530E11F13ADF6D061F14EB45BEBE6091BE94F4829D3B5B1E5159DEA37E7EFF7A1D65BA8988C77A8CFF8113E91B8039B07D31895D9F8F7EEA8B82D47ED57D9B2DEA83DC49BF4DA8863D9932ED5524C89438643DAF7298B89F7AAECA1C91D22BEC74EBE000E9E89013CEA06FCAD5B8C65C8E9107DD169104C0405463ECAD60C1BD4CBD7D1E5B24256E6834BEA9FD6EA38DE92C861869205FE984AA342EBD773CBDAF2C48EF7D1B142F3BF4C984ED7B278DA28A76F0F984180BA679DF0909520F05CCF1CEBE1BA49C7F5881FD9F3751E8D98F728947C85850EC297ECDE30F82A2E330D0E3D80ED2D8A210456892DB71B93B0864CDDC1E8F93E6DB4533E30F1603502A13832C698D48B7CE9E86D9F5DE1CFA0AD0207D2D246C70AF35F6C95215DFB55292546AF41DC44979D912D5C014ABDF4E15675C6C54E0EEE74CFA8C7A268CE06C4384631C485161F9F14EEC2539CFB48BDD437E9A7F7071C7638A0B5903EFA202F3A2B81BE0742EB646F827D4E11AAEC1F7D86C6FA28E03B67CBF2C564875CB1350E06C5FDC34014B897215AA2AA10D45EDB0108AE375B1AD3326692EF8DDFB6D646BCE4D8F6C3D9B0F35FC56441BA648E9E1027C86C8DFC39CCFEC3C7105D0827BB46889822AE2582E7E6CA8ABA1E614EBED27405980716D52C0FDEE04481F0A6EF690A53F92868891B2A0AC0B551E2E078B3C652840318C3DB50F72C80EBD288E9445B4193287E20588B91419C6E5BE719729D4EE34AA1EE72E00DD132C79BDB42CF66AEECB03F528F91279940F5F01BFF55BC2DA5816B5D029A1303D684F995740B534F2A17030EF2BA46ADF484E004CEEBA7EF4C7F63EA55397B7B1E59EC07A093CDBD5BDE7EC65B55593FA4B8F3964A74B326E46BB10535328431FEAFD3B8B9208E54C8AB7BE0B0F367AC95DDDD2E0F27EFF470A9F98F00C05D78CD28A5173AE67F95FD625B56248696C95A5DF46894CA012FF9651902A2A820BB8E435EAE5E731F989422B5EE485B06FA6E8BFC41A332CDF1DAA1C4DFA183ABBBA74FF1A6EDD80500B22708ABA6F212C89FF04D3FE26E06E30623D223C4A7A3D98F88E1EF3B5505E493863A02ADFD6303E0BDE99040EE2D182C87227BDA89EB2145733D786F8A285DAF40BBBB106750434B985CACEC62E27A5168DAECB3945F92D027E456B26589EB0C0AB5314A0FD6B4D9BD4C1AF458978E9B8A764052FF57E6E1F3DB8AB1182566C4A86AE3747FCEE4E97F98B9CEB9FEFDE7325EEC7781A7BE9EE1617AF4728A4EBAF98148026FAB9B8735165CE7F51AEF5EF879C699853491838988E11C62DC19CF16F9ABFC6340CC90C06992C4B7F3D00E79975317E11DF3F7D8F3D874C7693C17CC3539D9E45A07E088501AC1FE0376473AD3F1AAA86843F9B23BE92EB5996DFB34D03FE8B193F6D452AA3B2B8662746B05A4B5DAC285EB7A6010CEB91DF3DE424C6A8B9F141907427B56C1A5A20A787C69AF99747B428EF729A64943FC55935A7291C6AD2349D3EEEAA2C382C3D91BE2FE11C67A41A5CFC1469C2B59F3C756B30A8"+ "ECBC9DB7AAF9BF043EF2C011DC4A74E9577533A2C213B186F64BF4CFBBEE94190C9CCA44818B5F6E4CB98B5C520DE70073357307B00ECEB6011F4309670CBEC94F121EC34AE81FCB7DE4B190CAB2A53A555CDC79B8E32B2A60152CA057DD52FE4A560FF7473D5C4FAF2CEBDC392ECC2DC09AE7BA0A09205E096500A637CEF769C0EABF96E065D53C16E683D942359061BC646F4E4612E909EDF05C1D79723632052D11CA8F88922FE92DCCF3130E4F66837A196E01FD96D2B54E1F7295BF69E2DA97E7919C14DBD5D2714CDCE49E580C63F1DCE35510CAEDBEAEDE6B5D64A13509B0926C1AB258605E6F7FCBEB558093B97DD736F473936C030243F4743A44513D3732643112893B28D157E87DCDF002A352A5A2B6DF6C3B15C37EA4ED8032B558FD99DB5823E55F1F8E1661AFA83081A73175A6EF9D4ED6128A881771A9C93E330691AE4E983418F8943F05B67FC69DDBF04EB7BF65BBE9ABE8738C3CDEF45F94CCE075460C71F5E8350BAB86FABE8617C4F665095E0A147BD7BAC08D1D4D0440C6007505DD61D1555815CA7937C166A801A98AE334269541E652A8DEC74EFEFF85E248616787C04ED99930F34BFACE809987DC1B4D89220F62DC8D9EB0ABEF880916740641FF30DDC5D76D5736F71D73F05889B6D60715F9E98A619865852647F166EA696EBFD9233EB22362820C99C374F223D6F2E0EFF37FE2C61B226F90E83AEF63CFBA9266142F7C97C5FB94DBAFA58407F56E60375436A5FCD5D408A0D6781EF75EBB3F477CE892372A3525997BB79A23C2460806314E1E0F020E4EA766A7CCEA94109349C9474DFD7491857345A70A237B6F47A363B208F105088FC9E607C9AFA86A4086DB16930AE96BB9393F231791D835AE2F567555D5966E5F5C5136692ECBC5975C39E258299F9D18B8889E5F6465E0004F3594743FC77A68343F33A7022442D89F887C71D884E3C4CBB51C4FFFA29F90F8398D70A36420D9ABBBDD0D6468CED5D79E0204F5D93ADDA638D85FA12524124D9710D0F91A76A18D724B12628AFCA315BE7424B4EC270909878A1A06AD4CF6CE4097600CE3934CC2B7F11DE243653052DF0DED1BF352E4F81CB38D6639597E8F2D56C7BA4F8322AB901707D11F26BD15F1545F2CDB23250752AF8F053EDCDABEC33C08017A22A73E3AC83F185937A7ACB9CC43141842CC365741DDF6C15D74CF2F6109C7D276E13568776EA85A7C492FA55163B3AC0A20CC04DB775B570FB6A9719994A3674DFD1C7FEF1E380ED111CF9D34057806E568C4047015D093612486B07D796B4CE764ABC9699815CE243C19AB0BBD7C966626EB05E18ACCF9C7861805EC5C4BEABDDA5AB19CDF2CD06EA58E3C0E9719792A730DC8642926411548A896698373B37928432536A60A73EE4F2E08CCE141701C81886010C51201FF9157A6521B209CAD5A1E500836FBB2D41D8D772A26D36DEEFBB18920210E98784B5B3159DE917CC6582E9F4F306BCD0BA9811E5D8B4C5BC2B085E497E6F5DCFA6E807303065B563CEA027781951E73DF497016FD0806F8BF5F37B534CD67C8D0B0356DA44B167BBF7F2763830CBAFA65B9B2FFAFC58D2FDC24BF1A821BD5634E86833B90842D914820DC8A38E21D7F79BAD00931B54F54ECC5F2D1DD5F9C9B0A5F46A18C33B536C3E3B5FE1979F5CEF69D7A1EB6DFC39427106F45BAF075FD1FF6042DB0141CF3D886E9F54A76645C0BADC2535CBB910E8DF7B5A9278EBCADD9FF757ED6017BB8F0D6DA48EE518BDDB997D3779D4FFD9574DA2E2CCC0DCAB3CED846D6541DF3DF2460A2595005E9BB2382B6F2418B8620D6C99BCAE0A01893302A4DD80CAF5F6F8A1503246FD35E0FA40FCDE5A1FBDE1CB15385B48D7EF42884122DAF158ECBF16857EE1FA837735B7D2948D5C99DA55EDA6367D8A0C8C77834AC5BE22FDC39EDE496824759B880182B267C55F629D4E0F55C8C4E7CAF87EB9F48D729509F82FA7E3FFE9E7D9BBA66B2E38728B367C00CD5AC81B1D7B72CE23400E2141F33FCB34F30BCB735AA8C770E6306A892BCEB815CF65C4D52D12A3BF633480912C9CAB98C1AE59FE91D0332F9D48302F323167D09F4DEE9EC88CD34CFA5D538DD854B9265FACCA939F5763AE049D51F06FAE1CC6C268F5D5B2FB358C31F945F58E37BCFF38AFCD4EF83A399523918A0E1915364052D1B59543E1C61CEBB2CDD8EC874F8E9B6B0595EE6FA887FE49C78918114F0FC2328CDDC6585A4E35BBAE90B237DEEB2AA22F58150882DFBF7BEDAE300BD89A538010CB726A30BAC32A9E714F57F3EBF7F784F0BB928FD30DFF47F7132D68D42CFDF8B613D6E2AF92DAFAA4EB80CEF772F0DBA8DFF5884C3624A25519C395BDDF89F21E5F66ECC39064E633EEC607A04461CAA8DB485E1F0C482CF0AA7314891EE97AF32113593DA9A6E13E14849453254CD83E5CA4974A7D2321B8A4867A81A30AC883293CED1C90174D7282CB422DF0C1871504B783E76C94201E3FF954F729740956B28524BC51DCF5E08BF3EE9D3151A1F97D93F5AC618B085DE94EF985F7D9667168738FA2CB04B8FA63F56264F2195D298F4CC09FE571DA4237FFB11AD759D5FE68D0B68E8B31464E33DB2E4BDD304DAA59C0EFDDC7DF521F021AD41142ECD582313FA54FB3A6EA7F2A986F3790EC5D0C0103F7ACEC073E1E207A3575A060E69312BF282A4C5EAE417DDBD4F5ED729947807472298FE8416969F19E61F8684B0F9E5CAA2A23ED70832D77953D60980FCEDC2DDDFB6074FFBCD95C07BCE1DBC90319459DFD3175655E6FE2AE7FA52AB571AE804DC17B64453C59DB9F993BE075DBF74A582619BE4F29EF495FB5B1499878E543AD57A7B05D616BB7AB709F5497CFA85FB94DA39AC55BD7E571C8C5B0E4055C2EFDD3A01E2BB087FBB0DF0A170F61022576954EE4A7D3274D75D360D56D2C96B597AEC5A9E52EF508324B7DEC995DC8E29E32ED7ABAE3DDB37D853D20B8BA472ECBA364D853A058541009D87596A4D2B2F4298FB45B5258FC77294A4D8305020AC104816C00CD1C2DE738DD13E044DCC7C2B025BAC508334D0F75956FCE4F9CED1DB8CA636CA57329E454CC8E9F68721C99189D1FF21FC38B77EE0173E6F072A3C76C7C525165D06FD027DA8893AB3B10F98C6040F354598A1B1BD94D054341BEDD5029DA796B98F95990744BA9DEA7513B16C"+ "75A3050569757BF2A4D6E23458A194E146D60FBD1DED4A00B5DF1A0390FB8415BDA71A73B18229B8152A60C122E49860E0E50E062A56A48F17B787922D911C05980518F10175C755DD3A14C27C61B73D714795AC011C1BD6981BE878D706F22AECDD22DA220AA775920F23452D3CABFF2F5F981FBB09E2BBDF3BFBA8EA5965007F41FF2860E79D0A6650B5E3D7"+ ""+ "84BB2375AC1F8349002B6E1A353BD2A816A5C7823C134E38063F639310DE2046AAC3226BC934643652970A2396969F41E3DFDA53BC5925D6F6DBBD35B34F582A8C8C384AF9F23DFC05136D731BF4E782D05ED1F488C5B93111098BE233F389652CA8FF71DAD22879900F43D2EE1FB9EAAF3A06D9181178AB557C757767D7F8C9989E578B2DF961F2EA268B65A54AEB86ABF61404E0D45F6AF908DAD5611515C732A89AF830A851CB35868B9428B74951331B852BE78E9EAD81224C3B8FFB647B989C2D23D068736DE910B9CE2E0C06A3A5BABCE718C910A75C58350FE08E3A5FA14997362FA858087597CBA396F6627CCA12D36801563F20593BAC5D91CDE8DAE33872530D5ECEE00F605BC69C08F007AF0FCBE8834975C2EA7858514DBE71672229ADFF6A600C5BF75AFECF5C9DF763B78241864CA275937D6B14E35D2BBC203ED733BB34BF8D571EACD3666A52E481B5147E76BC83564CA5DB1A50B7B98119EAA217C674707B4C6CC21E024373A858803C5228A54E0357C6EB1297B09AD934E5C631A185B2E4DC4ABDEB540A32EF8EAF190242CF252CD08173A0495DFDA0353184FF50A4E209912C472E0DABB8F9ACC469B31A590EAC417220AE24E5314CB73D1C5AB4800222C9FEC31E420E5A5019108C9A4C0E88056C5F2A4D49BEB45404E807B9142B497C13CDC79BB08B655181628D41D825ABCDF441999D894D023E71FB6BCBCDE770B04E39D2F46509997A8272DF040E44E4741882E69C8D0236F111B15EDD8C2E7642BC81A87AC8EAB602A9AF92EBE98EB731DAE4F67AB76BFB825794716C621AC32FD6E26DD74FD2B3DACB85BB4D3D198A5EABFF530D9B4536C4B1E2D75E425EACF80546D6CCDA92267633D7071E6B61494B7F16D6693E821D1BCA4E044BF20D7D14490F49B46FCF7527184BBEE2C5929FEE409003C07B691031BFC94BBBFD07969801BED449DAFE4E435B0917A166FB1A2C458A6F8E21B3896DC0977D81B552ADA473A132D121FC160E5F155F0C08CCFBA0BF9AA76037F3BE5B4E4867D65D19FD61BBB3ED1A3F8AB67E429CBFD282BFEA052B8AD8F27540339CB06B5321F9E3E8DB5E804B02129C4438BC9E9801AA9326387739F988FE2FF2A962E6205945791CAFC7EB95F40BD5BB3DFF2F8BC8737AAEAD94830BE4C910B93393DA8D36BD01677729ACEB47FC1ACCF43AB55549D6AB97BAAD9BD731FAE62BB18789CE483CDB0424979A83F9A0C96A2AFEB164D954A8F69A1BF27C1D6F53D91370D37871D04080C295FD0923174BEB6250F9B397903F77533DE97BDD50DC58DB9617AF8FDB8FEDF975DB8AD8555CDCB8F72E095E8FE50BDBBE40659D7AAAA23DC112AC3B7DAC5C30B4C8BE57DC9D9A5D9AFB90E3D58850DA24D59C81F21D1EE5C8645999D8B1F480C13B602E91918FBA53C2335C9F2EFEA10CD278DE5FBAAE0D75D3C28BA43A0E7A8A7ED756E6D6045D84C272FE544DC3EF4D415070D1EBD120E519DB0E63150BD000AE2785033AAC5A52B7F4388D0154F4ED447B255086001EAEFC807AEF6D5D0AF5CEA3DD99E53A89D0905DDE9E661B937D14A991731F7F40B3AD56339CF0F1FAA923F68F1E357713D0D8A443FA35DA84F09A81390601696F83D76A6FDEFBCDA0D93435C86D4B1D57A925BC56A0AC9A0114FF5B632AEAA5F9780889EBC81B72970E0E0AA287976008CF9E8F96672209B9E1449BE7CFC32F177942BA8EB16C0D7E7CDEC6596AA103795574CEB25C04F1FF6702828B863239AB3C0F641CA494E32A25449C12F50F90BB7DA9627964E3D891B5470D0589C33C38A8CB44E87C7ADECC55DCF39F04F5E5AE0B4BF3DBA4E3792EA38D877F8955C71BFD1A4D47B60B2D305B61BC22C64688CCADDEB565642E0FA09845ADF39F80032669456AB6DC0B7FEF7F921E8F6B46775F176FD84652543A5664810A80F73CB5FA6CCC470A677385BED654BB11A45E3B69C4C231B2388890EFEA4082DCF97BDFFD784DC33C98A68CEF1ED3E887320AD21CD5577E68BCE8E6055BC5720A3147E377DA5D808B5BFB4A774DEB84EC771BE43CCDECA8CFEFE97D7BA39EDF421D4D2D1D6F55F623B3627F1661B1705D6DF720A9ECDB2A4162D5F2AD06A5EBE8F078EFA0EC0E42FAFA1C84BA1642304D940023C436B8629A730FDCC69C3A135812503A0295237822445097E8E9FF307C130B93224D786D6F3C44DEB31B50435027089782BAFCF2C4A9E6ADDAA6D5154FD68675623747219557E8113E95E8E8066F2349C6DB8AE99B4CBD08300BAD57CE20D1DC58BEFCAF6CDF38567ED72692122A607CD3404929C6B7B03CCBFA45E14F0F78EA1E34E2A82282D06344FC4DD0E640CFBD5046A280CD9960AA28C2DFAEAFA0654E06F7E8F70E6E74E3394FE3153A8200F5EBC29BDAA36A358A567096D4760218DD2CAC69586D6F2FB51C5EDE9CAA619BFF217F79FB1EAE672ACBB18B92FFE86A9D5AAAA1FE108D4C1F9A09F578BE3E5F884BBD910116D509517F34EEC8FF7D988F1EB9D336BB37C327CA1D624254B9BB18BFCFFD1084119C593EC98F8F12ABC1CA123463374FE770A0C8D212F5946D9CB01AD5F62A17CD848B2EC11762FC28CBE9F4B44B9AC92ADFE80C38CC7FD4C8EEECB9759CBD1D503CE5A5A18B94C47E3C5CF643DBA4EB4B63B7F5817DA772696B72DCA1C362B6E21585FE06FAA0DEC5C031AADCB91D73AAEC9F0EDACA3FEA9CDD3E9507014D3E207AE7E8525E41BF2A553236886BB6FD71A23A18CB3BA84007D25F980ED72E80CE3FC34A18081B0761023060CFAB59334DE767DB5BD46B31F149B27FAC368E838AE181A9BD4A50442B8F758B45DAA0690E9B51659CFC7E77C0F418EF0C95078F06869477BBFBA3EF6FFDB3C41448A7A5568978214DE3EF9D34AF8D4F6E806B17428E082F2420523A2C488814AD5FDB79AD806CC86AB8055F4CB92EC5756B2C875FCF1EFEC5808400FEA666EE705C063258C3586B09A96876FD14E78AB8176D3F6DC45384E52ABE816663A510C1EC4A2E754FD470F2CEEEFC9A9AA2D05D5D8740C46F689B7A6D27CED79321042A901699381F81F5443B7F66F5E2C19DA2AAC116F83AEC22288D24CECE84B0E4D97CE515A5E254C4904B54C5037A1F1DB2E580710CF38B91FB2C56C194C2282221DEE8D37BA8A73A5325FB9337FA229328D0792A35112B984F8A46DFDC43F42C836B73132834408456814222D431DE346712FDA8DE18B105002A6CE0ED88A6766A2B9644C822DC82627B699EC402075158FBD68A9C628377EDA2FC8FEE6CBFC33EA7EBE500E7DD884884A1EE0F052F37F7415B57E20FFBE63DDFBBF5550707023F285EDD26631EDD4CCC3B9DF41864769BE515AEFC2234A3C6A187A1FF375457EF4EF10DCBC69E48C135B4D3BC4273EC85E56E973F4C7E831FBCC6B187595BCB71D18688180D97E69B14999B202995082C327D14C328DE8096B76AFA178637E7A1611B7D7638E7747BDF99DFF69AAE2F73580780E4FAC8A0648C945353C2B42A20AC5B96767D194B1549DB4B0A843005ECF273BF4D25DDB64DC97A1E3A160643D942A85DAFB510B379397F665F8480A0D6EF5CE47AD28E0D95533F10A91FBA7265997B5402A8B652242BA70354930D88E6CC146A4F86E7C669978F5F56281023157F57D93A0336CE96F966A7F3A594D7D682B60D10BE99568F28DC47CCAEA73471BFFC2301AD5854DB5554218CC3A86AD1AFD4D031A25BCF3F7E79210313EC71531EC04024E57104A3C0C08FD4CB1E4A85E4104ECDA2007F345218960C1981C886D0DB7CAF690C9B475CCF15C93FD34B0581C27CEBEF01D8913CFF92ECBDDF215E029DF77434FAB30106FF7918B385DD8BB15B2DA9994EAB2B32C0905CB5E39793A7C630F298681DF1E6C505B9FEC598D8192AEE35D50FD3A5D6B6DE43D36309B9695B992F79FFF0BA50E5E3F26C76154D661CA89FE0541C4C4F05A10C9CAA82A0A1C9648235ABA5AFBBE21AF2E3309F0CB1124FA5EAD3FBA70D99C125B3F18DF5F34A212CE288AF72A83115CCFA1226DC41972DABB57ED97175BB0E0F1F717C37F95896D2BA5CCF469FE1A1724F9CE742FA8AABA33CF3A697C74CBA1BA14BC574AE514BEA4202D5CA2E679DEA386FC42D3AB48C5B9E59A641E8775A3DCCD26298570F46A4E49EE1E39F95351AF87FEEA90D5874D4B04851C389A8540D1543C6D4015D6D934483520253004783AC52DAF1F3D5908DAF37C728777EBB38C9475225F0CCE8D83B736BAA064C193B0BD0DC6EAD4947BD207AE82A34707740BDDD55D369DD9B22952F2956AE210B9F8C6E206B66BEAC1017418126199C7EBCD0B58196DEA0495C51146D32D230BB7E7095D8F41005C4F73C688099CD8171C78B2823603D0F6B508226644DA1481190C62562D19F1CFDA2087A78EA6BDD7E3081B2B9F0F53CDD48423AF20537EF50748E5B770EE659D28C6459CDEB095D986E43E9AA12726A00BC197497DC3293C741F39AB7AF406396E0B3892FC2ACC994454D89C1125ABE04A48329F475C7A09EA98FEE58543220A5AB803125F64A74574E78AE5E42773EB8C5B17346EF857E5940732A03E20123DEC77BB4DC2751620D52EA1F3A1FD35DE458CCBB9FEF2B2E1A655F1B1BECA0C003B8FF6C9BBE0AFD4BB36B78E0EC1116EBF3E8B63C62837A8A806C4E28019E9D600113AF6FA777F1F03F2E1687B875E3DE7570488B98FD9F40365352068759B2172FA3114011A9B101EA2D47245F115563A327C6A96A96428F2144A12169315C49668ABDDA755D973F9CF2229193C27CD4D3DD5E6B071DFE16BA301DF94A6EB2DAF3CAA8E19133C26F2A5440AB30481467E1371F3CBA8C852C01C3BD63D87E96707E45A9CE04FE2C3201B5B5FD7F0221A893108045889BD6C1557A3AE6A1C38EB4D17E4129791CD22402E3B546A2BA772E853866082B84CF9096A1915857D83A6D5B7E4F8CE3B7CB0C286D136172F7AB828BA37853F7FD709881E25F08EADAF83C69D2A984BE852FEBE104C140C57D9E06530A37CA9C7FA68F60F568ABF71F564D6CBE149E96FA71EDC70E04C1EEF47D0920B9DC01BFBB5B49A19970BC92AFC84C69CD56A9B6FCD0A359241D006A976322E8186ADD9450B20FC2BD2C2EC624981998E84D180AA8CBCB9E33295023B8E794F2E5D6FBE18E38F8DCDD99687032CEAF69E9D438E06FCAA3505674049B4E9AE866A87AB40BE7982788CAB4EE8DD16E637AA598EA01CB80A82C38FB29CFD744F3F5DFC5D4EBEAF7A3F8B9763456EC38E7B3E059CA8BC9DB593F51296D63ABD4DCA17C3F047A278719B50CD23CDD88735C64067B1582E94E05B08D77577F7B77987621FE76BAAD86FABA5604EBBC062A1260C476881E742BA6C37F6DDA5954E70F7A5B89C4D4D7E58FD94304E665AF0D0A486764E73D24F147A527FBA83D85CEFE2D41E70C139D1BCC94AF6A4E6887E5FB37C4ACC14B570685EDEA335E578D5FE72FA46B91791336759255193C9335EBF487A5B5C4FA2FC6F6DA112E0103F731D25ED9967AB42ED8573B984ADC715B1035DE162A3CEC7695A2AD6994A4982AA438C1A2E0F4DD4DAF820C09DFFAAB04AA2B7CBE51BCAC08E43F32D8C1E03C72550C8AAB28C62294C0E6580E83C2009B89BE531D5744D03F4E2D982E2E6DDBD784E03362BA0264F546FA222FF0D8E736BB9A82FA743AC69CC25E24A003074BA2CA7469AD6ADF9E36C9382BA9870F9DD698E09BA3A47207130838A9CFB7BA405896CC9EE1020B68CED84CB7DBABE018F53F1ED15342FEC459B0EDA444B7A014955FD5808B054F355735C727C58EC471A53352B526177B5EA267CD920EEB7056D21674E204A699A8FF36200BAD2B2D4EF9D5C503B7B0EDA3F210FA0294DC9BF4A6C24E64CA31B43F7BB2B7B72F3C299FE2458F600EBEC0F54CE569CFF662F5CBA589C2EACEA1F455D7B98C37C14FB4DCE7C0466794C5608F71BF8BF870F2CDB045946A88422E5CD5D9D6C2651B22D47C2B8FA76E9731AF48AB92C6E046E19ED062ED8B9548989E1E211CF8CFB636700125A7B439CC62C65DD830B45049AC93D1E687FBA0F1B2A3022E3342847D93A9B67B55493329E24964268A6C2E1F95048D07ED74BFD6A66A98196092956BA41EF2A895C76E58583DD2A4EEEA727BC80AF6FA699407FAA850C4202D1E9932B4FB6A3F8EC80371ADAF895B833EBCCB755C9C27FF57DF1B35386225C086518F6E32ADDDF1CF3794CE99105BEFE22C10460894C3B7F5305734167F46304E4C2503364BA0FEFCB61F0327B5C47F3EE5F758FC6C94D83BF6F32DEB3C2E8D924CE2699AFB77FFE0FED15D6DC6285FD78D85DA7A50ACB239DB50FBFA2296CE0F60C006E352BB8337A999CF596895F28E133BFCBD26802AA2C252E2C2396A219CAF6F876B0718D309A20454573A885D32429FDFB2052A80B241668A6757B523735FCAD73A4222303B4664707EA1CC2B425051537D85939AA0C010899EA1AAC7D8F0325078808692D4F4F7FE5EA0A3B8BBCDD1EFFD1A73C6CE7992CCE763696CC00000000000000000000000000000000009141C262F33373B"+ , SigGenVector+ "ML-DSA-44"+ 189+ False+ "12B197B30DF3DFCB40A94519CE8A60E7961EB743D3B59BD7EAEB80AB3D6EE61875CC49126B841A81168F7EC8BB06FD2B1EF7686AC6818CA3E9217620CE14F33D2CA2E31D93E1783F0F3FFEBE5E8FB7BB02FB6F505B642C0A136176C2EB01E75056A3C144042C95110903244FA3064617B56E25A2CC5681305BB68B6B78C3B410534245042682A0108909B764DA484482362C59148DD2908902250D5A3808004026A040260CA0288084111442009C968C2026245B2626D904458C40311C86700CA49004098C10228C8B468CCA32480987319BB6282181210023329900201BB748D126620C02600C164860026A42421094B2410A4026638010D180090C1828990428C220281A282AC3326D511089C1A28C62964591864C19384E9C02301B470E18A008214462E0900D82364149100E102911E332869CC80CE41460C4C28103024A9444461C2146119329994452C914458B36659CC00182A40899100C4326681931515816682441510030722186411A276D9C900C03876923B865E0026C883602DA344988A0705B4411893430D124655A4405123031C1A630E4000E02198D1244828CC8859B80455A9420D1222840A20002B980109561128231838288910051A4447002A844C8B83011C268E3822954C48D940232D0C28022B76C13048001044DA486500A3680DA346642360E5B169008146A04354C80C8411CA431898645644670CB4046C9366A4342320A874C1BC06C189369908801CC809143B66410054910040864C42404B220E3444D9C388920397203120C802861122909E3122602835053163191487223080A18924011B66D1C4144E1380409B6619844408938600C190210C02D61840D2221511BA48824496122B80DE3089162800812C29080C040934486DB446E64382003971023160D92226EC8964CC410068C1292CC286924B94109B189833030649010CA18060C4805C2284141B26054A0281200651A38700BA88C033760A4C44C20332193A04959004124B480014965C408214326295390644B0462008845A400250B8269190281D9268564080619917019398EE1B4416080845B4224D846324AA245E49808C9C6615C36612380911C1392DBB46412C31090486690204682222A20C1811C093104254ADA862489846198422841382A592602924042C4061164C2111832818A222D1A134021B87122A76DCA8608418061A384051981301123508CA600214784C0146819070A81300E64B0814404500B982811A864460669052632944F28CAD8E6C611C5C675A12ACD3AF1C8C21CF9BCC169B8D99BBB02279DA7A03AFB5CE1F4CDBFD673EB202FF38A9A47190FDC14C4265D4C194C7B098A8B9009BBDE9D7F5E5B60CA36980CED2AA531166FB243CE9F7A352F5CE63976169072DF0D78D352020B5F0C2CCCC873972CA6299EDC8311A5CCF2B7CE1F56F169E3C1EA636D4D2EA39B8B3C62E2950B95F18CB6B4569102AAB689E2E7308A9B8DB89391434E075249EA11C7B2688921D0364004652A3070FF888C23A9900ED8AC10C2BE89686E2B5D47CBE8355B89238D16D0CE724C377B30E249CAED1431586FFC4BDFE5A47B76022C31840F3617391C3A98BC2A7FBE040B5DC2D3B42DF401881652096DD434B0B742128A51333C9D7E245B9CDE1AA0BAB68A12CEA3918838397E652F2C11D999B399C1EFDF94FDFECEF6421DEB77A41214D362E9A38434E37477C666E5C1402327C0D8BF14512004AC262664A450C83ED9540C51A1D65E0A7CEA08A3771151885CFEABFE0953C5A938820D879FBD01FA4FD295AA90E80EFF986218AC0D5640CB1F38DC525A11CFCA5069B31042B4BF85FE7E66C450672B699B4E1032FB64F5B84DFF731B92174AF862063DCD8F2EE49BE4D7DAFA8481EFD231DCF09DB17B99BB52A52439C628513D2ACEAABE5532F0EFAC2A98CC4A4DA9CA9F8CE7A26D2F504CC029CA7CC8FBEE52025BEE43ADDF8778BD0F0B9D08F48A22C8777ACCD3639411AD65F3CACF1DD9C848E588E121CB0D5C797D595F07D08CEA32729A256A5F30FD2D8D3F1F222930F3D2BA8F9EA3EC325F2AA5A91EE933EB0A81B83DFAD26D0754350267AC7102202E7941149127C462FD1208E0D09BAE6AB921D179BE850B51C74ED8D09C8543E960FB8F8C1805CCDC9B83E7A97773A1BF8A794A497606BB39F1320EC841333228B8BD95272C2582068B1F6479C6D2EBB10D5875715DC80C08AFDA55792C0427034B1C4B03EAAA50FFA02C6B1B3813FC67C31CA9EC2B17F58934E3B350325E6667F163B88DD6368B153DC2935B6F45EEF4364808D6FDC04AD8E86EC2C5BAF15CE54E4F1D1649033BBD029E464416D1C35B66553C8053BFBF573BB052D1F9BE015EB6C1882FB4EA0F38AA3711C79AF4D816FB5A89B6E3E61772FEFEBCDD1389371FCDF52DDF06DEC2372EB76B5875305349754195BA3D40BEFC242049A01460633A36F680EB457E121A3086712C46516FF4BD34601C1E7936F18ACF893C2BF6115EBE6B2E5F5F3AE2811A87807BE4289AED0188884BF265CBC2AA78FE68508C45B0E9ABC2D4C466F908F4DEDB0F061350FBC9A70926678FC331DE6747E00BBDF9E5C7529B406A846E4E8D7731240C6F4AE32E76153DF6098C413796338BDDD9C89C4ED3B73D502E28644B4B3C899438326C023D0BBC70AA4C000E4F46A477CC2AB25DFEB0D946ECB2199F2625623DC213AAE84B66DE408A8A5E38A064A336A8C949EB23F0AB13A8A406A002DF990873FFF6893894DFEDCA41947D2D3A52F8626841D793D9065886F4A0251293B7AEB621EE2CF05420E0EAE263F53BBD8A995EAA023DF770330E0826BF63C291B8A769ACA8601B21FDCFA13AE0ACDC409944C8403FE04385D5487BA9E7FE611AF00CE164EBDC2599029C3B6FEB6A8FFEC418721A06072AB45442F41D86A83B24230B6958D7205A49CA5D9ADD08D2413B1CE9EF1465994CE960C12ACC3B4C27986916932C41AFE880C4DB470D83FEB6AD8FCFD8C5FB7CAC4A6F075994FE8F1913E49CDF66908D5089F7EC5AAD03B3C9A29A462975C62AB9CC9CA78EE0DD702A3408FA9F0D1BC37D2FF67E3FA93F926D542E6C8B5F75F36B62849A9B94C7574F4A5E6DB8660E0637B5BAD413B9EFB54D93DC0BE64862D18FCA111671F264C17CEDB6EBAFD3C2FB211FA5E3C918A7400898FC6CBEC172FBE0068565B574A1E409B9F788C52A89C2FD3BF7D804C8C6D1C98BC535374592207829B482A5C41B464C7EECDF398B45036472E69CE4D6C1E933AD2453F3F2F48DD75E80AF753CA68AC5E0FDE466E6EFAE4E6E9C3ABEDC3C8BB9E8D684E6512C4F10496475A3C133D35937BDAAE817FB9A4095500EF4C8B021FC447FC7A5DAB968B312AAC6FF45265F261226657DC3478080BDA20751FCE39774AD549383A3E0CE807B89DEDD425E326EF297F40212C348BE138DB230B86FDBD9A1B92E0E62FF68E6B9DDA679FCC1215E493C9DDCAD3D453E2D681D8138F2A6104BC56D5DA1DD901667E292AE8AC50BA8F230EF395C5A8B19598C42823E65677E2517E9540D85784354B8CE1BBDD2331F020DC5BF23B5E7602C8A728C57E0EDBE890359A947EA34DFF9B74203A5E1502AB8A58757A2"+ "E9E8B80328C3F6724B9500731E5C9DFD261E78714CD6241BD0B05CAD6AD74AB4C89BD17E98B62E23052272827A9B9F60F50D641C1A5B2EBA6750B956E7FB94BE187A0FBEBD920B8A3B92CD7B908B6B8A0643A9FE95622FD8598E11A34169B6D29D030EE625A6A49E0CFA7F4A33884879F4FBC09DCD32C12660356461311884BBDC0003D9045F85C6E176305689900DBEF043BBC62592D3A1DC1245AB6C85153D5009DB404639BCAFC1E52FA0D39D38A6E0D1E74655D5C4B7414912C00F47AC101703F4C237ED8FD789119CE94A14EBEED355F219962D4145CB122ED61046C15DCF7AB5BE8C0CFCD3EA7D74961B8FCA6F202206D0A57B8BCDC281CE33C07A54FEA8B77915830B11DA26979421FA03C79946EF5164446915499039C99EE6226A7F9F6E99E229105E71874B3ABAC41B068B1FB4552BF9F46C419AA326E27D99256052583D008FE74E9C1A4A58B1ABAFECB61B46370B82BCD8DD6F28C978AD79BCE0FCE569B1D222D257C0D8E8D1D1DC6A641BD35B39792145BB0A0F23613E0E74031C2A8D1613088A7780B12DC878E2EC928A0887227CD90A89FB7DB95C9A5309EDA161B4BD6FF56CA2F5D81384AB4139C36BBDB447FC1664D1E7FD0B7FAA7F4F9336D80AB2A558503F508338E0AAF28607FD63DCEAE63FC6AB135916BF35BC262EB9BD773249362747A79B87C3C1772AFEBC413CD1AE0951AD0B53E8CCA0F7DBD746F958D4F043493A358AF337ECD1FEFB25B6B710946D281192687692705CD29AD37394C07350B57452D13ED61AE31581E559287292AF4BDBBACC25A6B79B0B73600D1B0253E4544E2949C0DF49E2DEB70E2D2D7BCF32A8EF76E7A4C89FDEBD12B087B9F89B23AF13D4F8A53A9EC1C35F94C48475ADBBFB1906B62602F82947CE835EA00C41727C92A2DB37CB4CAFA59CEA9CF2E6AC40705194FB6B17B652EF0CC4F1E9445DADFC87AA30B51E0D8F1472CEDC54E7360BBF6DA2FEBA3EF926C358FA445B612174793E8428F16CAE134F4E39653E7AABE9511F520BB8D6EC80AF9C9C64315DDE4ED80C57FFD1872A12785AD5A6BEA9C360BD2CA239AFFA81D03E55AF1C7E266D25F5F80E01DA6816F359020539A72AD55EB81B8A9F82D446E70E3267E77A83147E947CD194077B7B47CB240EF6E18F02A63C299F92A402A1C78D167C56AFE21767F8E7C9A9D0128858AD2FA2249CD3FE8B8817C992EB0C655E25A2A790D329776E48AAC9DF1B8ABD05BC8B9E0A61C76A34A7507F27506012C1C6E5348E1744E34C3A04EB0608A5DC083F19F3FDCE2C445A31BFD5B675B27234B1D284F793FD1FAF896D36A5219BF37C07F5D9C395C3944BDDC78B37F46CB971A90DA2FC21E3FD19DC88C162772121BDF6C9232F231C4253CEA6DA9016FB374BCFAEC74358EA462A24FCDFD71762F1B758A2332CB7891F7151FAE6670E2216BC5FD9C569D97A5F29F6DAB9428D934EB40DE9B32CB897249418A0BDB7133C05DE3F6D6A6616FFA10BCEEBC4DC1E38110FB2F9A413005529CA3FDC697A852D14BAB7D77B9FA301AEB2FCB382E2626470811A43E528F7BCB6BA0C0D67C5B19DBF51CD26408FB92C5298229BD0B61036D8A7A11A18447EE2E0F86734CBF599D04ECD6A93193A0C129C3287640BD8B41DEFD8030195FF6B29E4E7604DA1555FDAFD2D7485D4000FDCA9361A836E63CA4145098A2AD097FDC2523F932816B548B7B6F5C8D519F9277A03D49893C2BD6978107702EA0D15D0C5028D7FFB7A12B041C9DD41D7DBDAB847F49740A05C0050D13B0E3BC7F71294F7CDA4AFD2411B114374F493128FA4864B3EF9FDA10D9D69B260C0178C5625365DEFD17B6225F741F1394D3EC5B2B092529A92EC2E7083C94806D5CA830D4B90B3BA46C222F621DE2B286D25D6E670872A4424D7E7D8F1A3B2F472840323695F5E2324363ADBD6777279C51F1336BDBB8BD9EDEC6DA9B4D7F0809EF96E9457F147815B93A1FB61183091819BB0E22350EDE0185B14B1E1BAB6C6319BF40709F41576023787AA9CA435029E4E646A9D221B920177886D6749B8E23E3EA0961BCFE0E5064E053FE3EF58A1C87CFB9A9CF404D2310FD6E9D91BC407766A9AF445FD7E99E8005FA2F44CA2D86062882B512F4DD832E5949968C9E665F0F98B85C7B90351A74CC9EACD37B7C5A93FB2C081B40102F061AD484F26E5B2EA2CD9AC124B17132209E052313E3AE02EE3C04E7BD192C9A69DD1DFCC255A889EFA280717E6A8698E2D92137EA79FC241796EAAAE84AC1333F092BAB4A641B41C5EF9D1007B5A9C8D6ED328E4C5BA395885850DDA77E798CC2882718235D8615676D938E0F847C6874821951C793BB66115F08C07972BF597DAC94742E1E5AF04EFA11F7C1C4E48128DA5056D7F729570C45368E5DC51F79655552E405A542C4E8F148E0EAB7AB57DD569588EC14BED090AA1960E333E197EAFCBF6D1CB1E7BE633C7B86C7B9F115B629B24F6CFA6614B26FA5ED53B6015488109AEBDD51718F9602835239614036977D16F48879A885E88CCE8E417DC0E69EA4F807F09892CD1E4AB915705E9180FDF76B149C9FF980189B74702E8FDA4F17F3A0CAF23E3DF291B38481F9C1116CFCED1E1BCCD8DA3F6FEA462E9B348092CCA2D681BB8E7E1C1327FD76DF697A5EBA0FA4AF22A6F1FE756C59605A412A9F2A53EB642FCB7CA19C30C2CED75FABC95A906FB5D3BAF6B817065D28200A4CD5F6C4D67B22C4FF29C5885A295F18F4D284897A9717116400DF1E391EF53229E65A11D7DC223042C53786D737BBD5C64015426A49F0A6E7CE9108AB759A26B0ACB005FF25691A24FED36B3E0EE005098A83C335A5EED9FC493740E6472F16307595C73CC5504A135AF84923FCF02F1F875FCA9BCA9C4663FCDB4F7F7C751D3807683E3DFF8C59451F8DD1FC42D6C798A6CD6B64571D98A3C3C6C92EA569912CF49BFAC488D5DCE2B421B953301A2AC5F9E4D91DA201C3F22CB703DC7F1D76C8E102067D9CC18D8FDB0D762B1FC55AD793483C1C22823F0B0997861655B7F03A5CC9289AE1CE0E3600FBAFD11CDD89B2CD59847F57D392099CF83BAAF78E7B7AEE8308A978282117B793EFB48BDBF345328D94D132DECD657290B312D5C5C9F0B02BBD0022CC30623AB74E153904AA500B1160B18F3BA69B932B75B88A6BCF05A928C9362C0E56A34F240CD97A5090E8494D7EA4CB73198951D90E896F7F864790B865C9FCE39D89C63C3498B5087491B4D52810A36FCE1EE2B619C9882CDF73BCC1E1EEE245CD716E12049981FC4F7B9296A63EE3430972A7580FBA64FB9BCAC5C695048417285B141B63ACD26576F2708EB523E7A53A51C676061AA0D18CC1CEE6FEF8EEB00A57BA9DE42D94DFF8AD72A08BF1CD6AB1C4CF83B49E7EE2E8F839088CDB12E9A30C70828EA979F9D58B0474FD89D9C5DA4BB211E281BC7AEA39375E177A5AFD12703A55C797B673D6A2156109592C20565082E7D83559DD62381A351D40ABEAFC1DCE19BA20265D238A474C6F1EB2F47E44AEC365471AE089F485FDE4E9FE2DEE848FCC5DAA74F523E83EE325E3C1E8360F2F1B7E8A6F7BEA2BB42F254A45527D6D2030F6E903AD501D4DD4C2F50C4E13AB482A4C7B9B1EB586B56147C51261CCF1E83231E5937E396B87099925029F8CFBF437BC0254559F52169A574D08D726E6F9C594994A07E2F43C90C12A60CC111CAE40AC3948B6C78CCB8BED7662660C483409F113BC3B4C907104C92BD25AEE0400C9B3052F5AE0FAE950C496A85408F5BCA3855ED011E66155DB9439B1B0DD801285DA6B6BC8FBC958C135929CA26C154E7ADCAFEAF089DB351EDE1B2449BC2EE35139047947609953704F59FD1B8023F6C8836DDFC0E15B9A53B2CC524F39A82B32BE1CC652EDC1E8AC60DC5C242EF7EB206A6DF1D751A17E0F6757194935B44C9B7E2C036B445126754531626F5B01AACC1588038643996EC1881AA404472DDDD33BA725D9B3F0B86DE43570B063061A1B6B506703977842A405C75BFCBF1AAAB5F1D565752D161C8C9B9B0E863D6498E417632EE1E3CFA8911C42EFD07D43F8F2BBBE24FBEA318E8A325AD330196EE6E5A7D940B80D666416D14FA6AB905ED77EDB27E5245A30BD681B3C5847D679E0C2EE803F44C2D41AA0D27F3C979999C69AA2B9B9FA21C575D71E038C281736860DE7403566E97A28951216E9C007DB7BA7546F8E6C87C80323E718BEEB693CC7CC3F347305311AFA311641C224F95CEFDCFEF554DF02C5325A457671E3AB3ED6E44BE32CE7A86A9CB94E9121CA1258AA690C5A2A301103AF5C2C1B7FDDC448C2480BE48D93C5EC37192575E2A2215E75C554462C6D86D2C8C976E2D3EF83C22F21335318981DFEAF28C7A112E9DD264EF00AF39FE379F59B147C9B927514FEBA485FBFC8605E70C8E439965311F745650957F8DE3E0A7EE10BC37708D6E2E8CF2AC30A2F323FA6F8D9319A625AF15E9B7572755DE4198D528EE1F8174759FA8EABF1547082CAD78EB5DCF7BB5D0DC2F3DACE52D16184B07F439CFA09BE5CC5D0FCF7702DF53959D843D6E29498C0EF78FBB6741D2AE7761"+ ""+ "C4AD982F61A2A27834455A6F81545BDD7E13AC4C1B8B40F7C298B31BAF23FCFE"+ "6AFD52ED68DA3355C947923F4902356A23F8E20D9C855C49766742AE6FDC471C8D3D37F6C47FDCDC2749652A3E5072E02DED5F71416C231C9D9F38D1569D0AF1A878EF33CCA66BC7F4890ECEA257B3510C1D7F2C3238508E099E6D34B743F23507F4641D20721747F574BEE8DE44C0F76A413749C0E83F97A94D13EC884FBED8407373D0508D9B0DC09F6CB60CF0202D19C8D65816F000D17986112CD7098DFF4AE5950A4FBBB3B76CB222426BEBC4AC2CAEA2CD833F1BE908A1E83BD7BD4720D6986D889A8E7DF6D546C138C462E7857927941BA6FF07BEBE11965FC503D8C39390613B1F0BF6736A02A52F03DC11C3227D4528C45A397BA45516FD5BBC66C0905CC517228E070412407B9A0627BF8EEA1BE53737B69371F42C5CE14F54016B80370FFB86DB3DA9C514D70B203107FBA69BD92DBD9703B3F2D5D0285789164E0CACB232E80685F217548C1A9C73C4DCA585E56B8834484CEBDB85859086071B35BDC872BE938AB2B10DF58E2663767C7B49A164838D0EB48C79E4BB90EB0D43D7F9FA040D2362161A5DCEFE23A9C3129DE9F3E8CBA330A9A50FA3C9203BA24D9E03E0D81761A09CA409D1394E22DF182B40BC0181875ECCE17F99013A4E1765D8C6D25FB9E7E8B977A50D1C1C6B72CB1A2089EEB3FC80CBE19023CBBA297A1A04DF965E726F797C7C85EC0AA5E0D20D13C0F4CCB28CDCCA61ED076B96C6CFEB8ED03A210EFDE0F98361A166964CC5D822199C3F1EEB5C83831C99CC77192844644212967D343F244895D26AB38B90C5B4392C620E6E12ECA393FFC79F76A9EAD00DC4F5ECBF64541E01513ED28D7B2C7ED5DB59DD02CEF66ADBFA04876D78DD834218242F73FCBF0D02991D41304F52010B4C8AE5251F564CE2E7035FE3D7401167869DB6A35FAF987B1B0AFA97E50481EAA15F32E236B9FB9705266B9084070AEA0BC70D3E1D3EF2A85EAAB4DB1B4DEBA0EB026B8A0A3BF5182872F1013C418D1BD70F4FDD7EA2459CEC89705AC1F0A25714E94A595BB21057AC334E4E6C6403E93D10D92D758787305723A79830E3DA45C4A284220687FFA67232DB8045A4AEF7F97936211CDB55EB2AD1ED7EF45C7BDCFCA9221A53AB1710BFAB8242B48FE6F9428F7F6092812801819BC57BAE8B0BE4A5EFE52D5754D9E5765D81F622D8B796C5E900ED87D96B60C3033FE3EA116A254E9B7604F18A364BB7CF519D3830DE845CA33C2AB08F657DBAB933E407B1C226889EE0C7D7B0E0CACB57501E107E5DC77AA88B6DD5F52BD30960F0AD3E30CD99AB3FD5F65DFC4E8D953DD4E6E34F2F6E9CC1FA8AC3486E744CCA14191E33D762125AA32D714D6BA6D40F7E488D33522543FFD7719040DDCDA1B330A492665DCBF9CB794BC60D7FAA389A6DC988CD84DA68CCB87C92D1DA9C94A2670BEB431B7FEF5A9DA4D9FD9812800B86229C212D5EB426EB821CD0FB9E79650EA51A195B5E67E68EE8E89BF78DC861BB28398A27A43C614B5E5C51B5C7F4843399EFD5222CB239887C8B8894E424EBE6A6C4460C791FD4684A2F4A1A0EDEE37AC0DBF87FBF0DA617611409671597E1BDF9BC03F8AC87D5366B91AB607506B04B52E85BCC8E8EE228455928B956B774C760EDDE33F3C0DB5B7A049652D5267D714EDB61E83D9C0E512131035838D21FD047F6C5AF91594D3ECC575C265ED96E2734F521386502F096D6415EE62B38ADAA73CB1F2C6AD48275A7EBC2DC260A8F415843D8AF54BACA40E7434F943671B491AF0DD312610423E5A48D48165EDCCAA9113BA35E507D565A5E72A85191C172B172DA82F13F5C927E093E9BC9AB886A7EEAB34275E4927A07C5DFE73920BF329722BD98F6A7B1094C52971031D4790FBE8AA7311209765DE096A114A3EA7F4B84E8EC8114AF83FF48D1C8936615CAF59228902BBD679163911203A7374B3F75C06F32609FBEFF5CBDA7C99569D768A9A4A1963223C5BEE342866DAEE1A02DE154A3B0510CDE0C66C04AC53B21B6A9C59EA3119C8F51CD6477540757D944E670519796E899BDE4BC3D5ABBA8EB7DAF318A98C62DACC7DCEAB56BFC5F61B95EF517FE95F08FB1ED5216AA4DFAB13C87F591447F4AE0D29F6DEC4BB8D51C69482E80C111745822CF110AFA5195982A48295BE60F432399B699E48BE01EF1D56E24E0A6F60766E9C89FA8B4055186F65C6DB9B4D4DB4B8C252850FF7B0199D2AC13DF3C8E89BCCEF2A788EC8C0AD4FAF558AF99D737047DCF5964E4506D9457B1AFC3E0211DDAB72A96F7CD5AD0192562931A54499FACCA77C96426B5C2EF785674779D1AD48C276A7F7113A0401319D77DC7F4B11A5AF0CEC95B5AFACA4D5448F81F4961157E3F9A0E1CF7590F979E6CDD2BDDE0A618720ED897A6DEC017038931ED10FD036E9864C15AB78E5AB45363AD88C496A1E850FA4D4CE10A80DE239DA898C1497370819D2AA37A3052F07D7A45AC2A5540CF6CD25449D6A7D16C458663852280C64E9886F3EA2689B3623D9996E62807B602A79B41D9EA4A4BDB9E13BDCB46B18547141B5A499B5B6B241DD07A97D0AD62BF6A50E552B057BC762F79648074A932507746D34F6EDC8088F06DB4F79C4E702F8E31A9F7932D1A5470B64C302566E61516ECFECBF74CE039EACEEE9B8BDB689540764F87553EDD408E2F9121F42F468968DB159BBC84217A55856ACEFF8E6D6F88280096E12F030DAD1EDFAB26D3C3BB21AC567BF97B145BD7752A619C0AB4E5D26966C32806C32B5D47E498F65DB729B4FE57A2CDCEAD3362A58801AE66CEABD3553DD510EAAEFB48FBC83A9F87CBFAE105812A14CA11F1DA5D9B9B0E40937565585A184D17A2F828A7E38E54FD5187E239CAA55BFEBE8DD3223D1E735DA7BFA0058F013FD0D22EA7DE3197AC9CAD773B540D9C7D536A9DC7650B4F0C8552A47748BBA5D5AEB750D2329BDA43DBD003DAADA8F6D90BED30C13AEEA16D2A8176EED21FAD808A79B4B8E4A8C3BCD229A66898243AB1CAF873A7D66E1B28B03157248646F0E3E90FA1A25DED784AD7D6FC896DD175E49B94A0B04BDE4E36F1E3BF24EB2C2F440C69C1EBAA4E9D2882373C9EBCD3884A0531CD4E06FD130C4A08BD2DB7C4EC4F15B0A0DF29CC6CE02699C988CC9FDF3BDDE0E166BECCA29A06B2733B1701E2079EDA98B306C4E16566BDCB56F667DDDF3DAA7123282E2DFAC95483329FF2824EE34BF4600C706D07E40C7198619CB55C533381B4FC57D4B24382DD3F844498CEC0FD573D353011E0FBB78A30252EF7B5310A5D57A66E2304884A299BCEBCB3D67A54B480474D6612D7E4091833677693A0A5A9AFBEC1C5CBCCD9E3E5F2020A1027282E313C5170B0B2B9BED0EAF602101617202125343F5053757C849DA5B0CAD0DEEEF216222D506A9CA3F0F10000000000000000000000000013243A43"+ , SigGenVector+ "ML-DSA-44"+ 181+ False+ "4F0B44AD5705B42A1FCC6D854D514AA342412A3252CB075379E40BA37555F9AB7B25F5F52143938DD768FF03E906C4180E9E04A252A1E70614046067ED83C0CD9A6267DDCA3CFE1CFFEC3B0787C7DC62614A3CD909D4168616D10B5B55B43193E9892C47E87F207956DE5BE1B426B939EA60A213DE78E0FC15671C6402C6360393140621C0801B95310A48022013210C06125B126609185184061009444111098ED91030C2008A8B006191464908B408891831C0B00C82A4254210105B44850A1185593226A34864244180519085D8C8509CB84093184E0C98654198446188490C06324BB47080042D09214899A448D208420008918C986819402CCB8844C1968CD120494B180A91468D4A040A00134D1A3101441242E3360A40187122C30984A20859284211180883805120C509040472DB9210DC8069A1300C42806024344C0CC84C2203680188084C24099A160A88A21100872C9C9851D8A26D222926A32464CB0625DB364D5422528930924C823014137100B82CCC164D61B6091841629CC08142906904C09001A72104C441233286493084C22644C24612CB860C818089140751DB16525C428020854900B1458CC01122172010048C91300E0BC72599040C5B043113C80C43A80561326A19C3449BC48C59088CC2B611DC98300C838124916C84282A898688D394115BC22518344200A00423282421B431603486A4164162948DC2C4080A03300494214C208A1A316451101051A60943304D048684149804E11880C1960099240004126A9BB06850981024454001A050138528028771224161C8104C634084D802282096849A144DE292210B21805C207012202449346C5BB82D02C6118038211411214408681014059C188ECB129160346EA0C291CB329258086D8C942C8C86304C24060C3350093852220931209140980861110404C8C66099186E181169939449133224C4A4514B84111A883043863058164111822C898051C34848CA2641023285D21069C4282AC3446CC8020DA0A28414898D83466012B32461B02D40B24CC91864C2344E04958D088451D806098414686318882185100B074859C4314CA081E1268222B924D2426522386251A20C1442800A86880247420C928D083606D180518C4024230872091784E042099802611B084251288644282C0BB849239985D11050822045E300124122861A39400B464444388060A0650A034841206E1A246C9118526104514B2870D824500AB63184B04C93148E6B6CFD9581DA4BF7B459221782451EA37D40B77CD24B96305AA08DB5E905E5C19D0113470AEA5F7AE91A5266DE20C590D70EC550BC213DC60571C40F09CB09F718F383B7A9BF4776DC8A34222685887FB13036A47EF0053C1986743082632C371BF00431D2569BDA36C0322099CAFF139D60A1966564FF4A315B1A8404714236481F3415E1E85BFFE410A6FD4F3F69294B9A430B18285E86A3F5E6D76995A2ABBCE6915C893E72F83E4B057D75AF4FCC2AB7088C2539026995C6DF9AFC20D182105602E4A6C8F0F8779B6E26119F35862D6FFA53F22D5A82C41F197F69DB992615338644106B382BE0D34FF3B1D02DB62B9019915F0BA1C76F038D094EE8EDB7733914E31C97CD2188CA3958E67B997270E87B22036B5308AF0ECBE08197D188D5EB1C43CA21B62EE5D3E84E88B3EE4048F71068688E8809AE26238F185C6CF21BBC78337EA403A60DD2D792F9632461210E4CA6BC98524CED9BFD43901E4B57B040568D9027A6BC1CC80AA4C549FEAEEE2C85803661E007FC26ED4D58833C6CF336BA326478D260A0A31E5AEAEA35A42F924FC0F8647EBC969037A93E88D56DC6D4831FD91E596C9B6A9A5F799289FA5EDE65A5F1DBE8CA65030669ACC683BEDE86799E5A26753183F5806C60D5B681FB11341BE8474C5DE12B76B5A114886E3F1BA675BC8D50E6BAAB9361D9535164319FEC594BF7D81B13F404174583198901EDE1BE03C933C652636D83C797E34A9FCB427B9CEEE2D5430AD3529C977BE2847B0570CFD97DD357C83905AD69BD6465672F9B2A84B771C49D52EF64BF4BC96FC08AD6A31849DD2CF27B0B02D467A23D7022B754391D9A0E5476873D1E30089517FD6C24B65743EC9DE9DFAE2729B5223D5BA13BDECDDF5D3A79F208E6541C7B55A6A7EEECB711D4FF91ED0B565C5C03F4446E078B4CF1DA1DC66F3ADB2C4B6BFE1C7EDFC3C547B63E2A03DF7D10CE6EFEF8E5FD76EDE61969B25B86BE4E6D11973DE0EFD00274EED9D4F85FC7798555E9460DE2A1AF5B030AB2C151110CAC4907F9BA940D2C8001D45E8212E47C7D9739DACA9BA9763A518F7340AAB19B2449E2526D2F08717B576E16690AB9AD6C68E94154529933EF5D526B784AD916E12CCB35A59F0FD32363E50B02339463C5AAD71DC3F474CCAD4A9C39EB1AC7E4E2D5845987D576842D7A19BD4BAFA22F0235D3C1E806362A71565D729333930469384B18C0EA242F2544DBE345D66C4E61A3006E13EF5274E629DA7F78AC6C05B73DD271B46DF592BA677354832FA29E53CDD99088708CBBB987B053D8592E417928DD5879A440A516AA46B98DA9FAEC09C91F15E20B9A20B8164D53481D9D7DE8C64B19473E662CF263583BE28D9B17B5949387D864D6757A5060B6D634FCA96E3A339AA89CD5E2130E90F6482F242D17A5C1375DDA4124A72B7B379033E7E79D4729E193C990A26F7539D86B59C0315A02C4C4B3F8DC1536273D9AFFEA07404EC6C06FC5414149AEB5C999443F390D5166AFC7956DF6D8FE00D50F09335554D0B3E6490865B06084DBC2EED2DF33763D9B8C28669B54E551D6197B6142947A36F08EF27A86664084D422373D36583C48C3C6EBFE5688AD84E68DBE8297906DD0698D394CA88065BA03C1BB812111146E25A91787D22AF88B409A5566BEB570F3616D089408188EEB6691B76BAA6FF69FFA81FCD6779DA047BAC0A83A847537196E9D35779EBF3942AE77B55C1D8238D7F791D7771324882EC059E00C214CFD13F04AC8086F9B4F09399BA2AF5E57723979D77DE247B594FA3E72C68FFEFAB4BB7999FB250F90769FE5F480DDA8A0F091E74BB5A75781CF858BD593EAF3C7F11B5742A03EF21A66811E3CE7148DA4CDDBFD085436881E74BE776C0549EDF0B516D36F9968B3B622089CB98A98991E7812E226B4443A6F34808AE9FA4EC1730E0D1DF49C609D6C060B6B7259FF451260F597EADCE3B7C0915CF8B7EC95E11B261F6849F7AB88976134D88A8BD0F4458897DA503B042DBA50CC5BBDB5EC55D31F7AA351BC3BBB5F09BAB2D81EE6BC6266B8BDDE4C4D32501ACCB1B46CA9B22A2F7D1EE37A50AD8D2CCCA9AB8359AB560B7D050505721F402756C594E7C9CA4791C71F75C046AF4B8F0B3640B2D24049D860DAB0C76463648AC302E9E85B16AEE6ECBA4BCE48896D3F8F544F08946032082402F658693F4EB3C2C481F24C1FBDA21A98CEB3FB040EE38622642FD7F7C4DF63482864484C959D7C787856DC946E12946C0C0A180F10827B248A9EEFA17F9CF6F4E2339E32EFF5194A5A89F792860E3C7BB1563C38F6F9FC28C26BCABCB4151410CADCA6651BBD170C885D5860BEE74C"+ "6EC7BBEA6D449E00773C9CD7589887C23DE59D95BB43CA246FC29052A85E6D602120C7EBFE953A276900C492719AE827BF04A706179D3C36B2B97FA76B08003BCF0E7240BF47C6E5ECB532F87A7CDEEF6AFC5AA9993D6E93618732C47A26DD9CF83E0F508952231AAD414486319DBDE57EFACE49A4F87ADF5EFC93E2617426303B649538615CE93845FFA1141CCD90F46A9439A05719008D7671AC738B68194F556153C1912879B79A2562A586F1500B6F1E4B7BC47C918F262760DB14845EA7A428056DD8FF12B50FFD1CE9A826B5D74DC1F7E90FBD785BCE2CC1A4DD89B5B70E0ACB3BC89ECC67075A383CD760BBE7CC3E65CA3B6DB26F63CD0BA4671A10483AF349640E8115B28B3356C12DDB28DAD028BFEED2EFAB586FD9BDDD51589B5D1C6618FF3A8C108B2C37BAE6C5214920AB5CFC02782CAE52F384F11E0D41EA451A1BE377FCD436CD2B381A80429EDC60F8CA1E3B9EB57900DD14D4715A3BC447B8FC85A28242EBCA989BB8F22C8C67C7F54B5AD188A342A0BE65B9E89ECF68E32A470DA966D2AC529CA7DFE055386597FD7240C31EEE8EC6A5EC742E958871777B0D2A9223643A350E78C01769E88C33944F700F47E4F155DA310DA008BBCB5DBE49C8E26BFAE66D8187A70F4469F7B38D7DE960E530C144F0968D0AD1BBBDB6880481619B1AB07EAF553F0828E1EB683BDF5CC57154613FAA63FD1B7996C2C18FCFF12FD13392EA425BB5466F6BA7EDB206A0947D4D867A3452942D04F3D9649CAF73D0CF9D6EDC73DA13F7CE96923E91010F52FB5F71039FC1CD400F06C4565283DD02A939EFC1589CF4EF50253C0293A434C9492BE879B21D09BF5FC77868C9641D0B9826E2C5CB587EE3D92420A9812D07886E8EBA6F8F75288CFEBB19D77FCD1371CFBD2F9B2CA258039B77EFDBA9087A799EF835DCD5ECC8C7C10F68235B1A5BC8FFC9B1FCA6A681A36B8F5C583DE99B67DE75F1DB95123B81B9D541AE46C393F8CF2CC5446F59EE90D4CBEF106D829D03BDAEE6EE122BF5C40E21559437566F5BA0CADADAEC5915C56303078BD749A20F6A7904B51E33C2AA07B29E841F5294DC56C65A1092077CC2630522D4D6FC76ACA674E7CB32C521FADA6DE2691B79CB86D8DE874DF648845145367B402D9C17EA1BB9FEEAD86DBADE1450581759C1F9DF455071EEE79266A45FBF38ECA4354A33843C2A145810D649CC6EDF5CECE3107E231F7578CF9C2646CE56FBAEB7874DF4AC45FA6A3F1CFEBBC8418A0DD834FA5A6DEF73F12AF7787FA66A3156CC9E8B577CB891E73E7E6D9710A070481BCCB0FFADC74D42DEC9AC3E2D39E9ADB30E5B3889A5DA8934A822BB3252BAD635A5B88F17E56528D714FD5641C58F603AA0E9CA76A5EED35CC8E412F02557DB20EE04BC9A234D9DD9D2DE7779365119BA12E0892DDAC5396252B365823EFAB68C768D93DE49731B2916A53CDF50260F2E6650EADCE352AD82E2060D3E77B47E0EBEC8A0467F7F0AFDBC6D7F88471C6960F1DD36BD09755417D3135E5470D897962D83FCB986010B9D0EF45138698095415B7D160E4BAC92BA8227478A1D19F10EA6E909DC5EC5DE08EC81699C2FD63F1B9E7B6D61E63D8036605A7E650AF71E7790800FA99C33BDC9DFFD4CCBE2F315008C51823B90818FB592A5101B8EBD06384D4C8F45D7CB6C20B7FEE375F313DB70760D22102F6CC41FC6E774085375C8C2723FDFBE638B0EAF61B1A855A1559C07AD6391578078772C0BF3D1E67ACED15F5484CA901B89D40864A288A798D01297CDEFD725D51C463DCFFA4838947351EA584025B3EC6AF684896673C6B64727532B6559CCB92AA07A37B6FA120E5E6ED8693EEEC0233EFC26BB905076967EF4CA0D5696024BBB29470149F2E891B215DADD0A45AA1134A9C268A587E272A2C32A9F8CC523384CE3AFE900F5CD0164CC4373C2185AA89356D27ACF4F2F6A1247F616F34D5DF6889CF565EE3E67062492B7D4F1F5B999D91DECE9BBACD541489D2767FCF1931BA035697CB133C6803F8E26A49078073AAFF3078668FD2CF119A798B5BEA7954A36474CF930698CE0A455E7E89EEFE9E152E60F913BCA63ED2F99E63E146AD62776C4937A595A20BF844350CCEB5B8BFFE5AC1F95CE0B0FCAC2A65450281024F521D24329EFCF8E727BDE9C4B94259383E90E4F385FF7F9B9EE8249E7CD9B9D218EE5F17EDE740E35E57F8DC8C5FE4D0B8D21EBC632B2DA1350229EFD0473F208A66149E23120A4D5D581EB3A0B4ADC8C804E5ABFA34483D8C7239E834AE4B6D0BA9813A18ACE5AE7E08F0AB58BA237EFA6E8E2BA43ACA0BA03D4B08B088564DF3676DD423A827463099D6C44FE86AC78AFA708AEC4ED4697DB4D24FF3163854CD3093C371BE697AF96A6A789437B0827D3E2EF8ED237E972AEF829D1310B09B127B0BF06AE805B9E35DAE5AD5FF48F041F550A3A14ADFC930FEAB77685CC897C53077ED380277DDE1D043BCECAF85D666FAE9EBB885CB9A6833F2E2C5E78A270A9B0A6BD5AFFB2BE119297CD5380D33E1716EF283D332DF9C5D9D332E68F71B4D18E4F9D37757C5ED6DBEFB5567E7F567214BBD02D74FFF94EDD32D75E899A0048ED4CD3D74552D3E61C9C3C2D7BD6F8B031BB8090B9A21E6227E38948637B73181C302C7730882E954C89CF93612031667B3ADAFCB31D791AC3045F74E81F70EC309E4DAD3D74EB0BB833AD32BCE26686577167D368A3B76ADE69EB1C93E7C9FABC8C75C8A789D8E30D42A06A875784AF20F35BB1192C78A566F5361657F94E589E35894091867AB22381FFE1481E5F14804215E13F672F45CC074F213CFB640D23E888E76B04F235D59A18497245ACB5167211BC067141261CF54483BC4F4D3BCA7226F5D3768791FF1C53818D4C3B118A0B718A372F7F4ADE5C5DF295F152291CA695AD5637C188FC49EB86BB1F8279E7DA750E796AE1601F7C2C95FF84DAEAA20FA010C623EB001AEAF4188F9FC60AF15C7670DCF438BCB75CD04137DB73F0617B25822A74D65E4B5103B12698BCF0BE637172743C66C892CB859C8136511E9C1BB14C8F856E62CFEBAD42BDCA883CF9527888FA100B13C7B40799D7A54AF894C17CAFD5B54237EC6B8380502D7D99645D42CC2FE5DF18CA8701D1BC5BCE526CFE703A64EC993B1EA86ED929D69BC986C3F4A5F8B7F704FBE09B6FF161DF5B44AC57EBF7045D59033A8326EADAF966F6F3339BB712FEC3EA683FC1469C76A2CF59B4223C92BCBA5FACB8F77AE16D3AA1E20C6703AFB5F8CE937DB4BEBC05BB31FE0858159B48035A374A6D0BCD57681CFD7D4C58A7A0D7B1C9BD849516F1A919C3C4D24989381F2E640FA48667C28B1813CBB0661F7FF6D051E14EBA776769DCD1D4CD596F7BE7ADE46BD79E7129931778FCFD80CD7BEE911E41C470E5049A3CAD16B0FADA0414104B7DE10D101412165133BF11A417F281CFA75D848C322BC579E8B3833FE54B370DA197BFFCBD397CC550A311CA13256099EFE0253B21B2E33DDECB4A91D767F5458924CA788C71AFCD99A53E1669A44432445BA7DA91C0326C104B6E6D06156B8FB97A6E5858A5F597BAB6FC541EF3894C3950A643F6AEBFF77D221D81086A8495FA5D0AF2F17D83EB9F9D8950B8AF8E2DCD76D78BAB65EDC46D5FD6DB9094344307B403AD2156C310DC52A9E3732C5753FEAD153DAD8C49FE0B2C49B48E5F3D4881FC7D0F038DDBE2B48517F563F8F0B614643BCDDCB64085C3073AFD397895F3470DDB63E36BB83A7BF2F4ED110D054529832C245385555D5F607ABCC757D5A8732902693D193A41898F523114A0E53EC302540E20091C4A76197C1DAD42C9CE24088DF74BC9477D924D2D70CFCB77DC7E28FE9025F9B001204E338596CC8EB7FF22FD8B7C6466C2927908D00A3D98A99F13EA3BF9C9F021930A5BB18CDAF79A1E38AD677516F06863F85A151A430B7B00BF608D67ACEFAD8973A16DCCCABBED8F221416B89868FFA151E8E9767A953477287C47C886BD334E6600DE1592BA4CE1D18FB07883A37C06281424E03E138951B6FF86169CBE1ED38D9A5C146F27B576C8288A04E7EB77AE92E486A490C0E9C2144E3D483015095C939763FA520FEAAFB9B5A027D69EA0F347B72CA61AB26AF195AF235AD66F2B694F61A287DFBE66BF8BA157FBB17E7138A814B75B7F59"+ "A4DCE602B7B73A674D3332C4AA54FB922C2855DC6613DBBC2222E050AB032FB6F6A4E3BA5D63C19CFA4C9D045105E74EBB613C14A0156B88974366795435B3EA3783C992A3B7F4693CF5F1D19C73FF899B34865FED673ABF37867251CDFD92EC1D3977518E838341C02A7BDAD4A20F09953476E622D123AED4581526F9B594B8CFC673AB7D95BD121BA83011160E174CAD6B0661B661B391E4F4E64C959FEAF13C92E9ECEEBD759C952D48D56307A3BBA4AAC6E3A6"+ "D61F6A291543B6FC8431F98FBFCA3798471EE5892894E2C56E6BB639E4E91FC6"+ "118731919121CEE068E6AB60F58D6BB18AC9AAF829DE59678A5437B8A6FC73AAB197FE1FB58CD531646F33D98E499A3BACD214A8FA9BF1EBA5E2A4A464C6443F71E2C3818381E5E07542C42AC52349597C99BB2C8B05CED48E8B4DA83D13AFD074701FFEC81870352C7BBA63789EBC134A6728317F7E1494E39BCD042D07DFD22656C94154417B6B2A801FE935F4287765C76C421798EFB4D4B8FB29BA0FEB09A7ED5830A782EB2477F93DF646680FAC735FB89F6B6222C46976892EE6D58EFDE51C07D56FF7570F58674FFCA68ABF8BD64DD78590F9862034C166B51EC8FE62AF8B34B2349E0787CBF71DC5696D958F5CAD9A19B0C1F18A58F882B2E3709DF999EF0E1894A417525A93A8FF3C5527414F9E58CDD9A0FEB63B6CC64F68AC13813B2237B7D5DB24C783EFD846A22A2682AC4396209C00425B41357E8F3ADB2FA4D463BED66252D18E67491E0623035B73EE2B11A475F6B00DBCAE0EA7BB22AE9B38C78915F4ABA57AF11B303292862667C22B7FAE7DE3F417F51609049AE3E90E6F7E07DB4E6994BAD80135BE281F1EF369DC0DD065AF4AFB7799EF16D739E8CC721A3C42C671EB805148E4958B6A3C38E6CDB6876651339A5DF9A6C461B5D6DAA4DCF58E9D6CAC0E07190A03D217D38FBA7F2C54FA8E3F04CDB7421D7461A7995F1BD1A6426C553ACAB6778075EF93B94B8038E9FACD85BE0E7188AAD76C9E8D2194CA18C37127CE3C4938FA9D7DC4BE2E4BDC39A0DB7C25086C77F2920F20F6723BB78255B68BE360CC05B9DC473E885534EE988C0DA48DBE075CD38A6B077A23E6507F6CAC5C4017A4C6F56FCA61AFFC9122512A32E50FE393D6E20A39EA1EC1ED1CFFAD7F8BB357B623D4B176ACCE9AA8C2F9FFB83AC62C551B6737F62977B30FF0707CDEACEC614733E6FA0FF2F686ABAB6E4BF356302B41EDBDA77E8F2083B8B6748F7800A7706BFF515C466F6B4779C64EB51ABF6506713A7A268D7A06685306EF7EE9147089A382DE8ACCF6FCB797309F364E8011902FA700B398F14361A1A9331709EF81DE3D66C01C3DA273368C2B523D7647A9321C3CFE831953FE6DEE13612367EB68D3A76546B3C22A9F77B58B1D274E9E88E433115ADF2E9520D56496CA9085960415670574E9A7940A5BD6BA9F87F376C9289FA7584BA2D1E501FACC5F31986315AD7ED8AF5E54BDD0B88EBCCAA5DCCD2ACDA93A81C34DD7B5BD10A4125B76431777C168DD34068918D7332D72F8A77892233043C36CDE46D22968FA1CAC4F44BACF0735BE0CD0F6126755D82CC89054F9CD68CF24BBD4E6C384756988877ED834CAB9082518DAD0F097B01EC8AB498D9C107986ED7865153E96B480986D3803729FC9302685D40D68026E1932908ACF5BC39CD88959DFD29EB1CD81B49B513D48E5B0628BCE04AE4767182A88BA8CEF07CF06D62BDF8E109F6C3D492D06AC60709E29022480BB7974DE4A0529B93FC0D73ED71128A2C1FF4F6E1B3C25D4DB7D8B2BE7FCA2BEC4711ACD865852B462D703E64B41C2700FD2CF477C4AD4B07A1434E366C1C559F85EB6A6365E130FF5E00E0D2D859B4CFD56D8AF44730A80690EE62825A389442AFE24A118269B05D6D47CB7F6AFD14DCD1EE9CF565E487429A934C55553329731D95891B770BBF2C6369D336B5FD55B2D7982FCF42C426B2344F61CE2C0CAF81122258ECA1DBCE1DDD227393241824D0643F39E7B1D5FE88721F1AE404564AAE283B9D38FFEDFD24989F0F8CC02CD237DBED3C45A6F89236A70568D3CE7E44464F1E4CBA6D8C878C8995590917E2AA981AD1F5058CC097CA3C2A9AFF44A0E6A77E97D5C61812DCD6107DAF6710A80800B727EF70F38956B6C85D320532FF8DA597843327E18F289489D7F0B1D9248E2ED2B93356E10D6F1356E01B26D904F4C49B2D7333DFC40DA8CFDA2063253B47C0D665A2B62F5DAF647F82E66CC5439E564DDFA80EDAA786421F735B12B2665FF6680DA2415478CF2B50A80C75BAD48EEB726634BB294CFDA96F74274B2AE62764631AD27824DF474194A7A925BF01CC77E6D81A84B9513B243755E1C3D09F6CFC255CD3DE03C24F7F7B7A50035C71BB1C39ECEA722C4F5A435659715B56659C3FA438FD012793AEE18F46930CB12262698B9380495C17D6FE8330B44E85F155DEBDF8C33AFBF8F8E2071FD2C80698C1622599453A9D1BCF3DB65CA5B13FB18CA753E6A269C26C89642E3EF2E93D7C0CE1FB24142FD814A963AE921709CBE2F8DD0F6B53EF6A48124C128C78979DC734B7C57643C407F4A3584B019B109B3653A0797482DE384A390B362E2147086DCFFC6CBDE28A21EAA4E36E36A618CBDE52D83C58B19BD3CCDFCBBCA35F04AAFEDFC2B337E4E07A1D88C1F7CC2C46075FF9924697F8529546570E0454683B565FDA985F284D490A619E1AAF4ABB58109CBF32323EBA831CDEE8A8B3AF53B36C2EC3022C42EF1C7D652AAF48CFF3CC28E9CE5B2F53705B21008F02219A0302C9CE784AAA28B662BA69DE62FB82A7A68EB99773427A1A41D40CF265C1D8DBD5CB1953FEAED7666655AD8ACCA80C2FCD5B7B508C9C946B88AF20499D94A545E9BC5832283ABA90A05A8AD4993216450C193E977945B76481A6FD57608CDADF84266290847C79FB6D50787D4BCC89AC9D69E4E42C328DABA581CE63FEBE8E5AFEEC66BD60299FB217BBDC97F7990F57D96E28F77B9A68443A82730CAA9FC719AE18051C4ED7C4419BF408BC12EB06CE17D470C6C423E780C0D4B39F788B6C6191D55975243611593706B8F9674006BD910FCCD5AE7402D283D55C06135E181FB7413FD44F7F425625753BFE8C4D6C7616BD01770BE0E40CD958FC0925C1DB16946DBC765653040C70421D62119ED1D4CA24E6CA7591D21E7AE68A9733AD4DAE6A3CBC17B2690AAEC3F905B996F19BEA89CFF2E4293A58317EEB4DA598552C5CA0C20487ACF183806AB16ECE1BF66DF386DF017E7B0642DC09962DE6614440C5A5951E8DD3418CCB361652234831CB77E7347DBE66BD20193BAF96F411817C8948C77FE270212296E820207FA3B6A00E08E79B001BB50E15F355F9FC9C38C1E5370A773F1E5088A94679F779CEF24BD46EF75975974DA7D850FB6B7953CB033D6A85ECB3D845B4F952DCE4D94D059B660369B89540F2983247329E124167F75F34701AF58026A9A1BAE6323011536445834CF7EB850949AC554F7FB577815EA0BF6311A101B4472DA510770412F4432FCD5BD3E0EAC5DBC38D44DDC41DD7612847BA3EF100682606D3C035D2C7137C3AD80B738EAABA32009143839434547565D6B77859FA8ABC1DADCE8F000070B0D13323B444B50565F72878CBDBEC9CFE1F2FAFC080C0E181E4E60638092B2BABBC7F2FC022C6A7190B1DCE800000000000000000000000000142B3B43"+ , SigGenVector+ "ML-DSA-65"+ 211+ False+ "E6B018E6AF97B7BC031EAEFF5BB7C1D7C0C6E7E99D26AD7982F086CA164303F6857FE824886494CCEB2E1C7587233A21D9C8A43FE72F40F20E1E818C8875B269D1BD79E35D99A737A319E23A8189513D4BD01622D8CE54D16F8244EB67579F3326D48CAD8CD0FF398CBF13C1558BBD5B7BBB5EF336A1DB3CA39A17F50901AB4410546146805704484358188205343435775365810270412131180120804507122731361844382271558373873051543234275801878474425612127621855656268420574057661416687326147410720130174200357401066161187212366571811631032327077866171217341246476726488814638881856524337016118600454057346067071241063471426071771181387157225771038224611454543507238074544358714676315450370406233228852676285487851613250216780243315303852746722221267061383400540165272717237145024412101130477147401646123346714143866483754524580775745646056406784613870312457081148550445178785753747030246330541703511363322802113418051867220145112651521711847627786376187625330522214146417637572781032524638355330808834430100508127570078647218173670788543156561783427270087104457227862447518405306045631358161143706677445471706857158277838470416001514348233055235474306367746306584858881446634574105224055654378563144221713308235170238816811148416782557832356155671153773077107663225447246348070087284655835238177033651453374455731644213518706823435831482305221638071802083671384801564824843855128020176782254420673527524130873768432883358566016316667787357548555273731662457578572041873282322876752467807201134511670363387024468408688034812801550164446607863361326382201621827443146007325157447537161850132777370067228671601838171260831210812527771055555538568801658558338613061204114885742780807405144442473621250216321238464008508862531043175741026205865871418778717100423113447570168844364438850352318808466746275126886100878428242827203378003384142420744703045567851451525113121745661565328753358113122830888284463123041284155552874803225364021351253068801288287026756771071230376126506116003840157202842525252503735710245302836186531770662353503546446707301026817745738057541368114851523218684786552455728430863877464845330271136242743378515740874658375740631112803847438375622448428055324317716870205254842008140535706651126561838260488262213371427256255852634674544256033837887547061355025705612378634720778260123421753476764446835403304681351632715172501060164545265123704413483037842183260620617774701418657003712652237760751234168548714186604556524067243733376755014625701381487162624270371215361786555383436444578686684421773521446823352864541637044256001806416130056340346168258514180007674305780860710272831053777171655664201562160667735123514053181357670031362703626178428082750078844008477601704304553840834241207434338084782584327003485001276630325441003823004201117452600113414832302365617264860284454112831688327220663760671655810875478227082201380178603633441772572184088070456012558817628876304818107064203607742878257274663004067626006527675164102838071063810712251531826554345142674230544807613105842256787572125455314257264333446771614160740181810482648667807858703616655212757355365416633073628052170138606517641430640383080634826336455853021248A49639FFE6F8A4C45E5D69BF9F59297252502474B719B44EEA80C22059C4A5DF6D6C17CBAD88FB62A5CDC8FCD9A9E9280C6FCE15D9F936F118D359DD39586A62F522DE7054767B95980B60931A563ACE1C42399B86B2345195A1A768824A0F77E702A8595C371FE878186D785504413DC15BA846D808ED0297F6F4855BD220E1B88664711F4534B09C8B0E23AD755C319102377304B95DD147D20AA6C263F84F2AEF78EF18109679648B56D184940D0CBF06E326C2FED26317B5A6C6DD6BE829621A49B3765328C26A8833571B4A43429669C7CE4630CEB50BA2C7403382C9956ED5368F00908E3C5675DC7C6FCAAE0D4B7266D1F39A615BEE858211BAAA768F8C6E10ABBDBA99AFE4F4FF3271D03444A536FAA2953B98CF810360767D45B5E9C743622590373A657CB01EB640524B01082822BF6559013A7862EA3098F36193C27D3053FB763C033BE38BD56D7E572020BF960D3A65274715CF79886ED50D4173F035BA4BC4A935D5F410EBE186924489EC3F24CF4FCB90F36AA331A2BB90DAA0E4430C29F07B485396B404AA3B72DCF05A9DEB1C142DDE42F05D5BA7D38015899AB96C4DA40012DD618F01A90F87D49F8D33F88B2A4705CDA69318587CDBECBEAE8B244B249FB8AFFA76460498B0CF64023791166812B268EF800478638250B4C639899ED0120562AE75FC95045AF9A8D5468196E28D97D60F4D09085DE52282240B2035686DD690B5ABECB6645DB35C9C0644CBF8F765629D4214AD52DF0BFD39D44491619DF8EAA98D32FF26D2DC9A8ED17648ACE6F6084163B7D32C7959E305DDA431F3332EBE697405011854D0EB9582975DB5C47AECF0DAA8DF3141E9E5A2E73A90D2B37C3E49B2DF79170EC0A10A58DCE1E106813BAE25C856961F373198D21D8D1A4D3F60102ADA354199386A36B3575E32F71EAB1159231F3EAB635B328091B5565E33AE9BAC192A0FAACA59ED96195CD41EB879E1F5FEF6B279A0931C2454A7DFCDBAB42175F701F1C59D4A57FB77C0FBB6631C28A6A2E6FBADA0E3356F5C899F7A3F9F1BF6B3BAC836C311C2C3415861C389133150C2055D36324544717FDBE90123FE8C61A6F2B59C51A025E3ECB2CC4B88110029501B7F26385DC31ED6B53192C522AB908FF118F777E71EFD061209999372B534E99EF360A35BB9A2168BD3CDB7701B930CD858DE89696A225B5E0280509C7BC97919B0BF58C54FE80712E40FC65C12ED652394D34BE60DC3E63606D550FE83CA72E0A1791294EFEEAF07BE46A5939A316605C5CEB744FE16C83813169C646529C9924691B4070B96E7E6C526716C69C43DE63A000FFB960E193F6656028CEE50ABD0A9E396434149D7016EF6B17D7DDDBE170061941F54B04BDE58937B8CCB87C61294A10649C46D5D6814D107B46D7421F4FCE12740972631CE40C55081E375B2D8D7F4069684FCDE88335C9C07D38DF2F4169E30CC764AA63D8CCC032100780B4554B748C9F3ACCF4AD25D9CDD2DFB457C42EF106D6B4C6CC68BC5E6FE20AA8A18ECC59C9DA093495B9F3BEA9759C0917DDA624D15955E4D62DFE80CAA39A9695261C01DEEF3B540BCC18384ADC710740EEFE79702AF54E70F3A499F644031291848068AD2A601643EFFA8E3105352C69F2F99A6E83309131E21B095852725D01761071B99B996E60DB1581981972D789E634731E75B5265B6394E99F2216D5C99E27458D9E4D20DBA1B9AA8A6EF88A3A70C3304051FC3F40CE03630969576A9C4DDCE130E4B5BFFECB92FB8D0ED64135B49848A98ACCF6F24F875A8174D081B538F8BD2F44CCEF08F4EFB89BED967C52D455D47C1568B9FF1E2F277788EDF835401FFAD8011AED103A66671E8232C25511060A8A96EE48D47DED33023F17A130402F37E2406C2C06803F75C83A66589568B5F9E78F885D18A81CD3FB646DA199D8936C2FF62B03AD77EB32A8250BC5F142E9F85E943117E0A825408C419C35B8EFCF93687B7D6CCAB2889F487E3C9B86FDE42320576289B08F1AD136ECD4AAA81143A876C6806E3BF23A1C0A8899711B1F366018844E2D667F74F999403580EE382718A439495A6D3F3A39913028749604E40CCFCFD57ADA9B831E5F40E275B680A9B665EBC92A6D50A1317E45C4F548897F7881E3877B08C71C163DBC79FC698310DABE54CBFEC7EA4C9372C1A960FCBB1A7DFA2523F4B1E1CD12F54E9E25A1C09DB29B799ECDB25E569A03D3F2D6A773765EAACB6645CE7FE5B7630DA3A2F382AE20FCD3F7DAF1FAFD2AB3C5CBEAC5ABDDD0B0E0C8489FC3A07659D64CE083901D7ACE838EF1B07859CE07EAE436C6B23735E2AB5C5E3A1C60D21B9B274CB91BCA35BE9CC63986B5E2BE885C695A61BC4214CBF71BB05C8F26CA976B6682DDC2CF4834F0FC1365F62CF32E94EDAE076ADF29F0603AB20F13957732AE319894838B84EDAA3070FEBC23BA5B8A1671F1092AFBFF5387CD0C509212D7A708DC913CDC33F21D966F6CF1484B8A334EDBA35C1D31A8B77E4B48324E60CA6BC3B09D2C7ED44CADFAC3B3F1509DC3A29F96D9AE9F06F9C0983BCBCEAC138DA27686E0BC75153FF494D961263128DDB7DEDA6224D91E1068B2D5E273FD30FEDB5060EE2506061DD4A48F87B35934697EBABCDFED5AD739116ECA251B60FBF43C8FBFAC2922F48AFC15F7694869AFD7D4539FE180FF060CF9910D09CF755417596E666BA0783AFACF0107BC0F692753D693C770EAF6C19836D65641267932431DC8A5074A9D2EF71F6D1AEF29F16371B287214537321F175EFE58AFC03D9B78BF393402430732F6D6BA0800C65E7FEE999880DF44E882E776C033319E0BAE3EBD914ED1B761B6277F2626F5D45259B4A47733CED5E81E23725371ABA6217066C687494A9EDEF3E02B6804EE8ECA634A0E9C5ADC53C54084479E742D01118E62C9A4695C8207574E5073BBC229D662CDFA951D9A71C351448427D8604F7C507C3BA072E345BBEED7AEF59FCF40D285559926AB3E089A4D318F57D5C58054A597EB7326F2D2180221866BEDB7D4DA45CAD1A43AC059394A9E042BC29F689245895C07E8DED4EE9B590159164BF906DCBBB4421804A1899B4F6675B281F80C3CBF15CC19D8633A1A7E52F82258A73042B6CB6BA97ADA269265E341BFD3EB89AED9460E4E010ADB401ED8BF2CA68A79CAF20F1313E7774200F1FCD41F650A0BD6C4FA7171E9B6AF5811DA077E9516A14F08A3B167BA13214B6250FCFA439F8DB8820CA2D8460D0F9119CAFD9E04D674A0F1EB851E03D1DFB78FAC5F05128FE409372BC174C0DA5EADA226C74B23324F9C62998D5C467621EFACB47CFFEE2AB9E6E550925CB200EC2C90E361917C79E81E77E7F9D15143AFF61607857B063F15CDA5F7AA26A20759A7F4CD55D59387D4F9F00EF0B02C598A8375662C6DC2115740289455DD24BC4A21A2C3B093458B4D9E2D7DEC5C5B306E296D4F1D8C40578E57D0BC30F3F881081EFDD3A5F8D59F68971E48EF3BDB50373D235B48DAA0ABD394E6889AB348856BE53"+ "7837B09F687F6DAE4DE61176A74E30B6E2A992EB2FE65318DEC63F79837A238D4640C2F11143F5D9638335B7FBAF19715A82A8D97CF4233B28C2E25D58D132839D0D25C0C08A398B820FE052217CDAC124984BB5FDC09B6C01827B4BCE78E943D65734D6713D09A538A599186025BDA40DB1D5421508DAD1E396FEC92F72311C274077DF21E873F9DA60BED7917E03FF4CABBDFF9410844E3755BEE437E305F67D2B9663250A50CBB16A539A561127A17B35AD3468AE48BEA49DEA71B324642DFA14E9B8207E35F9E026AEDDCF29EB21C771089C10FF0B67E4B6E15A0430F1604722C34133EE1283D1BE3474152C7486F7EEA2A484AC4FBE551CA7320207B7149AC40A44B0E21069158D4860BE61838CE9E462BC448C2451667424435828111A1EF3D599B95E6A59AAE730A9FEB6327A5F005B1321D607705EC3A21B3CBA5046374616F1DBE38C44F783924CFC8215098F60FDA96F486134E4B07C46A262DD3D342D2C13BFC2DB5DAACA5B820AEC21DA09457D1AF78574635AD5B4B640370F53069C391D5D1E4E0D7BF52EE3ACBB099867BD7DB2B5E4D96D4B89320DCF5653D8AE073E1485889887F733FFD1F2BFED8CE2840D9410BD5B45B0BEA6F5A545407BC721A39948EC1C74518D6F102A8A2CF17FD85310D11AFA51A18600A1FDEA146744295C1EB16D653594DFAE0A0FFCCCB160A3E0BAAC84A524405406AD22B15400977F0DF63C8C955F92FF94763E3AA6A834225351EAACE6512098AB46CDF7352956ED44B6A500D3B3F5CA6693D6C5BDEB9AD4A558DAAAEEDC35B2F610408A2AF9936B3CFB25417CD97A2DFCD9881C0417A8EBD0F7F1C99492E96AF721D94EE1F1A3EEB0DADC65D76BCC96216F3DD8462DD4D05E217BCEE743451961FDAC486F9B9C39B1CE4AD8E7070D4BC6D80DDAAC4B67A36A49EA7E2A55BDFAB3C46DE4B7926E2E49E923AED13E19489FA623A53EF9DDCFE263DBDECAE622D57D1B170150ACF2EA7D4BD4891A49074627B4644D2A3A16CBC5BD91A97DEC791D733EF8D88440FD04F34EDF54C4A38D730C58CB4304951475F49C414DB0E5395D9F992CA7BD149EF2F869E33EB00A68070B428EB64A19B6A0E6B5ED2333F66DD68D905E2DAA98B59B0F155DBB35FA2BE1C075441EBCC6EDFEC6D74072AFAD0755241839DCEC287C4F604A7D37177D7727289CD13A4C92EFC76C4B5627E815E68C4C7438AD22537E91B090B0AB8E59491CFF2988953954CA5DC65C2F6ABA2DE011F60B9160997650B0320B8E16EFB5239CB61054C826DC97DDD5D2247304B19D3FA6A061B1F5FF2C53F2AF71D1319F2FEACF341C576C86B157BC8660C6CA4360A4DCC0B51FA0F3D1D268C04FD83354643598EFB6655579FF5BC002A4C6D195144188A52596D4C8DA869A71D16CE27259614D315BDF85E84A015689E87B796B43392D314A8DB6B266F4CF3B4D9F7B56E5A97D1E3215CE4273D233DB483BBDA9C2E7114CC2C71BD6807F462A21F9BB7DDD9FC4A6243B535486F0F965643631AD1FF135081D58F5AA4FA159856A54326D38971279E8DC9AF7F02EA5B8716A78DA04B07F5466F315EF18AD82BF3244832BA86BCFDC10365E9E459BFC32480E57D1608CC9ED80DBD8087422B0477BDF92EC60EEC7C93CD966CE9E002F3EFB8912C9F5FD9EC530E31970340BB57704FD9E74D9F7BB5AD84DCFFF51FAD19A4B0D132A99B5441A546CE41C489F170BD8B8C86F9A7A4B0FB3AC7E3A18F331150062104C178E2FCC6C39EF25066F075687E7F2830FD1FD5B592BB9C982B1534391EB8C50F69FF3CB3BD74198D6A43A6C7500EB6B36C3008CB206EB0B5A6BA4AED1C5CA41DC7D57735329460F5EB06244371335265F25085C2C68CD4D1247F482E79F4838E97180AB274B913F1F2CF57F54FE534EEE8B4D19FEB3B35C287EBFE468663C5330B7428A6123B2AD3F0610DFC5F5CCD12C5CE923F278EF2BFBE2CED9F8D046300CBA7520CF7FABD7AF4092EEDFAF3EF757B7B5AE96490DA5D712090DFB62EDD1F5EF8F04EC2489E10B806D001DC73E05EFCA3BE6A23B209BBE62445961B6581C40B0B51DB42EC1939D57E9B6E5CC071C6B6AC4537F2E81AB0D4332AEB72739F43752CC25E3AA70B31EBA1424E62DDFF110FC809AAB8F6F0DD699C824E2DF4EB27BD0E4C4BD1818EC056FE3EE3C5845D568158E06DAB4C51C178A3EFC97752301A9507C2B0347069E035A4F70AC869714965F9905E1F703F7FBE80CC4B61B4754E08F74D46CDB34011A6FFEDDACC104234EF418DECA03911E24E8073849210E0138FC31B0AAE609AE3D6409721628E0022155E5C51D805111401EBA7589726623E214058CC4257934BA355F1B14D13CDF5D90ABAF45F0163E89F5D6231BC3E39D7F1D4453083D93CC052AB633BEDA071742D5386D10E54126670DCA57F27F5F70681EF61BE87FB596FEBB9587177D4C534DC9A27A344CBB75392DD2BDD392889AB1CEB10B2D54E0E5506CBACB5B7B543ACB9DEA7B19113E1A292370C9640502B986F0026CB718F13744BC8B7266E499AAF637C3D64B0F8BB43A69231E8C25A1DBDF21B0F5254B718F73368AA5C579834C42FCEE06E1E2DA36E528B729A7646ECE6251EBE931DCA6E53DA5B8264671DDF192CA83FAF4042DCB0BE1D1CC30AE342C5CD017A26A1517C6F7E20A0FADE518B7F5E7D7DB90CA9F79F456FF49A29DC848B4650B5F29D2D969A614C5937E21210A2B746AF9F6E991682F4075447FBEDBBC768830CFEF442B5FE0CEC42232CEBA699A7E077E593ED1E535FBA35532BA386B37F761388BBC8833F2C437C8CEC763D295A7FEB82912B3D7C1BC8B78771E23189693EA96A15011993631900D1692A1C1741A7356D2CFFE218E03889C55BF6483D24D5A7088D8ED09FDD187283442DB4A58E138DCA1F666F103726E8F590E6ABE5AC54FF1DCFAA2ADF7154C39C00294FC92E2ABC07EBE48223F366A2B85E41E490A482AC83E6D2C03A1B69BEA18A5286D9EF37377587506848D8E5F08FE446392FF64DFE97032F0065903358443D6B09E3F6870E8771DAEEA319AFB2F748507D73BDEC0C94467D60EDCEB95065F381F89BED38E051B9F71BF2D23514641CDF1EF0995BD635F6F72CE6E8C88E8DDDB0E82E13DFD6575106CF9422038B736459A65D372E16DBB01C6BB3903158898B3DF66DB878CDC32C461EB027964262BABD18414209BBC42E90A6D192F2AD6FB6F72A646646308C9AF856E2F2456C56FDFEB63DA11F70A5A25CA3E8F558BB833900F593FD306170FAC950B2BF5E20D2799C3A4E85D670AC8B17A4C7B72B45FBAB16411317E5F3284D3C1421CBFBCD47246B143B5466997A240D89E068AEBD1EEB1CBE7CA5553D8DD78EA51F5094BA04F80D76644A31186982E6684573CAB6E25B74B67600C5549ADF5E6C6CC02C509C52FDAB1F910883A192BF41F9E00CEC33FB653BB1210D02932D077CDDD2AB28C4D058526FFEA2578725E773035871924C2C4F07E8F130E4A55DA9C1E88FB99915C29A6B82491883035E2CF215A2504CEEE00C876BB0EB6F4BDA9B7881A5A5A4180D256395608248B3BAC8D4508CCF30C62B679E33D1963B363FDC0481CC3DCC228D7ED4E32668467E405901FC21AF17EF2D34BF71D83186D2089A0DA6F0891CBCB4F60F94AE6E0ADAD96675687833C04D73051AD8EA81289D03834131332FE4B5B17A3FE1FC9AE4381D99F300CE9A6549CB40D6D53ECE26E4E7FC3F82962874DC6733F2DD8D9B8E59DA49A7017600993305E759A7FC4BE9B447BD7D7AE4BE81B4AB790157DD7B3124F8C32128827EA25577144127ED1045788BD048CEB0772D0B72397028660057570F36C841C47584908C3DF211C9F3EE2B91775834C2306BB114E59E170F2E8369261B1BDB3BE6EFE4050AFF1C2831F110408F56F9798D622B01F428A46999AB7B4A2A43BB043EAAE860D1B2345B6DB53D9A07A0AAA9DD8379C62009E6DF40C5C74F5A89DD6F8B17F13A00C70CF7384B1BDAF7F52F50EAD5D3F7B3461C2FD4F937DEB96C968980259F36F1958FA18B2D1C3C39BBC8789470ABF1036D952A4EF2CF8078882444774596B07DC3D186D6D76548606B129479A066EDB11DCF26E60CFAD31C5688F0658CA74C504228FB5E6D1ED2848B2739C5EBE7018899D12E91F6F296BF3FE3E5B9F9D397A4E87060A1EF77824C511DE8A3BA03004C32EA39144C24D0A2DAEAD5EF5B239273CAE3D8EC334472D2E57AC6C7A571AA05D340DD9C475298B16C93F6524A67B717D59E155418E38D10867066E4E64D46C13D1057124B7801D037E1DFB054F9534C5FAD04FC32D7163A5329144922D0934C544B4B9DB416C4EA8A482B2F1DE0499AC999B579899B1C88AE1B3910628C186310EDA0AE841CE615169E974F6FC4459C3976D54B33A43A04F9DFAAAA2BEB8911978AF3537C83019EE4F92DC3A0065858B554FBEBADBFFE04412D2195BBDB0E34623745B81CFAE13562830B31B88B321B6566753A2FEB58ED43105E6DB87421E03AD64ED221BDBF3A9F1BD6F69D4114AF9AEA09DDBB391AA1097775DC257"+ "270E7D9A16E4AFF615F9DF70DA06779F6A92B2ECBEB0D5BF24B305AC03B393C296BB521E3CB2135FA97B8DB279315F0269A611FF4286167A2CBA4D2A445871A2AB59A23A08E0467B91"+ "418A9C20FD0D9D9AE226E14CA79C4881ADBCBDDCDDBA0CABFCBDA35E6B7FEF5A"+ "D7D02545BDD400F453F23809842DEE38F54B858B06D970C1583235A426ABCA9D1D9158E3D10D3F9C02E35EB715246C0809F1DC24FD83860583984FEEC382CA5D91674B338D4260EEDE26312C9C2373CECA6331D9A898FA12683A943AEE312D978124D0D7AEA85AC38A21B74B57BD6B089D49F0999FE3178ED222CC45C2E95315D32776B1837EA1390F9FDB59B9BC90F5A11DE59065381BF0844CDE39C17F272CB836DE98040711E51C37A696B2026D8F38FAB751E946AD1761FE74FFC89026726706F1CB0EA28095555A06F322D97E2FB821A07D8508D2A3DC58F91E0867D5C81FACD5600D206C5C02DB181F881E2D72540B24D317E4FC4EC7AC65AA2B8F350BF4C22F5322332FD130405A134A3D0191180659AB26506E23082FF93932EFBE7B9ACFBB08035B94E16DD475D0487334047A81E7F1FB86168A3D7DEA446A4779F1D7293D7D40EECEDEF1E183AFF67F7E368628ED5D957426EE960973C24AAEF9432A3912F0884A333B495EF3345A753194E2911306EDAE80888A59028C8D59C14C0077123A167BFD2500EC86A3059C3C5149BED1AA69A5C1459E0A05B201E5A24003C20D637FD556C49A5DA08A96D896F198B3F4A14B337CA7AA06F57B3795131E7FC0E664FCDB7251CCAD56E2AB29EE9FFF23076F256098BBC10949E2CCF66AB6AF1708AAF25A93C91F694315F2FB572FA4ECD7AD4E1222FEE9A3BF85F2AB73DC13C22AD463AD8BDE69E14F03A6B410A79F5552D601A02E6965F15BB9BF30B4DD7BC9AC8D2D33A806E40555BDA85BD595C5B5CB436805C9327B96102D1FC96E7D876E2605B37C914FB2A102B8BC66948F4D27F68E18699405A08D9C7ECFFDDA4AC46DAFE89B913C59758AF6309102113848AB59562C552FDF752FB8A0823AAA0610A3D1D6751F6FA75A9F5205CD84B4A91438059E2A6F89FA7BDC96E4C701EDD93415093E305F610B3F59A7C25E2C8D1923377B94BE62234D8C0DE1B359309B7F0A1272444FEF393E08B6F24CED7B2F62E8FAD92614045E0E4B16DA142F21A7F73C554141D301EA133500D97C492C724C07FDCA987AFCEAF09BE96E816E0BE557882B6E98EB90D3398B5B2A87FB582FA4F82A8A7AFD5612546758EB8AE06A848A2E9A0EC6E6B06874512C141E3B3FC3926A4B87C21B16679898DA727B8AE3CA833E23A6B6F98A2B07362693C6D8B5A8C449E74C35D9C19F4DEA8B74B3306CED16876902ACB51B26A7478A32FC65D26EF893513E70D639E098886196272B7AB7993934D9D0EA27A0DF2272F4AF8D0E092530430666D8CBC3595B1ECB82A8485FFE44A64B1704C85F8A01B0F6B7C608B0F57960A833F1FE4193CF77FE5AE3D3BBA1AC171F7B678049D5F72690638952435D823AF1F6BAF403AB28DF6CE94172DE72A4743926D9A105CD322C226A6CDFF0D6A43B8EE940723FC2B8BD3543F8456D0B3C077AC4FA6BBDD6CFDEA9D26DC60DB166602C53B5C9A2E0ED1608BE76AB60F835AB9FFCABE777CCF34DF8BDC22BECB50D30D9EC0DA2F862C7516296B6FA4B07F4E840FC80A38A717740CC217DDA72DD2B3846566B0339D5934ACEA2A2F746550D9CB05EA7A54A40E4F636C11E9DB6C9D35AE5EAEEF1CD49598CB4681714F81051B89C73944E60F65FC4E75730C07F66D77A372AB69B6425467E101CE3156084068367D94F0BE3E07289B4C188E256D5FF166FBF66D37CE86B78B03A08E542249436C553C872BE65A1B97109FE59392546F82771A50AD2208FF3B8A831F645E95957B0CBC4980BBC162FDE352AD65F8F8A512D2FBE8DEA4A826FEB547C866587F9DC2568BAB4BAE6B3A50FD4D197DE3B44B387251F20389C45303B27C529BEA6DCE01DAB7C2EB2EC6A3888610FC201684F8EA567B4AC4C5DC3FA2260EC9BF0919E1658D22EA12298E033F44BE4100A6A9514D439F1D8D00825C49E3AB0D7CC769F3029B2DF5DF740AEB7F7D9ECDC01282DE66026E74C33E3F7526242F635D1666B257FAD8F657E9A70C1A551657122BECE6BC388FC864147C0D59453BA490EBEE70707FADB41B8ED0FD2D16D48FE3EC3967B0662539B77ACCDAD7C0173AFF1D82CE76B2DB028EA62A2ADF55555D38B6E129A316AE2AF5AB37E72CD6036E53824F48BC2F0575C9863169F9678FF20B189554D736A6DF964480D3A7C76DDEF048D144FFFC0646648E89DBD40F3CDEC9EE01A19D4D62B4FF1E65460DE8A8284FC86750F5F519E303EA103FB7795E9344649C66E53347EDD927DFBEA055AD89040B6BE41E7C74223DAA115ECEC52FFBC0392D3D2715C7BBF2BBC04F8162E5924072A69C9F79E75EC39D6191FD6F3197DFE77CD7147D76AC39BB5F73C7EC1CB02B3CE8C074AF59420075CE2B0E53C0195D50DF4EADF5765D8CCDFDD7ABAB7D635E06A6513840C35E611999EBAE529D45326B58253BDE5FB75ECD58E43A98DB72C50AF5A72D9CE91BE63BF68B8EAB8C8E2433A70AE9A20F6CCDEFB53754EFCADDE54844BC2D9D81A30CB5640184573CE3E3DD8ADA6855B51F52818B1E45152E8EEFF94122EE906E985DBE8F84110B1F34DFF67A184B5C10B54E093387ECCF71AAB4E698E315CEE4349A8E372AFD276EFB1640E1209F32F63E66B237D375CF0A48627D480FC4F186D48B52BEF1682A132AAA029CEB07D4EB95E6EDFEE8935E7E6D589A14F12B7DC5770BFB8529DCF9A8EFE9C9F76A09E61B7D7DE156B1A11061A3FB2F672428D722CD3077603A0DCDCA5589FC5015B7323ED53155F0A95FBC6B2F928D0B94D41D6EA992550938DAC99CA34C736D59461023141C5945FB3FF9352ED63AD2C56D9B3BAEBCDF560D3523CC0D5D7B49F85E51F233CB0FE7FEF274AA5135A3730BDD1EB74773E7EC97A711DF4E2485501B1CA0ECF6BDBE1970C91E08943A72718A5A60C468D45744D946CE129B6D26AAC4986CFD951BB5FB813EE46E4AC36CB5C5B5E663B91A5C6482C89BB7127886D50E8590E2133CEF820C44E6CB43EF43D33014180D9A315DCFF17659D9DBDB8F57763D9CCF4A7DCD2492DDD1821C3517B200DEE234DF7354A35657CD5960DBFF1E3E733EAC19768931A29E5992FF0B6BED6944BAD1299D5AAAB500A0DD0ECE9211020BF7C3A9F822C73596C3CF95B4EFD0C148D9DB83526E3A46D19E9A83222570E182B5A90DB1242682765F807E36D7421BC77E3243176A44B9E312E14907669C9E1ED51A9DAF5D0B4609FBD111FDE00BECD49BDD06E616FEC70E05F5AD107DB887587C2DAA02DFA161F74EAA5D5E8E3EB18F452BAFD7E0FF1E5D652B798EF0AD19EEA74DDFDA7C4585C670E299D1C512DF09DAF58F69267E38D5D565ADE68A6AA351C05AA02A2580CC3B8A7D5F483785E38F16F8A5784DE0E6A654E8128AD2E68C423635748783B899ABCD8DCDEBB4FB68F190BBF9C6887C136CE5A13D08F8B13ECF9DD547C0134A0C247C22EE2B3E5050CD9E3C77CA1DAAA4A5453F944A1823CC877F76E49C1D45ED71C094AB83F9E7E4B778913BCE4D075D060D7CC2D006995E1FD664D9277C81D22D6E72E697BC1DFC9829FC3FEFD1C0996CE9C631ECDE78CE8112AB61BFB22D7C4834029AEE178DB838A92F2A0B27852280B55239AAE70D03C902999FD09A24EF1EB79890678E6D8927ED04FEAADE840A387A3BC40BC7BBF0FC382258D3D155DD04D2D7B020D2A39A0E66D16E850A91F78194381A3EC1D5491FFF53A26E691EA7BD39BCD23CF6649CD389F34545A5279D2608F79811580C5F803FAE740A7782A8956E675F8CCE4E6DD88505A36A50F13F879221C4A9418D270D8C187AD0630464FC0629E14BE0BA06DAC26F71F5DC2B3413A9ACEB8124B44155A849EE5D9C719262E4319ED149E964F909A9E71F3D7C43973D5A268FA681A3B9E0F0C2FB5E7E4DAC3CDD54AF9170E2C5BA391FF1F1441FE32C81C70B3A3A4F2CB7B06A50E0224C74B5F0D3D7E946C32C6A3CF6D7609BDDBF386229AA1E850A0C011ACA09FC0CB7FE2A26F876702162834D5976643863BA05BF1B9E97E3DD8F73C096F9095A604BD5F4D896B24A4CE7AECC7B5BF85E6736C84BA1EA6196CAC705898AA13ED75D9007762B91D175A1DFF3E4F381E5011ADDA8FDECBDC4CBB0C60E0B87D1688989015DB2FB38EFF3A77C4D27DBE871D889BB52F814233A1D63ABD98B0E57ED1A28813709BBFD3730CB68A5A7DA7B4F1F0CE5794F6E3C13C172E994AF6C986622A681F3C2C1FEA081AD3366865E40FBE23F1F2D4DE7EF5C11F710BC544AD34E362CBC5683E95377A9C66A5DC5843F92666BF4D950653313B2D1EAF106929F378A9BD8E78091FAF7719002FD048D5A12FCDEF8E0A6A531A0B767BC5ABB39CF3E26C8C8CEE8382A2129E2CF7DAB882C4217D07977C652404AD5E93FA10E7F5F02D87E0D1AD01BEE6E54348599AA82BBC5F1530CCAA9E39A15F94AF8A16ECB6DA606837603536E7D597E5785C73E3552F37F299D9F208A20AC2B6B5C1D0AD74CCE5D8EE8A7914C8C8AA6278275F08CDC7BB79EE7CBFA44A59F177EC8A3B5386FD7988AB128C6DC2D89148F4349BAE854AD421C57A2FFFEAE88378C9784B9C841382A9EEB8931E18D74B3726AA41A551AC3D8149030A23B1723FCB105C727597BF06194C6E78ACC33E91AED20CB6C7DFFCFF1D477779929DA3E2E43941A0EA00000000000000000000000000000000000000060D11172024"+ , SigGenVector+ "ML-DSA-87"+ 241+ False+ "326A32B5141B532480E5754C8C9588BF99B10C99DDF705C3A4C8BCB1BCDC5D1759511C267A392B96B5AC1A1D7B5A394F56DB0A3AAFD9ED590278A61157DE9D17A3FF17C115F241960267E257BF284B0682F43A63EEB5335CCBE697BE9D1397CE9E34F487926056A95D69931DF38C19F6771A72DD2D1C391191034F525A9895FC941064A0142101474A10315260342408151201162D1BB148C0028112268CA402520025528B04441AA80C9A2485209985E2184DA2864148B08CD9488558464924062624184214A28D21908413054044824C080968E3489163304D422810621608A336910A478C43283204B500E3226112A644E0022CC0243219A5310B394A18B31120204AE4326D64104E2320700B19699020089946688B425040482D881049C9A2889202711B198050261003C981A10620134532CCB68D14013003812921B74854468440466419909159262910A4299A2491C00682DC06408B864DDB164123842D99C4611B28291418645CB04890801104C68423452883422A0B9101A08090D8C425E2940CDBB64909A8614A8011E2362092344824866CC83044C1808942C285129010C140610B474D84428162381113232003C0895B346EE3308A20A8112446020B035298B0114230651415802422911941524C2204A120422197451A121119150981C00C2108315222716138865322000C0664C02680E2A8681B248AC122861130091A9360C2262021452991402020400842A80D2400711C200009898822B86812C4088AA240C8346219882808406D94082621115113994012A56D14372D14C911DB288C84122ADA9080C3C2504940081AC1092406111A078002118D541804234421643286884449CA360910C8841930241A452EC41488632228E21471C8060D84A200C02222828830CC1029CB9424D0A06D23060899B200A0224512266A089021111989549691C2482D19013248200E1C064000328ED4C64094164E01118018920962B05159002683B8001309610A10210C940859086A14B16DDB268504042D1398606202809C942D808451001630D0420C08104A04435201434A89024113466DA02490D33245039844229371909430C1387152C880A21870DAB080D00204CCA48C1C120063B24188A889504481CC400E51C21020B4001CA32DE2B04D5B14854AA88514186A24A3084C42720A91041494694984608BB42551846CE4C66D0088490220829328200B3226DCC02514004C5140919C920CCB462E830071931292E0228063401209254863308912248E4AC46511400052A68061960DA0B8242139604C9249D3A66CCA2090E2922CE4C865E2306E64C0510CC63004238E11307148A48441162888208250B291D01622CC0450DB0230821640932471008970C04050D326409CC86101134A1A344012463091200C40048E5C848819062E52A86121174ED238491A4111C244604B04880B414DE22070630450C9466940846011296C21086840060D11A3445A906CC3425091243189468ED0068A14174951342204316642022660082850262884208C1187290CC0284B204C51428D039431909465111922922824CCA06018154C89C4908AC66C99921061A42412B30D1C38061CA168D31292CC9800039831044565C30202E3A0810C860C1487308A362681469258B664C2467113128682482EC0326DC34871921484E3C069DC306D18A6649C2012248730A1B2652028705A4005033190A0C02091222D8240899A0000128028C188054932910125511B3600CB824CC0264E18100609A1819B980D13418C1B864CE2964810B790989441D82052DB100414046E8BB289111068DB322E4A028E12A43050C46192C44914492A9B388160842199A20519288291A4314A041152A885C044658C964520C781080448DB386864048A09B624D4C660C812400A264C91B000A4262E09098E02378ED03689DA46902337618C202AD3206D59145192388154B051D2906D129900432809CBC62122C67014B209D8468121C28CC132250305058046410487605C3491CA2651CA9688033291D4182019946112B044114581803285239901223024102771541645C91065CA104112892DC2360402922D801282CC309288B61048220D98288562828010258D0C342918810421403083327264C64489340A0B15808A207012A14940A20C4096051A0531C28481019609889450093184484445749697E96509975B51BEFD2AA63C2E98952B5B50EEBE3D7624A254838172E568BE420B6AA3D6554BE087A08BB09D14C15C41806F8F5309E1776107389849650A0841058F1CAD023359319DEE38DAF85122F2FF6DFC3B9DD444F2BB1AE2CF888B8F15576B1480C9727126BAE6D527E54316EBF6AB31196E4EBE365FCF6006921EA3AF403AD3731A35F31D0E24D82DF96AA2F7BACB0184E6708DA086A96A987C1BCE93764EC70B2B9D0B413F668A66D24AD8CAB0B10B01D1CE57EE28E4F44AE01A2C150AB1FFCAB46468E9BC342EB36BF242F242393E7554A9C96A96EFC2F7E3213EE084803F76A7DADFFCE63E90369D7155E7219D183D261B62C12FED80E4F8C950355B8106C869BAB9627517BF9E34E9886233855AC266FCEFD3928628F95747D3BEEB9D70CC12808EBB10E2A3235C8CA7CB45C31E2EC381010BB6702548CC41950BF85AB06430C2E39AB35A5108E13701E190AE90C95B8657A3F63D90DABC2B9D395E73DE2D47AAD0330890DB2324AE8E1A87C397B4E1D9ECA14CB4DB88AE67644AFFDC9067F40E0D4607446E1B351113E67D6D7F4D3E5394EDD5A0F7608334BEEF67BA4FDCDD73FD16A7B5D276696D132A65D40797E85C2819B8141C73451D2C4D69227FC282CB5A4C8CB9E0B1F6B7DBF57ADF9D5BF01EA0B05030656175D9EE00873295F34E3B85C64F3CFB28B3402050A0BDCC88E6D64799D208F8A43575BA5FE84C1DACCAD5870813336676291295A58477C86FD419465E3CA56AA3153427AF83154F9450166336764D7A59677ABCF09270BB5B36E53012EBF25DE6711F13CDAD210ED2B3AE190255ED1B4B1DEB79629603265F15F2667409E6E6527302F42B4ECBE78092D814C73E2565FDD1CA9CE4424E3F4C99DC3FA82D2DC1D64E90F1655DD8B9AECE4595CCBF00D9CA5326BA0B3659539AA68ADA482158AEE3148D7E83001DA0E56DE8348E8F01FCB857E6BB9C590CBF1E6B2076E43B992BB4DB601C6171126656F8E3172B7F796C47D010A6176D93E353ED0AA1747F0F8952E6AEF7F902013A2483111CD5B5DD3AEACBC76BAE1E900A125DF4ECED2D98A381DEFE7C9CAE562FE859F49C4333200B799B9CC20FDDF8622C822E33C05164A85A99074427EA725E8C4189C70F573091A0B4A472F05BBD2650F7A20F325259D4E49975BF523051EC37B47101148A56E30A2C2FECEEA6924E3033C4FB0D9D70D397FE9A9F5A32BC84BF819402EB050F9D3410F8960CA45C31F98AA88C765316831F38465F239D911C8A17642624C31811A3031ACEDF6E27D5D9C9931416AE688A465BD07A38D38AE8BAA1D1C0B568B1A3DC6F16E9D83324A959E2CEA6A6426C89323EBCD4E3E7317CE74C556271A488A49BEFEDD5C2F8F67DABAE474511AF90E58E4C509155BCF4755E2CAE74552A9A83115989EB6ECF103B2EC4909616596B9A0631B2D6D6B1AECF57F30546F5B1237FDE644402AA8763C023D2430A0E58C22400E2855266A70B56EEC792B2A458DABE7C835165A11C368C331FE3FA7CA8832BD54C01F668E7377D59D74010B0DEA6387937F29D7A878CBB5C1D5738B6822D0E8696D17F23CD2E7A50AE1181268C4F584B0D2E246B1C2FF029B20720ECA596D0E4C946C015DE8D7EF370FD016CED707558D52B6223177D65843052359F488B5423110D69E1CF7204D76019F5ECEEA89C10EEC07DC00B0F9C57EEFF7A74064496C906CF59F42DA843EF00C3A19EBE15462322CBCCCF09516B853384C7FFAB23995EB8A4BEEA9AF0560A3476C2886AA6E1467C59B84458D9D90DA0FA7552C4D0FDCB330CFCE0B05A53B41E75958A162D997A17F48F307B5F8872A13C03D1BE785464FE1E5DDABA564D3A50947DF15AB65675D895DC2BE8B5C346E433B2DAC739C010856680504E9BE75FD01090C9858F9BB49E27E2D176DFAFFF3D953E9E3DAE031159813AF05372EB0B9AF203485E1F929995F62C1AE3769EDFE724BF38A3F88FFC8CA4378FDE0FFC1631E6A878D2232C67EEF63A7A3F5B0ECF08D2B3C5E232E8A28A995264EF6E63655D5CD31B0F17E4D525E76AEB58B715AF30F3E8891823FEFA51DCE32958984B61D7E2C488AD70200D80BA16CCDF2D96559B718F64CA63113A442803A776E81890D89B171FB05945C0D7539DB748540811A6396C9D07E10384E74D42375A81913E7D0B61C796CA54CC025BDA26CFD043C0DA42FCE5355CB6E0F03554909911245B13D5987BC70B6188DAE392EF23C79590BC06A14151730D89BA936BF20EAD62237C9DB120EB91956F0861C92C37C7800447BFFACD3FA50B8CD1043CCF164AAAD2F128AB030E112CDC36A8AA1C8D60DB2BD9AC72754E508E5D803AF95C73C1EC2F6C5D736A3D3FCA4793E0D03BBA67E0DD5D4443440813606E9F9FDA3241E24A183DC3DC7B86712EF0E2387991B7F65FEC9BD856B90706DD5C15687FA2ACCE230E72F4795E624BBB13AF7D38B95DB1E8DC9D52A906C786187A62786372A956DF86963EDCC0BB4B54F3A06202138A0D186F54A7F69F8D05DC06A36ADBA2C23AC91EE66D89B5A468CD52145C30A2B07412158092F5FD828526E17D747EB9F7CAED977377F561CA76A22A9305E8C36D11092A0E00F016723C1FEAFCBAEA0F88583732B1E1CD5EBC40FE32018EE953EBE566949EF39A9265FE5786606F699189D0E5A7791269C9E8AB21B53DD77320C228DC06BAFD6EA1EF7DBABF582AD09FC752E768B88DCC61D9BB7122B47E49C2179467149081A02065EDE3DAEB9B54A436B7BB6AE9902226EEF366CCA066E6F1214450A4FFE1EA78E900626973C33614B487530BFBCFF2906D3A91D70F6D00770FC5E17E37928236EFC885ABC62D78CD1485CD11F2417292FD6DF4BF5FCF5A24F53450471757EF22333661405230FA161BB98E32E6D692769F3EBEA117A725E2FD1DC30825A87EC44949BA05FE4709A517F17151CF43D694651B2781293C1A4E358F84FD258A658210A979BE5EE6A127153908F3D3A4A65513C5698E6D559130DCD75877AC8E654FB5B5CC027CCA158F46C356864667B7E0B73636A9BFEB6F53E4B7C8B51BF50F4D8E376F8AFD92EAAB741F09E4058A68F71E217BD10DA64AC60A31C1BD20D98E29311B9A305229D4C6DEF150E7796C8FC2CA0F786553449E1AAB04A61FF395E4DD31B75718FE13338CC617E4D2DF00363B56142F2C09F4D443A1B9AF0EF21EF343CF8B8515FDEC8BAD535FA14BEDCB66E00AF5334F11C3CC192351729543D03EDCD6DAC2306F8B9C926A176AF6B36EEEDA116C361646A3AF6B3F95148C3CE14E3C8CE3D44E642CF5C03D9ED7E7BAB56F1CEE0DFDF5DDB161CA9207FA84D3DE47E967A308C56BF9FB13E73D3A8B233AD7D4850D24B24456E2600E169D87DE3523AD2CB607357B4BC339EE36A6F9CD3882720B156FCC1DDC24D09FA12CF03223017E32FF176D0BB71506F01D551A0790492D6C765E00CB601A9C36B9CC72D5D54B1EABD6092824AC54E8B49E954D0A5A163B48256267133BBF10322C92127BE6B6586456B78DD1DF6B5EECAFF7B2CCE8FEB38FDFD28C674C706B729E53B19226B905064262C918145F1738984552BF99788D5BD26836D07455CCC2EF1542280589BF3BF871BD67860F35C27CA8D7046089BC25082050C2F4611492912C03E54DEC0C436C01CE20337DAC3E9CBCE14630162C709B7A4B59597B0939319BD7BED910859691148F2F774184E258F5455DA242DD32837512D12EFD110A765AD6219C0EB1EB4CAEC299894C8802A12B5E7605AC2A2ABDD15B1EF9797F1FDB29D852FDEC92B4BB82B459FB33A1DE3662C1645965B2DBF51483A97A4E9BD430C5AAA815ED66589A836B96BBF86810931506ED0026F949F0DFF9C1B3162F2D890C99E63D5C970A3594A3661A10FF225C4B44E4E2B678C27735E924F09091CC3A97A64E74C178659EDAE4290C88CB30FE190EEEA3ECB53C8565C3D13D99747F3914A296FE3CF0F3E5F28E151A5F6ED5D357D5726451165FC3C07955878AA86DD4709A08F5DF5FD885431AF4C83CB5153E7C75004EED5EFBDE2FCFB1FF4F13F2A119D585FABE0CB5FE6339F682AE7FD3FB20FB1CAD4DCDD8A1BC5909EDC5E3BEDA7D1B117E498E0939F87703CB923DF9DA37058FF9500C4CB9069343A36038CB67AC9AE5DB8787F1DED5ECE9DBA7231FED0C897A47E0C7E4B50B18B3D7F04BE7B5C0D6587CA5E084B38B2151D1BBB98F6515A1AF48E0E31C915C7A6586D723E1211F061702C0D38E50EE4A8C269BEED63742F2A65D0B621694F222504D822A9909EC45A889FDE0A0A04EF3D862D5B757F624060EB51708912B2510A001B689BE87823041F764A6E62828E0DFAA1D5910273293C5642368593ACED4AD1BFA9B41C1C9920BB86AF0779040631E919103309A69A4B4B114871F0965680AC69A09DBAE204CF9D15D5AFB1D11D3BFEB49FEB864C7798CE12C4141FD0950AAEB549E63BAB02F0159EC9BB7A20F4630B1DF088830B53FDD415BFB891F15815416C4857250F9073C0D0DDFEF4609B178C6DFAD676642EBAE50C8731767EDFD1B9F8DCBD595C81F8FCFB624C5A9DD359C9DC48422044193632F9D29A1E2343A85363746DB79ED8939E527EB8F983CD5E6761652BA65233F6882FC6AB93C838CD8710B39B2F40B272655B238D2ABC52508095533DF374FE08630CE29939198C51568304D2DED9E6CE4C4F17A7E985E03B623"+ "F0136C8C07CFC92954E22DD375C72DF823EAEC62D6A7EC30ED5348C2C58E901B2EBEFF5A3F6190321E667084F8C221C61E31FBDABCCCABF06EABEE9801E78FC5672596968D4C291EC5A82E81975BDBDBEB49AA5553F741C4548DA7688056DA4CE808E0D49E0D0FC8864FD2FE91007D16136EEEDC3478CD47FA34A39D15B06E98A6C0821571EF4AD925DECD213F3BF9F82ABC18EEEE810D6A1D52C3E8E735A7547112398AF8A8D22C5DDC5D41E2EA99BD1D8AD3BC04DFAAE04A7ABD441CA7B68320B00CD9F91783FD781DCE986157E5DA96DF909DB9C8057DEB33A27CA54B492C75BF99C8C5309DAE288C84095C9614EDF7638F66D32B42BA78D20997C5AB89FD272B2E7544DC3D254564C8FB827E4FF5CB67282F497157A435DEA40355E312F1880787AC8BDF541F92D8D72BCF77991DD55F7BD89D1204AFD7BA5599FCD0FE1D8F98731B9B2555473FC5414E202F5E8A78A1D35401A289FB2223D1E12694E5D426F75B2E0211BFA9B37C7A260E27D2A99C7C2DE1A42415EAC6762C5ACE09C3BAEC4F75F3D9FB06A5837878168D50724F7EA0EAEA4ED1249CEB342909B742222C9207778076514987667BB3C8C0001EF07FE39CF03656C163CFEC4422826D2136413916AC40F1331B9021401035F4883056AC76D8786366A9B4B871A96CDAA8A78BE8BD4930759CF6BC15549CFF761B4196B1FC2B14AD1FFE834721DDE8EF30E5BE408651351EE153D78B859762E9F0D69559999DBF66E7123AF26812C9738242ACB786EF4CB19B2F4EBC92FCBF59F1EB59EA516A31DD9C4430D202D04E300F698845F053D549C0E6C4CA7A5BAB17E43CC3B643D8DD0A879D0B5B390D975FFAC9183D14A396B5A29CC32E74B87BA7A069DF9A3048F927A69CE8870877E684143983F5034B81C4D7F4FF3C0083A6702513D22091AC987DBE1B8995676DA9A1D9C4B11B28A0BB738A7146C9C5B843303250F5B218C049C654AC95BEBF0B351EBDC5F543B4D9C01AC6EBE0DB72ED2954D948A19D8C6666CE0866F4E9BA815E6A4CB9B6EF0D1C79D3EE81187E8F512C4996503F1D7682D874B402602DACAF9CDF696F2D9F37C0A51966182064DB1BD16994A1995760C8DA7A59EEAFB236DFEB0EE6F39960BBC0EB2FA7A0E550FC316264B5970D98BD83D203D0E86023EF4DB94C681C963D808700A13A56DA33CBAA62B9119FFFF02B542B7D3119E8BB62268581D3B1FA5FEB7404B58739EDE800041DB8A3C24975CF8DFE637C82B1843C87DA273FDA37E962067C5B14CE7E2AB7A700E39E25423C4EC55C0B9B5D6568CF18840563A49000E7BBDDE6FF973A2B9999C0E2EB809CC18D3ECCD0DB1FA540EC7CF24EFA17C4778AEB633AE6CF7A1A95382B419E52E4C1677F1D271C1AEB4BF6C4B67AB2D064C4DA5E35CF420340AA65E8EBC76B9B04A3F6A41F5FDC9ED6BFBAA63E277AE002800096DCA2F6DB6686DEFAC6F2DAA6F235EB305AB550645BB7012BABA98A1FA42CDBC56E5DD0F0D451596FC43856D8EB664FC9461E0CDE466EF3AD93CA2CC0F6A2321DCBC60C41FE314F679396D24E6594F6D86980C1D649A68B7A9ECE2224B9ED069686BC6CE23EB94908F55685D1E9927CB5EFE6AEFC39B9D937A84BFB473A45BD1868BCE1E24AE5BE187CFB8F23DE33C69A165C6977066E52689EB6C8FCD1C60385796305AF2C9311585DB46B8C2178EE7048CD0D3DCA377C6E6B0E755B3E1BCCE888B19BCDA40E67249A34AE82DDBE76131206189EAE8B361430A9BD94B7AB8C0B36484A978C0177606CBCC9014CAC1829A32C07CB270330D34827CE1B6FD3E06335FA2FF5E719401AD240A2DF5E7F3CFFC52D68980FC206663D32B1CD1B49A10E9C288088ACDD560BC5178D2E6A6E05FAF4EA6510E27C8CB583DBD33D250631606F0C4E647930EA2B3E4B159C788A421070D6D34CAA111AEA8DF9B996099457DAB306B9AC7F76D0C7370F22FD661AAF3D72DCD27CE0E46628B8CDDA4AF8213A60487CB628C7B4ED4253B3311039D3B76844B2B844F49F56692E80616CB9247B68205BD01E82F3EF2A20C71C374E94A9193A936E748A5BBB189A381CB6DF0D096FEB7FA4E3EDF1FCB226429FB0B7F1A24A37407AC39366E8C53EBD107EA6156DBF77D429F0DEE15AB45701DA7220FAD6ABB7C2CC4DCD8143147083C447A653B401B5BFAB8E9C590C048E4744D77B58480305B6D15A16237F2262C64AEAE5E8FC9BD856CB68FEDB61891DE3704CA3A5DABB6EAA30A2EE8D578128A969FEE329E7FE94E87C5613B6AD3659C0E86EA4A7CBCBC42D14CDE7908355ACDF1FAB11BA49055114F36F719EE88277183372E3046047557AB82C2E39138858EEF5FC39254E8C32CB7204EE20467D5A4307FB19DB43A1DB58D160467CD0C419E290B71B43C67E144C4E26A63CA026DBFFA9FEFE2D6BE8A51140FFA462D881F118418F47217C0442052984C4E9B21B5875C21FFDC2DD5B71D7D9984708843DD012105858F89C9D1FA7C1DB88DDC07FA98708289FA99302719987E3BF9CAB03E1B16184685CA3609E62C991C44FF4F9BA9834F30B65EE329779B5FC593A00E939852A7DD67B18DE7B138E8EDC2B7B2695E11DC085E5890B2E38FF813B1F742CF1A5BD40758ADBD859BAD995711D6A96FAF01CCE67F28EF6C6A007B253719954A9FFAC46C27616B1AF85B214DB50AC36EA53F2D4741D2C9DEC7FB78C3C8B5D0BCFB5E51D5454EAEDC490EC948F7B757D2D443DB17C680A8E1D009BAC63DB429B19428F51CED97A27056F3C7EAC4A9E872B1B687F27774D5E06B3F7CE7BF05C000F98DE2E02A19995C36429CAC6104115CF179CC0DC5234F77092F3676E3D8B2E3A038FEE5F6D9EC8DB44D15B6FBEF0F91A660C071E97049F863E9BA16975D994BF9888B70A5C7BDB1EBB57E0F2BB360B335DDC48D24BC7C3BB9A4ADB48BD833324C316A5CA2456F6AEDF52E65ABF3AD119E45EEDA8FE58D63E52DC61532ABA71A1B2A02C8C4FF0914A85472B21051697FC9D05032E25E6C5F0A3F4BAB5A17BF60A74EB7056366F1FE78C138FC34879330D9103CF20C4E58D22F662FB6ED71E5E140B749A382329FF6C26C714B209BB529E6391D40090017E0F23938036DC45AFA0C5987F9B18969E66035F2243F4B1B67699D7B486BF24AC8A0FB2CFC18F03CC480B0863BF32EC76171893243C8C8EC2F3550F3AA88EBAA92957B2878EAEB6F9289CAED2B6A40B86D3804241684D043D97417D78882C68F6DEBF07CBB9BFF65959FC7241C9260370CE97661185996931236B47840710F59BDD77F8BA14D91E3A91C7DAD27F27A8DFC585897DC7466ED77A1BA5D7FA872134CEF1687205C9F557152E57F370E34CC842410C2945C38C46E9FE8309313368E9BB4AB5CBEFEB20D2AECE41548C58D5FD24D60A6371DD7E84BD585D269BA3265682E44B45E1719F4DA57D06C3474A9E47241FAD578AEBCD0790963D8F39C550B592889E2B9FF91DD22E954E3EE40E1AFBE796ECEB59B8DF200FEB3FFE43CEE3F3BBEBD2624586DE100B06106A0CDCF2CF94815D1CB90CA065EBE3B17D0A28D83FFF5B1AD7DEAF7C44B3A6F8FA1ECA04C0684A261C0DE24FD6C2B112C2CB91B603002E34119D5B19BB9D6AFF09E26316FDA67C0FC12BD86E346CB833F76B1AAFB2AFDB5D35C8AA3D780B52F57B16BB8FF5A97DDD7A4466DE5E63FDBBDBD3250D92E615C47259BF7B83F753270A37D2322893061D79AD95C8744DB7F4442D35CB335C2C4D49E422FCC6F78A100CDA2B165842E9109F586ABF335FA8945686720D086002364D2A7412A4A576E46DC9271DE10E7BB73B3E2B35937A78BDF6C6A23CDBF7D934F70D24D3F9254ECA8EFE39AC1C9F5BF9C24E05E294F94372FE6E37F27004E2A67544BC3C073E054120BF2FD6EC772B2DB91241BD9E4E78B3492B11FF676F8ABED4262348CC140C040BA4310CAD8D9E106D9E61FAC3B6CC988C8951134211E651CBDD648A8899CC1B0D04746BE7043D9B3F54104DC0559B38B0ABE19930190880101810EA28DD657DEC22505EEF7D446E22812C4A91F669DCC88C67ED17B93D5E6363E92B735F1C6C9FA1442A3263C51898F9526D776CE3F944834A4449041F314C83C2B495DC6775EFA4B775F0AE6012D1B1F136EA8AE126187F08DAD40B10561634D8937696CA8A2F3CF85992FC784C5E1F505A5A36652416394B606697B1E7FE497268FABE31210B4C6D4FAFF85F64DF6E7186F20EA0889DD252A8B8D6D69D8CA751F26157BBCFA54369FABF02F768E104363C343E8E4F3BD3C0B58ED76575462D52A660416815ABA6C5815EDB5C5D88E07F1D4B1DCADF4DB233CA6C66756BC4C7D61B607F4961228159FF1EDC016E9D2C1C0442293D1D49B325AAA28B9605A5E35D604AFA8AA4E5C247883A1392735DD555036C1B4E21282116EA242F236A0A9412B04F5E44F330881F914B653F5F9D4AD9C5F5A09AB794ACBF8FEDBA2D055393ED7A7230BD369A4AA406D2E2CEE7B003A4513476FDF70E85E10D95DE3ED48FF2731A4CF062429B0E3BF298A83C75BA556D2B8EBCF7C6AD3A6AD5725945D53D3CAF0F58883C1E2F330E2B3DAA32795A5366DBB47B103E01CCA3880F9175A93A789DCA2296C9213022A79FFBF99350C37966E8037F4DDB07D1A621A96C41D6D20925D0FBF5F2C03C6A5A02571810A7EE4CF5506848B35F4899C01557068B8A24FEB70BA0819DB8C9CF31B6DE9674C7E6FD2735F665D58A14C65199ADBC8F6E213898CBDFF7A51582EBB63DD3C7259A326DD2E774ACC3BDC9970246971FA4800C057E0330A635B8E59CE2B9239E21160CE8B9523C9C739F3CC97114B08AB264A5D9DCF5DE3182EBC00E84E84CFF863BABB2D149844A59E8FA5C35B391F5C769842EB9AADACDC9DD323D9C96856EFB717D030A77F436935F83B6C2A70FB5285CA509564871033EB52AC57F17A9DC9FBD034468BA526D73DA762BAD1883F3A3B1554673252BFFB6E5BD5CE56711098A31CF459AB2667EA3182E814728E9FB120F2D9121340E6E039DF7CF38F585A25CBE5397149502D043777A81942E9DAEB5F5516B096DC8DEC8E3DB9D83A9F51E99C2B0E4D79EF1637A37FFE912D652DE3B7B792E131F650DA22B3605F97D0742618D58E60DD82A2E23658A6F7B6A09FEE8D41A8BEC56ACF2A9BEBC01CD15C905A67A541CD53BB356066584BF110ECC5F2C031AC283AEBCB5AA2D21C02E3094A8259C7D47C9350203969E39A80C5620CA7FEB3261BDEEA223C21D52D73E49F4400E911532D94B4A5E65D8B125D17ED3FC05600A2271A2393BCE5692139467DB47C30B27DF52252A890EA304F729717A5531E63F1A91B74B3670685704D458026ACFDA964CD404E2457A9CDABDD8AAA3265B676C96D1AB225A58FBE4356BA525E960DDA2CC97D5BC9E7D11C5B412A431559DC3A9FA41624772A72E2291A260564050ED610AFF40EA0462AEF9191DC9887A8C8795C1E576D6C1E8A10D81DA1F4C762AE37E9A1EBEDC32858E7118242F13B4B0FDB0EFD68023CE8780C9DC726F92D3C61D49317E39783D766B4208445E41B27D7FF16DF212947314FCD397FC539EC796B57666E19133A16AA3026BF437CA5B17C62F103729893C842BC2B2DAE2E078D737400FB95978AC9028977C4F189ED7AEAE79DCCC75FFD7C52FBF5A3EC4C9011F5183930F780A90A2FB8536004E6C43850176A4E4FFCA5BCA4F88C3E91AA8165E98CBBB5C23D924525DDC96D0C9264DF3A974E6DA2899EA0CE0B6E89D5CEE7F9156FF279C5FECFAE850CC7EEAD73B81382520D899396C5EC0671B491DC1751EDCC66D0B4A5F7314810E20468659522729D4A9127BCA45FF614A8710736E09089752D83CFF6555E1389CA3DF39CDA947788EB826C90C71B4026D70A040ABB025EE59142FA0CAB6E2FB731EED7E486DD2EAC7BC2494F762AB827806B647505456892227A6A0940E34B24C1E2BB7F42D0923B8F1950239471600B4C3AFFE692E199ED6E1AB51908ADC5DBE5DDC0D62604628051F766AF32D6ACE79372B706CF310282B8E183A550E432E5B8930A4158F37E0331CF17A57452DEE1B015FD4B837E6EF02F19D9A407945F69ACDB0211B86CFD80BE4A853E2131AC82959B39B05DEFB0BAA4B2BE56843C4DB6FD48E710E5647CF9C64B2B443913B2DD705DE0555B5CB283FBE7763556C3DBE98EBE51D41F1CE075E0950D0073BCA22AF8052A5025DC755F1CF63DC5EA6D9ED34BAEF0EB4EAAE29ABE0631D0ABCF8A39BD60A2E5A727888FA9908F01B79E8C1275A8C00DE66F8B9AFAC187BE6C2EEF9C07E5646680047C1BACD74ACA45ED5C35C49DFF8B6B921F11C722FA23433566E46AC4FF13D440B075342B9FF58152BDF4E4CAE3C3134A50F6F938B73079B0474CDCCC2D03E24100A86102D5BC99185A8E5373D39A7243B087C8D154D4F9118809E8291DDBD11178B8555616C3D229D85E0DA05201BA3F75FAA7FE255AF9660E87D6A7E66166682EB5758DC9E84E5885EE3C4E3B07A96EB1DE9A84340FD6FE462F4E552C681804AC48A7FB607C0E008EDC0DFC494F855CDD030FFE6602A0D6B0E5F9D7E8CD3483DA759DC0547661C1CCF096E4D687D44FF7371250FBF27A3A18FC71C23DDB0115C0E4742D8F8C83FF53FBE0E03CC9D322A837B2E9D2F5685E931D5CD875B5CDBDE0D3AEB9AA8C77AA10F55664411D4765D4E787FB98E1E97F9E2EA9F736A4E4AA60320D842FA80A20234BCBCC65068D3D14837D5A602DFFEE57DE51B4ACAAF5ECB1BB6ACE5A5BF95241E81988C0414DFFE346A22C91A665CB0057E018FA5C184CC49C0DB2F66083B6B00F00D5E8F03637F75C10C4E7CA88CF7E513EEE12A95142CE02890689354BF41B99F71A65E41C3F45F2C25BAED64F5D9286E84254AF50A72966BA053F7C3AB9B7BF9B37182B466693938D63508AB97807B882C3EF03A3A6D10F35BC4ABFD2C36612D6585E5706154AD746084621B65F22F4C5024FA0CC8E07E99E2E40914DE99BD38506D9DC7CCCCAC264BB91BB906F2D765D88997E96FB29B034CB23D0E7BF92D7B3E690D3D5708D5E70E6ABE64E3CB4C1FF2850D0B15FBA25EABF5CFA2B4BA900A1DFF845311CAEBE695ACE5111046F81C55F1C90C044B7BC2BC1F0742EE6B94F37BCFABDCE7A110BB92115690BE5E7DB238BE0E48AE0CD9C8DB6BF521857DF8EC3DA3708DF930DA643A7719DD13562EB1418DC0E658D297EB5DA40CC2064A33ABEF6AE7EC10BE40F8A5661B45F8366CE82DFB6CB929E7572590804C392880CA55BD1214803F5CACAD46E9CEAFF8276CA9DD3542D0A0F0BE04EB2649BD6B86E17415146C64E04A1E70CB0CA9E7EC424CD93DB8C00AF4E0692977F5CB5DCC78B9F0D1BAF14400A9629E4036B0091A35F9B68415362577B22F07586D0E43E85EA77C15C41CE5F9D0EF73DD38BD23E7C2978538B4BC73F5CFB5EDA0EA15F01D40D6FE5A98CCB1BA316A40892BF42CA0CFF26E5D2B83BF5C0115E276D7166310799FD2655EF2D930F5B1BB7318750BB12BC5B64E2669BC123B5C4D72FDA76CEDDE59FF37A55865854194C7E896FAB94E88DF1D4304426978D678BDC4781AC5BD99526ED1C493904809BA9190E2938AF820C58DC264053F4C2AC3CA36A06E40D224AA7B21F3D91415A28149438E943CA1AF5D32B9EDEA67147361ABD8A933C952FE1F251F343A55537F7AACCB1AB2ADD69C736969C13FF249CAAF5B492C93904E4C64F63C6BD91519C6F4F8DF3431C4CEAAD86C398215B7F45C90F949531C487759E0FDDE8158AE08ADEC7F78344EDAC3F8AE66750BEA2F494278951AE2C205466D2D2D0EF4BC7D104D7723B25FF96DF1AADB0B5863B5A487EFF19BFF69FC80D09EDF72A140849E0406C92AB0711F3BDE77FE166293A325E0CE6E9393C183DA782795D19BD8E4F0C03D9FFB899673F682855D824374A7666F7A4A853045DCA40152C12367A1C03BFA418B0AF912A02B549FD86F65852D394BFD725B13DE324C48990D29C1C7955AF5C20870AEF263E2D65E646391243D0F7C82EAFDFCB5FE02B239FD5DF47689B8026A94A295A9BC35FA6F1111E1E48545F6029F4A8097C589659E1A6A285E51E76F4647176933111BCD6B95C13F63A1CCA425BC465EF8FE4A81FED6A7C58AA18F31517582B12A6AB248FD437FE147883B45DFF5C9EE8239E4C8F941D4E65A1911EA7609C79FE6AC99EFAAC897DDC743C3D3BC01DE9C8B063A1914695EC35A5D10942147250A5BF846BE9155F5BC52F93257052F436BAE33319C1323982D6A2A459F85E9D7FA9BC802006C8DCFBDA34D3CAD96C8EE9FFD967EF841C1D35DF9B34B6012990A8FA6C81FF3CF40012D2C725EE78191E0B58B573890797DC2BBBCD1416EBAFB300235BE3E767D79F73626F876037D8F2F3A21055C429B3F4DE53F9D6DC6772825A276DB76B0CC01BBE79AD3CFC76301592998B896E09EA6DDAB6E0B783961920AF6BFD904D6B8C7F3B7E6FAE978EBBCF0B7D6CE58811EE57C7BB22F78BFB9A1E2834CBB8353B930F7F039B4C51A218FDAE6CA32DA431556B1432CBD06B0B82D6CAC150A08C6A3BBEAB1116ED8D028A939B9E030ED254DF78E5581270EA3E1095F1284EEC45A48BA17B758283AA4AC336A3B61E8B9173FF19AD9655181BA7242D94C13840393E49AA199472C16E8D1F4A602E281E5BB5572968882A9F5EA545FE46F110C9952354CBC46A30F8AD5791D825B3BD51957B1EA58ACD639E0A1E045A8D2029005544FE2D8FCD896A6AD1F6DC1624AD52FFD31E6FF1760E8350E4895DD49BCA9433B7FCD316BD34B3EB001B8950C482936BD8E657B45CB8521804EAC471647A765B479098B8193BE63AD2CCB9C826983BA90AB24F1225EBC450B4EC42936F86A7C9E3AFE8F02B9E2E28087DA021091704577E33F80C346A0B3090C4463763EEB872EDC2DF3AEFB9ECF2CE837C211C55150B23A68227F9858CC2DAD8BA0880ECAB84AEBC762E12380E005BF191B12CE6094ADCD7240F63572F63E0F162EDBA38F46044A91725AFF1D257B5F57F28F96DB44DC66441754CE02DC5D2262FE48FB3F5CB2A8B2EE133CFBF1E5C70FF911D018A00D5F2A1B0041EEB638A4AE55F103EC8A5869C3A737E9A40682445943853B800C775301358697C67550EAD7B7437D363E0A943492DD90F30F95B0B2C6B63EA1105CF2D22741F71E8CFBA1372BF939B7340B0F82264BDDFF4A7A432DA424C2168715D0F1FAB2414E8DBA806FDCDA6E53C255D789CDC6B6E04DEC84698F84D90C6B7D129A4CFB0C0B94DC48DD2C2AECCFE4D3B17E312D9499B01DED06BDD62AD598AC9AC8AA532674D408B6616067FD3AF3423FBA5A71B90546BCED7370E0144B3EA8442965DF12238D0BF0EA33DB569E905DE7D121687DDCD6FFA8BC77B480C4F2EA3CB6058E98DB146123D74E8F05EF786428F0A506357A5E49B8FEBB2685"+ "FEABDB56BF7B4E5DC3966E89DB7C72447844D26198A0A062B82E3EC678845BC90F847D4E30B170728BBD1ACDC38C2E35327AD4192CAFED7CFA0C0730EEDA1BDE6332160AB1A5BE4D5DFF10861C9B6FBF5E2DA09363BFFB258E2FF4EFFECE729B443668334C8D62D30E37D828D47BABEEFC05FE60755F14CC6562971479CD843627E501C78A6BC55D8E16F3778B618825446DBD8205998146FEF29E12EAF0EA02AE64895DA1EF5CB98804FAF7A513BE0375"+ "86616EC2771AC1C55B85A900DCEC382CA02A257A2C61D400F66C024FBE63A068"+ "35616474832D3BAB031CD04BC85E46776F16B7C3166FED646066C3195CCDF135F23D583CCCCCA20119A1A428421312962ECD86A23C68040BF76149323BC550030BDC20F8E0226830835225EA2488206D44A7675373B8DA215225EF037DC91298A05D87927D8F2500BE4F71C1ACB3BE7665D41E50DCE9A989D2329BFD0A1563ACF0C102E4C89B0E608F14F5FBC6B65779F4E3B67E3BD11A8058270B5AB6075A5202D9A1A417691F020C99F1BA1A76741A8C3B417524C1D62E6046A47DAAE1385B8EB9DF70D08A4F4B73C502E5E9AC23553D353D5481B7CC481145A233F14CF5DA79D4131B143F38E86B3D4CF9AEAAC0C3852889D0D9728C8366E61CC0A9B6B0765A03908E2D94293C58B461E83BCF7473686A27B885B0F3FF9B88799B73476E4B6A1ACE17F9568EE7D86CA3D45ACE12974EF43C1BB680B31FBE9C41449359C9A6FC048ABF087C23FEFDBDC6BFFABCD698CE693F4CC4DA64511A90DB6C4246DFBF093A540DF96707CE128FF871B61F4C01028CA3459895C65740C62D0EFAF878065ADAD6D6FF545311D5C76716DF5D438F8EDA960C0DE457309E3814BCEC08AE6669AE099A24AEB757587E53928E5427A28656AC3EAEA93AC9C38FE5298723B8AF69F547708DE6D87368BAD585700DFA3C937A5AB4BC7D2E2821EF338076E34B7C0A9559BB64D165054C6C6A554DA4CB5D85E4B47F47DB094B65049E53299004CECF85AA9F2EE7DA6EDBBEFE28BEDAA1FCF63F66B174BD95808B875BEE6BA89331BAD8337B09035F8BE6B53A2E8B4105994D2BBC0DFF833ABBAFD0D1405975B898464977168010A8E395265C24F94B214372E21905BFEEF12763B6AD0288AB134C4F87366AE8EF85F90BA0A38E9EB27579BCF548FCABC3CB4883577D71FCD9514E135BD2188F148B418F17E13E6ACBDCCC672EF46AA0D732EFF6D3503D6184ABE1093D4705644F7FD1DF85AF7F4A934834276409C2D506209F4B366CAA5168A0B8370E8335D30673556A356965DEC6ACDDD4832DF7EDA404C3E5ADA9C17421BA911D505407221A5531D92A6BFB25709A3D1C4B8A746E77F51907120448FB8B9F71D58FEBD81CDED3A51D3C80D0817631F0B29577E8348BA611A5EE6B053E5B094887E1DF2E141984580E502199967FEA6610F3C8A2A689D3BCCFB2E0975F7200F4F7EA4BEA70601E6BD9FA47F4BA12CC946E4B2B3508FBC6EED8584D9C2E801C1F7BB14FA20D481E8A5D511F61DC27D403994EF23FE3DD8B561054DB53D18A3407129FE6F17F3468A97B9E8484946B988ED269299353AE962236230F1B1DF460F823BB41EEF87193036B519F72257908BB4835DD92F9BED4FB226FA35986A0BED4A616378CD9CCA7C9E50BE7461EFE38889A3BEEFD0C93FF6949FD2813E9BBB0466C8ECBEC808AEC6A8B87FC144B01D4F3F106D9F2A1DA977949277A0A3B7421934D554871EBFE48DE7EC7FB5F2915FE6891FE9A1B97DC0E4FB323C8FF8BA9142F6BF48C888CF1AD8A5C1BE1A2639B1A52F4000F712554E90BB0B0E73369A1E5D747C71DA32CEC244B6D0B554B7492692D4FBD0141A0198B2EFE64F7EED9957149D431626E4D3B3EA48D3D83D3070F58C0BAB7DACFE8FCC337EBC44D4D2A29A7A39671E0B9AF1CEAC5B9D1004AEAFD2BFBFCFD5FFED1A9A2DA5B55562ACC3574BA04D22C3B0303E58918E469A9CE039505468E6789D7A42F7B6D12780E01679FF00D05A71B8D99712299E0C82BD21E6019D91669EFABAE51091D52776686D3B76F964809CE1F0477589933274581EB3A4BA331E9834F722CB72AFB9264A7DE2733F2C2C5E5664EF31EC7587B92784020E22158E00D8D38209C74EDE8D40F3317C2ABC748CAC82B094BF1561AC0D54B7E1654A04F5C956F674F6555FC387320A2840F2B963EC33F0A57DFBB33F6A5F8C1FB760483B3D05C0EC2AAC2AAA1FF6FE10202337333463761DF07581E0E65AF0E1C00FC1612D4E41AC04FDEB52120321C77457A2CB6AE18068DC5261218865C5F78FD2C5DAD6832002F7B7535FDD19CBC8E2B0794737F4B3B79F625B25DF9CC831FEA38B1F72C43DE95597BC23BE974ADA3F714E6E77EC88027BEBCF15A0E8000224808D4F770DC8867BF0BEC045B68C22FCC4BD889FC27D88A22AB99C70C2B63870C5EB7B976C9882F368A91B9AAB617FFC2474DF61A8C9F99B29616CAFCC7678E99BAB978BA1B843109E2FAB9AB15050134207072D6E895128E6A030A25178D350642DE5AD1322EB3F2687B18070DC24DE3A803E5792782D4A91151CF332A566B6250FE975AFF9ECF8665D0C96C78016D13BF12B90A82D3D1EE45E449273D3CDA9564E451C0B0A63C715CDCD9A9028F0B9973015947C5B015E872C96C054E1ABBAFC65734D34E107AFF670145086652F86B78C27C5AB56F91B047F446D0965C406E4560D56E969BCEAA1E4C535E5F8A50B2FEFC83D43B2B5C082C469E2962B14351E89539479550B0BF1777F0BE3BF1B27C8299B4E3D9E52035E8B156C1E779547D513D2313C25367A0CAD45049BC98CCE4BFAE54D46C4B397FFD0D7E8E93575D458A8330AC69B204043F6A5F4D49FBFA77DF0EE8CBBA28D8BFE92117DB29DC91DBC96F91685031E4331AB39FE493DC7AF777F6C2380C278A77DFF0B35E47DD3A902FADEBB69821D58F4851557E0BEE576D8BD8C1CB21F0EB296C13209425A773A9B3E218A7ED99FBC602BEEB48B141B699FA9DB640BD0153232193C9C545D4601DB6F1B81D1ED5A3FE0C914E28EC4834BBCB3BC8BFA9B67E3556D591A6100C4CED5E1F3025DDCD4DFCDD9D6C40FC39AE851F8D64F98168A4E5F7ADCD13726AF8EDB24F93ED2D151AA080B4DE9234CCAE209CC3E4605C1600423FA61F872DA07EA21EAA1A8E572F271A2409F4A70FCFAB8B9BC28CE765953650997E600B781E91A15E7C2CFC584A66DE7B477F2DF8404349294F85221D3053E78C03CDC07AD1501E7029EB9366FC523A237F5058991F319DFAD59C85F2C8678E4A36BC58FF85BBA569FE29EB717D51467983D7337C27116656E9ACC9A82B9871FC553472F48705D4BF438318928311AEA9449896BB09381E2A85A3C4863747551EE7B7CE978043E2595A1694F2ECF6F3665871CDE2517F493D2CFC6AD82AF0F160E1A3B94DE751C5AD7FFD03FB70B219BF303F2F4ECDCD7C7D5040257175527829CA29AD28ABC434B58ED461D97041E05D1D359C9048A3E48BFF328F132AD9FEF25D19FF2334BE5E9EDDA27758A3D6B352785A8752216F6DC15FEF79BC1329EE03673F3282A0CEC3C1D42A60B236288E9D11E55B9E553924164E1EC536F990D97FD8F6B010A2201900A7A4DC73D6792B48C8913A4954DFDD437A46D44049D8980B7450338832C79CF8D5488ED68D17B7E678AF26464E5429781CBFAFE24EA60B77C1BD8213AD21569C04870F60D7B44C10AF7B483C6BCA620D933A42FF94C87D241BA5FC8E8F817754245551D6EC485CD0B2A60584EE7D76DD0414CC2CBD51F4B11FE84D74604B2870B6E5A357EC9ECEADEF1150356378E73F9A841C88F016AD23FB064B9024B24749B821827F38FA55E425999A2F91A930D4B92240F375D295650933A5C11DD9AD8C7AB8F27EAC1831FF79DAC38F4F3F41F6721EEF780A8A4DC750CE160494D0F9C5297CF4718010F8CC36F423B3B7362870E980BBFDD5C5B51C4E59BC08DCA7BD85FE22E792399AA049B03618703E807DB28CD0D6C4602B35E5C864345CAFF421BD37CCACF89945B1D92FCDB763CFEA5A16AB8CA92FCE5D5A5226A7951C8DB0AC6F3F1F02C65A7446487500636ABA4AED155A42A50E36C71D3F6A00AEB2198F27AE9D25B821B6EA3F0C096859955A2F1BDCDE091F19B28023C654420997B058972FF5418380329211E26580FB1D0431F11C6059CC696453E05405B6CC1A5C7894C2F0992388D8FC46FE3BC9A6071564ABE820E366F2667E0F913C287A5C339FF07865489235D43457847B075C872945427729A431CE362BBB8BC66FF4763150E1E64B806CA0F12A9B8F085F61346ACB69ED7A68B8402F8EC18273950504293D69FE1165383A18AE91F9CA3F59E90D888651512C915C70F9EE21983C8A956413EA58C3E409E6E23CFE9379E2C1D5D80F74E892947E76B55D74B8716FB8C77B5A4FC8F481D8F2B7BE8A301FEF6470B5957E74C219BC850F5BE6F7EEF41C8DFC6B304F4DC1C59C17F858E5D2E3BFEFBB9FA24C06FC0E41969D7E1894FE626AC055629952D9BEF2D1BEFB888155A5BF387E741323F1D1ECC54B660C379BEEEB18BC69A33D79867C01A963D2DDE2B46C068A9CF788C2AB988BAF2661BD371AA49CE332145E1DE859798EA8200C13577F0A952F60652FE7B993BADF1A4D9C980BB21B96F0294B4263478F9951C96EBFE6E28023C78CA21C6E16C9D0FF4B045B60A9DCB9FFDE4D3620FEBEF5CF4CF0290FA595E8882BDB79574EFA6442451D206084603830514BDA2A859F66752B15BC0E5B1A9E26F1CA8D350B25941D6AA291CA1EE0C575269AE52ECFE4D7EA52C6159AC4C7EA63D27E8EDA208CC7D64224AC78C4E2AAAC1C114A20E6BA98A3F86A47B07C9CF5DD99FF97FC1DCC2E2D3C7CC4A79775C0944E036852CA524A720773CD527099F989F08E61C29B6C4E09D90B84825D3CCD329D22B1E5EC42DE800D5760851D5B7EE3CE84684386F95D67C7237E9AC1DFABF44BFC5EB91340362E0825116715838E46123D3D4273582146E423437BDCFB340A444441B0732C1FF6698396AD4EAC39803165DF4E174CE979A61AF41F30D3EBE022902350F521EDB1C0083A4D1F25585EF6D6EF296AA39AD31B8DD43AF711846890BD7EB0DBC6C5B5BEA5E612DC1197030AE1D7C0F5A8AF2E4EAF298B59145D58B5BBEB1E98C197E7E02270D7C88C2AC6BC6F40D766E62226B7E66CED9E58F47AE2FFA9083F04BD810754DCF1A628EB0C7FCC404BDCFFCD48094D1ACB6D42DFD74CBC4FF36BB5100C6D858119F705B0A10BFA98AEE43B12CF27C36B600B309B0A8B0177A2551A7B9A7B0FCFEE8CF3E8C9339F7A37BB517E3A63F83C5ABE83DF98E1D8BADEE09633467CC5915A8DC4C9E00A3FBD915A5487B2F7949CDAD6BD84C31A32FF3DB9E4DD867B2BB417DAE07F23FC428A7D143C62ECE97C7CC5762308D8E5C16FA8BF853D47CEAE94D2FAFD90AB5D16114D16E1C74778AD26CCF493314214DBEA1DC3C17D294BC87FF10E0C756E4330CF400FCB601FB3952E94CD9D06264C00DE6C6E2BE44E6F93D5E39EC56B0925001783ADF1CBB104223D3C8950CDD5638A834E3B08875EE85C3D579111F34D4B7B8172A7F164E0C44A203FE765B89C5A008E504F25F7DAED8CDEC86E2DA93525C17F6D500B4DFB6F2D2CA677BD071E2C2F5629C91D9BE5EDA44D313E64510ADFEFC72DAED7161344C32D0381E416E87F4C8E56569D0AFC7C4FC7F9C3B3D50BFA698ADD10089ED1F91A8A0CBA4BF6785E7E24FCDAE61C7E7C28AF496E4D74FF0E55B01C13A2E9A8A73A7F2AA70B1065B0B2E439EAC2CEFCE3F6B8A8F0535523D43A36F791450BE8A040DAB814A329CE5230815AA1068554AFBFA9A45B05B634801DADDF2A84B79D72C322EE429F2DF6495C86083E6E733AF3BDA3B846970AB5AD8E704F855879F90BD1D41614E37415AAB711F9D7A285F570BE24148D22ECB8A15E461183FA5540BBE4ED3405F477E6EE6DA37D2CC0FD6F3BDEBB4559BFBBE97B77AA3B66507EAEF7AE18C764B120403ED7B1C80F992A012AC69C0F193FEA90A6A629AF7945C1728820091012B2D30DDBF9E0CCDD08CD9045B14905BE7D3F1DC8127EB8EA82C9174659BF3A9AFB0608C4D5B1F77547B62A5EE576FF967A32D0363E84B722DD8F1D2E436A1CD7B408C0BF16A6D185070CD2A9DDD6011B88B02E7E0C5EC9D5243BED5A7A41F54B6CD24034AE47DAF2DA58898BB2626408509D04B2303915F29F8990A4AB18FB6E1A500FDB71B2375167DCA15A15E48904CD8B2826C6E92FBAD5F61C2467CB22FE76F165BDED3C41A048C81FEEBAAF14FBBF77FB118AE1EF1C5439545804A3C3D7EA22615205328EEB02E382999873E1DF83F30F1698B6C6534DD079FDD25DF617984C59287B4E73C15A0D3B0DE362EC2E67EC12CCFEA891DED0C99FDA7A3C90E096838E7507F6BC86E7BC8BBA1A782ED685D993393F37F5F7B176B478DA4469BBB3F6E3BE87B94E2E3E9E564B7ED33E0C1ABC690317776FD55EE9106B97BA4C4CFE79199DDF2327E7F2A94DB016F100409E224695471D0E048584A4B0270CA66612B7F2480B1987B3674F280C445F027E387558A3E26C684CA8FBE28CAFA72E1D558202D9213FEF1D8292BBB9D9B69105EB3987D9159BD15E7C6321C1E1D7E066F04025F0B9A819E20CB3C34B43581E603753BA1897AD8223D94D7DF1C451F7097D486A0C2BE10C8CAF3CDC063F5E375C0D748789C35190B2C9CC0D5275B6C6E609447888B3B9BFC1E2E31C51C3D609466ACB181A30348BC4D3DBED062335718AACBC000000000000000000000000000000000000000000000000000004090F191D212A31"+ ]++-- | ACVP vectors for the external-mu interface: a message+-- representative in, a signature out.+data ExtMuVector = ExtMuVector+ { xmSet :: String+ , xmId :: Int+ , xmDeterministic :: Bool+ , xmSk :: String+ , xmMu :: String+ , xmRnd :: String+ , xmSignature :: String+ }++extMuVectors :: [ExtMuVector]+extMuVectors =+ [ ExtMuVector+ "ML-DSA-44"+ 91+ True+ "6BEB6817BDEF24413265898BAEAB86C94BF2FA5533CCFC0C0D42E92F34B5E7B5AAA3E9C5F83E71B3A69C77FFD8802D23CA07A53EEDC12560CEE54F317C9247E499FF7DC3FA2DE3963DE39F405088D8DB8C977BFFE9E46E5F7A1FB02A000452A5EFB967C1258A339E5FB15ADB410A5CC77FFA26F3491F6DD0F54D46BF9B3946E208A86103C740CA146C114812C40490D9844804A048522640A4122A242085C3106A23068DD9A205502891A0961061128D991680C2A86803B28004C14C21B64964B429D4968589B200C03622048240A0068E231644E1240CD434711908680A4304802830E43424E2A2899C042103340898006582266D1A396E00862D14A42012A52D98B8440A944011B52149368A09034D44066C18B8701A454901B5885A324414038218B08123462E22A82140A26493448159C60051464A1A124520461061B2019A08498B246DD9A86920B5694318611246454C282A08A0254A88494B8684203630109708E3840C521241021461132571E048858C02884412890C12920CB24D12280D1204421CB70564165208178251942823B83063206549028ECB1842D4080522C1601B252A040609C4227049162A0A062210B36C1C30211017800895108A9005402828840064810671521431C0466603880049A00008C32DC09489DA024A19288A83484A13409014080D0B1629124092D20862404289DBC86DD298898838005BB80D21126D440869194188941030E2202589444A8BA06004070409142D22166C24062A0C430088B68D43808C094012544665111784E244611A086D4A140981069024B80DDBB0719B848812A38008362E023488D0846003114111C14981B660C81672C9064D1A136D23C708D1340042064982C009D84428D0B268E2222E589068A2320000362851262D1B262D9C948DC4C491C9A2841137900C40259C2850218751C226468302820A0452DA3405901420D2142E8388219B302290184851166212245292A20800372288484DC2446619080D0A05821C34269096910A06044B1211203009C9866908852898204D48B84448008402B60101418E1C182190B00589940913224E1025328CB644D840241B474153C22C613601C08880A348288210801B09648038619930318C220C10C41014A54C60A8081C338554205111A289D4204DA348441845410A118620C16C60340641348960902910B68154320010105261284820B50859006D23C7916486458942856134890A972958404E92266D58C440AA56B190495FD1285669BF1F0F6BEBA53A52AFF58C90ED5DC9396F108BD5A8BF6F4DA9D8CA99F0CF13F1DABE503DF3612AEEFA0F5F8521C9B98536FDC1A135960FD342AF86B8AF34A7496E723FBA8DD8E4B5E9668EFC7DAEC509C42908036346C46029C74C3C401243D303117DEFCD85FA2D33A8E8F09BB402F1C33DE1BBC4C604A3F289A8210F7104E2C6E1FBC1EEF2ED3DCA16DF6264FEAB235E3A75FD800C4FD028239456BFB8A141F90B573897FCC6A2BE6E8FE61509911733980861917C915BDD8D42BB8C36BEBF80EB82920539DFE783094691767AF801FA93DDD3BBD6041951075F504D3EDBC580F506506469FD6C2F75992C2DD1C988B50A3096DF7702AB254340D6B06EC8D06F8D67FB3E78083B1E31098A5E0F02F8A7C0C110A99B8D0D619BC1E7A5C7998C7801DB708499244AF761FABAC5691035F97C4115DC6940F6F51BEA5EB878E296B166E1DE0AA5D7A3C86B6DE58EE4F21048A46436BE55A57C483016E4AF2FEF7EF51ECAF42FC40961B63070961CEF97CD5FC63372EFF708D9F9398A06CB279A5F5C1EE3D30CA952B04B2503C31AA4784D279E249C5B7DC2235ADF5E0ED0B561EC1D585065B73F1299A082D535BF0613808870D5CF8AF39C14DE9E385FDCB95A5661ADC0BC7CBAA4FF3589B3412A1D324FF3563CA895BC17A6CB2DAF83B704BDBBD72EF48D34400B2698CE3FE0EBC525D542CFD2C6D0B955DBEEFE69A0407D2DD475400947179E583BCD11B521BA9D2578586847C7C41EB50FE0018CD200515CD7165E711DA5B4A9DFFAF6C27E2FF8F8C18B9CEAD63DE207FAB139E8DD6E4F759020E8B35C0E2CBF1287F67BADC2312A3A9238F87324ED45216ECA8D2504B7A359B2A9277474BDF67D0DA9E5EFFE4C0D9F891E57CD78A946FFE6C0F4CA6C9BDC4949D7F0957072C4764DD914FFB3D24A03026A9DC3C619294456AF6429C33256417932D9C15C47F082409760262F896BDA28E1E3CF9BD7407289485AC0F8832177E79098F602E98582AB2463546A671CB329C6DB14F51A6B51FF74A76DD0A91314BA2B691FD314668BB8B1EE2C8E2BC7B9E8C97E8D6105EE3DB67AAA7ECD9FADF7BF2AB94EBD4DAAB42749EA118634C729D0926A413EECE91C4689563F8825D5C61F50CBB3177A20C293E5E667F346F22C48460A650494C17839EA5D388D212D9E1A7DDDA0B975EA546977BA2175339E7F994DE50AF65DF644DC48DBC9641CE2816AD53D0BC9757BD8F209DE8A65726C489532DAF43E4F51F0CC66B45A1F7DF7055CB524F968B8A3FAD56140204D2F8C123180E7004DC65EB587320C72CB89CC3A7FA4F60B907649F2E6A2BBAC02BE39906B669E18FAC21A9470276813A391C385958089D8FDEFA383DF21C8BD054C48F418391343DFC9424E749C55FE0D9CBD7E0917F01ADFFD622ADE02D7D811BD7F64925A67D7288FC8791422F0D44D4DE8E740B7722DA907A772849BE8C3D624FA2A4C1786927754C1796E1E305F781CBCF4F78C06326CEDB433BE9F9CF7A05D0D387ED62B1DD6F7FD90A0BB9451D3C7F7D5CF5DC741157D7DFC572886B81954CC7217F2EB1BCEF02CDC034DFEC787585806E78C81F8CF63DEE8A872ADA2247CDC57AE9E210D54B4A611683C2806D572CCE09DAB6395FE6DC3051078CE9EA3058B4E9657E6C28144C4C54CC35AD4D99CA224E20B840BC1F8E9F983C1C3CCD93E7CBB3D8D47C4F6EB97F9BA0E3CCF259ABA488684D9B60F0A796C80F7F4B9DC12029C12782F9EA0F1E78CC3DCD0E0F6267CAE3A10CD2C57DB71044926A3BDAACA6D9835A941FFAFF43107D7C05EC42F09F30EAD88F863689EAAEDF4DF9107F21BC39993988B3A907F5AD5D89535CFC2011560A15BCD1BD5175C09E972528C778BC6325B87D79434BBEFBC813C0565A959C687E0718577A91B2F031B9A278D80AF8F749B9ECB855802202AAA6B6586087D6B2C939C81840180F5070739990C2FFC913F8CDCB55A2E965BB186DAC8D1738AFF04046EA10849FAAA627CEF16798CFCEC8377A937CFDBB1804F86FA1815EE4D32B429AE016224ED67D310F7006A3109AAD22798828CFFBC0BB65C2AE5EB0A7FEDC7B9E2F6D810E5538BC20AEF00C531CE5468BB49D97976540310462560AAB008A50C7BD12E73F0813A3089C1E0C24C54DE1DFE81637E4A45511002FA4EE12629EDC887C4FC401EEEE8AB22987B79C7FE07BB938D9707F971D22F5817F5B65CFCEB62A0033CAD9B83F56AE7E4F0F09AE8D091A380CD4A13533ED9498A4A903501CBC0ECD4C787D06ED61917C6E2119A79EAD8C9C21E6C25E18C8ED2A7BC82E039BCD9DA09D8DCEA9D47E4C610CCB31FC92ABDF99BA06237"+ "3E240AEF582154F8AB1879A5B6E0DC69A5A214DA86BA5585B4268C68B5449A81E20B8A8CBAD23E37EE42C9E51A4892D76859143FA70B51C9B0C4C19758433A74"+ ""+ "3327AA270451D3E5AAA3E1E27A8414138D9270C1B44710F249693B205F890350C970BC577E0D2E9FFE641147A7E877CA86D6B31C64A8D80B9CB55A3FB1EE6C09E0E78F7E01C3A576D92FE1C26D745BB194EFBB375840CC916595DFF2BC0219642439E9A85D64EDE176AE94E51CF6F9B0506E2595C934DDF7432103CA8DC93DD3E2087D9FDD0746C8E14255194ABF853D9BFC7A6D421C726D2ADBBC2AF70D2CEC2A59A12A59CBA49EA33E015EF457FCEE9D3E4A2759816267283E0C2E9850C36E1192EBC92252F11B0AE6C029F5F1F8E9844C4A1D5252C6969B42F231461992391F97067F2D98CCAF750A80E582D5992E684E7D415C486C28CE519A3506FB2CBA2467EF083DE349D5CFD8DBB84BDEF507F0AFD6E9682A465D4116531552E12127CBD721CFE10C18C20904085AD05D8BABE2ED9C215B1F83FC425225843D767DFFACBCA2E9E58E388B689E208D0723E6124DD8CE9E31D1CCBEE25987EDAF7BE55E1E0D8436963C98C99A11D32FEF8440C44CDB83DEA2481527DFA349C2B9E2F3527D0887F1F718B758A7AC3163007ABBCB77BF6C1CCFC23DC155DC1B0BA876F34776D35A2ECC44F4C7D996C4807965F2E3584FD4CDD8BD50F134A78AE672E4D7CC10ABB16892F68D83486FABFCCED9316CA17EF5109559A97D84B6B0835E36E0635DC29E5159F872B3B15227E28B7E2971371444D1918301A7A0BC6D682DD3567D22B055D5B339A3C09C6272448D52E4E9EB59007FACF2FEC44559966599D1EA6772A9B08134EDB7EC533AD84709A4811465D46EB26C8A5F7AAF648AD43ABCA302BB85274984F8B6FB8EC663FA0338AD8F2584A0BB0070ECF58E3D8AAC634BB1B8ED7CDAA8EFBD461F49513B1538FEBE6CD55DFCA881366C3F146836BD4EACA6FC31D56814C058731DDEB45DF78F86615FE2A10683361F6D76BABB9D0653CD51C7D81DA0BF05E937C35F75B87C9B684840159A7EFA578F770E2F777FA7C190F9DC54BCF65690327142612F7F9D1AABDC846B72419752A9DBA8D3271DE3DA8D35B065091E9B601B00285FA7162431540617000B949509E24E6CA4913B316F157A5EEA97AFD9E82489B26DD8213E0F1E62D5A96BBC94C8ED9B4E1DCD57AF129BE73FAACF704ACBF5E1277D23760BE2DD4718A744F5B450410165E61F4793D9CDF639985F1DE12C981BD2A4C7167E681C7AEC4D016B6839488F9BF8A77F9A9795894E48DB3D7456F3DC7F620E8A2266720D1C3C325377D9528EC4C82BD54EF806BFD7B001E87E6FCAB79011FD9827ED7C9D954D192FC7748C2DFC2D3190110201AC8C84E1B72287793CF57941CF555BEEF8F6C61D08F0EA3094949384277BA95E2CC47CC627E071ABADE48077423610FA37BE30E264E40C9F58A59F0944FEE0D15983A2B31F72F591ED3B04B1FD5B82EBC6B730A358638FA97E67C4D1A79DE19C73B9A561A70DD221E94807A840A2138FE88F5E083FACC31E6A08C2729810A22C6653324094D2EA278D5D3EE266EF10B4CD79C00791CF63FD16CAD0EAD7EC9BE7010A80F8FADABBD98740CD52138DED90443EA770ECF2B0187DD33DACF6F4F034D364042B80A58156A976B94DA79436B58C6726A7A55AD58D9C85234E79CDDE7058E318AF3084DFC203E23AC33E4691AE923A0CB17C61B2432DED3F61D74D7940B8A1AD7BAE8DC524EFE024A0EB178508EE4B55B864F78BCB5E6FEE4B04BD6AD94D8C366F7BEAF8C161F286750EEAC00FD79C041BE0E424680FB29439BB8A0027AE1EEA94D11E32C07EF01745A335DA969FD77E3E8B2471441E150C40E66979DBE238801AC27F623EE12D9854662E310A613C9FB0F10D4821F98B6AC5C0F2CC61202992FC10B0460C82A943AD7D87832FF11D4FFF800CC0E7F91659C53E3ED9740C7CC24BE4684EFAB1E929A78F2B73B48ECA5249B56B320F5B07D398112985D60E04C6A83F2FB9F042700042AA3252108C32770D149A7E905E18DAE0073B3E2694489B5FD9097F3A1D0D77B4CF9B7763EAFB6BC8F01C13624F4AE7EB22A2E22ECDDE934F6105547254792B95EA91458562AB39AD8FCDE3247C91DFE3C5037FABB273468A9D7C9082066A7519B2141D021AFA8CF3EA64439F019FE18184179F04AD744FB9A1E65F39EC1DF77C275497EE07F666F3200D4F07F0E9ACAAA721B543D965D162BCC82EAE574B272443297C5DBBE68565D22CEDAF11CD07E2129A5F0B6580A1BACE3A2C69946010683C046A8779A0F8AB08E813CD14481B0826C8463AB06B30F23CD6A289FF20C68C9A0A0C0B73F6EBD4BB31EB6FD11FE4650BD5769FAB9A2CF71233DE64C7FBA84B10C2B172F674B3ED2D85BED4A7701B785593F2729C3D4E83C46B8211D11CFCF33A9198AC9086FDA7C97BA07C69072427815076A39A825BD53572DAE9ADDF9D4DE6A67B51B8C785BB2402021D369AC58D42BC09C4D59C3CC1ACF41966D2B19165723D9836F50FEFD4CB42A8FA3156B96423B76FA40A3120840D874A0B7B991720AC95F15D1B856D8F4275C7C136257F49057B78CAE101C97360D3A90510598F99021F44243F62151B13FAB61BA3947B9F605DD7C850FE9860F6B9A10843DEEA99ECE225C40A5DA1CDD026612A3088566F798E59082221924F823CAEDBEAEDC3B12BA29ED400387AE33F201DDAE901B21E3343870AD91C7C9CCEADF8DEC329EFB9F2FF8EC1F8E103A508C8E1E938EEADD307455153F892C612B4597D569AE72250EEE17DF0B97353C05F766278E91951A7314D05C52BB32A4D35E752BD799F11A84E5C1EB4CB9EFC83CABA48B74BBEF5D53B796E93B0178E72A46E3A7010A900D4AD3A20724F84F3CA8CCD5720EDC4A3E65523AE5F167451EDBB5623738FFB626D6287A249C763F6498F5180F79F1A9137D72CB52497581217D0FD0BBB05183F63EE788DD01CAFCCDA55BDADDD882EA716915A888C6E040F4A29DF30ED6424330FCCE93E3C4B204C1286F357AD0501662AE6EF6A56F795E65505C7ADB776714E26549B41EFE8520846E46FDE5B1A06DEA4774477213D8DBB410B6A2DC9063F04D48E7AEA33E03248178F025CA5480A91279FDBD57639D28FF8117455E8EBBA89A0E983ADB00145477F29EACAE2F778FAA5BB2DB971812C7272931097011B0943EA530B5D3FCE1EC77B30A2C1A31DE96239594E63519CA451BA614388E65E590385A0EFB6404EA6B8905852812F4D6AC7C45F15179449FBAAD3564ADB018B777317A1E2E940EA29CE40615BFBE3C319F3CAB0597C5DF6BB11E729379B2239DF06856CDC3FD1DCA85D28A8CCCA86D0FE0B9970B22D23E7141E27282E3A515770717F939699A0A1ADBCEBFE02101D2E303A4345465887A0A9BBBDC2D5DEE4F412313D41798590EB152C373A3E44525A8F9BBDCBCEEBF000000000000000000000000000000000000013272F3E"+ , ExtMuVector+ "ML-DSA-44"+ 271+ False+ "2A85D5D03CAC6C8977EAF1C19847F8BA8D53E5D1C4A9CA2D02C2C4A7E9838BBAE95FC78DB5A422E1F747EE3CC59290B084383C52BC03906295CDCFD8D5FAD5507813FAE6B7DD5F3289528AF354ABB98056A8C233F49EC8653D02DED6CBB7B43DCED8921EA9F2A03A0EC1B1D409CA80A0CF4825F804CED52E3445A56E3235ADBB0C276D03450E42842998126D08983053484CD4906CC1160104196909804854122962362E88868DC284099B046A1BC65002B91043A06D4824480013265340289B1090CA34120AA441D1A690C8302E01122D21860004B4900B050D4B32841AB5258C344A8202429C088519936542C68D1B19444A08260C2660802806D1000E08236843984403341202290494368E94368C44A640233792DB1871E4B27119238201B68408986881984863280209266C4C201193C41093C2659C944801486658862C0CB42D0180441B498699346C13994C00454C9B428ECC186E00849123044DE2042692A020DB002D202140028929C80082C0340448246C5C026A04164A22116A84846151166641328811A165DA006964120CE0406C13312E8B048941A269003785042189D49660C02662CA206103A22542861113C469D2B089DBA8501AB48DCCA6885C12420437452101904A048D19A70440982913414951A610C9024491288D63328604448E9A047149B8648A127298840519B66DD01252E2822CC8464A23913089262ECBA08951042003490D21458A63202400410561C088A324908B1646D48668032492D000290428269A088C8CB409C80808A0068193C00D02C6891401520BC849C4B84124B851CA420958A851042841D144229908500B214214300C08989118076DC9080ED4A0711CC649A4C62D0B336603262DE4340E1187840800621A427040B26194169144B82C010142598640D330240C487019491121966CC2002884862C843240121548D236006300861C3888142922944820D00221E0320604108209B08C031541902020E4B845913021C0228A5AA22589346A1C14105826810BA61148263222B86858483162A6851196290131219CA63001188D90007209B44D13202504C76C89B461014089134125C8C40188C800021465DBC0604B82016084008C48298240728B88718AC48500254424146C61C41109388EC148851B1752D4144A8022222392491B36420492711AC3410C0129A11270DA0411E330925A2290C0042D1A928512459060242D8824061041405BC42484C4906122885BB828431621475D2170672CEE466D30FF84A2A6D043AA46635A5516F1C0AC3FFF2AADFF5C05A1A164CB409A1AD5F417E49BB01C94F5168D8136EDB7CBE1C61E5C19ED45CB0DC45B1D0C2C98A9C05033C374269453BFC458847BC9F3EE88BCBA992D46F5ABB7F05A8A3E924424899C3808C05007D64FEA4D8DFB3CCA275E7CD3830E225F3F4D8F162CA01618840AAEF7346204775E6EBED692939618DA470A7B9FF793E08B04589F8D6310AC73A30BAF9B786D798151462422010861E0AF7E2219EBC08BE234FB05F80AE6A04EDFE89BD59469DDCCC21E2E4CD3F72951C17A6F178A33368DE11BF1BDB1EC6ACBD73BDD8E058B8E0BE35FF72502858EA819BFB6541E6550B5B305C73B90B83C83EA143E9E871775E449A39929E2CCCBC71F867D4D0022606D778F32C437A649D9CDEC131F2A8B06FDBCDC8E7326A5FF973AA60E4E6BEAB5033F37481EF4ADEF1A86AA3F8519B541561D7473B5AAB33224E9A7F1685843A80718C5A6E6D810FD695F07F36EB725FCA7AAD654668089D0A77DB39F841F8D24474E17B28718EE1FBB1C6DDDE71BD8E6812463CF91542EA42B70C0CE545AB2B9D693CDF1726278F5DF850085802F16F7A6DC0CC694AD07BDDC90BFA4B1E429161D667896B51014439643D689CFB93F7201CF676B792CB18D023CDA85A7CD5F6DC4497E834D4B5D416A6D55E56B844D33732FEDC6EFBE9F92CA0C73D6BE804128601413DDD55EA965F9828EF7AC40A1F94B96B0620C57ADDED7FCBC0A2EE199B69B0C6F2E458EEE4912CBFC9346051596D67F4A7667C879C990EB205E087B3EC30152B196959D3AAACB7AF8F41A03786C3A5EE7DC0605B232B84329CB85C2563FEDEFDF894D78267A7CEDE96EA9FC19EDDB6A3336B9F355EADF9F37521DA91D214FC212AFA67DEE0FCF4B16BE5847981C1A72531F4C5122841F877E8B182C9EBDE70BD60AE59F977552686DAD425C9534C8B8F358A5AEDFE3447BDEA63ED18705C73DF335A1AB5F0B2F0CEB8DA38F7CD93F346AF92400665F7D0553FB322F4A5E35521D69FE425E434546301943621000972B3E4FCFFF86C2175AC509D123239CAA9FEE418A8E375F2BB92DCEEC5C8055E33820FFBF4D3F5E67260E4C6EB761DBE22A97255B66F6C31DE9E77D8B0671CF3573B1F644B7ADC8A4387EE2564D634A7D748B25EEA78D1C60D606B47450BB0478323D471003DCC5B12EFFD72B30A5EB21F600DBBE672D377EC64C942DC5E9B4160967560F7C0DD5931BBF876AB1E596C97FDE14F2E1D86F8FFB54FC65CB1AFA5C92344F1CACAB1A7662728E813546188A3E0EFAC48E60CFC55418481F88221A371FD66B41CBD2B901153136A007C7D55DD25A73AC5B5393027117CDF37D380DEA394754E9A4C10FBF43466FB892D3C119E9D486BE7453E963FDDC069790A6FE5230D91E28A9DD420F9F8B6EB112E15FFFFEA67D8C3DD133017D08A48C24D426673D8FB445000C8A2E9A015AAB250F1D7AAA6C7B34E7871FBD01416238C417B097F7BD7C9359A5544CD68537C231F3C49EE996FE0E45A76E1705B041805415D3F73919901BF7055053F91F5B446C49254F3AC5616FD695E661EC73C70916EFA75E1C23EB94C5DB4692CBB2BE32244A79204E935D526DFC13652F3B7F7B01F0B6E8458AAC49927502E8388E5EE5369FA238A5CCEEEE70B6E6D446FCFE6016620DAC08A69F97EF8784AFFF9F28F0F68024F366F38A395B0639A959164D3CFABC1F720B8A259BF98A41E8E548BA7F5892619304E7C78643251176226AC8B598A248CFBE8BAD6AA9E6606C52C2919860323D2A7CDE1B9673801575F6BEE95A1E05952AA9D4664B74BB8AB9A1B11AA71B19442E9997CE05B2B840C26FB85116597F47BD03DD1DFA08560D37CCD6B5DEE4E351C683B0D1C42BE627DD7027A5922185AECACA3131FA98BE62A2636485A66E45FC09A04AEC79F88BFB8CAF25F564FC47893EAAB3BFA8F5BDF24CE293484BBDC8822588458A26FCC3594B539CF6D547016F564D3EBAA0FCEA8CA431D438C5E3D99B66B4148C9A8A194FBDB2D6521F9075C658ABC50564F0039D3C779CB89FE7C197D1CA07E74DC3741D02C5DD8F0A7618A9EF8B2385192720979AC0023D75322F7B899B6C7302C8802822ECBB488EAB2D8AB0987B7666484DAC4944E980644024E03C1736622099A0F1785C5DF972C3E97390DDE0480A17BE2501115DF0A656ED9FCBEBDBFFD88E9BC5E1E9D4E369A4A67C4D9399FA011ACCB8D90B07B80E3262F9E06D3A9C3CB1BB09A8C11426548115CF5ABE4191C2D28A4566690BEF7FD8417BB1F6510C7F5B12ABD396E40336E7CBD0A96FEACB960A38BABCD0B2FBFA7810CF5743"+ "1D24714C69DBF0C970D540CB572020257499D927CF3B7C99DBF0EBE667FAAC0FA3ABCB36B2AF351C8D013FB4E708B101667847EE18782CE95A23FB9EF77DCAAF"+ "288F870FF69E89D6CED754AB158C3A9B42653103384E2CC0167F104C4209D580"+ "5C04A9FD9BD3757969F5099D5627D3B0A1F2CA39DD4ED72EE355E568EE8CEA68B91EF0AC2E0003C8B84F03FA40E5C2D74287D7CEDB6C2E14787C6E1E6006915753F9F3795533053E50CD771B689CAA446F96C3A53265568320921690848AD3EC7209B2B05545AD470E49B8C226ABE1E8DFE4D5D55DDEB72BD2871580C9802145E13358AE3E5A4DF855E26C5FEE86C41E032325138BCCC1E58911988F5C4AE8149FF0C1EC880567C475A445C5802763E0D770877F4195475D3A34FD39B95735BA6F71B2D4DA59999EBDFA0DF2D989C8AFE7404D0C82DEC7643C6E0FAECB0D7FB5D44CFBA0DA77BEA14279623424CB9B3DD41823B02383E0C7394169B562C012B28C5382EBE4EAD06A1DCA7CF5E2EB816E33129736CE65255EB4EE02028835E4C8C3F626D3FB1C97F05A7E65C71471188F59C4684F768A9E6C7FB31DCD7726F301E4695D06D52D5D91C80471150117AB5E4688EA333D44D3D440E60D580830812E69716B5AD30CA8D2F0C7A8502B7068008917039B930767AA9A4A54B59E24B6CC89854EF9F430A3EA6EC8FFD9D3A14171011AA8CA53C77293EC81EDE75DE60FB193B427B2DC67F095496391E31CC0D75D53818DF88A5A10EDFF994ABA36F4D6922EF185C543FBEDE5FBB445BF84DD2D99D6CC8C845F390872EE5A6EC08359FF38D41E0017A8FF20C720BFA1697435C5ED16A852C77150DDD9333DD5D99C53ED91D1E3DFD4E1F5F1977678C83BD0C1741109500C71FC29B97ADE88DD1A2E95C15CBB10748504F1D54ABDE13FDFB39EDDF4CFA9ECE08FDB9A0D54C2FCEE36D506768414D393AD6A4106916B213DCF71A028DA5FD43B6F115ACC6F46D406F9AC36C305617E76E22225B6E79A1B1EC82005F492D08692A127EB77CF798DBDBC93068E70E128BAAAC1129BD69736B88D55361FBC4DAAC8A8192B59199F802F8B80F2C2875E88777A823682255730C0E0D3260A1F6779B17C806EB17594A19FC681C8CEEC7E71EB3A4FB3D276A988FBE0563BC0E10D34AE24FA0C17CBE6B3A2833E02E44E56BE07EC592E8BB24F3843D8176320391E13F1BE33F810CCC6FA486D3340F98940B89E1DBB3BB8C3EB360D450ACA8C5D2E097E86A3820E8A9722E9AA65CCE411E879BE9FB25006AE83CFABB4B9317D99D76937F15C45AC2143BB3797F14678D60E4DF904852A06FD9B1979504DB89FDEBCC9484D1723814854856B2599DE24D9478FE6E43E2C023F982709A8850059BDF05F0E716FD368EDF0A85151F932E4BED7D4E0687A852DFF40B116F0164C03593B09AE2F1F069FABE592F2DC41B19F579B20CDB22527147A6097206C7C76E1DEDD595B7E1F0F94CFDB4F55A5654E4D544946843BC5EEE859A4F2ECA0C765D0992C6A8D7CBD7B63556D7DD7D18CC067EEE0D50D1F7ACAB5E8588E1EFD785DB02F8473D9712C86A4768B8472DD31A8A0F5EF15E63C8222BBD79570FFF5941DF891548C397404A60C6CCAC68A9ECFAA98BFB2D5C84FA806F76BAAE8918A64C9B0C042294EBC7AB28F5A395AE7D359DB23FAFDAD37E0DCA95D609636B2F3BC1A05643251EE389E40403FC75E5657515C1741DA629F68307BC1AD220E69EF62AA8114CDD633CB33479AF35A671015ED06496271578D6D67BFEAEC7B3850E3472DE32D086E2FBA936F7CFBD24603C2B4734CA228C9B60F46BDFF9A58F8404D66A12D2A69D24924D93AAC8E802ED5DA9FE14DBFE1480D72198DA4DDE65AD9559259016FDE9C3B16A11EDD8CB0758BDF2F969E3445A651796F259D0A66A4D732363F381A691F1E2444ACAC20AB0E0B96C4618FEE33B7117B43CFBAC0247072105F34E76CA3C34EC4614F08B18437EE01835295A627BDC1F9650CB3AB394D47928AF3B8E5AA9E574D1F27A47FD1BDEBD49FBC5DA8EE6B9C897DD23834C82A039F48EBCEF771C18301C789073C457AF59FB5D5200D0EF62EB055C6C08DE0DF25B2E34ABE200F3D6C9B1977A89A671668AB4B10ACE8E78A10379FBBBBC19364AD7EC795AE26C207329B7C4D0233132D21679FDEA77F275B0CACF1EB6D6FFE2D42AA0D8A750CA441731D812BCF098032C83A0B5FEFA7E86B10205BA254E1B789B59D98CEC272A179A695EA3948B1597BCF2BEB653332FCB5554671EFC0375CD907D281F03108BD52ADD05DB18561B6AE5A30859BBDA23329BE4BACF3DE9B0891412D85AED9D2983CB0D434BB6BE70B43F139BE09A02A7AE3461179D19200FA016FB9396199C181BB3D947F2BCA6E2ADC899EC2ADD4FAA68ACFC400338EF7A3EDB8451F0D9D55B8875488EB940FB897C5EA9F42A833DE2BC2592F0A2A4976A55DEBDC3C7516195D54A0E9035A97B54B6B45CCAFF2A32546A394499C7B42BFFFD7BB84EA7262AC8B86951C75009146546F4E4320BAC89FE08195B14AD089A6B396A46CC5ADC509AEF6DF3B83194689258C66C18D27DC006F1ADC1C74F47575F64C92D1C772402548A130250BF68F0CD669DE41A8218DE7D7A31FCF69D3D4365B150C866941128204BB92E03BE1E98FA1C4E9D2A57ECE743ED18C04987D62EF3D1E7F5DE6E8D2896F17D204BEC7CF2AB6A1645B8B8912CB085347F9A104C105F76D83DA9B24A8382A28403D3EB9C81102E11FFB94D478C4E9598371ADD165181A9CE70483D912356AA062B8C0A3CEA544D12AAC653BD0C797EA23BEB8B035C78A708539F1694E72B2F7A4B3AD60F14A4FBFEB118FFBFDC8F53657C06007CEF8A64AA1DECCFF14C64D1807141850354998CEE19758F0C7E386E5F9649D8D6C964A329281B8348E9D7A59C2546F95F91820EA619DA1700DE87B9FFFFA959A94D7F9F298DDCEA6D48D0159F9A6BE63D84CBF04551D0DEDDE5F3A6B662F586DE92D2B6368E96C9311205C5B33C6EB1AC5F49D3A90A0361FEAC68DE73D32BE91DD33F00133680A1CE1F36D73F0FA4EEA975E4078ED4C6E6E6E81BE55C5DA1DEA6DE0D81DC8024D3C74508F8E64E97834BF6D3D98B5A1F64F18F38BE13FEB356785E37FD3E8D1E25B51CB69B1066E6A76296FC1B070694012B4C513EB88FAD48E0DD28C384639E73894AF1D0EE58DD2184860CEA9579FC8D5E7AC69A8CF85656BD08999C7FAC4E6CE811F2F96222D9C45C91DF101417E2FE0B5C84ACFBD14CA27109CCFB5CCB4E5AA0F5D93B077BB46221AF87550B9C3C53CB43AD525D0939FB950B9973A3C452C42952EC4FDDA481210F26CBC61B314C41E6EC23F8693440470A1C6E88C479C32158A415AC6E8C21B8C12A146D8977ACD7EF1480248D5FE3C58511CF6BD53E045E313C3DD0367E082B14B6189FDC46334FDD2082548497B95A1BCBFC5DBF8FA1120344858596F7280939FDCE1E9F0262A44464B4E545658606D708890A3A6B1BCCD124F6685888F9BB6B9D4EAEBF700000000000000000000000000000000000000000D1C2F3C"+ , ExtMuVector+ "ML-DSA-65"+ 121+ True+ "AB7A4EC2C0AFC8498CF7925E3D60A93B6B6EB419BECD5EBCEE1C2AEACFADFEEA8A2B94ED05A43494E1AD8DD53CB6F5F5549C70EBBE3733A30C19D613B3458A3CE9A87DED7D27E65ED4B9C3994D0B53344EEAB0096715CD3BD062FC5E493DE2181BE3B5FD96B924D3FAD5B8FF6C77DCBC3DA1E9C027A1E679333A5E35AB8B004A71586147841003264887731527328242807853235432038134051624202346086281176674880504571712611408565817276707863652441688234632531151414526434617442002301057453217218367123071021625107782412184428103038308130725463267333656581642752387685224571011883663382427660465565077413126127088487371543152880358858055051321376430111706840712724011545160338687522458026627145875285466703575688838580725232114023676876528480353628014710087842704177204183453420133664574077156886430715815673571314212318682224003872761102200576220438207405870026044121220200405883612301645714386406554274811828087515712770137862343607458325464163356456426615881684413768834457427451266667600867488673163253071556440082208035245580538754132375068757580665877131711486083864852472865210463525434668568031038764681301800841364027417747453562101681174623311603614120344231073470172104555805401341184128332224726687437860070812682822613011600714074403761324886008624433871350674831163132028630005855118412075743812074472750267308047510812847348845324822742168524417756540033734323600025665083521001221317575617605841284204586125815083553413152117710830802537755523165610726814433568321687118558661788005721053672766307677133313243572455050333471710188355120823878375658652570245425113815240146136810168632252410380760825757811251637523116158642342881660218647256242603011173254121750446688073286831603534178815551641581108763652568816886585304108446403036118130251384038481425412270328168587704453773656063074177375202314481714308072344401132260444773371278212485536412335073463664611471813806370751155852353663864561740764065218333351808330238148832376144483802846675306210781571511883028755774567757023810855211164202637348323001427744707070526046047426584777272851434153757335183865451567108672466843235573268086586212333318753488316753775012228237556036055660815377356285625313636214862420805678175736270058525615080488388070884533847462441678711815703458235013652682614481455234613860664080054443642163031825603244451230373708462650551003371800327257012130288034210067206634560675565806768871325875382837323747630882351666287123354045786817244850237114283035884267331780642008756187156471037281528463313237550204487316442105658023144310606503758314377232117268378327374822115484276110277355308480518408664808353217107527618428323183230735430353214402450801445187612124235810535657576412370863712814167847778253250212325721772781614438328108773880615462073865072481506564745133245713750614866018706440633682058014728641814430204722287080214700430471427261478316565412850287855315764540074753325287577144387387460452020077684135743551700380331863856257066685864103828167481406622455436571500500225684218255500620135162856367085467247548812666261868304358145002476357835523564632807373376612363648656084410211042107500014524013518714666368834284772016188331990D6F86BED8A287C3246AD255FF4F4B8BB43C0EAFCC23FF5D1F0416877E341575E759406EEEA21ACDD7247D7A543CA079E027EB613EF2C1764782EBE086F6BF4AF2EADC49BDA16968BCFC2003C97888DDC76C1DF1EE69C543208A43A8C28AD3F0D0C184BDC220FDC51B5F32EEF57C5B16BB695CBCE6D6C9CC6A5B217F48DF4AF809463B387C15776E2054B646237CCDB7B3909659B359F947ACE82AF58AE9ACAC5AB4AF47E5A6B7CE42FF257C629DD89B178EB218F3E57B5A7995FF41F68B75EB39AA5C77BF0097FDEDDFA89BA5F51C762C991944F40C8FF7D38CC5CBA92B7DD709322AEC9A050C524540C853CB5A690447207315DFC962C4E389253754908A93EF6FE5DB5CAA6ED428CA80C3F81F3C5544BA8376FF6BCE9050AA9475B411D983EADDF8031F9C6D7062E7B02EA2DDC0A9F1C3248329E602CBBF7080C66D42AFCBE8F628DBF1F2DAEDDB65C3DA527266B4846AE8EEA52C419C91CFCF42AE3621F87AAE6BCBB91FB49E3CE906B39ED3DD354308FBF373A6E636962A7023A6EF05512FE325C24DD6A120FD60737FEE47E4263B86FE28B46BD3FE6054D3B2EF88B7E20223138D9E6080B8880AC9BFD3EF9746A32E0402BA298520EA0C5877222550F1CB80303FD4E5B1D06404D54958A2C30902A454EDD2F62A4F3F2C85E953840E17463F600D7B19EED1E2D15CA109BDD9EC0435A6BA5A0F1B46C94BADDD1FABE95BC8F1F997A5ECC3A69F6F70EEC3D3AED3573635BAA621E756E5CC5BF72D2625CB079E80AB8359CD0BF8B70A431945EDC3553E0B49FE7079783E7EB28E8710A67B2810C6B78031E2FC5DF61C0499ABE789F1CCE15B549443B1B5B6CAC0C495C600D8877C5A4AEBF76953CF98C55720F425390EA4FFDBC94047D89C9AEBA3268FFD953BD5BAEAE8F40EC91C41E9F89F40D610EECF5FD56E543F2C1AB9B5280A4265D83576DA1DAA606D5ED9442767EE63B3D79BBA0BF68ABECC637BDAF99F367F19E82E148A0803BF29318FDA961C415087E3DFBD070BD96F5480390BD05112BEA81A104BFEDDCEC6EB107A38A2498A64DBB6B9B77453782E06A75658B0056DF1D71E44471A616B51D24EA6D942C23E1F6F50AFEE0E1FD94E74EBB42718EE0BEF962F73FFFB1CF073CFA448B1EFFD41FC3249EC9A249BC1ECC5635089BD8E2978DCF329A75417E34A5FD1B78A6C983E6DA4552DB236F5A4B018B6CBED747C20B3404F6F7A2F2D3FD7E43A3D078D2AD1EE0B28DC9E59E593073137C2F2891E3FB7FC7F9FDC0C1F5DFB7E83BAFA2D28EAE363EED414BF38FECC8976ED6DDCBB3513CC75302C849559C5880975247C55459B4ACA61E5D1855EDCA14D6C91AA479849045EF8D30A05ADEB605C2D50940524374C332C8846030CBABE1EA48FEB1265ECF913369F32EF308459C9177768BDB0F9F69A3DD171748F481D4C266BC13C9A4611F6D5B666CDFABFF13D1794FFD2438514960F209F23FF967C22A99E0D341BF61EBB494E3C27A45E4E674F854C05DFDD3C61616A00A3564037B6A1D665F99FE955322967C30E37D077C731CCD03F213B0B66FB8900F5C5C64A800E367C861C941F74D233D5C01938872E2B29C7C55F50ADB38D453A35EDE8B19AFC2662230E8FF719D87036751FEB3C310E08749A1ED60B791BB268B362BDFADF3FDEB4EE75C89A87E3837E1714C29BD86651CA0724CB26744A551FC0F4B60B57B37FB420D15F0735DBFAC893BC321B63241CCCC789395EB6E545FD6F77B087E08C3053E9D65D29F6F4418D02939A2A197126A75F6EEE32514BC270ABD990E30CEDDCC7B31D77898AA3D4BFB84C6D3539C894930544E52D107176731BF48EDA32C6855C04094977FADD763903361EFE04141058856B9C089DE450D84B6CEB08F8B3D924844658C74B4A520726B6789E116C0D7E91024CEE1625302B2C6E12E0127B01D0FC5EEB9ED4D9F5BB1676E9C37FEF16A6B2849D09AD5369F5AE49AF1ADE0AD6B5E248064AE29BD57B00F3301491B34280B2A671536F72EA305E896E637E8B662593602AE928EF0A3EFE3E510A193FC18E1A386E4388BAAA278E1D6C9C6891D19DC082E084C9157ACE6060AAC13F08B0ACEA4AAAEE3B53DD0E2E6B2FA12EAFB3751316DAF990B2282119CFAD216F0D567F0B4B67EAA3A784592D311FA8EBF889A8F3E828F41311CA8A96554543644A8694EFEEC080D596C0E6C84676F448CA0C3865F687D4A9C56320C36F32B058B061377C8FA02E07987EE67991E5FD825BC54A457338B82154B24980FEC04919755B777C19F5A8827DEED048353F7A5FD8FCADF061335839558067010BF9621BAFCE5BE0AB9C5EA2B33FF21D0319B4A1862866B92C179B262245CE52D8C931FAB32E5858470FBF29F9FD452B7BE84E6D71558495D1EB40C896D14C3C833DE0007F87F2D512345C9609FD6F72C312E883A4D0F2D23F1343C3A2EA446F18EF67FB6B5F77080391E5E580BC0AB0F78FD05A3F0E3F9281612DF9B7ED63A47665D519E865BF20E46CE98D04B147D5E711D5C6D46E3071FDFEC710F39C2745273D2296395CE21AB4C423BA46E02142BA4EA2021E1110B1B2F65920CB7A137E89C3238A9BE7A01F8094DC22EDCF126197D5FFA771561CE8D72A47693D883E2BA502B59E312FA2854EE898B3BB0D098755704B6070AF9653F89464E4BA8E281CA7C1F4525F61ECAA8ACF516A631379B1167B669671906A9BCBACF61EBA9FBE2B843057750A27C9AB60E78141B281C6A106B1608A2A76FEDFEE895B89ED53CC2EAAC7FBC4BA00B87B929BF50F043FF776340D633FB6ECEDA3C02BCCF791918841D0EC39905591E3A76C7FB32317C42C13B73E16CEA5796834D15E5CB5973C3504E1C1FDD693AD486134CDB358F76AEF38C1B9BE67D5FB150D130B53CA80325F0D9F4AD1452817645144AE1233D92162021227656967D76F3872639DA284EFE048490FDC7581FABDD0850B9680C74940982D9C94BAB52600CD46E1FFDF0182104209CA009C658F455404AA5EFB3E6463D80F4A76BF13768612542388CA2857D852135E1E7D102BC99AFAB2649DEA08DB41933EC07EA61B1728BE37848130E4EBD5135ECA073E8990E71EF52E7CC78BED909B346AC651BF8AD885C36FCF0E6F4E4C6E245B0A5BDA75D51D5593F31DA7D42B40B6B26A78237CC98A62BC7AE21D06E7BD68066A9CA9D4E507064911B954BE137CE706641ACB4F182E085ED9BFFD903D670A29CC484220F8D13B56B05545D5911F4C8A2513166AA0AD1816BF186B7FE7F2CE393515BF4402A8BFEB4836CB902281BD6D1959CB12A63CD5AD1D9D0F5EBB42C0ECD49A9BCC54A4B4C5DDAE10387F7D1802C07AEB93F3E89EB641981DF126F500585B85D23ABC8FCB5F05C32675CC345DA250F3872DD5328E551FE09EBC41E2E886D591CB2EA48E8B6FE3D09C396A56994149503E84031AD9DD600933742749D8CDC0A0738FBE40D9932871CEEBE57EEB0670FFA67108C7CFABF809B38C2A6074B3CDC4F5149F225086BAC3F80767D0605DF799450C147"+ "5524A9CFA4CCD49DFB40A32FBFCA132E14DD58127E556C0F1C517EA0E9494DD96B803217CC34A89ECCB25CE151D18F9989F1940CB12065883C76044C6BA05057"+ ""+ "018EC7BBA66F8156D1132131F909F72ABF83122C0C9EDBBA032A851CF1195ED1EC8752DCC7C72AF830299B75B2E7AF49C3AC3CD03F266739C723B3B245BCEFAE6CA6BC627F2706103586E72A0FB0AC5168693210AF8081938A0E77801EB745A130AF21D513EF75B55352C515894DD8C35208A9CD6955D3B193E54D742835820DD2FF3110CCEE17D92B23ADDB85A3FC0A4B2B615ACCA4E3078626F9904B42649B7084A756F8518C3E0C3ADACD3984668282C3918087C10B4D4DD1100063B3756AC224717F3B60DC4C9B8FF1A5D5DD1F19BB0144349E67A28F0C6D42C0B13A8F83FE16418BD58E8C06752D0BB59A1BB9F091261C28B7E9506F85C5D2B3555CB62BC763E60BE24A6894BC08540ED2380E410852D336A135B4822F52C1FAB1810F7E8531676CEAE926A1247E607FCD16738D1F565328968F166A490441F8CD3FC1C6C6D2D86BFE5A209C923C99BA701FB89B9C719BB69CC8C32076CCA9FF19D961F8A2C2B1C01697719768E134ADF6B28E5F20CBFDECDF2157A961EA2253B7BE3C5B2BD8B1381705866B3BD9FFA9EEAEFA56EF022FB83459ECA7342314B8CEF9161838257AA60C0483CE652D1F58516F929D9016C75DAA1DEB2616AD41A7E8BEBE9E0A834A5D60CA4C26F1F50C1A494B573917F4A95323C4006D3668D416AEFAA79D6FCAB96A92EF6ECF63DBA0CC7480D16DD591B11EF359D998A49F27E5287C9AA16EBDDCD3127C757F25B3221C19ECBF6A7203ECEF69F584A43C8382D513035D361E0797594FBC85AF686D6D6BDA7F2A42AA01E8AFFA32486BA3CC00A84B716C20792DC8CEE14226CC4E0A15A97225FD66C74ECB8D2A5BD22A04E958A20862120BBBAE092DD2361569F846B84066DA972E3CC5F0042BE682BDF58738A45FFDFE50CFD6FE165A2B42A9054C0F256B42532BBF83775D47DDD9AA4BCE6145175FF8F3D4EC56EDCE2F9D152A9DF7001674A10DE4A0225E144ECDF0B6DB906FD63C44F5CCD159C17B6B0644CEFF5F7E509CA5FED88EE51404951E0A1B8E2F7C6A8A2740FD304A68B8D072F4B350E1A487421A600F1D8D93771F18784A4390561828A1B1BF5EE5854DF461D58496DE217E269871F9D80F432F15C87529F04ADA33DBF3497036D978F1D93D37B7C56415D7F3BD95D5FD520E1AA3595794FB51481A838A97E08C71E950865170DC1B9FEE84F6B2E9DABA41CD3970613EBA01F7369BF68BC78AD3B21CA90F2E00350F2428DD3B1EB3136902C98C13C493D36CE3BA7F80901E01210E47198AD524808DB36F74F106A96480168B8D8B981BBD635B9077EB463C59DC791F4113DE6013F2BD5B6897AC7F04E7A58938E684BEEBEEAE0EA93A6FB4809231D4770972EF987D5F6B361EE16F895936965C6B981684794385CD0028375E8CB5C11B505D371DA33418457173E368BA11361BA461742B4E118C6242293F12E56A34EB0DE9446C45FA0BF591BCCB527A8434F67DC3D53633B50907E76A002671517434D2655F7EA1B5089F628047256BF9D079C0A1A73A951404B6BBDB9B2BD5979EB948494EC73D3B840080CFB79B3354A9D35D1B9D1542F10B44F8E55A6FB037C92A61AF475D1D6007712407F43C1E8EF609FCE19EADCEF79E4C2CFC5B28CEA0226D9BF0029C70746BF4FB752B28B7D3516E6D86F17F8CD5A9CB94A384389384A9975CFA5A4960AF5397AB9AD38EED806AE8F1E19D2840D25766615DA9FEF4D6C52C25AB9F9740C063FC8DEA4312A1EE7CA299C73F7AC7A1544844F27A3FCC29A92BC544167C128C7564C178DE48AE9807E62B4C5DBF5C0C19607D8D9F9221FFD04659CC9CC3C9DED4A1792EA787638402CDA2995200418EE3F827F976B73A53D256305D8FD1744FAACB4DFD58D305BE14CCFC2E11635FCD72937C4D4DE3C095EB179E21ABE5D6F9906D96EE97395568343AF974A529E80ED22CCACF48C1B39E170696E495C9E9DB40AFCFF1043D38C093113F0BA8E30A121D93756DB7A3498980D527EA12D95CDBE609216705829332B0030E3D0C15934EB6547937A0CD9536A1BCA184E187955CA94C6E5F7F55C7883291BD932C69CD46D7BC6082B58CFB374B044F068E5F308285F5F33C70F53E9F92B55C6711265DCE124A6E6DDEF65724FE2B291B282906C06C7212AE353BC2356E54DBAE8B213693D8452C1D0AA2AE8AE8A4409F6A03674462D791C28E957C06C457AC158A51ACDF773CDFF273DB0543233A1E196FE51F30CE1A6C9C9A3DF081BA5D2CCF3066BC8F173648748FAD50D06FA7458DB7448EC4116A438A87CE43C37FEA690A2A619187964EF2165C50C1A860B6FFFCD815EDA7FADF1964AF736C9FBAC7D37ABE03A8D2C534B476044B0C5E643621EFA866A637727E6034CACED93686862A7985783E418187F47D567CCF81FEE7954EAEB3284652F24B84A1980AB1D26D878D8757EB8F721F8D126CE6D5F77518165524E6AEA9141464FD23FE62F44A0AE64038E7DFC01C6E481DD05B5E8636A9D34242E99A4C0438C5B9592D912158ADC1B633AD698DC31930E99DB8F21D5F4271C94E7D261938F6291BDFCA06579FD8D77409288BE72095113F8BD07A1F5D1171940BFCC15432FFBE0035492790070458A25466411D53D65329500AAEABC9BA7272DE27AC6E95383C3D88434A1CDC1CD260E660FFF808D7E59ABD7D0B8235210633CD67B56376A04325A3AFAD3A919B3D8FABA1E68246B72D87D1C898FDDDF8D255EA27E0487E8E9C068F1045977D86595393E183AB938C5A95E5B8A15C3CC8B526A28FEDD01AEF2FE244D79EAE1C2FEC714261195809A9C69268DCC4C50B42D0AE8C21DEFB76110F0FDD3E49189F61FEB94FD7DE0A1F91E780105702479F04071DD085B060C2DD388FDC068F4AEBEFD55D332CC0F6970EE3E289BF43A02782B6E09DDF5DA004DA158BBD37B7146B8F889B5D75825D2CE4D00A871C71494185DAC3CC5BE31C9F1881A20B2F1F36B957A606A5BD79600CBE862DBA025118252AAB049E961D0783839924C19B0B0BA65392414719AA79B714DBE3578B71512E908DD70B90D083987296914A562CF2444B0B005C4E9B9F6E615D7323DC1AB464EE78858D4D9CBEB79556CFCE0CC4D78B6CEF3BE34FC8D0F7EA4FDDFAEA18E6BDE8FB44EEECDBC19E21F53241880994D5CB39DF4D254ED3EC74E6756B24F2FFF205E2E3F6E223FD1B1DAC16A65181EC992303621F81E1AF4A9285A952387455451489D1EC3A4DEB5668249C03BC33F6984F56F0BE9C880082476EAC7B1AE2804962C9427FF9E5D9FFBEFB9DCD2A02B0EB1C25DF1FA18CC5EA1C58E7378B8CCCE2F760232B896506C8FAB679BFBF7A773E42399092EA5C038E8E62972A3AE82656DB7B64E096603542DBD96A8C83EBCA0683052EF1C3CE5B46DDEF8D71CAEE66B0E09F011A546F7017F1E406B52C5D04F3D81A451229DFDF50381EF7D482E182246FEA11CAB88ED6C3C8CF5AF53AA9B0ADFAD32F4A256D626BDB2AB15DB686DC8C421EBCABEA0ABB79DA7C86FC922DC1C2E62F94A644D6007A4FDEF59DA86A35B1887C7B56B98C84EA8CB82A425CA0018160E9F1EC9CEDBA6CF0312B0D4734421DC3F621B8FF37F0ED45F098E3FA772CEDD2195DA11070227005765E05E75FBC5504BE7C2327D1A63635515D22747AD76C9CA3FDB8E9EE9FD911C076C74FABBFC119A46EFA3C4E903207E1218CE012B316E49FCD3127F37C5FE66A6E5869248B1B3F0E3CFAC235B7DE2945CBEB8DDDFCAC20FC857C914C84FA1D65A2ABB5927F145251EE5ECC41D0FAACAA03B668EB015A313303BA3822905A28E5002E4211456A5B4BF27130585CFD3FCB3A261D400097557239894763843B4490D616FB3E0B6BC6FE8C77B75197C1B6247A11E34BC6C0B5430241D00378ADAFDF315B28B6759949D6DC5E4DE1D08F739AD1A4FD07BCA030E9EF1C153BD1760BECB965B72EB1F567D82A40DC1D4161CF58BBFCEFC2A9A9988F5B019BCC15B3FB30983FD2786C7022FB0943AB487DD2FF1A35A8930C832DFF2DFA4C6310DF5B71154C0D8C8C2A4F0539ED22563B4F327DD331DF5D687243D19E4522FAA94E01B655BE918A689EE554882D59BFDF5A4807677396EF9F5F9DB016AC1CAFFA9AE38E95A15EE8D2EDFDB1F955A6A5D9320CFC730C846346D237D498A52CF7D25A37A378EDE231C4525F5240CABD265007BD552422594852E012074429CBF0BBF5DC3F606EC4B9939E5BF51310C1A516F91929344EEB05B21D2D1AA38C35300CDE0258B788A7EB57129D10EAEF9B51A4172DD52D1E7914F733094E81FF946EEF167E140DDF25F9DBDE233E546117BB756718F518086A1BE2D5F3EF7DC7BD9B493B9E30CCFCE9691EFEBB65616D8F0ABA8CE96647219A00883B1334A987D0E89C44F6F4A15726A40DA3E9E29FE61FB5424F5C6973170D587E5D889FF69DAD268F4B94EA3C86ED54BD27E2B9E10AD1C3FBD5FCFC68B498946AFE657A09778DF702BACB9029539AF69C99EC98F812D649FA9C8DEAB13B1E24C2F51E75FD70C24FF362E3D8E7E9F875D53749A3B4D94AB5CF253445C6BE8BB5CA527323039479981620A8F5A938D50ED2C1A9E81CEE3AEA77B4AD2B6A804A0A373EDE5791090C2757515E6489A3EE086F79A9AFC5CBFB32485355838EA5A8000000000000000000000000000000000000000000000004060A101820"+ , ExtMuVector+ "ML-DSA-65"+ 301+ False+ "99259B67E253EB22326B0F5D4276DFDD07588F2B010C8E41411B4E623C5B285BD7ECD0946A042B660557B1B650B0278211AD31AD25F201C7D66C68E3A2F9FA1FB79DC085922321AA2178CD1F344B1FD15C7366437536D6A0E62A43CB8976D1278558CEDBB7FC4CDC746B672FFFC0E7FF04AC23D687220CDBFD7716B7B2BE9C9F5436323341012526304751178784346354365266180061628256170683080376663588271672205532066537888455555837188244235611837077113032884571828175365720315653671642517885454551186606283882111214507411706518530748408458285027548020617185522486542012656151033036576776481822653305681111408641468800328783753311405150841354627800236743238324667881314056888427280106004300471873781412407667330251485114213472437866200462624111676084424122420241628383324437266525571212631618328558184042356162646345533522433386175883753613518047263585633710375032253800672176338312313224278467128421703484676561805032627285502234822773727261634710517585168858033325601477732738157266560675587102358688207618013435383308381230688344875331488524658761065123178852442682377470844447516187082365612141562280774428557010442087230266105188305426640311356205671274687140685726088410821575822473311063264540066515166154771028561777041736023803604564263258874616801067122655484713216415023351736088832234173620757877602810662818768866080385153155677438331173708481742271715564642260546385855548682378055584806350051175707686853464276567171402087733133363367701666630136524613272365728070673034580783046516786707671183412461402862261522746001248735226566323436125738848364012175786708636780361103401567284266708760561716371228328724465630322187444828444846388366178526314683218720337676406448178140646332057560460462665715538430347478842808513837484458441388562330201310540041762325027877541288877645584531203550116266428773606586364012454415403475817461184778247854741301057125044512851326048465878142078204521002034105441474712483574337442586323207351126877371506537330662521247020627446643602555175710131522456662784888783264756057458371027773008662488024542552558240076050222228711207666786500823435600803671106623756638432832600545226756344151616833027874344211476556401155070461064031065186663101041648167031818202442266843455164620843482212757625657610174771446766320413055648781578761382323823851422800810021518618006661864620611614485000856326102775028273151581516641506855366633608431287403410057785631844523706465466812815180274558866260060408275836082043441422344166706470184527407862020810670688028327364746604441327303228614526873146341447182114484064436363637377033224220461350533578431100056500546883254113652322723242314433640646643878847125082155657747022147783170867631237216203306864758401117524684187224865862181271456617230442540462478244680148542628180568125480385506385546087007411545122137366785881812513550003430470082042580301773104346104627673216635057133262163340445681380360138641831685654410115218106540322261472130612772503862676523325725454574705127516652837676654184837142885158761564366705132532683042413427827156683823133016152881142313806311547256036683474676621614427053028120007124164356421445281555253013218288067861504B976F2F81B60D8FB3682E72F4B9D63BE886E62045C88982888CB64D14A9AB7E84C7C4128671B28F66057E232FCFF40B1FAE2E5188786C20C4F6BDF05515B77C9C474EA1D002387EDD61C30CEC50B9C3F34A66B82296583CD659BBF22CBDCEBEA4FC7CC29BE2E51E22C6D7DFA18001226ED0FEC499C53BE21E4C6C39B2FFA98611477E98176E8BE0C1AB567B1625157B152559FECE36B2B78F0857D5C7A717177AFB7DF768CC13C02BAAC553C68FD0FC103562491F576D69B19F17428254853768151F8AD91DB98F549BFB452CDE66E650A4CA4516C3DF33CB7A3432FD5A76D786376C1BC51C29FAA456EB60727EAE059A29766BC07F58075A95FFBE997CB9441493335060F2E8853BE632466D465B576A5DFF1B234091AD48365C2AD3C762083C5326E19C0A6BF55FF3036AFB1393ACF6DA5C9DC377004DEB454C0E06A80725BE1DE947DD1606D8FF3E3F509FB5BD78D1F57152FC86B49698FEEBA48E99E9A1EABED2598A9A21E46C0E26D82BA83B4E908A67A37E7E922DDB16EE9EF9D74E3F9CAEB9D56008456A0D1E5A26B5CD67C2B7DB29C0E55DF937EEAB8E309C795D7E6BCEF8342FAA5125FCC4AF8BA61CFFDAFCBC5725A32DAFE0034B70D995AC3EFF70B849E7915BEEFA9519DDEB8762F57F449DB2E79A25065536680978890EC8ED471D4DC01D29D0388E1EAE8EC94200AFE7A34D19E66CE10389F74B4D447D06446460547AAFD109439E84FB223A0FE5E446E938383C0615E60A68D6F63D5F9BF69CB674344945D2D79EB5ACBF751A01765F8625F3E42E2C319FF5AFD81FCE7789BDF2B0F6054F377EBCBF57F97A1734263730D14666C0F4510461973D57169BAA92C36EFBC9D18FCF707C49716238937FCF8FF3A7122E7C5B0F7589DA38E5421C318BB9A8CE7A2EB2A8A760272E60C0427F51D3089CB36119C459FB86186D11EF2E6E19F43C9EDC45D6B8523BB175C63669876E24B6280A0EC3B44B902033667DD495DB6B9534A7BF05B2359F19876D9747553CB176C8E56A15012EB6D0A1B2F865B5C8BD717687DDD1B29DCA7622A4D914A60559AE5DFDE29C0F51534F44C4FC2352C1EAAC3023528A49858223A4A449A3B8A66FD798DFA0534751E22233AB53E6F6DFDF8EA73DA82B5FC9B22BEF2DDDEBC498CF397EF2A2E3F8095BAC3456A327B306F2FF96F9269FB420E32D4AA2963E807ADC1AB1AEB731C35CF3E7DD35202F1B100024677184BBDFF48D3805A62616C5A8F1ADCAF353810C50FEB699DC10872728552BF2362183641671D34E821BF620C1B990A45B973B388CF18860980C8F469CF955B6C8905A8539C913D1731E1B4EF058C893EB9A2A3AD6016BA5DB29DCE869CF461F38C3524D11EBA59ED19F278267B04DF1023576EE9FD2652101F7D9E5C0B918E07312B7D8682DF8B5263ED17A08D7F071E343D970234348ED546D2E9440AB0BC3AC4DDDDBF914F036AEC9084E1095CC34475CF67F661A6D8256FFAEC8A3FACA34F1F281F0FFC440DCA840055AC65C9602E0BB349599CF07DDF14F446E08DBE1E2AEDB401C15937BB218E708F912D4EDB974D7B99D00112C22067B3A410A58EBD90BF815FBD90A69652A0350F3840D88FFD12A22108C93DD8557CCF10D881FCCEB4AF521DFD993EE199C621D13BA8274EBAD07AEF3D47DC418670D57F36687A7FF8E6261724BC0CB57B3440647BFB35946BACFC1B583EE83519C3AC189B09D002B26146B682BFCB0E60EDA634850E60DF239F7AD509C34F85ACFFD18A2F38572607F48008B023B32FDF03B6BA8B05A3DBDA1EDF7D56E4720DD1B70AB5FA014B34669CB086242F695C0EF459B88726525D0338CEB4F5C5F002797F5C89D4E925A378F252CB7361057F678C003B491C6A52A90FAA2F99B418C00C6779F089DA552624E4E6E5AEC6022C4C5BF36109CC8A7DA3AF08629C72B9804004B4F2E7ECDA48DC381E625BD67D6B3E3BCDC937BD1284D9981F7C0F340AE4FBE356E0E419E7DF1F097ECCAB1C764EC3D01C00B79A5936C69B8A0F297CA39BE32D44940C0E231E811636894C575F3B4AC2E4F6110237E784E7EA72D88EE4162B6BC606DDFE7D7A38AAF83CD6497B531CDE9BC267B270B139579539ABCF73491509909BCDBC358FE2BB1273250A7029896632DBDCE54C92726C4804050835C0DFC37C21EF762F4D5BF9C8FB506ABCF13644609B05ED1B8D7E8DC48FAD5D1ADE9394F0099DA60F317AB542C7180F3FC263CB946740BB4207AB55224BB3E80CB6120666D4C7645CC06778F7FD1AB4B64FC7D6F2CAEFA172DB2DB123058D6D9600E1117CB2B3F2680C5F9692B4B1D0879622176E0503FE9FAE96A241760F02B68269E231E01AEE22167A783028AD5C43B0B3CA03528869D293F5E303B9192D7775F2E9E6AEDDBE979ABCA0C591791B36AEE307AB04D0FAA8CBFB469210094B211B4A14A60739931E0EB012D2E0639CCDBA3ACE8FB26AA8AEE54DA0E5FA466B506DB7BD9B38A1F38DD440A4EC166EB8C094BCFBEC30584BB7FF47C2D0B40D62E2F925BB12F52DE35D352DAD1963266DDB131DC1D74AE29B848E1424F3CF0FC3A57A9FE829B30223284A9276CC23C034F96D829D833F995315B4A2548DED2A728C081D3AB464C615D80898416E7BA007FCA2F903A21CA44B4A24786C5468596661238FF70AF25723BC67B07BC8E90C4A716A7C292A76018C971F5C0D8915018A4EA545AABCC003D27461272A1A5DB9366131F660DBDA5B5E687905580C598C1BDA6ACB29812C7F15BEE3ECF6D3D316315C444A546345BD82E926E694096B681406722A8C9D7DAD233FABD0892309804B8116C922D58540370D3ABBF856C282DD54198E4098834559F1B21FDFBF2B81B8D242D78E1CD635466796FA6661984F0ADFB1934AE46047B149587FC5E9C9115009096048258AEFFE02349FC5C71F484D438D225AA2306E5CDAE809944476FC842B10E3440E84C7B03445D0F6DAC79E9A8CDDAD265F5E305A8D47F7B329032C3F3B26C0318E2992FB79038253EF9D8241983769BD739AC35C64FCEB0834F904DD4CBCB5D55D4AB38E1106A35CE7C43E4EC2C7C18EEBC2EC8341E3727F2AFD719D6183D26AD35CE5BD6F498A3C7C61D80F98BB79AAB2F582944D574905FFDAF82042C0E0D72FF9F308D94DAE78F65869C25E879B62BEDE1E32CC919E3E432A1BB7F4264EC7A9E515469469CE21F1281D4F1FBFA5591C209AEE129E3656BE64135B5A977A49A9BBB475E1FF6742A4CDA1302BEB3B9022FFD770B57CDE8079E8B59B8389D310F4754486B3527DDCD9A2434892987E8330FDA910D22CD4E7432A8D809F433B807CD75F8F391B8FA0ACC2D168650929810D8611CA818190EAD237A8F715CCA170FDA02050130210F5A61AA51C789F2499DAECCDA83A0E215D464CD1ADD237B89E280D75F881ECAF7B9CAC5400A21863BDC16584D085A762594C4684D17D04DB806C04911EC82337658037038E363688D393FB84BA37F9A208A2686F6D687B24890F0814BB36C9A6FE7304F8BCEF7D43EF"+ "DAD12B228F03FCC52986A9E70C5BB0CF6CF15BC8430544709F7007DC2CEA0724FB2DDA5FF6A2CF19B2D1AA31311AE26872615E2FBA3E7BB43021C45F1CAF97D3"+ "447DE003E066C1807A13217FAE2636DEBEEA8EA4EDC17A72619535732896434D"+ "4D5FAFEA677654549C97BA59CCB6A8DFF6618629386CC4CC33516D652D9088AFAF466F34966128F7A10640D93116C0736C69FBF6C2F1E12F36E53E2008AD030CF543D9783B0675C964295D22E27A14C0CECD941DD5A26FFBFEC9588FF88C58758D66DFBFEB47561603B5123D99789D678B94F6F7D122BA621A1F7366EB18B114E22CF0B5CA3A09E47C2891618A9B1970DD5B18EBFF1C897C8A1E34A7ACE7021C982C135CF6876686B927A9A6BB0CD6F6530A5001534386D81296E96BE0AEBBA410D9CD45F14FFCD9C6E6CD461A20D9D8078DF048969BF1D7903B55BC5CBE5562EA9CBFC12907ED05832195011F3A762F0EA33EBE8DE46A75D04EA64B4DFA1162F046A436D739E1655B98BF08C542771A5D154E911E14D012AA0F2070DE7FC4F60B835A600E46E8B4903909349DFCD45E6CCA19D7CCCEFAC60BD0844DBEE74EF21C738863918C048900429A1BFB5A223C75A9C958217F599EA54600696F8D66C2249D878F348A8756AB3E973675A46A9DF1B660E6400FBA9DAE48DACDC40A2D298D4554CD2354819451C8E017D2738AC0A0DB39082D07EFAFDFE3B343F59400AA63734119E83C5B6A14FA96DEAAC024CC9AD00E6E92A17F1B54A6DFC8B380945B356160240FA272D6082FC886EDDECFA9A21B2D7F42B10D86B404BCA3BC4987885A4F7F9A07BE46BE245BA39B36202ECC76228F6DAD289B7ADE83C20CDB021E8402563381EA427F5EFF7D41E4BF8C67B106096AC1EA8025FCB58F72A351334FF0BA39592F6E2E72116F0D05D375815A5445BF28A4DF060F4C09C4935B016CFC66C41EAB03E35B6189C4C0B0C425F311A22B14422B50B3238E053136AAD0BA350D742695F35066203BA9135D6F0EF06CA65F0E7E0BB53E01A82FE6CF944B7516F9B19D92A0684DD97672C70C589AB87847B2339067F1BC0C452C1A97EA8A82E9BAAE384829816BBD622F310510BCD40A7D0714CBAAEB52C9681A6FF2E16D74EEDA2C52C2B0DDA560AB225B99EC0093E0E4A49FF748A733740E81BAA7A0BFB6328CC236880D4B5CD634FDBE0EB24717D53C055EA4E41B7BDA27AE7644643513B30917455BB6C456D96FE4419CDA05116B67D4870888BC2FDFD8548FAB73758DD52A64F8EE112EC365E6F6DCF0B373818868A0745C0836D265D69704BF6E7116E538510A699550066244C84C7357DBF0E375EDDC39CFE9B37DC5D26CDCF88B978DEC77D31ECCEDFD1C2D649B4C364FB732D5674ED6E3975FCEC1FBBD6BE68A8C41094D1AF52D34663CF6DAE4C1C40B0D1FE474530F50508845E714782187A77851630A2C1A4572A9B2710C67ED44D09979680825AA6582BA2D6902D2F30DDC21A9797D361BEA010FC081DF1876A65FAC518FD0F51F4C92DAAD50207C4B1401E0165C36CE6AD8EE2BF51FDF2DE387844D0F38C02E381836105A291937BC3019AD5660D159CB13BC06BFDE6B26B7225BDAC775D63B5CEFC2372451ED27DB9DCD6EF6C5ED7A14517F5F20307765BDDA233FB1680182701AD9DD95E241AD8B869B3236AFB234B200C491C75903B63F33A1E978060A89E7A648C50CB74AEE751B49AC756D82CA388B15936F6B542010B230BF72CF5564E83BE7AC8834EDF793BC15A795B41AF36F035A36D226E18801FDBC4A8B7C9741E7135D900CEA7A174013FE81E7BF0ED72FBDB35A891F727E18639B7C2DE022541FA2E3F9BB1329C9E696F845CE4056C7966F2E49E56980B6A2F3D0DF967F486A59218EAE71AE746E8567441D8F707F622B3B159CC91D9263BC15658FB0441582EAD3FDD46601BCB3A5EB71EE64A95A4FBB11A64F3059B66B25CFE847416BCAA4CAF6B614F5BCCC40B990F1E565855CBE13766C7C7CD465091EA125FD715B3E30CFB4CC3DDDD3379C0307B3D875EC894409A1A30A93B8EC08F194BB94E06AC39E79F68C3A6BE8AA5986F21DEACAC7CC3DEE8C15223D27746DB20A18588CE30AC0CBA03E6555E68F54CDD570FD8AE59B9C8E152F9617A1129EE0EA04D0D7281960DE5806CA578E3B98D86A6204885FB5FB90FFB7F3E6E8388CFE5B21E829FDDF181670078FC8BF628920151B21DA77F8D52205E1F09A70F83ED69A38326378DFEA53B347DD7EA62E14837764BE04B6242B8DC7098F98C2C04B5C33D35438202348211360382045E8437E3E34B231AF5F5B0D3CCB786040C1B98470B8CAFDD2F489859971D8BAE5489CB7647AD1BA3EF85C7D2332621EC73C977D0092FC36930C65C825ACC30DADB5A512D76C39D3661660DC656EF9D60EF906461FC78314813B961D9A1D055C40A24EDB3AA64465AE8A680421D619D59751F75A4B50FD236976B619A91D0CFA022B6759E9972B0B0E019987C3FC2C3EA61F7B0E8625EEB07B1782AC7445578C5B189FD8C697062FACB18BE250396A5B5051170470F11B0F1E431BCC95896C2ADFD49E45FB92A78718035E892F4DECE6D0EB2D7A02DFA2B3733D3A538B0EE4B94D2D1807A2DB879E42781B3F748C9E0666CC098C478046AFA1792CEAA7AC5AA0A36A8DAF8C85E4BED9E040AA5EC33D1292E825811BD6936C53C3C1E332BC8080762AC51BE8EA98726E19AD342D269E76D5CDAE83A696AFD6DEE58916701B8FB0764B70638A24A872B44864175C1B9C4A137A5C46EBBB84913CBB39ADF92ABFF84C769689EC64E476CA225D7ACF8E4A712160783598F1B7EE4A79949EBD2D29F3892922D0A8023C0420E445687A64D07115851B27DFB09ED53F071409B8217A937363AD0E0629BC29B1CF1619A7C35CED703D10991F77DFD540244A708844DB97CAE24B0B2F59671A7DDF1DBBB635395A88E2222C3C104BE9E447504A578C08099A34EAEFF49E987D97B7F1081E4BEBC2CF0CC5F3C298C7D3A459536B80CFF4FEC23D493EAD435CCFE6F8EB220C8F66FC7C1C38BBAE36272D1EFBF8F5CE6F46DD7763CE06B1EB41D1FBFBAD8155EECEF416B4386781B27673DE69A316E59D512B3C1994F2361B0133E83C83FB43D210B95C5E583D211DC7655B1363B44D384EACB20E1DADCE6ABC87C15FBAEBF6529919078A02FB0C606AA4DF4493ADD707E3B3DCA40912831D2927A5F8444047A0F3FA567BD12FE85CD6B6E25C105251060E8996F14391C837366C61840143EE9FA7991FAD2419A137928876379ECC56B559D60C52360BBE2359EBBE17FD9C426D669DA482B7F85311121A52502172FF8A39F580C6B2F37C2EC20DE4E68287254B9EF8F2D1AD562FEB053ECEB243C69D4E7F78AC7E076161EB4432309FCCEC3728BE9F0FBE76AD21417D176769E8C8D9BA217A124A93DDB8736EDF3DFF4FE1F603187F2ADC40055B7560A9B339842387545815E61D75072BDAF161505B329126CA1360BEBCC9430681F92A32AD80B573E2FC1D38EBC03DAA5815FE951DFA02D160D90E36053D33A80676F086AFA9CE8A01BFBB169F10A54A0B1FEBAA1B5709078E090F21307DFC9A16A117038E5F3F2112467DBFE08A85D95FAAFD99C0CCEF5CB95CE099E3A7D1BEF0234853285271FFA365671060A4F6DBA320FB625E6A72D0D06AC12C0D050BBE14E7201C69BFF074432C89AAA91B802397494568E01EB6AB197B5AC6D2F3559A2FD97CBD37675F84EF382301ADA8C0E173D1E55624F22BDCD4DB6065D7A32704EF1687C8CA9BC3F9BB3A2B238C9EFD605BE3CE76EC4C9B2FF580BABAC04ED83A47EC403B112D1038210685B212F6EC0BD9B312F9205C5296709542B2CD72831960EBAD6D3CFA0ADE4255B659C2D2985A94FB2F2A139A64C4FB88428D5A6EB80242AD67C4BD82E72EFE740400DD22B1796F2967625FF2753E57EFB57ACEEEEAA07A4712E01EDD6C999B8E2315A900BD14F1B06CA15C5DC2A08532FD674D68FFBA62D49158393026417865E65E657A14597FBF884A8528DE60E2E0D0FB0265B40E7E85360D371F2EED22BB76754AD7A092FE93827F8BB162D430DDDB7B217C05A79CF7EA97569CC588180E37F73BA10D00AC752DE08891247B3F5CC1358588C6AE7F9F4F305A848B921588DAFEF5DF74F2100DFE432863536D15CD4ACCB2AF39B566BD8BB12EA0C8C17F84D936A97A175EF12C847DC255289F6EFCB47D9F6448D3B4FA21E5C8109C24524495D6F91FA0D7491FC14EA51A74C77DBBDC7B04B51683942CCE215A7553186430D7421ACE4262F3576E3C71A62EFCEE459B0C505EFEC6B5344FA11437F45FEFB16521591B814DDD79D3126509FE9C9B3E2B29E63B0024596655CBF9EE0E11DC2A738999ED00FEB5C3958DC0B631B45119FAD15BAC4E942A9115B26592C3A8523A95C3708321FBF235A2F7704306B7DCF1CE9692103D522F24C284BBFF3D6826B258DAD52628D8D33FF86FE34F29C25E06144CEAF9507E3817A52A0C57D32A25FCEE50DF4FB7CE22F5F3AD674EB6DA9288C20997F0CDD3575DC94CFBD56173DDF0F5D730AD4BF074D413426EEC6EC21A94E3577D4B12AE165A07DCA3BF22BF3A3096AA9DA127881FF746B5903908A73DAB1F9DF7A1F16A87E0CD90EFEDF3BB079A63BDC45DCBE4C38A25A14F258CB9282A29849DB88C708CCC4E953DE205D857D6FE177DFCFC95230349E841DEDCE24E3FAFFB848EA5FC1632183F10293545676A77A9CBD61B1C79ADB5C5D5F52B49E4F32A6D858CB5F52851647C81D1E5F705186E7A99BECBF500000000000000000000000A12161C242C"+ , ExtMuVector+ "ML-DSA-87"+ 151+ True+ "C2AB8038C46B135AA07098E5CC4168A670837D3878BC52B24CC5AFACEB488E1EBFB207207D8CA0415CB502F25F08AB7150B736EB694A35CD4D9B8E02F416029A07155C0DD309BA69F23D20D63F617A51888419347D5D85F24F252A3C0D6220AC962443E186FE559C81B85A6A2300253924E9BB7C269DDBD5B174EE5FF60126C0CC864D1C31515092910388111C4386D316601842701042712109711B8149D1C26112080118B861093560539081E12290A3324D603289248269D334099B842948924CA2360813C800121270D8B48018032CA0121048C42D00426418A23102252421296D92A27100B48190426100C3299AA26050B881941021081051E030624C1889D4947020B9715C482948982882B08C91000C10A30898142880A08552068D8A04415324060306056282405A104E63140650282D9AC468983211D1863014884C61042E11476214318ADA342903488E94808183B245E2400AE216660C94851236109B2008230242E3440962964C23C86D4B26310A45928806059B360D040086E006510A486593104D00358E94283219B72518282E0AB121E0906C61126A9B2626E19671420691CC185013462113324C1C90054224104B04001816848CA09099B42152A809509044904470011281843861182804E0404951B668191024D2B040120586DB94204B2284A244019C98811B158D5B042542408C908669A44470949228E29250504248004312DB1044D910621929321289801CC68010A2845110514A140A43A03121A871DC242ECA128183C845E4846C9190209282414B20851A088A504446A43442404889DCC48111C851DA22301AC010E38091641626601212D8246ACBA64980340CA4480219A94CE1028DD018511A096A92468812C30963388EA23811421626043306D9422C00C48D13342A14B21122450D5120420BB665A4022A84089284906983824DE1388E59B0808A32109CC80C1A2661D2102D44482D2005099B328A0400089AB4850C9010DC866CE100249B100688384050488249444E4C944504A391C840501B150280204E60C4805A8065D3B4481B165113C8000C380A094960C8082DD0126588004108A74018C36C518280C1B82C944006C1C63014232C0294210B95200332465446860C31902083508C800CDC407162480D04864D134982CA346E5C040048A87021B5211CB68C9C004912862C13064E21027023392DD092241B040CD43249E3C85064C240C0A65050B81009B7009BC2091C11909006484818886138640AC26D44108C8C18300A40124B9890989070931808530252810684A390402421882380301293012309291BA745D8262E8148661443490409326242852084858C2211E036710009689A946D82162113938841486A54960803032E1AB864C3380C633625D1202E63B4302121651A354844C26024436602207143368C8B22825216281B1126D4B26820183018A43122C589D9186E44002622414E433066E3A2019B483111442252868C0C4308D428111C0132919085414086E4B829542288142492C38651228691C4C01093948412070C6426020B2601CB420159488ACB422953324ACBB4911C3724C23640D2C84894C48923B2608914490B36091A016692B84912332690C8911285915834525B36491808328A2472CA204A53261209A48D23A950810664CAA66D941482CA444D89001143B030422832E2A65012B400DBC80961088020C4512005400B282CE414480AC08C4A306120C4515110529B0266C418302003501127061BB304C2C86C23A9809BB2684CB00800B7480A840C5B9050011991503266D91230014701CA8661E09421810631000652C2286124A305200964E0102010261061868009C640A0066483208001A22C1CC74CD998440B3151582289E2346E4412322115619C80251A08614012821A020C5BC64923052ACC10508A363048124E4C886C01B5091BA788CB3650C034909C482D10424A88382C12344559A0488208921BC670CC88819A200422478A939861800400DA148E50266AA3B6614A02459B1009E1C840E41680E39629D3A8905B326E43A204038251E3086919250004246452A2088104859122719B402613922443846C99B481031386CA322E1B266E58A62102420211860D90C48840100E13236E98320D483041E0487092424A18A6042382511896048A3088180889DB242008182E00C9488B9645F0738C2672FA24FB0F493286EE777C0AA428814994C07322341DDD11E6938E932390D3786ED3F2B23B3AB38FEDB1A144317B12C3F9E5E5BEF0CF921FB3C711CE739C2F4A10DE27B409EAC0CA85E80555BCE15011995DE017C7E4FACA100E2F1B1A0F2485248E5E230596441AD87620D2CC4C70ABCA187E9F0C116D61B8074BAE8FE2D164E899BE2B5621D8A7A07D11179E7A98C31B6F2D8BEB87B7C7066F10E7AF5ADC086F24B4417C25A969805AF6E2BFE48FC60DFE7E169748701AD8FFBBF31778EDF33DC81AFCCFE48CC5033E8D9CCB6FCD35B35E6A1259DF79A3F27454F319364E29DFC92B79A55D40E1AAB1802DE0B8FCE2CBDE7E7D641AA4B36DABF68B851693C59FBE18CA64FFDF3A5671EEEC952C8C82D9C8E07790FC44B957D4590B25BC02DB0816F507439225F14D56493F993A741C5901739A1A98C1062C23A3911B017601BCBF126B363E5D0FA665EA9FB4D34B70A7F08611FBA7AF4F2438A741C31AAECA0809ADF919F2A000DA0E0411323D773FBEEC24B07756F37C9C70A9E7DD76A7EA309AFDB0AB259686715052D3673262377F96FE38A543B40CF51DD8D8A4C4E6E3CD98EAFC08F8B383228BD7406F16A4DB483C571D6518BE9AF4FC34706C83351819E3C622D4E84B61923699335AF0080FCD5BA95E74BB8D33B6476FBFEAC97C167AF09242AD7E99778B9FDC86C9B4C2FE7AA2E0A57D186A9AB176089E069F9DD472652383D507244A81D87C99495E2787206196670C8FA799941BF1EDC81A8354E2CD2777B59A250C5524BDE7449AE0BC11A1D985615BD98B36494BC450C78A231E98289BF11E43313F121F22B5E34ED459D812BBFAFDE9A74A684509702E02FB048D6ABDADC167537D40E4F11B4F996C17C5DBF2919033FE9E00C99C536C70ECF539F3CB12B8AC33682A3405C1B9A984C07C6113A63ED143AB977E01696EEC625204BAEC4910C7AB585299A22E7298AB06FC880A126E2D5E01DD4DB2E915F327FED476FF95E8091469A5A5E89A00CB02800E5ACE958292F6290C0B1C1CAD512915B227CBC4BB958E4B098A40B2414F90620CFC83221B4A83A72D7FA9BD932D24FD58099352C7092F0CE95A1F797049812968D54ACB0765B716EB97D78503C46F8A94301D75DB577548BC8B6624074807F46AB8EEAE283E9386A0B8FB196E2DC8FB740C7C3B993FAF1464B442BC4E5C6FB39CC322603EC4E91EBBFE6974F7487CA167D1E0772B8A05DB7CE1742E85C00FE8C2CD35F6AC2ECDC2F18C6AC6B8B9A5A572123E284DDEBECCB9F8C3C46113FE8B4148E68EDE4288DB1784D1EE904A654083EDB82F9E071A602996FBD1FDA144767A463D6FACF600220648D1C0DC9124347129F0319015DC36BDF23C493A8031B4F2931D4AF8AA9A4FF7C90F28392C525F45E35C4C04CF85246E4333AE668B084A57AF4F080FEF93C0CC61C0D64B5690D3F99F5685DCBC42A4AC71F48968A1D6DB66D780C04E27CA5401AB23D8A7C7D7F8DA76D69748CFBA49DEC7548736B55724DCE1AB3B4523828B12BF040C834585E6B183154EEBDE8EB9573D03B0E5B22E0F0A6AA5096A0EF1888DE67BEEE1A52C4220ED798C0ABAD5E86A2D89582358AC296B15FF994CCC08F1A8EE8DBFB84484ED0E100236E826D98CD8DF6D0209CFB3A1E6719DC4432E2C9A1215180F72FF76100BBFBFA6670EB64C63AF31EE808152C8B45402B85681A7A0F3F507382B4A17582E542799D55D7E26461D862BB15BA6E04FF73E542FF377383D1D668BA21A0F06241625771FF57D82D41A661544AE5B654FB19A7E17A4CD53A25A7EE16407F5547D2FB7EA8479CF1110D9486D282ED2906829019C452C3EAE64828DCCAC6EDB11267B87FBEB0B59F51A6BEFFAF5ACA032495E4CB53021DDDDD8165D116547D4CD9DCCE492C4614C0102358AE258667D55DBB422B88A303852A0E0CA85D053ABF3F41D19DC8BB08D0B84D6C433C062D55B6A36AD34D9714790B6048A31416C1989208D964C2EE0FC4F24DCB6FC146FCC64F79640742E6D099091F266F8A48A5090A87A5573DF829CBFF09DFF23E4BB9B369FA4CDB524CDF3BFEE1C7BA6E02AE31199C762CC5B35E5F55A13DB20193F437FE80AD17DF5E2009089A23B5F19BF445656E72E94A4F33E710E5BABE9E5FBEBCD8C6C0D60473D1DF50EDF466A2705C64835BE9E57E554151ECEE5018FC29FA5F184B784192EAD8A4619BA16A20E2925D1D1C85404DDEB1F14F178648993056ACA0E43FDFF2ED95EF7779AC153EE9B1E61DE4C8D93432DC524D8C13D709E7294826315FE38AB66859EB8C81AD8FABF6FA75A189A8963977DB0C457E9CCBB54984B07E2638449C18997A976BCEBB907591C944C1D0C0BCEC1DC3F73AEE13F2E23125D2DEA7AFC36EA8D9D6E8C65342B5BAE295CFAC0DB1AEE40E99863F4B8D3DF99BB69DF9983DD1FC1CC24E213AB91778F9379FBAE001ADA48D7E1A2C6766F5D94E19C2CF0E41CC3133061EF2745C11D5CF5C93B3F182D41C48566BA9D2BFAB05325BDDD0704CBB9268069534BAAE5204A7CBC5C5DA2300A1AD3D531C230D924C47DAB8BE125A14D3D13F050F58CE8C63C0944D6ECDD348675C5C7E910263203CA3A1A75D3C9311C516691F0A8F342BCE49549B2B70BAF8E53958E66123996F236A0D432F482FC5DE63DCE8ADFED89CCAE6CC4D5BF82C714DBE38A63C9272439C63D798534C39519783BBE8234B510564BEB13E6F4A415939441ECFE7480FB9B32F781ADA1BBDE9BA87FDFC572B1FC6573D56E4D4761EC046C98AFE4C020244E119609AB7B7A537464C4DE449E2B87C37C84747522914B7A5970AD16127724009AA2117353A098F4C7CA206ACD6ADC511FCEBA92B73340F3225DAE1785E57F97F7130A4C4D950982439D9CC6617F8B46C8F81D921C4EAF69031E21139BC0A473A2FFD134A48AF56880EDF19079C6B86D8E74CC436652CFB630C91BD3321A7E50B1AA351C3E81F13F3DF81F16E53361CD8560D62300BB3A09735DBEA5A819C3DF5E34CA306C2AC29734508CB6CB7FC7DCF55C87E09ED6C6B87BDF00C84B9B93F57AC4E4840057AD5360A71C005B71A5A4E80CA1B9FCE26590A489F322BA5C30E7B1D2E18E33997CED864A05428237807137C318241DD951FD685E0B9908C4C1BD86F0CB5E4C4C5D91E4E95C9194573FAF3B6DB50A345A64B39480D70AAF590763D9202470592D78F7906E23616D6517C65D77556ED087411EF6732246949BC260059430316F91D5F9185DC71795F705CD9EDBA4C72F33FE9C08E5C1F64CCA074F292A711A80BAC0DD53BD1D82149EC470661FD63D6393FDCF01F2A946ADCBE0AEF6117B51D9915D8F186CD506D62D5008D66C28044D558B52001EC212714C59098F51D5060F8A510E9C30DE4B37D4A1BD0F8F9666380F75C5C34A31993B82814995D2BB7B2FDB0C35EB3B61F7D050AAB401A7D1A6E29776C1C90BE0BF36A8004CB37A817197BB5BEE7E56DDF08235F284F74A3224E35ED7495E6B4C9ABECF340AE50CC459FA77E646FAEC3BC451400209228B4C4A780AC1C5AEA00085DA44C75BE261B8434E8B84D4245B6FAF19983B96338EA4CBF1DD17D17E753EEC23AD80702088B4E01D4812AA1F8C72352B528C457D70E7FF19338A7152FED920A97D7D62F1E2C9F93C5A8FACF7F417E08B1CD94ADEDDA43AC8F68D5AACD31D61180EA69B36AB123E4E23BE1F380D3EF41BA3627424537F42C79D19EC316216CF602379B00F53E71F9E807190729E23210750631A20608AF474E50295311466F854AA8F2053F3ECBF2847CC0411F6AF4B132CF8E2444B059D8E8EA7FFB634CB7808D90BE3DDDB5269BE030658AD390FB9B74C6E40952DF6FAAFFAA1811E6BD621B3EADF77A566213F60885574D87C2B7A085E7CECAFAB3FF0B3737EB76C2A65907BCA040AE43667927E8BD53215DA42D5494AB76AF43EA96BC5841356B6F520A513AA1C8228F14AFA4091D02589456AE7DDCC47DD320073D1C8CC931A56F6D930372317A4FAECDBADFE551611BDDF1D84C4920F76C08CD75FDB9351BAD5210AFCA72CA72F29990A50217A0F89EDF36B5676761DD576D5F4ED387E002444C6E92B0FF6341C580DDD6530FEAC68C49B41CD8C57CBA44D34716F70A91B5CE7C579F832E21316694FC180C1D0C31A64FC5D23DF80E3DAC1033300A1C61F9CC22B8083922E7B7EC75348DD000B256024F4F77B3F1E5EFC8A08ACCF2FB58D8891911C8F2B5D1E666493EFF44926968C05263A4DBB1C1E9044B2983F3CC9D6137086CCD9F4D5024FB73077DC79A50A42D1453850BC2B9F04351AE5580DA750E66333902F9190E99598546627AD7C22DB7DC24D1F8E538D350179712F24B1CCEBF8535D2099E802B1DCA98A67A410FD8B158B31240F71C1751759852139A03D210C1C633263716426C1B4B7439E8CA004665045FF2BC90A65C566C04657DBAC89237217E689740BD2203717EDA2D05E9A41C078112F01A87D1D94491824882097DD05A72E9B78C92E17F0E1EB274C2991D3542864BED0A05ABFE0EE7217B07E3FEAFDDAA85276F27B650D5BD530387ABC568AB7169622C0F2276587E9A1EB1C6AA91A6B421C15813C583295AD45E38650D2C6A0667D37A3044D8142CA63B0FE942BF386ED59DDDF232CDD2E96D52CA59EDAD2018D04F3D30A9E1F5FC4A5293DFFE8F2BEF27D3A0FF48B49FDEE3C2B4EFE2A800BAD"+ "F5F6B274070FA8CED053B71F1F9FA37CFAD56BC040D9F89F60FB565BC36E34017037B1108D6C97B7F98A63FBA0C16E20FCFEC8A3177981A60740E19EC6A25934"+ ""+ "2BEFF7FC531C100CCD447C941016866E76ADA7D820049488259BEE24927DE880F48CF336B230E87196AFB6B479312F98DE970C1EB693FB70B7AC6311754BCF050075AD1D9431C237B0544BEBEBEB478101E64B03DADCFF6DBE7E70675E2E40CDCF6217BE73D618F2E9FCA20F98D1C6E9DDB633C172993E1B110F394A5207F27579E7B07F6CEFCBDD82B6852024A073AC3F75C2BDDA00297DFE760F73E0BCB1904B91C76BF5FDACABD6EB278A271A6603740537D7BBD2D6BFD4201A8541334CC34DFC10C27F4CF31F5DC5C96BC8027A64BD0E982FB52013A1CC60D9343540EFF87B90916252D366D64559399A47EA51DF746038340F0566D10C261E22F0151D319C846C32C064F5BB9669F74189135DDDA548709775A4152564447E606572527D3CB9515379D8B49E5BE053225B3DEBF3827BB590DBA73B50BB53A67C5055AA3BE1459701E45C1B9E6CC1F9BDD4985970C68CC22B923F9131ED6829972A9FC7A8549C5E31DD97C9BEA8CCCFE34F692BE66A97D63CD432F010845D792AE12741BE1C2114288846CFC96B662C86C9DACA5A80B08D87FBC6F2C3606FE76C9F0B7565B4D0FA025FE4861DAA33961DEAD0A68B53AF8FE912BD4D11C994075A58D903811240560E0EB7AD9E10E7206A6B8442BD9437D4846E8C5A3C807831CD1627A4E776DDBECCAD5646FA86766211DC2DB56515A5032D948A1E9D3704725B3786BA3638B398D1301B723B4C5B8A3A04B3C255B9E62AFE0DA6D94BA8E53BBDF478AB35EBF308E2B84BFC1872360AD1476E13C5BD264C8B8B4FDC81CD6EE0EA0D893F0CEA2C395AAA14B090107AEBE749D04F25AE57096091BE98B3ABD5422534C1E1C1DB65B8302204CB4BA4DE7CB5E447C373179A5AF4E7B8A13579279EBF91C955EBD46261D5E0635D5C2E648A268010863A630EDA4D7F834BEA5C4385870B5EB5D239E83D831593A7E37FC384C24DF00A862240E69C01FE1E36A0C194BA579FDA06E4DC1DA8B6C020066FB4273520D6D2C3300768F6E696CDE46733580E5DD16B9418C4A6787AA5AB964386A735C9F2D78511DE53E006460748CE501915444056E3AD1E2FC55C0B2372F82D38EBA136C2088B0873C14C6F8C03D241F1EFA99A9B2339C672BD1ADD5EA959BFEEB01BB51B2C7E870D98C54BA4A5E903509AB1CDE32CAA50C1120AEC7DE5C63CBB5BE7A3AA97FDFD2E1511388F669EF8ABEE57285A1A746F9FAA7B310FA0324F76E2C3D9EAD55000C7001FF129493B7A0E4A1B149C94871B67BE7BDFD4D0C3DFB7DCD4943A6C584690A860FAB36A7AA1CF01B8E33486AC7C59240CEDEFE849F33AE9EB4ECE145FBEC8B0C1CA2D6DE1D9F54084E5CE775D4AB8DB278B9B896CC4605F7DEAD5B0AD25E22DD63CC48EBDD5B4D36BAB58521962F1D8C6D2F792AF96B4460D4FDAE1A46E235BCE5571385AA283EA5F2DF7A47B00834FE11C9AAF165FB378A0A64EE2F4D777DB0A45417B77B8D5D497F95672B4ADD7DA98815D1148902590BC27F5243092C8FCACD8F6918A58691983257582DB4B0017F58E6537061750875DACCE9A3618641B58E03E238F6F2D38C668752A0128D44EFDB846E208646B0E556C932827B37814160A614CB19FC51ABA8AFA5D064E2BAD048ADBE3C18E7955B1F101C8460F9981F5CAD22556D148D36BB6A7802ADF0563EFBE26976744D9A1DE4ECA311B73D3389845D3B0BA6CA5AE6B166E30FE199D3CF5DF1B44BAE2821D1CF54E21FF74FE36DFC91E3607E5E5207AC7ED474C1094975C3C83A3496C7028DCE07B674FEDA56B548C86B5EDBA50BF8BA9BF954FE06CC524970B644E92696F66B48FAFD5B31BE8DCAB022496A322AA2ACA1920268F3B614224C0D870DD59E4BB95C400189FA2B9ACD4E2984067F325364B60FCD1B42990AA665F7849B1A9C1CB8B08182FF747D15DC339D4303D526D243E8FDAF3D34062B22DF72F29B894072F43C204DCB7095DBA92B3D71FE6F7DA9BFE13863835F1E4ED03BFC88A4339EA2AF778DBEDEB61766A7A1FEB1C96DA8628C5F8F54DD75BFE1AA014D1E100BF12ECC3F1945584450AB9D7A74EFA3B804EE444B040FD7C5FE785E54D5640B8C18B13F8F36824AAB7176715E09C428A75AA9967B16A3BF9191AE5F73FCE4ED2D2AD7ED1A1F25FD693CED39FD7AA7AD3F7BF88865A6250FE50A7E5537D0B4CF45ED7C35A5933C72F6D5DD03CBFB58D1CB285FDCD6EDC57DA80C84F44AB9154902D1B693BA8358E172E9771663CC6B8F374916F4CDEFB8EF33F297EDBEF928496C7D26F98F4F96B7FA62A2FF1F3D4D98FE3B20E8660867D55EB5E60C374C9DBB3C1AAED656D606CB1A609ABF090D7F62EA4C08BEF832E19B5F327CB4B01C7FA3C0915EF3CF0E5B002E370FF4C6BC0693DCE6E9B999D19254B206655F767A48BE46DE6B1F3331F70F0E21B889DE390C2EFA43E98427FE05111CBE03B05B83A2CD5974C7A60773E37F5CEE08497EB2C685ADFAD2156273AB22971740B81906FDF12A134A3CA1DB7EFB856E23824AFDBFA3AEA78928C47A0F978530C0D75EF7B38F3A510294CE47E1C5B25C6CC40A3E40984266844B78EF2EF2D1119670141DBB8EF92CADA1FD72B59FA1D74AF82BC30BB20D338EAE371C0D30A9D162C31F09CF4F43909BD62A0CF8F4C528BFEC99796793C865E4965C328E4E8685EEB4D33B0BE2F25C453F9BEA11615744CE80FE68319A49229FA56C6C1C869684CAB447EBD1D1C08A94D324D704C4602A4B563092D6C636CAE27D17C00E7E8486FD7B86592A4A9D2EE93BBDCB564EDA2A1212BDA368A5F9AB7698C985D13043968DF9F1386CD151B3DC0A38A7C0835D1D6C3FD689ABD3562E3E22BA0B551E90CBE186DC8E54757DDC446E9B813FDCB7945E38D2C0BEC200DC5363AD73E5BE43385862AD2EE57D40B0A56F33D2BB38E21234DB51467E77DC133ABE0537A37310A7BF3D56CA6E73E28887C5EDA1E86B75DC785267CB9ACD2C0C6BD54DD12418087091D95D1E8E8372C58D5AE95B698856964CBDB8AFFF3D19E0AED06B1FE368AD5F293CD1B9BBB067EC993A5EED6A580B82E82B749B656B0F4A1AB9BC92F81CE08B6D9AFAABAF67F590A46FC2625E79B859A02133F0365F2D9EBD7EA241913ADEDD9F13674A69653E3C36D1AB33356F8A6C1967E81D4AA702F9D28AD2D68A62336278482B02E3A758C158D795CAB8CDF2604F8FF21A5A6699CC065FDB10BA9D1E4F46F8AE352F3B94EA4B7C4B5F6C6AB1440606FC153A07B558B53E8552DDFAAB0778A7C69CD83760D1EFADEC9878CB924A1DD4FFC23B558376C322D243458B11061598460011E2363C882F4AD992763D9678438651847AFBBFC540E274F173E0CFABD0D28C69D1CF7250544EF60FA032293660BF65E6CF39D8D5D26B9E6C7A654EEF1A9DCDEFB642FF56C189FC2B0A55A672FB421A46BAF463BC2C4336E3EC93CAED4F9CF593541122E4055E70D74B68AC85EBFED88EF015333514167537020A1F04A159D9E4B5F708516FA220CD19B6D42A733BA6DD677131D459C08495F85C974B7FFA5D3A2A6E3352C06E681D9778E580F8F638919AD904F68F4CD6DC869EB51258695AA858A8FD4250EFD36FEED39373B80137D19A5EBE88AB77E34D630C171EDFFD0CE649EA0991AF9108CC8C855165FC2C34E8BADF84FEAD48649A7249CAC4248F1C02111223C308B978F0E0D3C504613723D03FB00DC4A39B4343EC0627C6BD16DE006BD158E29F6CE63469DF8C3365F68E76DA2496653DEDC801B3CE1F0B8D8C27F71CADD6462E0298BBFE782E819D511EBC72CBFF516CDC761089B44F5EA7926DDCAA5501715C74D24B901834C7C6DE59239EB202B1EE365F152C4B0AB149691DCEE542B34F870408346D7BA325566E43B4DFD8B7C74294BB8571AF781D5C257BA7EE3489F3149441921D30E62ABFE3A5C5CB51200C587E2702452D5B3B1AF2A5E56B59970607A1EB91D9091935E5E5DBCEE0A56656595BBEA88DA24686EEDFBF58DC67FA78CE05B59B21D28FE68BC9C1E580293C597F7B3BBE700C121ACCEFE250467176AACDFF4E679E1F074525729C245B124F922575DB32F2A9E12F512AB421917289CEFCBC53BF6C5C36F58F04EC6545350AEC82F6851ECD811BC3870E7596CD5FD5A3DF3160E554A20F07D0A003B589ABF23BF03671325C1ED0B6472B2A29A47E38029387ADC62CF5193FA49DA2043CBA460037BE46A965D12E63DAF479FEF850A43D8E1E5DFDEC01B7D0D3ED0A224D27BA32C4A3E4CF196DEDF7DE08FB5F959AEE1EA34A9F2FA3D2A2565B438CEB0819CA5594314CEC9297DA8C8A2360A55D4EDDCF67B68E2D70BD9A8BDF3EC98AF11C4C21A861DD9E9D5BDD7A78542682BA7F5266B21EBF36B8626BC2BC4FBD3718BA00DACC6F67225C38CAAFB6BC79FA49F0298D9B90AD95CA741387E12569F2A3EDFA04285FBDB804C08318976D57D25C5522213AEF6C89BDE012628E9BFFC7EE67A47994EC02659E057E64CB028590ED120833386CCE5AD2EDB82E8D1EA5BD8C2E254F7FEBDBDD05298F667291B91F7E5E606E752DA5E7DB37216344DE2BF1D05BA2C040D829352A114E893C9927956944CDCB32DFA8DD6922D394F38E2A46EE4E8E59C3A71E1C7378150A19D7F50E3D00F5A7B49A61E8439B2C29758F421D56A588D588492C2A0621BBFD87F887F001E216EB3BEC22123D3FFF9704FEEC08C71189471139D312D20BA68532FDF384C4597EDD49C075A5E61C5DA5A288426A0D9495525BD4E1B9FFB07C545266B1FACC5BC424742C3AC443504EB69EC56179C8004C1B9DE16E1244A3A662B156E04AEA9025A2385727C6162A60540F6E2CA851F93692CBAAFB23C4CCD6EAA18325DE4310F997927E2C0D69EFC1E5C42E11329471356835688FA7EFDF6BB12AB5983062B4AF5F729FEF5443B6811EAF41E526151663742F3026C961CEB882D56B65EB8B62785E1DFF34F88B40B178BF5BB84D958EC1A980C9F7DB58D98D2D46901D4599DAF1D94162E5B83C0A473E9DE98F1D6CB290FC87B677A6376E6139A0395F61751262CB7BC136F96DB0FB1A0D3285031497BEBF49CDBCF74393CCE395E89225BD684DB9AED5623630BAF564E319EEBFFF943D92F6F9CEA9E792E180BFFE3F799DF825A48CA8DAE4A511E1834139644F3BCE31EB8BF05E957904A8208A4380157E44148AA6226B5FCCAF7183BD41045BCBDC0E263376D85418A260AA640C4735CEF48BA5E8DAA08042CFD402F9A0F42B6C13D58B6FA3C05CE8CFDD92F98FA281A158E6B942C61F967292D74A3C893BD725DA10D3053208F530F4330C5DF76B14D3451B5F7014DBD56EAA7BF3C2AD848857E17012D0BA0FFC4F92F106D0306B8840CA8A83912126285AE3F737D696BA267E60808C82108A78B1D6638E547B3DA1D2B8F6495C9E6FBBFB3C8BA66B5427A6BCF0AB70109BC4E26D6B4EA0F5FBCC75C0754D115F929B729F98C03EE585ADD062796554991EF3F5964752EFB03FDB2BEC4EB3C20732F2A355FD1BBC7247F2AA129A59C1936E6EDB78BCB4EA103F876492ACD4333310FA0794038D2027E99377EF33C5B188F8C89C2C790683BEE55CC0BB9C3E1BB0ADA6D3FB5DDF16DB5A3797D5818367586161A8D97FD746DF2455D609DDEE42AC6907A3F5F48ECAD3DD7EAC202A54007A287418CB8F15BE5B3014953CF5EB3A6E59FCC623CDED91C259A5A60A89B9CC2A86B7B88C7AC4CB4FD8FF2B773D4AF44388F644576D2EBEFC315FDE36076BC711F4803258AB493C516004D4AF110B9B66CBC7E120D6876D97A09668D5E4FB63452C06BE936EDAC134B4232BBA8B29710C19ADD644448338FEAA240AC23A0E1AA434377BDE5C5499243A89A196C97F67D76C8304C186D8CCE329E7281636B9880573160583D904F204EF96F825D0F382C8DF600C17868313194972256CC3C7BE3727B2D559F5FCCDB0A37F65E9B4EE9176592C2AADA40F41F93680DF6D85DADB0270E3595E4F717D2936C0A3AF68B2D2349A2AC24CB795E82B80EE2DF6CC05FBE5AB4EEF2BE089EF64D8E84282ADA273F3F79FF624F8823010F6228B78E499625CF9A110002511DE710C9010158625F1AB8A89D195FAD5DD02693555CC9F68EB7F4A8D9E0DE2AE90ECD37032370200D7AAECDDC6BA88EDE19A3265BDA88F97A25399DC34594B7D2D84AB00875C3DE01366F7CC7316CE9D832732D0696B49057668AF4A3A082C526F8D4F296ED1F03E196D4A8F9ABBF604547125D4BCA9DEF2AB417CB34A3A81931CC7A9B474939EA54EA3E53F74DCF114F583B5A2946577D2E87120762493BDDF9BBD03EFC2E5035862C8F5F6C44D25ABBE65B580DD3F880287CB2DC1878BB9BD6046BAB6C0B1F89B5094B42802F5EFEC6B9301FB85477F9AD219E5DFAFC1EA9E1D04681AF62EC9108FBAFED944623D562F41CDD07903E816257E7C179233E447BA9AFD0F117494A8A8C99ABC1105D69C439637A7FC0D5D7DCE4F5F802405C8A9DCDE1F606292C536EF4385E90C1C8CBCCDEF3FD25475889C10000000000000000000000000000000810141F272D373C"+ , ExtMuVector+ "ML-DSA-87"+ 331+ False+ "4B6CE73916E642641A4F515BAC5C66D2587A570AF8D2044270D54EF27D2C847A9C232B0605C2AF78CC3C05BD9F9DEDCE35F8BD9F0EADE9418637A71334FFC3D816AC02D910AD0667F4A5D0C04F49E082B4EE0F4AA24990AFB631283E34FC1E99F56F96E6D69AD62C875862E4CADDB993C238D2D139685E7575EBFB06F8DDE3B50085654A360C108765C9823010A96850200A09B700211442A1266008B388CAA08C21430E04101113C20120338DDC826053228410192009488A0139011BB41141482621494D81B0298A064C40C260D2168CC14422D9B04503488018A0459842062149200C02641B121022412ED9B20490C66C64B44024278861A80014C990108585040302DCA669A0120643482C53920808085000456E40060919C2092349680A368D82848598228902962149066D1C084C4A0402D0C000D94002DC902814B08408A868420006120861D2168844801153B085CB922D8124055B064D58448594C444D0846453A42524228554A4915A264C5C1086C3468612206011B56112313240888C4A3291D026890A940DC096241A456823B688D92630DBC471110872E4849010156EDCC6415926319A38868A822413B92CA4484619873013A004193584A03630232111E420314842446142696314510A056DD3420D498869E00282CA488D21168D4C042080266ED0206142320D24A640CA3051E18844821286D0944CC0880C4C38054B060861968410124E19922C18464014226CD3125203307020340A0B8810DA14500B012810B0290A3572C1460162062CC9B82DC8A05122410503852908074A5A860C1A962DE38428A2422494A02D62020601A108D1420153A229C1C46DD3483221166A1400308938041C386063A225129208DC000D81302D044889E20421611204CC046CA1226A13488A10476A23C36088C87001B56923B4291112920B25326332689AB04CD8C20C93924100376588A20DCA8440839801C9084DC0C288C90606E3A8509220719A924800B0890018064498201A3322E3384C091368204520A3140152465163842CCBA4511AB3200338510C32100181814C142E13206292866C433422E29208C9B40961C450C8148C09B66CE11648420222A082886320229C448C222712C3065094B605880445092369D0262A9A00214B4049138410D4163202130E9C882CD88044A3206DDC428464B6114912881A920450442A14401011238C2347299C46910A960DC3968CDB804D23B840D2863162A60C50288E9414101048840BC4245C82801C30828A32429492500A874911B724104325C8422D20B36020C18818064DDBB6094C4664C2002809B301C28471C10288049160183531D08005C04051D49665483601DC4865E3487204A089A3A00181B648989231C9122890202D5BC421102262910485DA3840188408DA449122450103078E032040E4322C148568E39665DB226519098ADA02495144269CB08448A86D13086A41446113B70DE036100C11808012328C2265A0A045501872003150812685194849412012100432D4B2482328400A0292D2128981384222482DCBA4001C08711A392C2124860A160022B569CBB60544120CE008612328729A444D03042DC4380590C66114A07089064D6428120A13129B006881C8414CB24C101740C0404D00860152308C943044020566D9204D9130310808321B960D834684C4120A41B060201410108308941805C40025E09884D1004208A205A2008D8BA421514226083769CB180654426021C925D10409A484314A200C99888509491124B180209991614648D0246C133426823842111584D9A268A482912130520A164DCA402E11456D48C225DC080222A92113394A09B44041924853B664DA44495A26060C1242944808A424299324248218300A348D64322513A190A240011B1820940650843085584649CAA809C10248A406840CB231E3922409A76C0CA8680B256EC918218AA0619124321CB9800B2512232624C3A881D3A44D91082C93268018A024224924DA486282060C9008429C46621C47924B3649D3C46420476863389012928404A80DC4182CC1A28D19367064128C899284094072D444259C90802224450CB94163484CDA902DC3144923C0051B3085000809E046905026510CA38D4188644C148211490A08109110A84520A90124186A18B5641A867151C460D080095A0446D124681B36915C40492B5E0EF07A11E1CB556039E2CEBA7BA8B9C353399B5E267BD91F6A25DEADB18F543C36A8E597BE36F6D88C75D810AADB19B7436AE309899F561439CC97A4F65331DBB737A398935C5C47D06E52F15394AA235C72D81283C063A3E6C14C2D9E0D355443D645827B8A07DA9D3B9DF49EF158630D6DDDA82C3F05EBFBD4359FA752A6280D61F3DB5DF226D5661653ADA02F5EC341506D117FB39365E0C2F3ECA47CF9B98780741759D3BC5EDDF27AECD11960356B15B8B2169C6400B7C388F28B00E8691B10802E6BCEF22B4297016BE40F89CD05AC45801564C580FC7ABBC2249AD43F59457FE083698FB4A88EFB996DFDAFEF0FA02853B76310F2AE13D6D77DF08B425AFBDE363F6F84529B47645EB3E9E669FEA21A1054D16B814E8712B75035110D5ED2E850E1992326C2794EE8B91401C76C78F29B6BE9337D75C2673FEB2D86A882E2ADDB65FC7ED14C0DBBEC4C01E339146988EBD6E689BFF4EC0B5361EE2B39F5DD504F9411321F227200A4870F7D46A9FB12B1202AC11914715A1FFAE19537CA1A552754213DB50C5C400864661C5C0E274863A31251EE5D8AC68B9A5C1B0CA63F03C542F922F3F902878BC0076A4F7759B59EE3800DCE029FFEEF5709221F26DA979977076B845C3259F2E7942DB2B1E7C7B377AF4ED7C8256719ACA3C9EE79A999F2FD3F7A4977B03B2C04F4DF00868EC208DDE1B9E64C46EEADDC03C374596F573CBA9863A6FFEE60C5F22D0565A63A4DCD037AD5899533BA3808F7BCB9A85109E88DFC84F82B76E0A2429E6E8B1E5522501ED4AC2B14D0ADEAB6953E1956DD49A2C88D76689C36229C89764395770E71DD58DC87A920BA8CCB9AAFA3C1794ACE4C8556D5F7C2D67678B781817678F9CE5619AD773E00D1F3CD0C28FF52366E6FFBFC2C453341A38CA0DC331C023977091DF6D0967DE437C9B0C9C04798390405C2AE028A93FD56A5C4F6493754A2B15453F2C39F71C5857A0756E61466D2E3AB1F4339DF4020B76FF1B0C363950B88414E8A5607E08B21E3BF7DA749FD5FDF4AA6B8ED5E43943E8052CB422585946F977D9AC37CF7D11210C9218A5764160D2AED79544677D1C5832A12B482973BF618DCEC5D6AF5C73DB0FF126E5724B3C8867B336EC236420E60BAA3AF36B1B2C32D27D28CB6D879F66A55B05A6A34943045813A9CBD262EC79B007DD37963A609808EA8F69F4C320946F6746FC65E659AC1C774F9205ADB08C15970A9B7A16ACB20A0A1B236FE81248E516AACCBB98B500CA425400912CCA8C0789CFB29C3674B766EDDE528119E5EB9C8363BAA8E42D537698B95D7CD08D8007A529A04AD452971B0FC5B176DEB613D6636290E025E058175D1ECF9DD0FFC122D8D10D6C1A0560E649299FA9D4F94C371A9DF1671870C0AD037108DB26871CDA9B9CDC4375C01BF91E1D18712509B75D9B1A05A87636AE952515B2BD2531BC948974DFB8EB940563C7EB71F0FE584A89BE0641A8F0D27AC7DE144CBCAC175790E5BBC541369A0585576059459BB2F32A556270A01E6899769B0C49A246539F839D41A509413FADFF9467B0219FA3E2D1A3D224404AA510D4D469544A8BAF6D562DB1ACD4D772FBF16E4B6FBBF5AC55A2BA1A81953859DA0D11827DE6A37FF2F38F2D0BDEB1575CE2C92B431B3361629E6C5F1AF24642A7D0FF9651AEBCAC0DACD1F7F8C19034E12E70350FC89170179FAABDDB4834D1C029D94844825C10A2040DF3CC247FE60FCEEC731DC5E7C932E020C905538C6BCF97B64E9250EDBFA785275359E8EBB196796EB708FF826E4FAF382D0BD15B8A4A87413E7EE5E66321AF6BE7273298FE67E3679904571EE55F4A748620FD7541EAA7C3D1046499FEF901E5870BF92B0284248F3CD3E1BBA3D0CFFC2B4EEEC1AB6F09865BE105ABE3F41E6D9D08DFA23FEA6476E2DE1100E3C4384CFE21DB7AB9B2FF5BF98A5BC2944BBA01B46D87CF2A4CCC7337F8882FF0D7898F9F0948D3E05859D7F355A4CB74964DAB276D32A1CA45EF61E1D200744B80C591F0D282386FBC2026FDE95BAE97D9FB5B9B5757C78D16CE6D599B9DAF1EFE16F21F9465342E25090516BD45DB7B4E6FEF79C1E0CD41675344BDAE19AE263686B77E1D2A6717FDC0F57E16B7B33D5DAA0B82EDA1B4BB910FE0B8BBAC1809D02681612E455B174CC59BA4EE88AD97DE936E31A1A2C69194BAD8A2DC489C01D4831C03DC36FAF18370FE7D78B7F0C216945300D22EB7DD8794736A7D3A2F28D86C7082B6E089A50C87A501939E5709686FDD2EFB764C546E962C161BAE37B7292322560F60EA95A6A1D83D028523B2F27B3DEB249FACF5AA95AF0C1913450036FA4DAEB3124ECD21C753775CDF9F974820843D3B2A5BA9430FB3DE5A528C42D1CF5829E0F9D76C68882BDF7D82E7A215C1E578B4BE338715772E11514C4B277EF0A43A2ED913612C09CA42A41FA68AE77ADD3792B2737034D69D84CA5D20CBC88970482BFFEC8848F1E8AFA051E6A3D79035ED583A62382007C6BE5D1F8DD506152E9C0943D9CAD1CF72EBC07FB74874877432DE374C500866D75AF1ADACAC71A8963A5ADB69E68FCAB2DA6D1D35C880F8CA4B3913EEE6308E3E47237B783A596787404D0D8049F277A17A38687ADB6514A8C7D9553D5C425416F14075FDFA63941CCFF85DA0D491779505100E9ACD817E492178E78746F28A4D2069F18F10F6CB374FD4981FB1BB098F066BB1CC1501B2FCFD3DF75521645412FAF5154B4B1973285035200BC35747AD83E057DC1059429A7192C953110222CD5C5F219D55B160A490AC98D4DD4297F1B5B9F68AE84FDD913B66A9A1056EDB1D2534F5BDC0F8A1D9F5F63E812913C7BD81C57071B1F08769CACFFCA2E9D867FF2B29FE50DE92C7A657542DE0CCFCD9C7B8FA605E0FBEF58CB165BB2DC505F015BA9F39FE10CFEC48C9E7E541685320D009D47C149AA64506F5C1C0F218C9529D6954A328FFBF7DBF43CB574190FDCB33477FB0B44FCD0A6BA14BA657CBB0AF439D64B84FFEE9FEAAD50C94FCE914CC5FB8DDAB8CCF6B08DB1D361B62351ECDAB2476ACEA0E3F5DD5C4D22913C1A3FE10A3761C42FDF39CD49FF0641D0EE8463A188C5E6E30A00D724CFD87D10BB576163ADFE83AB2C8A6BC3D9D70A81AC4DABC5E828C249C10C48F79B325E24B679FFAD23395A177D2F3BDE3D249392D0AF18F19D445C6AF288E87DD0F76445060D4617F778FB9CD3B201FAFB87D507A9EE84E822D9ACE3F7BB56CD6A85B33C9E83AE56D99AB270802A2CB32816B39675836576A298B9C8F009719B75190B09EDED6D6996C4A295EE4568F7ADB3F0EC6EE2F6AABA235DFEE6CAAD71D8353D31EA74EBB01F718F87A0831236FA7239E173A38CBCA77F042E69E616985733525C614320B78C0F04B266D3DC5806D444D1EB66433EF8CEE169CE7DB358A7CEFFC7FEBC33AEBA6B0BF2940BFFAE4576CAC3A84ED7A7A3E7853C51417E5017A5D8555F71FF75AC71AC44CC5B147746613EC1AA5556FF7B08D133DFD4D082AF8761EA406185C8B45AAD5C1EB7946B1A06CDE00708B17635A2F06FC0E7B45D90EC98E9735913C1F1C223C177FD1D53F139704D56F923D0A107BBBE0F05B5B6A4BCEE19BDC56AE6CD9F4F3EAF0425F0E6E018133DE9B1D5C94F380E321FD26563AD8156C9B5362FAE55F32B7AD4756AC8B80D1AF58C88D8BCBF35E1BE7D7DC09E02C87680A94158CF8CFA3726EB13E57AF13916F0A184E94D0C81688022B691D4F925451883525B05142F0ABEE8416A27A2CAA6F9708C1CF6738E0F3EECECB695B6159BA3902A23064638E4D47A8DFFD036D4EC0EDD4DE980C1B28613ED785A5EF80524C20AE6B4C8F8F53BA65B661A0A6178EF0B008B2D20268C1C38D9C56B1FB7C466E5B113709517D36341E8A18D67C4A70674E0BEE94FD48954C8D0BB2494F578CD05ACFC7313C9ADB836D9F22C566B5248B9B7349D04CC722F6F6697FFE085CB153AB7EBAB6B8D8741A0B9D7E32015D1090747D2BC5D0E73BA49BD860C3CC74C3CE838B13AE3BBEE2391C1492131172FEC746F7022AA26D680F56CE0CA8B91D34C7BB70ACC5F60A67FCDC439421796AC51BC7E67D1777C5EDF51C84B7AEC50E06AAA8C91A78F448AEE98B4ECC3A6D8BC9527D57BDEC4356D24D1672F9904636C25C91ADFF0669C9813EB468594A6768537E222E5AB01362C36CE52395CC2C23DEA3EA3BF1867DD6A78CDD21F6E0D410D4B80D309460218DA4AE070239E5BC0E2D4D13D44308D6766D317F44483E84EB22967F7E568901E3A39F53E8916CB9596F979765C53B855B67C45546F63B27BBFAA3E9ABBC61D6969AF88686BB7DFF76FAED2DC2FB4F305D28835839F3F6288C531FFDA62273870B0FE60888508C5D8CC0C1AEA80C48EB01BB5662C7B112151A4A11853374C333C21F48C0B05053E8198B14DB7F1290976F065ED5958E5405B349BA52C2D0555A01A54AA371FF3D941B6178895BC4D5D46339F5D62BB5ABCED8663038DC42492BEEDB8E5CAE9398E83A7DD83A21389878AFC3CF2537B7D6556B46D7C02D423CA9ECC1EF41C5317E688783CC6575B6C92F1CC0D2A4159EAF7E404209F875C382F11ADA2282726D2A06C4EA7BEC61A9D06EF02A35E7C4F322CB4750899E927B73CC454BCF3D24D76FA409D6A4B82083C6BFA38FC4A5F7C22388BFDFADF88CFB2AC9D07337F7C3246CB"+ "1CF7A0F0D4ABDB6F8200DE6B356B82F46CF3677DA321BEFBF0DF8E49C7AFA48678407FA519534EEDAE92F661C539D03E2EC9E3EBA6AC429C9CB9442398890279"+ "5B000AE26E572CA3A70C0B142224E75A8531EED4A2CA48F0B0E4013447203401"+ "612B292C10779A6FD1D1414A8DC3A3A1ADD4B13FB18EA56D479A3E10733CC018F3BA25994127D2B84AB1936E9E0A7CD33ADFAEB76BE5B039B9199C6F915C296CEC7A6871845FEEF86C04E615D05833CD9A3058D2138F5456E79372FA7E7DF78B0C81CEB88D9B0387B8A1C717FC707C9148EE02F12024144147E82DEAF075DEF11D1D3FF16D78719121566FD03195C099DE3777B71FB2DD51E063FC85FD72824BDF8C25BD9FD18067F0470D7C7660FEE73D4D71FCF9310FE5E0512BFF0799486B3ADFF0996CC69758C22ACDEC29CB2E75F8399E2513C57263DDA5D2E2E0253958807444A042E97C4A3BDDF913320A582FA41143789E2DC6EE27ABF532AD06E501B44C8769B568CABA12707C16B3DDCD9A0B9C8232993BA5495F8A2D87EA886E25A739289512AE6C62F53101CC2EF21A7778BFF3F7B4591B698D84BB373761DBAA7A908C3297162B9C5AC72D302F1503B1EF62C8732507A7EB5E207892F20987D5F49E7CC98584BCFC8A0ED8855A07EDE9F7F6B0167EE0A88C8EC932756D769B319FF2E8A83175F4260A5C3D20394A26D8DA271C215783A8DB2B1811A04F7A3AED3AF55CA932E0E32BBDE0596885CF6A533C12BE49386255A0D69734B88573D1E5F9F79A68D723EA699FEE5859E9C3EB3A9E408B92E89871839F571F2DFE4549CF3C95533E14EDE4DF931ED4460FFC3B0D9B1C88EC321035CB055072B5A644783365836FB4A13DDF185CE80F7658991F61C1F99D6B4D4919F1889F385B6BD523C48C154FEAE1C531EEB1EE34569CAD4396AAE0540EC9AF3FC1476F8E4F1F29ABCB2557C8AF43A7EDC51F2460EE135317BE25826ACAA669D4552E03CE4A9F02D27ACBF600DCFADF705342221116B2F38EA4C540D3DFE16D74B0CAF3B1B954635C00EB981F33EDFB836F46607723EC8237EECF11C57DF8F7FD6CFC4819013909FF43D7996A3DB5BC5B217D440D4E455042D5658B4E27BF76DBDB7306E279C358B7A9D22BA18B7ADDA564D6F2489AF68F11090747DD1ED22C03FCB816F6BF2767D95CF1D1E6197087A7F872EA2A84AACD36CF5374AAC0DEBE7EAC3B7D63407CFE576BA13DDB837227E1D97FD05E50586857985F1F2EC18F61A36F39B77883483A8E2DBD07049DEE6C8214479610C8C9F013DAFEEDADA87444734B8321A6D1996C26199D79F1BA59022AE5CA29348456EF19AEDE6BEA6379EF8EC0147BC1494559FA2642E1E33225CD8BE28DBBB714364F7D9E5CACFD72A1B84B1A4A596A87A52025B7234C4BD69ADB1318BF6E49BB4BF172E725BF6DD08B3AE150E8C3EFB27F1ECB6248EC0E7F1B3C717AF5E3A9D04BFF37B93B6E51C8933FF0FFE09B6ECE237081586417E110811EE994E6F7164D9E264638E72B8E07398C41C323ECFD897852A0515E199048E8DB1D926AA17626CE8543A429C0895D5CAAF83A4C7A9C18B466FD77A617D8CA15899803E64EF265AB378D7027DDED4A8A0C4AB11F804955C1440A71BAD2496DD12807A01CF78AD0FF8FD495E9814FE11A22370852A0C85FE89EDF099D4C8B96F2CDB0751BDA7EB5AFC733F8FE3AD08343FA68D384497858CCEDBD7080156CDE68D826827DB12B83A5BB533AAAD5306F25FD1EB735D8EDC036BD339CDBEA44655EB384C8DB1D7DCE9DAE578275CA24FEDA5D0EB15179075EC05FF458B55E7FB1C1E9706AE9D83973DBF89FE59E8CABF0AFB792D88805CE5E2908B2EC6B8FD8790360DB6E3722AF0BB3049A95CF7FF5D4290006F6D7B077F7172B0D4BEB84282EA4AA6559788748BCEB16EACBFEFEC89436B7337551DF1450A679C15E7493263CB252C1F0288BE5B46253170C4870AF0197F8D387135E0A6C6E8513204FBDCBA477D3C9A3E7708A986A26215F3115F004992D12587A68901571FEF1DEE1560D7780348B3157F0E59524DA89F88DAE639EBB735DA6C807A56F25BB95F4AA4363A7F89429C221B106C6864847C744B0D44A4112724805827244BB06952AB995EECB749544866EB0EEA9AB34D54EFF252ADDE20B210BF14F7ED753615BFEB9D8293C2975319C38DC907EE48A206BA99356DF5BE324D6524E0D05F2ED29B17C15EB2BB334E0B98586A56BF9934973EE6879C6C1C5DDDDB01290217B4E1CD7468D704C739EF88A91EFB0A4DCA6894EDCF36067CC28E301E02DE233233C41110BF4EC710A2FC26AD47B0109581B78FB05D038786D8407DB96AD459DEE5BA1941C7340251FD2381D2EC7F911DF1AF8DB258CF144DD83D69EAEC22CF582B783F5757C8A9F15D20FB2CD27DF7A9E5E610B1F4FBB8019F82FC2EB9A27E85EE1F73B69BA6D4BA6313E906451978497DF208B72AA3630F096D74AF418A2E6E3FA5A4C25E2D97D501903724FE3346EF8FDCA42FD018BDC7B1FBEAD58B90B2CDEBE702A3CEAD2ED936BC3535607842FCB6DCFF7FD768AAE8322BD3C32E3F791378BCB77C63D87D112ECE628D570143A628C5A896113BD56B5C645DFABFD818926ECED4CAA6D2ECEAFC6C7B4DC95E3D11A1CF536A9D9B4D5FD410F08B0D1C011F01BD5F57EB8CD965A07673F9A5DE98702B460FD62CEABBFB509652520A5BB9B89ACC8F35EBE56463220E0C6B43B95C4C4B0C37492E72D20216B876CBA74A8E7C2B53FEF54E01BA7269A8AEADA4F2006DD279A38F24445C45B2083BDEBC9740BEA4601AC8F68E1FE1F1E7582E711EDBED1B4A2469BAD204FC38E8F4190529AAF9C0BBA9CE922ACFD7EEAD758625AE5AD2F16DEDC441CBA9E254378B4C85271C254C4996035BC8A0495D57B3ECCDD373ED2BD6DEDBB82F3AC637F697F7DA9ECA5ED2079258D7CA89DE2656352CC33FF176DB0CDEB1DE69C30A6E6527B686F5429BCACB296A1572C0674E2F6EDCCC98181EE07F76D15CC0744AE4ED1FD96E91E105E8D376FE97416834A018451D9E686439907E278BBAE676591AA67B5E072F0B39D2AAD4F897D4FC2A1A4DC31F22C598AE2A9FCBE82AF016D3CFCCD1A924A3EEB7364122599F8E86A36BD44E4EF2C61572DB8D2DD131810E7D5F1238F60D13286D314F7A826AE63F24DAADF1F6F4EC5C49541FB9F4D0FC7941BA6AF2C5B04589CE36F29D8B6C31FD747B3D1489DA004063BD381737BDB40376DCCE16D4B48B3D42EE7BD484F1343FF6E2FD3BC83C2A61DAC88A5E8351A94B26A51A289236DB4E16E3D5BE3CCBA00592664DAB25F299FD22FF870C5EB36C73336D5D15CC0435FA46BB98004958B61887060F147FFC09DE787F6E3D7B3C6767880016343778597580160C783265803088878DCFFB808C9DA851706BB4614122B5A530B55AE0CA2B8C276CFC00822EF2048AD946AD0C81352569D98977A2B83236C8DF7CC50BD4A1C57AD4D16CD225AAB2720D583593D65E6400610739612C00E58DEB8A464A3ECA51289C637F34D1511840F4DC45D11CC9C2A104FCD5F3E4384C2229E4108EF532726ED297EB25333466BA727BE652C76C62136480EE437FEE57E1A492045BD7D0EFBF841DAD6540D620F7854A3B379771FBAAEECFC0BFE9DFF266C654433BF51E2786845A2A92720AB20DC28D4B7D0FD937F3CD35E4447109F9D55CE52D203C821DF78D8807D656F47A26ABAFDFC19FAAFE9FD1FD0A23C72995E90EE17E13E8ED8A1998C4A5BB90BD899C1A737CD8524CD76CE559804E13E1471805F5EDEF4C1DB312C7B712935F54EEE80874519A35A5437B5C8010FAF8F2535AC26C8B13AD3398298F1F14BB0A71ED147618D8AF724B4DE41F72BEAAB27072FC4B150FAC4FF6E12ACAE8309AC652350A523FB54B816ACD2B4B9E670EE916A8CAC58021139FF2F5A47A9B9E3C00196D9A5A0BC9B79C542CEB82C02993BC645F9E6BCEAA6E01E799824D9119AE426F961ABD2132C4D4782D670DC3C241BC8AED847CC4012341EFA9113882DD455A1D0CEC6D971857C17A187BAC5E2719C467907BB888B8F73111282E9B703AC88FF509B992DDC0FA6FAE5D1B44AEB2F8798C8D93E2B76D1623FC85C9C20B76466EDCD920F843504D94C393017F14403D25386D988C9EA648A4A3E786E976A9EAFDB6493B40AC695AAB88CD89990714A4DE95AAF9836C36F3BCA919218BD2779968445F31E71AF2AF863AE9C21BC63E3D26243B3FAC925B9536CC338959499531E41CF36815EDBA8965F5DEF30D6D70421E2BEECD0ACFEAA5935E1F0699A2D05C1948B04B65EF4668A7E2C25F7B73D78298C86A56FB0BC088A255B19671BCE09BF238E69FB09D84A8513359DA87F35AEABCE8AA3D770B315CC34619ACE898BCE91F1D8304AAEAE350005ECD810F069E775399A600CAD68F53D01305A2E896441E37DC0410CF431DB61DD9BB608DF8D570CEA939557EAF6A74C620EC3007EA3EC5B0892CFAC774323A69BFDFDEC2279228041A32A2F843A027BCE265C46441A9330AFEF6B8BE2FDC8577BC2F195EFCA4FF02C02C54A748E74315692F9AF4FDAB7EF42A119D7EEB9F6DD648DE965F2D78F119EEA6E62852A6669267D572B7163A67E33A607EBF1AD09BFDAF9D4808A936E21A746D251D66DB86ADF72B89300925BB6EDD4868455F157613A1CE30ABD1C26D79B4D93FD9003CB4F4A3E994C69F2760DC521BC5D19B0FFF44AB4032F65AB5627F238E40D6DA327F56897C12598080FBDEE9C830914BB08B324D74F4439292E319B0B67FE634FDE10823B47E4413FC65ACC6BA581E1E452C580CE8EAD13DF8C3EC76DC5BCF742418DBD6B0C9803176EE4E0C6F849C93CA623C44055A95203E5344FA73948BB85790B7EBE10026E86043BEA93E16903FF1485964ACA76A1F4B6935B357ABB5E180C8117F82935A4A1A2EA5F8ED8E2A3D75AAA5FCC6A67C96A02BADB6B1AC2AAF369240B718117EFC8C8CF8FAABE3F3EC343148DEC8294B5A31B9D2EC2B9B38182B9BFCFBF0CC06169EC86849F64AF0B14704FFE142AB5217F224605EC5E4B9B4C7F59A14B6C1FDADEF35533B33BAD2398E2A61ADEFBED2E3EE6025AA2A573B0F125EC6E63DECE39C4C34E746F9165981D8A38F53CEE2908A874716953671C6AA7B06A1855AE6B62243CFF4844786C2F1F5D825EA746BD5F2C93A90DB32F56497707512BFCD5D79544DA19D0999570883774BE1317FC9679FC24C7C2E9125714C437D071984F88EEE976C05981959A4ACC6BD2C0D574B23057795CB8DF64893748AE29411AF6314A4CE9E0286812EECF0958714D9892A92E02ADF25DFACB6AB1EC02A83682AC75435BCF4ACBAEDA7D4840D4FD28983C1811BD1DACB26588B9120F0ED4A58269F153B8FA01238525EF544CBCF7F5FCE2AF2C3133E796B336BD951EB2A663D13D084B6B01F8C3B8F5DAB6F53DC090C202C066D2E13ECAB8E45053ECA2CE74AC936B55587C18B641D40649CB7EA0814CE495020AA35D1157395669CBB476C99171433696DF098F64AFBD51C21A67438F2D3796E4DCAEE07AF68B422969476E8DF327F94B669052B2935A4473B847CCD698F936C08DB5FF53E4B0AB55DFAB1DEADED56B624D861E019BF0B658489AFD1EECE990AA574E06068FF80C44D9CAAD4689E1DFA23C56981F5D4BEF9841C71EC70805C241E89A6C4948088691C3572BF20C91364CE37F0038F38A5B4AB47A6EC9863709397750745F1A8A427B61744553DEBD84451B84F3FABFE12185011E0BAE2275EB4079E927AD1FD0DB0999A5B24E8D3F5DE9974D7621094850789A5C2DE48802FDE6157F60EFD9DC7AEACB58B4004CD3AA9A9D430FED8C99FE05E9797B89092B5511D1BD15562F39A2B602A0C3A65D12683086DDBB4D5110A26F3FBCC2EE24200C9DA5D30DA39AAE9ACA70CC1C00CB335C9C3BF68E22BD1FEF1BC1256B79ACDD4CA8BCF5D788729F932E3CC793B475488FD204379D8FF5F92CC34641B6A40A404F26F0561C70C8E8196CC3870D7A551F521A2B9C7A5D778A592426E8FBFE0606AAD44D8DA6B237D1EF3ABF929D920A1040EF9BA599B1C0E9C239C293347AEECAD751134F2128B2651DC7EB253539B710DF120E6195E606EE79ACAB150829560D5003DDE113301AD097A3F5A5AD0BD9D9C30F7644665997E009A24AC414EE83843110E918D1503788D5254F38D03C273D82C3A26FA4B2458CDD837A1193E59BA1E6870E1F3DE6F86BDA7F324A3D75A8F0D73A2B6BE50D8D6322DCE308AD5451A3CF466DB178383659F726CF6601C59A7A9C48CEC0D398FCBF1697FB35F3661C4E65A438F138DB917D9E770799063E6E1D3FEC55A458CF602D456393DEDE66953CA6A92D77BE003A780E8CF00D23A3C02F0580EE7A7F4DCC41766F6B7164F9095893A840DAC93864B362D7568E917CA1BF232225E9CB768B731C90FD519A37D0CB1CFC306D215471DC23E78CE1440853CC6E2B98BA6EEA6FB63ABE99638600E8994D69C4E62ED548D9AB78007B35D97BA139A2D37F157FBA37EE323D289CD771F16C75C7B9B39286A3483105F1119C50E40C5A082330509AA8FE0E193E729E1329344346475B8AB6BE17607FAD212A2B5171878E959FC3F86DC5787FABDEE0FF4F5E696D85DB000000000000000000000000000000000000000000000000070C161A25272D33"+ ]
+ tests/PubKey/MLKEMSpec.hs view
@@ -0,0 +1,184 @@+{-# LANGUAGE RankNTypes #-}+{-# LANGUAGE ScopedTypeVariables #-}++-- | ML-KEM against NIST's ACVP vectors, and against itself.+--+-- The vectors are the point: a round trip only says the two halves of one+-- implementation agree with each other, which they would even if both were+-- wrong in the same way.+module PubKey.MLKEMSpec (spec) where++import qualified Data.ByteArray as B+import Data.ByteArray.Encoding (Base (Base16), convertFromBase)+import qualified Data.ByteString as BS+import Data.Word (Word8)+import Data.Proxy (Proxy (..))+import Test.Hspec+import Test.Hspec.QuickCheck (prop)++import Crypto.Error+import Crypto.PubKey.MLKEM++import Imports ()+import PubKey.MLKEMVectors++hex :: String -> BS.ByteString+hex s = case convertFromBase Base16 (BS.pack (map (fromIntegral . fromEnum) s)) of+ Left e -> error ("bad hex in a test vector: " ++ e)+ Right b -> b++-- | Run an action for whichever parameter set a vector names. The set is a+-- type, so there is no way to pass it as a value; this is the one place that+-- turns the name back into one.+withSet+ :: String+ -> (forall p. MLKEM p => Proxy p -> r)+ -> r+withSet "ML-KEM-512" k = k (Proxy :: Proxy MLKEM512)+withSet "ML-KEM-768" k = k (Proxy :: Proxy MLKEM768)+withSet "ML-KEM-1024" k = k (Proxy :: Proxy MLKEM1024)+withSet s _ = error ("unknown parameter set in a test vector: " ++ s)++spec :: Spec+spec = do+ describe "ACVP keyGen" $+ mapM_ keyGenCase keyGenVectors+ describe "ACVP encapsulation" $+ mapM_ encapCase encapVectors+ describe "ACVP decapsulation" $+ mapM_ decapCase decapVectors+ describe "ACVP encapsulation key check (FIPS 203 7.2)" $+ mapM_ (checkCase "encapsulation key" encapsulationKeyOf) ekCheckVectors+ describe "ACVP decapsulation key check (FIPS 203 7.3)" $+ mapM_ (checkCase "decapsulation key" decapsulationKeyOf) dkCheckVectors+ describe "the seed a key pair came from" $ do+ seedKeeps "ML-KEM-512" (Proxy :: Proxy MLKEM512)+ seedKeeps "ML-KEM-768" (Proxy :: Proxy MLKEM768)+ seedKeeps "ML-KEM-1024" (Proxy :: Proxy MLKEM1024)+ describe "round trip" $ do+ roundTrip "ML-KEM-512" (Proxy :: Proxy MLKEM512)+ roundTrip "ML-KEM-768" (Proxy :: Proxy MLKEM768)+ roundTrip "ML-KEM-1024" (Proxy :: Proxy MLKEM1024)+ describe "a ciphertext that was not meant for this key" $ do+ implicitRejection "ML-KEM-512" (Proxy :: Proxy MLKEM512)+ implicitRejection "ML-KEM-768" (Proxy :: Proxy MLKEM768)+ implicitRejection "ML-KEM-1024" (Proxy :: Proxy MLKEM1024)++keyGenCase :: KeyGenVector -> Spec+keyGenCase v =+ it (kgSet v ++ " tcId " ++ show (kgId v)) $+ withSet (kgSet v) $ \p ->+ case keyPairFromSeed p (hex (kgD v) `BS.append` hex (kgZ v)) of+ CryptoFailed e -> expectationFailure (show e)+ CryptoPassed (ek, dk) -> do+ B.convert ek `shouldBe` hex (kgEk v)+ B.convert dk `shouldBe` hex (kgDk v)++encapCase :: EncapVector -> Spec+encapCase v =+ it (enSet v ++ " tcId " ++ show (enId v)) $+ withSet (enSet v) $ \(p :: Proxy p) ->+ case encapsulationKey (hex (enEk v)) :: CryptoFailable (EncapsulationKey p) of+ CryptoFailed e -> expectationFailure (show e)+ CryptoPassed ek -> case encapsulateWith p ek (B.convert (hex (enM v))) of+ CryptoFailed e -> expectationFailure (show e)+ CryptoPassed (ct, ss) -> do+ B.convert ct `shouldBe` hex (enC v)+ B.convert ss `shouldBe` hex (enK v)++decapCase :: DecapVector -> Spec+decapCase v =+ it (deSet v ++ " tcId " ++ show (deId v)) $+ withSet (deSet v) $ \(p :: Proxy p) ->+ case ( decapsulationKey (hex (deDk v)) :: CryptoFailable (DecapsulationKey p)+ , ciphertext (hex (deC v)) :: CryptoFailable (Ciphertext p)+ ) of+ (CryptoPassed dk, CryptoPassed ct) ->+ case decapsulate p dk ct of+ CryptoPassed ss -> B.convert ss `shouldBe` hex (deK v)+ CryptoFailed e -> expectationFailure (show e)+ (CryptoFailed e, _) -> expectationFailure (show e)+ (_, CryptoFailed e) -> expectationFailure (show e)++-- | The two key checks, each given a key the vector says to accept and one it+-- says to refuse. A constructor that accepted everything would pass the+-- first and fail the second.+checkCase+ :: String+ -> (forall p. MLKEM p => Proxy p -> BS.ByteString -> Bool)+ -> KeyCheckVector+ -> Spec+checkCase what accepts v =+ it (ckSet v ++ " tcId " ++ show (ckId v) ++ verdict) $+ withSet (ckSet v) (\p -> accepts p (hex (ckKey v))) `shouldBe` ckPasses v+ where+ verdict+ | ckPasses v = " (a sound " ++ what ++ ")"+ | otherwise = " (" ++ what ++ " the standard refuses)"++encapsulationKeyOf :: MLKEM p => Proxy p -> BS.ByteString -> Bool+encapsulationKeyOf (_ :: Proxy p) bs =+ case encapsulationKey bs :: CryptoFailable (EncapsulationKey p) of+ CryptoPassed _ -> True+ CryptoFailed _ -> False++decapsulationKeyOf :: MLKEM p => Proxy p -> BS.ByteString -> Bool+decapsulationKeyOf (_ :: Proxy p) bs =+ case decapsulationKey bs :: CryptoFailable (DecapsulationKey p) of+ CryptoPassed _ -> True+ CryptoFailed _ -> False++-- The seed and the coins are drawn as lists of bytes and padded to the+-- lengths the entry points want; there is no Arbitrary ByteString in scope+-- and one is not worth adding for this.+-- The seed generateKeyPairAndSeed hands back has to be the one the pair+-- was derived from: expanding it again has to give that very pair, not+-- merely some pair. Two generated pairs also have to differ.+seedKeeps :: MLKEM p => String -> Proxy p -> Spec+seedKeeps name p =+ it (name ++ ": the seed comes back and rebuilds the pair") $ do+ (ek, dk, seed) <- generateKeyPairAndSeed p+ B.length seed `shouldBe` seedSize+ case keyPairFromSeed p seed of+ CryptoPassed (ek', dk') -> do+ (B.convert ek' :: BS.ByteString) `shouldBe` B.convert ek+ (B.convert dk' :: BS.ByteString) `shouldBe` B.convert dk+ CryptoFailed e -> expectationFailure (show e)+ (ek2, _, _) <- generateKeyPairAndSeed p+ (B.convert ek2 :: BS.ByteString) `shouldNotBe` B.convert ek++roundTrip :: MLKEM p => String -> Proxy p -> Spec+roundTrip name p =+ prop (name ++ ": the two sides agree") $ \(seedBytes :: [Word8]) coinBytes ->+ let pad n bs = BS.take n (bs `BS.append` BS.replicate n 0)+ d = pad 64 (BS.pack seedBytes)+ m = pad 32 (BS.pack coinBytes)+ in case keyPairFromSeed p d of+ CryptoFailed e -> error (show e)+ CryptoPassed (ek, dk) -> case encapsulateWith p ek (B.convert m) of+ CryptoFailed e -> error (show e)+ CryptoPassed (ct, ss) ->+ (B.convert <$> decapsulate p dk ct)+ == CryptoPassed (B.convert ss :: BS.ByteString)++-- | Decapsulating a ciphertext made for a different key answers something,+-- and that something is not the other key pair's secret. ML-KEM rejects+-- implicitly, so there is no error to look for -- only a secret that does not+-- match, which is what a caller would see.+implicitRejection :: MLKEM p => String -> Proxy p -> Spec+implicitRejection name p =+ it (name ++ ": answers a secret that does not match") $ do+ let seedA = BS.replicate 64 7+ seedB = BS.replicate 64 9+ m = BS.replicate 32 3+ case (keyPairFromSeed p seedA, keyPairFromSeed p seedB) of+ (CryptoPassed (ekA, _), CryptoPassed (_, dkB)) ->+ case encapsulateWith p ekA (B.convert m) of+ CryptoPassed (ct, ss) ->+ case decapsulate p dkB ct of+ CryptoPassed ss' ->+ (B.convert ss' :: BS.ByteString)+ `shouldNotBe` B.convert ss+ CryptoFailed e -> expectationFailure (show e)+ CryptoFailed e -> expectationFailure (show e)+ _ -> expectationFailure "could not derive the two key pairs"
+ tests/PubKey/MLKEMVectors.hs view
@@ -0,0 +1,195 @@+-- | ACVP test vectors for ML-KEM, from NIST's ACVP-Server.+--+-- Generated from gen-val/json-files/ML-KEM-{keyGen,encapDecap}-FIPS203, one+-- vector per parameter set per operation, and for the key checks one that+-- must be accepted and one that must be refused. They are here rather than+-- downloaded when the suite runs, so that it needs no network and a release+-- tarball carries what it was tested against.+module PubKey.MLKEMVectors (+ KeyGenVector (..),+ EncapVector (..),+ DecapVector (..),+ KeyCheckVector (..),+ keyGenVectors,+ encapVectors,+ decapVectors,+ ekCheckVectors,+ dkCheckVectors,+) where++-- | The two seed halves in, the key pair out.+data KeyGenVector = KeyGenVector+ { kgSet :: String+ , kgId :: Int+ , kgD :: String+ , kgZ :: String+ , kgEk :: String+ , kgDk :: String+ }++-- | An encapsulation key and the message in, ciphertext and secret out.+data EncapVector = EncapVector+ { enSet :: String+ , enId :: Int+ , enEk :: String+ , enM :: String+ , enC :: String+ , enK :: String+ }++-- | A decapsulation key and a ciphertext in, the secret out.+data DecapVector = DecapVector+ { deSet :: String+ , deId :: Int+ , deDk :: String+ , deC :: String+ , deK :: String+ }++-- | A key that the checks of FIPS 203 7.2 and 7.3 must accept or refuse.+data KeyCheckVector = KeyCheckVector+ { ckSet :: String+ , ckId :: Int+ , ckKey :: String+ , ckPasses :: Bool+ }++keyGenVectors :: [KeyGenVector]+keyGenVectors =+ [ KeyGenVector+ "ML-KEM-512"+ 1+ "47B893474672BA92E4B12EE44FB32953AF8E8503B5FB471D1614FB8A021A660A"+ "1F8CB39E9E30BC458A0DC5408884B1187FB217018DF760FA57317703B844A0A9"+ "28266A088B3482439BCA01AFB7CA5C6136A979B5159985A9484B36B679A5F7B9819EB63577891F7BB9CB98413CCC434ADC79A16D6AB3076569CE6291C59B5D64612A7FB0C15013200BC8BEBB03A570174B5E4363AED86EB02A220D281FB5457F0A549FC5051D49A6B2015259A2C3084F405E1769952260675586A584904059275A265234EF3ABF88C171A80898FC783358BBC9803C8789027D917C9EBACBC568CC18DE84C85454B94249586C0C6E2B8A16FA789C51212DD1728EE9B8C6C40528BF93826FA82368419623032AF27B5694305816811D3CA85805100E9C1A9621E5089E54CB47F5A8FEA0B49EF81C6B5187F48924C7947D6B61697A4A8A18452EF803336AD4BE503275BCACC03C181405F7B1DC9B47FB169EB37BBE27E29C763A4E52B9A42520388CF09B8EDBCDF41CCF6537190E6156C37CC1AAC63C0F90CE78D0B9B190C548D71B6F26CC8F585EA14004B5B30AAA100B2ADC1263828833B24E46163B41446F98C882092A39941867B80632E2097674A793935227DB0B8577E03A69C50A514C7473C892E3FBA7C4316BDABC952A70644176687D4191323BAD93D85A3CA250868C0747E6C44F6126C874AFBEC0BDD4503CB2C59A69816E7D4109941467579A1FFE6A4F50FA379051729DAB6E2F61432F15BE67D667C7CC1054742B2B953078A5CF88D9133087309D88C61DA240D99C59137329907B47865321ECD5564E987333B4CB607B0AFCA86769DC95B2F921357213FCB80C3B152918E9BAB2228C0A1B77897AC68CE55088165F87F397DA9790873B62C5383C0CCC370F0267CBE195651CCF336182C22AC3924B76C9E779B7A271D166B6D24B84242B7E73CC723F764039F6C851744034C3304DB0C091A5764FDC9D593556FF734B82A87CCBC38CA99564D988BBD2D1BF071BB160722D365104FB27610651A8ED817F2742A6B5A1273A61ACAF4460B0AB1456A9922351400A1C7D95D856D6E3370622C9C4164BC6B401435624A98B95CAEB274F34CE92038D785068CDD8CF44C38D84ACB2C466A2756C870EE78C26E738CC451002304EB8C90AB24B6463EB124D779F937A2E3692611D2E34D57B36CC4B2CD3B31FF485C6684D408B972E0D5CA7D2224AAE4E"+ "89C31D05611AAAB258F78BC2DE0A80D5914BF80C376A990D33CB97F4F2077CE12D2DAC559DE3400B0622754A2E814730BB7C2B401A076CEC9524654DDAC661B2F2123EA64D3BD727CA42C8D2725475BC0AD4B698E61001D031105897CA746249D24538CD63849E874EB9449FBAE979CA2A3F357B7D87F112FB16AB8BA20BDD315688231B21A4083277663113BC70A806E772917FA95743A01F138BC7BD5CB3B21CBF8EE301EB0BCE71A4BAAC3907B469CBEDE767A55AA194930B4A2B227E633CF33A3FC715454A32873B717FE6A21C01989133E42113D985F46807A26B0DE105CD9A897334D9816075C3149A1919CC2DBED56A23F3192CDB68F6E305852B8B864BACB5349CFD98419FFC9486C6B07F788CBC79B7680441B0DB1DBD92AEDFF49DB6F2AE52B95FA9B966C6CC08714A131770903859B823D2CF9731A36D8795B60585F8F1BF1A2C4E544407D5C1359810BF99A7374511A663EC1B192B83E27C4F4B6C3387692E4D756A235C14D151056E829FB79936A0C8A52EE128F0F369F8EB7390783788037CE32B021E57C45EB968CD5B9A04B02F89149778653F208C14C2E786D4544749572BF0D9A31B13BBAD60B6756B9379FC2E1660C58ED373C957309EA213213A5988CC59B1C37153811F7271BC1E729AB1629714D1955F706796556213B08E3F766867D1CC02648BE0574BB8818D0978578C562FFB6795DC065037129C5E4738B5960FE7E2A22570C36571A4A7A0A5187609F8D27714CA299309509C7C8AC2776E6814A262359B971B1A153CAB81858D96350F9F2CCC9884C8F5509E237655B4B2B22E75C6919A82EB56B095E3C52B55583F6855E186CB75DA3A20640762922062721D5C1AC51B87B10633BF43860814C89CE16BBF66667C2101BD27BA80969C78DC487678989EF5AB22D9B42679590E1855C94A9A90B4E021333A4532C7AE238358646094B598AA431486D0B7BBD82C5C4EBB6195680C3E5AA806B155AF7004318867FE55B1996353495034DF459624F4B04B51167A5148499846A36C605AD85B997674DF241D8EC59E56E22CCDB342F535292AA444B69C3E7A252128266A088B3482439BCA01AFB7CA5C6136A979B5159985A9484B36B679A5F7B9819EB63577891F7BB9CB98413CCC434ADC79A16D6AB3076569CE6291C59B5D64612A7FB0C15013200BC8BEBB03A570174B5E4363AED86EB02A220D281FB5457F0A549FC5051D49A6B2015259A2C3084F405E1769952260675586A584904059275A265234EF3ABF88C171A80898FC783358BBC9803C8789027D917C9EBACBC568CC18DE84C85454B94249586C0C6E2B8A16FA789C51212DD1728EE9B8C6C40528BF93826FA82368419623032AF27B5694305816811D3CA85805100E9C1A9621E5089E54CB47F5A8FEA0B49EF81C6B5187F48924C7947D6B61697A4A8A18452EF803336AD4BE503275BCACC03C181405F7B1DC9B47FB169EB37BBE27E29C763A4E52B9A42520388CF09B8EDBCDF41CCF6537190E6156C37CC1AAC63C0F90CE78D0B9B190C548D71B6F26CC8F585EA14004B5B30AAA100B2ADC1263828833B24E46163B41446F98C882092A39941867B80632E2097674A793935227DB0B8577E03A69C50A514C7473C892E3FBA7C4316BDABC952A70644176687D4191323BAD93D85A3CA250868C0747E6C44F6126C874AFBEC0BDD4503CB2C59A69816E7D4109941467579A1FFE6A4F50FA379051729DAB6E2F61432F15BE67D667C7CC1054742B2B953078A5CF88D9133087309D88C61DA240D99C59137329907B47865321ECD5564E987333B4CB607B0AFCA86769DC95B2F921357213FCB80C3B152918E9BAB2228C0A1B77897AC68CE55088165F87F397DA9790873B62C5383C0CCC370F0267CBE195651CCF336182C22AC3924B76C9E779B7A271D166B6D24B84242B7E73CC723F764039F6C851744034C3304DB0C091A5764FDC9D593556FF734B82A87CCBC38CA99564D988BBD2D1BF071BB160722D365104FB27610651A8ED817F2742A6B5A1273A61ACAF4460B0AB1456A9922351400A1C7D95D856D6E3370622C9C4164BC6B401435624A98B95CAEB274F34CE92038D785068CDD8CF44C38D84ACB2C466A2756C870EE78C26E738CC451002304EB8C90AB24B6463EB124D779F937A2E3692611D2E34D57B36CC4B2CD3B31FF485C6684D408B972E0D5CA7D2224AAE4E3A389831056ED8FD81476869245782689C84B3CE90FE6A9E78D0A380FD6A15731F8CB39E9E30BC458A0DC5408884B1187FB217018DF760FA57317703B844A0A9"+ , KeyGenVector+ "ML-KEM-768"+ 26+ "E582B7D75E6C80B05AE392A1FC9F7153B12390FD99930368CC67A768BAEBC8A0"+ "1CDACB8740C0B87C4A379575F187B367CBFA3B300BF591B109F79816E9CBE8F0"+ "28C793778741B80B02B4339F2AA4347255B099F17264E1B8CC0A2C7C2A1A79F7997B907FD0496C6E6C8AD7714F5F339D75F11F625591A869BE1175AE47F05FD4313468232BA6957D7807B824F445AC99A0D568AB1AD54DCA8249D1482E61275F52248C77F61A4248753188CD1794CD0A465EC0DC4B025985C461B74E76286E4C37E77405695CC9FD0654374B427A20343AEC0FF1A187768273BFC4905472A1DA387F14559D6CE87313F6A5B6138434539F9A13684055B177E543F8B40F432ABD7CC49989A50A9084C660913F45A8593B17499BC4CF936C2BC1851421CB986808A0EF30AFE97AAB5B8B8EB3F0B3506A95B91563A0E57DB7231044987EF141BDAB3537C316AD16F17805A81F29329879A94E96157E4B7447F7D59603B21BD896CC47B7CD4E232322EB9C5D2215696BCFFCA3A04EFCC4C5D9CC39AC9A6E8700D38C244B0169E7FA1FE81B4B10365E74E6A1F7F756D11ACDC84043F81006D62995376C22535958FEB53F78117EE0F61C4C862640D06DC57A2B8BE62A41A642AF3BC63F6BAC98BBBBFF70570F37B8F8D9572F2735657A6C98F96CAF57A849868720B2640B8BB2732237A1F984C18872D10289CE43C952C9257E06529AEB76AFD127B17596FD25C5216C9CABD9B18EFC50E87BBB04568BB7D5C4E9288C006483AF5912E19108573700BD10CD77224B80659EA75AA74270B33AC4008B738BFEE271E78658C8742FF13C96AD0781A03C7576CA26DD58B52980BA58C0505E446AFA140CDCEA0490DB1F9B18815D4314B2459CACC562441C91F4084E5426C88E632CF7482E79907911D06473260835D7B85E7856A829AEA0381707B939CE86882CC09C4448C6AE94A9C303107C5667EEFB8DF7763CC21189A3C590C40AA51F491503A7935EC08F4FC300CBE607ED8C9100C29FBF45584B13C8D780069337AEC76C36CEB70373E2AB6E7B934B466F53FB32EAF040055496B8540E23A2A277E534468608D5EC0F8D38CEA5BBB806C1BF4F164F6AC826FE733F95461E29DCC11200C0AADA1B8332023EAB329718CE25CC0A09555903F3578BBC863B1752CA94365DA556DF54C3B7E05CBB7115FBC1B6C57A172C31B9906560C8FB54F3C563A2256CC073243B8179B4A28D60E086CF51082EE429272996F0AABE03BA0EAFD3C8E7D954BD0933E2F60ED0C32CEDE7B820A28E48F3CA3C40913CCCAE2337ABFC59843F08C9863325D65A4E9E15C1F46172B118B2B5EB0F1D5158A00134F27B085488C3A0621FE4E5678698250FB74EE5152E3E35A66544A05D279EA99131FBC15165060B90F88EEB7B20892A4DE4CB1683495BD7DA037966B47CC040F1764C5DEB06B5499D4267391CEBBB47F734D8539E39528436A1858182854BF20B1F93279AFB706464C65CCC5AE099B37CC03556C26ABF4C3F8B9BA3A936707211A49A59B268F5284F7970C77612719450377417428C4BA47C9CA115CF95304C4759C5D8859B44985C06A6C924689237BA320D610960D61C53E85431789E67A40113F167FF93429C264F6CABC95448C903437D39A6577BE0CF0012852AA476351A9046A110A1A625A3D74C910B78BCE9CFCA735E4F91B8A4C57DBE489E849446098AACF73070AEE638FCC8896473D3C159D3AFB4B687B40DFBF371A9C2644B605187B71A14BC4C8678FE8247"+ "3808B98D9A093C7853B0B814D1CA5F392677D3D0A38F81C852F95B9A69B374A24588C0ADD5B510BE567C5A24688EC91ED0F28BB4C86978C09793795C36B94E5DD8498EB4353FF40EA87B17E921B5B4CBA08E5B7BE5A9C8BC69AC5AC3075DD947D04B8695097AC39790A0C8ABE11200C9F136ED7B0C10077E36111C1F139D9BD27142993F5883925C4918413CB8043962A28567397A1C7E003BAC30644055155CE9464E623FC5E2160CD1143FD2C90C03A08D0B333FBB40A308CACD81618674B406809790B2538D30431F064F65EBBA7C41B2A53146A3ABCA66754CAE591534045C1D640D6308472EBCA660A8A7D372470D3869115B8860B577311980B2C66B8352229ABC17B0C31F58E5A5EBDBCA6E3B157D9CA79EE890778A4CF28429C2A82C80DC9C524B0864E385E355BD4E65732D395E464070F61828E57C3E1BE2A50052C4FF1951F5D0C89720987D8941F085871113A91F518FC79B7189056990B2447BF54FC170C932E0A4E0F7A262542F56D1B49EEBC392106D93C77DE9186DFBB824E73B2F7FC796B312875B43414877AFB356214492C19C748DD381BC7A9237BF097CC606A03119A240A5536AC7844F6A79D0E2B821B68D96542450179BCCA231F2DA68C5EB6D118B99A1B66EAA566C78A9009008B0D66155BA4839F8E518C650177DEC170EB09E8BF6A89905320590C642B801CF5B5151C2AF3E7271CB9961C65A5BD17479429A31BE9081F2767D94816C16D1B04772012382B689AB2D3FB9BDC66547BFEB23052600BB369771494D9C914EC93A066ABC9611940B947310DA04197312425D736A1933B785C95DC430791E42CD691C79BA63BE06EC80765C9C07053EAC706697F21720C672B9803DF9935532197B0485BBE42B0DD16561AC0605E9DA73C189C9E29ABA3AECA19C21621B209326418543C44B88843EB3720A2CEC3DB52C59D6507EE5CB03C268B31FC4BF115695EB0B8AD3F11FEDEA1A5D0724A5D5284D0C10E9844D0D32BB5735448421BC5317650BEAC4829B234B787339812F2316ACE8C22C42D6346214934FC0B49ECCBDF8DA19AEE64DE9628C4F7AB3E512A399B0C227B677AC4A69891A6874F6641D2CC95460B751E17E434C924B2947D806856665696E3CC4DB9553B81606C31C3800B48756A073BF685EEA20899C176769E902DB971827D153B9C33516E959C6B469841946C0D921A371A5DE9740C6AB9CA272B5850BA8753CA023A460EBF7BD573C0745F40B90F105EE17C19B6832E019B80F8858BB515F7C709E68F29C1375B22567AFE7B528F4431A94B553DF825CDEB84B4EA9296EAB9AD66271EEC5AEF6F79509A20A182279FDF92F87FB6E7509968D22CA750B5056974841B9654E5716BDC33A2C6A116A4117FA757A1D22710668412B8878E134B70FE32158A2317FBC62F01371296AA42E33C903E0C439F19684B11E2F911F7FB79860C8800F9CA146EABE29DB07237AABB503A9BE2AA7263A0626C162A27537775792B2E7B0FD347929934AD8F521D4159059F611312AB903879490059BE8B38920E2CB4A256B8B35783E909346D13E9888BEB9350369F7C1A8501331110C651621A616B365A026D1CC47DF440C9E650DD0C0BFC8295439114528C793778741B80B02B4339F2AA4347255B099F17264E1B8CC0A2C7C2A1A79F7997B907FD0496C6E6C8AD7714F5F339D75F11F625591A869BE1175AE47F05FD4313468232BA6957D7807B824F445AC99A0D568AB1AD54DCA8249D1482E61275F52248C77F61A4248753188CD1794CD0A465EC0DC4B025985C461B74E76286E4C37E77405695CC9FD0654374B427A20343AEC0FF1A187768273BFC4905472A1DA387F14559D6CE87313F6A5B6138434539F9A13684055B177E543F8B40F432ABD7CC49989A50A9084C660913F45A8593B17499BC4CF936C2BC1851421CB986808A0EF30AFE97AAB5B8B8EB3F0B3506A95B91563A0E57DB7231044987EF141BDAB3537C316AD16F17805A81F29329879A94E96157E4B7447F7D59603B21BD896CC47B7CD4E232322EB9C5D2215696BCFFCA3A04EFCC4C5D9CC39AC9A6E8700D38C244B0169E7FA1FE81B4B10365E74E6A1F7F756D11ACDC84043F81006D62995376C22535958FEB53F78117EE0F61C4C862640D06DC57A2B8BE62A41A642AF3BC63F6BAC98BBBBFF70570F37B8F8D9572F2735657A6C98F96CAF57A849868720B2640B8BB2732237A1F984C18872D10289CE43C952C9257E06529AEB76AFD127B17596FD25C5216C9CABD9B18EFC50E87BBB04568BB7D5C4E9288C006483AF5912E19108573700BD10CD77224B80659EA75AA74270B33AC4008B738BFEE271E78658C8742FF13C96AD0781A03C7576CA26DD58B52980BA58C0505E446AFA140CDCEA0490DB1F9B18815D4314B2459CACC562441C91F4084E5426C88E632CF7482E79907911D06473260835D7B85E7856A829AEA0381707B939CE86882CC09C4448C6AE94A9C303107C5667EEFB8DF7763CC21189A3C590C40AA51F491503A7935EC08F4FC300CBE607ED8C9100C29FBF45584B13C8D780069337AEC76C36CEB70373E2AB6E7B934B466F53FB32EAF040055496B8540E23A2A277E534468608D5EC0F8D38CEA5BBB806C1BF4F164F6AC826FE733F95461E29DCC11200C0AADA1B8332023EAB329718CE25CC0A09555903F3578BBC863B1752CA94365DA556DF54C3B7E05CBB7115FBC1B6C57A172C31B9906560C8FB54F3C563A2256CC073243B8179B4A28D60E086CF51082EE429272996F0AABE03BA0EAFD3C8E7D954BD0933E2F60ED0C32CEDE7B820A28E48F3CA3C40913CCCAE2337ABFC59843F08C9863325D65A4E9E15C1F46172B118B2B5EB0F1D5158A00134F27B085488C3A0621FE4E5678698250FB74EE5152E3E35A66544A05D279EA99131FBC15165060B90F88EEB7B20892A4DE4CB1683495BD7DA037966B47CC040F1764C5DEB06B5499D4267391CEBBB47F734D8539E39528436A1858182854BF20B1F93279AFB706464C65CCC5AE099B37CC03556C26ABF4C3F8B9BA3A936707211A49A59B268F5284F7970C77612719450377417428C4BA47C9CA115CF95304C4759C5D8859B44985C06A6C924689237BA320D610960D61C53E85431789E67A40113F167FF93429C264F6CABC95448C903437D39A6577BE0CF0012852AA476351A9046A110A1A625A3D74C910B78BCE9CFCA735E4F91B8A4C57DBE489E849446098AACF73070AEE638FCC8896473D3C159D3AFB4B687B40DFBF371A9C2644B605187B71A14BC4C8678FE824781E66EF5A7A221619F6A64039CC369843E10DF5C859F6959CC3FD8E5272330FD1CDACB8740C0B87C4A379575F187B367CBFA3B300BF591B109F79816E9CBE8F0"+ , KeyGenVector+ "ML-KEM-1024"+ 51+ "F3A706FAF090C03DB506863AB0B20BD8A1627956318E88C67EB875E8E7266009"+ "35D2BC43DD1CC879F765BF2A0C5E297889DDE910E57E2BB0EAE417B90AB7A275"+ "8D0923CA8A2DA2B4146EC25321122B8A5AA8AFE0C03415273008A46EE83031E98AAAA125ABC75D3B30322560C197E75DD0E48A348099F7B2144D7B8A8660A4A97BCF19C0583BD9BB2123033CD7BB5A14B08B817831A673A28170F5F6443C0551913A327CBA18C3A053C4040250403B70AB9588832403AA0FC37665E04980FE1602E7D2715D9CBC00515DF432A4F5B32B3BC92AE3F31700166D498123E94576509B712B18491B1435EE7AB7AEB1AD30D72348C3CC083ABE24A8B12097BF32F792476288EECC3BF630ADCDAC6ACA7950D9839501A448500742BAE37F109203A809B2B960A307E25347A32C3EAB79288173A878789B296E9E8C1C28C5BB3AC472601C9765F7B77225A810C7B85370BEF4A5B079D2015ADA54236B8F33840675F9B2EB427A1B5974CD5C61B24010886C5A7BDA5BBED974AF7217F3338AD719CB308A8BCB1B6D6ED2A1643736C29095E8A8452A3A36C7BB5AE58CBFDC61529466A90F454ED6895B0861083DD1371999B2F559A3A487CF59A074FB49215EA6A6BE656F9AF17B121A2447CB7985590E9738842B899BA57AC311810AE2D9794F37483DD6BCCB64AF6D56588AE94665961C025C3AA2861974C236BCA4BD8FF5509F7AB774593E7C5549E57C2F18D15C0515094AD9A0DFAA0601E524F8231156B627BB25A0DAE04DACD0A66C041CEF400583FC13BAE640291A39A5C5CA8BCA1AD5C683CDD8290891A76940817DD8C9F52678780548E37A05806600801426DBB950C3B2BA34E24CC77864DD91B39F1408C716A69DF63342854E50A245FB50977B9410DED2C93F86B1F9D5A78B87BF81E51CA620A7E8566B19AB700964A40E3266415228D432156E5CBDF52364A90483A55C39B3FB16FC7465A3F8AC801B70B9FB28B583444BA5C1A73722D417A9D6D9B7DEB08BC6B330FF27CF61AB8831E27758C64AF3B12150CB7B33ABC29858106D63686D8762459ABF9413850AE53ED6313F76F83D0FB8AB34374E7DF693E4A1B3E5A8AD0CE820AFE1CF401ACDE650A8101B0946022D52178E19613C42B88B07CC04EAA81DFB28AC9DC076236B67219A30F8F945DD57BD2F335C52D59372308D38993467DB53DA3382B74867B616481BD0091A2232C1116DC88A589DB9107224A681008C67C589186A6929549BEEF92253DB02B0C8AA9F9C875A670266C72BCBDB4F5625043703C1A0457395832E4C335180462ED2220C59E7361903C107D85457F6CD82EB820D0855D97675C2E0151CDB73C2885DDB7849D74541580124E890116A65BC068093B57914E20C937C60A3EB25576F1A976A9583839B672144CD4A45C3477A45C29B4E0BC2BDBD206585C9B7A7741C8B6B5793A92797A15AE7A5B73A74B2971463634BA52AA792AF05530730B6D0A89A346156B733677932BD36593A7496130CC458DCC5CA987C21960604EC8A8C5396056680CBF3F1AAC4F401AA5029FB2150434BB4706C31A2D54E4297939FA7C9C6F85700613CEB65C7F03AC56EB86E2D27CCC6DCB7B9394DCDB942FF222D86958A996C0CB6A8A44F97A70441C95FA71250116EEC20863C0B5A643458788AB001F8869D909922F51EE547A1E889255B3A0599C65842E5AB8D73872F053BC62392EA53896D328102D460BF1609583C22C3B43780EC6DAD0319EB4A5A65B4756C3CB40EAA935183BF8BD46ABE76BA46E199103A5313C3235F49C915E097BCA804DE680781D8365731BEAC6789A9203FB8787C4C070E00A13A6722A66A28236DB179825653D33CCF898B72C6B8450D97AFD3276BB13340519CBEDA708D12A858F54C49F4547195B7788A9150B2649E36AA394121926D568A488B16D3557B2A32AF57D11FC3373F80A28C0723273D362502E7C428AB44D3CBABF9EA585FD1BD0C9846556A1E196B78CF951592984A0A8487A78C2317D7ACA4118E1049750A0788F0D66AD9E48E34731130ABA0B427360A856D96D80B3F028FDD3ABA9035C10106BA1C0934BED36C6D7C7434249654EA89FC22137F4AB903653B75FB25B6F01635E6CC7D39CF1508690562826B49B6FFC59E0DD35022E541F8BA0D304AA5B4E20606907C424395666C54ABC2B8FB009847C86317685000C231215C8C15945860F6A85DDB98A8C3A527F2749D3C027E694E8F0B0F0FA454913AADB635AADD452F7128BF7752569669A8B93290EB92E78F6ADFF23E89F57F3890753B51F12F3F3A8A654E677847"+ "6CBBB491D7B3A3439F53C9027BD727A0E1C78004B4CDB66760E50F1BF97FFF0139F82862F1F3178A077B6AF77735CB802E91316BC207B9F1485FF7A868891E003BB9C6558BE6052C0826AF1CB5836B94C7D5D714EEFC7CFD983E9884B71348499DCA05DEF33B49D1C14BBBB95AB195BD72219B18BD0C704275CC94AB06653BBB9E0EEC74181903A0081AAE068B81EBAE8925335B8134B1480A60F758282C2F011B909641266774513A19090BC672EB3A0708CB577F134151C155C1E560D9B335A35A4D75F223502752CD250C9A7778EFC76A52709813A24679B987A6172F1714641EA6902D364F96357E92B44275EBCCE7F6207BB0A215E98B3AA92DAA46A39AD9132F3874C6239391E716BA600E42BC7084425051F865756A89DAB46A76104E183A12031B99BE86257B583C2E88104FEC3BFCAA50F238BEEC20ABF31219685885F2C3CE2F80C09B874E73A8712D35B8EE819D84B1C59A37947C0B9E25F066EE030ED28172FD5914B562408BA9B9DE9BAB2B95A4F30C22ECE458E133371A693D9458058265A75AD6CA5707617B38A1B6F396FB199CC1C77100741DE37500F1BBB10B9204EEB7C3F5B91A1CC91B6D4072AF410957358ACB0625488B9F6A9915758B7CC9B4AFE7D5AD6A95231A7A711B3A8996E23AFB3695B45B6D919C771D7685D90A7550A9B5B1E7538A365451A5A815009E49326C6541CAF0BA685056BF82EA4AB504161B555568583802800554A9480602437B0B9677FAC638A2508154A658E0623C62666E078D21A5A545A818D66C9547B9749458B735A4AD51812950E8A625A1B2A753ABBBA16415006300AD32D34C8B0B3B2AF4D1CFBDE08B4F06C70DE4BA28FB3763745CCE133DD2E488F89A51E7A586A86AC928867B0541438AF5645DE3211AA94ACA4A1095896EBC363C374A85565C1EBA030B5EA21B9B8A140E131ED07580D5B8590446988DA65939A823A84661B26CA12461AFE373519F43C13987A96A9069CD6B9177B205B25B5197DBCDE66C6457892C1DF96FE50AB7EF3B3DAB0A87436B3451476C44AA22AEF918AE5A1A8AE46920263032E3BC5DEBCFEE43B8AA290CEDF06966939EE6CC7C6AF83CB9D90CC889AB8B5A3D4C224E73916AE057BF5455528CACBED31B554043C6B0EA196854461D2644461470EFC55080E326FED49D62B5A4BB06B14F505EB4486C9A1B033B927836DA656C2475E5B8B823986DCA99B298512D94B77FE0588DBB96BA53A67369B82C2BEAB2CCC0103E549317A99DD97B598FA0A506184EB55C96B17302533C7EC3334E3CEC2BBB22BE0F89442840884FAB4A766A41BC8061D4ABCAC5190C729309AC6151BD434B818BC1E7D0CD435C9CA6DC7182228FF93867134A5CC65C6167C4A8BB830E7FEB863C15BF87B259EB0518FF5248792A625F37C589589799746D36197DC63AABD134A00EF1B49C79A2DF67B28E6B6E989CC9D0577EA5AA1B2FE64B322C7A8B241D503523D875CCDC479D28E3C01BD8383DCB2B684281E6949171F518EA97009237B5CD42AC8FC8A571716EB784B336529DB67728499988B73B80EF5C6D33F3B42ADA7A0A36A7C6A8A0DFB804B3D7BB05803957B913827A16F52ACDD7650EE2971A85DC21F6A94A5B31988E4438A9C1733AB8C48C32653FE800BF81BA66658CD2001ACF3A83E3A0426AB6C3B2600A53621277A43D723AB96758B120F120248C15EDF4294FC38F96EC490BD52094655841D9355038A379D8B666875171F39218E21D37023F5108B6D15389F07C1241E15071C43CD7D160EB5341B719C2055424EF6CBF66815F84943C9481CD1DE685A74C5B9DB37A0713952FD4A67BEC8CA7496F6C9C5FB84A8EAFE68D54654C6BAC9C3AB712C2BC32C0CAB072C18DE3CABBD9224DFAA305DFDB3F85019D7B988577B40F27514822DB42D87489502C53CCE6A095B486D9143D6D160B1C7313BF288913657698117EB145C0C9C55D66C15BF1A2B268F6A5FA259139551ABFE3468F1CC49C01B4232543DEF0BF9AE503B998536310751A8BC302B4B4440B835233797C967E7F9308155B27BA63C65E08BDD1946828196B30C042CDA0B56E4008215AC51F373604F5930BB66297C32BE41A162CA21471CC86657B5CB5307D89091DA9A5137F1C1433AB8AC1D59DE421C7C17A883EAB99BA4714F841998D0923CA8A2DA2B4146EC25321122B8A5AA8AFE0C03415273008A46EE83031E98AAAA125ABC75D3B30322560C197E75DD0E48A348099F7B2144D7B8A8660A4A97BCF19C0583BD9BB2123033CD7BB5A14B08B817831A673A28170F5F6443C0551913A327CBA18C3A053C4040250403B70AB9588832403AA0FC37665E04980FE1602E7D2715D9CBC00515DF432A4F5B32B3BC92AE3F31700166D498123E94576509B712B18491B1435EE7AB7AEB1AD30D72348C3CC083ABE24A8B12097BF32F792476288EECC3BF630ADCDAC6ACA7950D9839501A448500742BAE37F109203A809B2B960A307E25347A32C3EAB79288173A878789B296E9E8C1C28C5BB3AC472601C9765F7B77225A810C7B85370BEF4A5B079D2015ADA54236B8F33840675F9B2EB427A1B5974CD5C61B24010886C5A7BDA5BBED974AF7217F3338AD719CB308A8BCB1B6D6ED2A1643736C29095E8A8452A3A36C7BB5AE58CBFDC61529466A90F454ED6895B0861083DD1371999B2F559A3A487CF59A074FB49215EA6A6BE656F9AF17B121A2447CB7985590E9738842B899BA57AC311810AE2D9794F37483DD6BCCB64AF6D56588AE94665961C025C3AA2861974C236BCA4BD8FF5509F7AB774593E7C5549E57C2F18D15C0515094AD9A0DFAA0601E524F8231156B627BB25A0DAE04DACD0A66C041CEF400583FC13BAE640291A39A5C5CA8BCA1AD5C683CDD8290891A76940817DD8C9F52678780548E37A05806600801426DBB950C3B2BA34E24CC77864DD91B39F1408C716A69DF63342854E50A245FB50977B9410DED2C93F86B1F9D5A78B87BF81E51CA620A7E8566B19AB700964A40E3266415228D432156E5CBDF52364A90483A55C39B3FB16FC7465A3F8AC801B70B9FB28B583444BA5C1A73722D417A9D6D9B7DEB08BC6B330FF27CF61AB8831E27758C64AF3B12150CB7B33ABC29858106D63686D8762459ABF9413850AE53ED6313F76F83D0FB8AB34374E7DF693E4A1B3E5A8AD0CE820AFE1CF401ACDE650A8101B0946022D52178E19613C42B88B07CC04EAA81DFB28AC9DC076236B67219A30F8F945DD57BD2F335C52D59372308D38993467DB53DA3382B74867B616481BD0091A2232C1116DC88A589DB9107224A681008C67C589186A6929549BEEF92253DB02B0C8AA9F9C875A670266C72BCBDB4F5625043703C1A0457395832E4C335180462ED2220C59E7361903C107D85457F6CD82EB820D0855D97675C2E0151CDB73C2885DDB7849D74541580124E890116A65BC068093B57914E20C937C60A3EB25576F1A976A9583839B672144CD4A45C3477A45C29B4E0BC2BDBD206585C9B7A7741C8B6B5793A92797A15AE7A5B73A74B2971463634BA52AA792AF05530730B6D0A89A346156B733677932BD36593A7496130CC458DCC5CA987C21960604EC8A8C5396056680CBF3F1AAC4F401AA5029FB2150434BB4706C31A2D54E4297939FA7C9C6F85700613CEB65C7F03AC56EB86E2D27CCC6DCB7B9394DCDB942FF222D86958A996C0CB6A8A44F97A70441C95FA71250116EEC20863C0B5A643458788AB001F8869D909922F51EE547A1E889255B3A0599C65842E5AB8D73872F053BC62392EA53896D328102D460BF1609583C22C3B43780EC6DAD0319EB4A5A65B4756C3CB40EAA935183BF8BD46ABE76BA46E199103A5313C3235F49C915E097BCA804DE680781D8365731BEAC6789A9203FB8787C4C070E00A13A6722A66A28236DB179825653D33CCF898B72C6B8450D97AFD3276BB13340519CBEDA708D12A858F54C49F4547195B7788A9150B2649E36AA394121926D568A488B16D3557B2A32AF57D11FC3373F80A28C0723273D362502E7C428AB44D3CBABF9EA585FD1BD0C9846556A1E196B78CF951592984A0A8487A78C2317D7ACA4118E1049750A0788F0D66AD9E48E34731130ABA0B427360A856D96D80B3F028FDD3ABA9035C10106BA1C0934BED36C6D7C7434249654EA89FC22137F4AB903653B75FB25B6F01635E6CC7D39CF1508690562826B49B6FFC59E0DD35022E541F8BA0D304AA5B4E20606907C424395666C54ABC2B8FB009847C86317685000C231215C8C15945860F6A85DDB98A8C3A527F2749D3C027E694E8F0B0F0FA454913AADB635AADD452F7128BF7752569669A8B93290EB92E78F6ADFF23E89F57F3890753B51F12F3F3A8A654E6778479370FE5B05DDC92C939F62CBDE4C0FEA36F45CD20C5748CF3AC891A4C260449635D2BC43DD1CC879F765BF2A0C5E297889DDE910E57E2BB0EAE417B90AB7A275"+ ]++encapVectors :: [EncapVector]+encapVectors =+ [ EncapVector+ "ML-KEM-512"+ 1+ "17E5129B2029F3281987D6624725B64C51CF8DCA3562372BACB7AE15FA9F2FF6AB47659B7D305B61F55F571315FF69AA49E1388100319A650E86C59BA3024C3DEC83D4AAAB661C452F6CB6A8D2638C133C599045329CCC8F677D24683DF1146EE1C7318C3763A47ACFF81A03927B9BAC5A49DC285C4EE204C4D72DBB1C97FE53C622E621338669FCFAC1E30A36F0A9769BBB2BA408787EE629CEC383F29AAAABD46E22133F339C08E29B82C4FAAF0F676E1C2B04377975DC3A3B246488EDD1636C58ABCD65B8198B6FA8A8475EF541DAF34B5ADB9DB0589D8958B62A88930EEB1C0C352B1E57BC7882C89EFA318457813F2B55B96881E7F75C3DC97357681E54553AE099095AC38B34C00199952602C481C0F1CF74B550CE7C4F6C5876AFF076DD921942DB377E1749F0306F77443A2F058D785854C7E32B67F49BCB99A5E18AAB607C649B892B4DA7CC7AF31557149F02B19460FB49E5051A7251ADE0083CDC1B0B0CA9633220C2B3C532FA2CBC0DC6CFFCD4455E0005D06CAFC727C778375CB67AC461B3627A653C843BAEAEF866C7F67746262FF6D76536255C89045A172CBFAA123D6ECA3E56FC922F437A14F78D1ADB54B3E54E8DD530450B8500E15A54C97471CB26AB437394480282BA6B7A786C28857BAF387725BB4342B4C3ABE4CBF90612B04C8BDC1599C2D4C66B80B1A1941440040A42AABE03E1CCB0B4190780B6603BB69EC199B063826BEAA5447C88427C6BD7A06E4164A490B955152979B8F5BC9417598AF01BCC371577189E1775165B9610B9F75AEEA277018B1C2C04391115B6F353AFEECBAA18AA4001DB94FCDC5F46EA520E9836BC5A4576B58C59FC3F75634946D8CC1B1081CE93ABA8566304169332FA316D4BA675A086AD87937963511E4A1845EA38689B564E11295672098CF01D5D4BAB0338121128A682B048E6AA31B83C65FF861E123010433A3C7607AE5E90C438067F637B44ADF8798CD0980EC83BDAEB4D427A8F8D88C519E52542F07734A645A2A5BD4D6521A4B64A96A10F35002780169B35E35F01CA74FE6207B8B475EBC079647CC10AEC3C29683E071D87B82ABBDD6D369E326E475325ED5AE7ED232B37F49388C06A740D421204"+ "DCACFE4DE1C115DA106ACD1EEFEAFDC7F0F4E5707453EE2D6B0D69D34CC0EF4A"+ "1C3204A5A2C031077459E24A179FC80F8833E19F36E7AC0D3071BFBD2D48FCF1352B96EFD0FA3195B44A27EC575B2794909E4089421E56409AD00CF472680F438E0A6D39E88FE6B938EF722C7B7F75F714264C8F22C528A63985C75D2412278B137ACD29003CAD1711A2637C630164507B7D3C0ACD1DD3BA6E689411DF6D3EEE410EC8C93EF27BC82019C3943B85E645519BEC1105D4738388C7A5452A67880DE88D65C1626A55A4565B5C26B20BBFC33F2DCECF938149D8B58B19DFB5F451C3A9FB5AB3DC486C435F5397B6E32416A9306D9869B91231ADFCC9A4AD3D956EC49832C3EF2A6ED50638F6633D5A8FA7BB7B04EF45FF9B57D6EA7B771D57C3E5B9BF96E03ABD601BB46E5CE3E104233E7BE642B082A610FCEBEA684C516AA39A051DDB87025CA849816E177C17C10115939B98B5D9F95323328CB5250C4E38B7E932481D20BCBC66B0BECB3DC1AA196F5FE207B28F36344A3C00F2FBE179878D6C7981467FC2DF70D079088F09B4D2CB56D5DEDE593A77B31930E8F6138DFC7F9A7295FB372C1A33713610B2BE51F311E7CE6050EC276E5C3A50790EE4CB81511052AB659DD54F4BAF211EACDC2A4E987C7E2EB8B384F06813C2542BF765C3E6B142968BF3C66414D01A205964E743777040A6A9879C84681B6F6B3F2F2B0B0455952A99A21CD609C1FD7BE71BA6A1C9589445DC69CF1D4937763AC640DF853FD3E6DC9143467E473FF4B89975F24CCA51506830C279AD8BF806611E836F405E8C1FC36208E99ACB764316BD7C42D2E248B1DC3D5C551AB77FC168296853129C31F5722707791FD5F9AAABC4FD4C82A99970A93F274FD0D6252EDCAEDC076BF5984E6E28056BB831FF289CB0F6EE87BC4816712DA0F322A51765F5F230D0DE72CBB4714C9CFE56C1BD727DD7D3F7C1B5B83029C81A22B8FD430B137003B63F29B710A1A2C549F9AB0B5AC8A123935A1A3BF627690252AEF65F20681E2E9A664321CDF5FDE1943C196BA2F04E42E35F62CE77198F01A747A14D57D7ED46259189A6AA3363FC178A8CC5499F033F9B8D90D327AAFAB7DE16831E8325D337AB3DCDF1"+ "2D74374B55D29AA585E144A29BA4F0A96537A73B4F176C5527075F66E38E8858"+ , EncapVector+ "ML-KEM-768"+ 26+ "9C6839354086919BC7B8A547FF18B523075FD1706856400952489E4CD13ED8B17A817A3076E0947B860EC9558BD9D3C94DF5418A7A7301763999C09996715CE32AACFEF767A44969925949CC9B8120CB78953BCF6FB8282427CB3F4CBBC50AB23274493A8BA7E5C09088160FDA731B55980867912BAC53125B27073BB502AFAC5E74973E0F3601DC50CE08CC7F3FA44CE331A43F9673BE745295E693077264A8E238F0A912C0A59C44D561E2A8BD831541FE869B3110B051100DB3517F87510ECAD8651CE4998BB648002BA41A5351FA2A4ED94A58CB707214A59F8876176CF82FCEB99364971C7C0896E8B62C0544463FF101983C2045981CCD9173EB33BA18483ED915BD11EA950DB3A8787C105263242E3A64C748B8B1A5947880C5BF3896E08B4B9916A39FD70C2FF8BD041A2B18E251D16A276E38100A613EA96A86CB3652DCC83F2D31CD9F8B79E42541A66752D191436CF5B5D8350EE7835CDB122855392C1DA5C33FDA12AD1028F2806861A2AB9C63785FD11E220346322C250936AB819254734A8DC9241971AC90CDA6696ADB90D1E327DC977841ABA38324C8FB2092784942003A0808E5546D034592F03BACC19E3138C9070CCEA80A933D536C49E4771091AF8A70207D474574C64B9F1C8E47224174061117E98C18897A7B1B3C4478C6558C321A89A84B6C508947CB61C89204FBCA43BCAA1D688F768B02E50BABEBF0C5FB8962B2355973C4193FA92E0CE4B9F8473E154828C986BF63E1CD985C6F0462C9B4374594D226C5E99B8A844978DB8DD2839E27138AACF75E917635CEC4B407A54F184C22F62981B9F4552FB334FF2B7FF68B470FD61B993C32581CC0F39881A5A6A769A60297F4B0579ACBF52A8ABB143885886245CA2636E20498959E4F2857AB96706E6409C4164BA88C7193E568E4446120215A6D8499C6B4BC559628A9F481C7DB51E2E10DF1F4339A647B4CCC0405B79F70A90CA9A8AD07789222BA6E66C3AFF42B61BAA637AF15126C88B2E210A7CAD9CB08CBB5123A7F9BB4AD63DAAEA6B43F9AABA561212823CB898A676C967B2D64F4B6333439F8993BD7B19687248926E077A9700431A59157B9668EE022B6D31343F15C08F14488877DFB30A552C06F96A16F354B9A3F19090B32A043B43256BA5B1088B56A4125588C8746D4099EA21C4F33AD8C5C7C32286DEE2CA76FC2651D57AEE9D847824AC609B5BFD5C87316C2AFE0E153D6326AF89987ECB70CF0A0A77FA000E0E0ACC2685E2317429C3A7513F42873CB98F669703AE9146919226DB58D6C2B93B6F50C7CDAB677B505945A0F2523886C1B82A6A7CC7EBA9251096D88360D7EF6473ACBCD359A0D20414B95C209B2413B1483A3A061386377B28E3896BCD8C5E2692681CC803E1694EFB6CE353082FFCA4079EA9BB7B2025BCB68990466ED2803AB11516D80325DC4CDE3DA9E020653E084C45F837C88711DF3AA6075741C1283121C8B1C691013969A62B2563E556206CF0810848691C7C408D018617A7300AD943E189A9B44512EF13B669831392F1329E2AC7C623AA044223079139B7F1C213E93682ED86165468377D40884310163A5539F27745E667ADF506896BA933B002E4D50B2497ECFF09D0BBCA4F7E6F9DB9E10C643D23701BD6385E163CF71C1E919A6E20A"+ "4E77596168711E913965D8175AC3BD76AAB08B7F9385A02AE883CF6C6E17DD81"+ "0385E8044D17E2B96B3F50ED28C2502216322AC33F69D2CE34F0A11E9B3DE339AEBE98283A6010A34BDC98B0E5BBE142A38575DA305066331D8D11D161F512B4B56B60FD049DFEE3771473253A0310CD842C71CB7734474BFBDBBADD2B9B87C37804D8558B00D77B1F79F0FD2C94552A8D3B19FDD6A5511C4EBEEA40AA839A489AEC636B510BF51BF3782F24F53F752EE2DB9FB4964E8379D0FA70603CB8C2C7293BC35A56B818257B65FBE6B2C98178B61D7CEB58C7E1A9100650A945C739E3603C188BA3C179823950C6D874D7FC8F8DDB7A9EE941F4A78A0F8A1EAE080D4D77893AAEFB0DB452659B0275D171736390A1B23661C69058869A8C4D0BB46D1655858162012A057C7C067FD035D47D5D627E3CD27CD3CA021A34F8DB6695C7E581ECE67D898482D509E4DF61DCE5BBDFAAC89465A9209C547B4263E9155940F8FCB1DA3A4A7D25DF2969F31644741A8C6FE8065D5F176214A4DB8FE76BF41869B0F22E1F5CF5C34A5067B1D9F39F3AB4B0B9F6330D7CF6EC7477B53D9223FA70E4C680D49B32883D92B50C53CF9E61915437CF57ECA49D6BBC88C64CCB222D4F20227CCCD4F7E7C21F95CE74D6AB5DECE17F600AD1715248648E4D1AF6FE41A02023EB5CFF871EC654756B7D156E601E82827E17DD0FE4F5357136120A810506A40F28A0B860C3CE75BF7796653327939B507043AB8BE0175A9B312319818063AC40302EB42452DA0B1CB9E3C2E43C49D740A03AC1B9057FB44EFF830D548D12585F50DBAF7404C2DFA7F4403DB8B7CEE65DF528F575C7F552D0632553FC6C90A3B09C1114492392C165E9B1B0A10E9556C72CDE2BB23DF6DA784521249EEC542012EC33599B97B95FF5234158A6D553B3BBAE9704E630D64A66D658F74F27C8C30237E87686FB04551BB128D917305173E2788883352A0216FD8D0035D748C9B42A600073ADBA73F851BB5ECA287AD325A3A3561769868C56E6D84FB661ED5A41649342EA342F82BE528FD9BB4F492452C0A9426ED8677113EC6F1CAC51233F0520425C3B27ACFAA0D25A07BC2CA01115C90432ED14A1F4DDA693C6E704A16BECE8972D8218B62CEC4F255058988B224D86F2BDE4B221A3C672CB744CA8F9158A4E5C97D0C56AF5DFC1C5F0191F64CB6D7AB86B14432ECEB5A3ED77A8D3EA651A07C9A0CF95E631F41C2AB29191F66DD9A73271F47A455999C5FC8391550BBB346CA1BCAD598A32A50CB52281156874F36DE9D8EA5860D99703EA3F4907C640B47A75A285B4347845CD6C94407995F569FD8A4B12461D8236E4015AE000E2ACAF2BD9F8D28377EC0D9E4CAE714C481C5ABAE37448B3F84A291B84B4213C8FE1F6E0FDFA1A6D1A52C974F795C5947F667A8DBE162983EC0C71C0F0237E6ED8BCB7AC9E4C73B5A1ED85FD5E0FA2810C842D484D6F9DFCC8472F05683D9D1144EF6C36294BD06A019EFEC202FC46288EBA2ADE82AA3363FAF32B9B027C75A7D615845CF2EC538141F1746009737600F697CC8D90C80C1E19F1AB2C08646C0E3958"+ "79D74F6C6C2D916BEC47BD828FD9B67295A37F54927FAB1263C0D122F1C6F1ED"+ , EncapVector+ "ML-KEM-1024"+ 51+ "2191ABB6D6BEEE29C5780758A970349879B61A028DEEA5404731292346C81EEB1D17766AFBCAA68C867D91132F34A494E28CAA767241B50902F4825771865FC8D633736248963A253DD52C1A07C7177CB6DF74C43C3C74D7133DA9DC915FE14B5BC30B86153E7B86BE04189F4CE8CEB6E5CB69808D35640AC50335B1633162E28260300646153BE7B26A84C1C565956F8577C95CFA80B68978CD7A79E49C7E808B21CDB460ED8388C4859670333B69980B2F9A0F78D7A5B9D8554893A0B5C52447B15318828989816F43406ABFA4950607A5A57CBA62F5C4340AC3D294191687C7D2B60EF2D83F6BEC15B1618129B78577447BD1E36A87142B9C552C8CD2304F35B81B3BC5196C3840C0346D513EC8862BA14A172C051402F7398707A6133C8E88E5B57F9B9EEC18AED1404EA1C47BE5169E5D558E262A231A75925D25C499650C8AC031BE580CFB185A0E0362CE2194C53713A630CB59600C1942908FF5142B02364B74B3F5474301AB2DA4188DCA640961142F28584EBE810683C046B59CC2CC07025DA63A5E8C5E99E24FBAE183329886279C2D58D0B8CAC1C7B3033F05E765C75414748687FF03C00B36709D41C5968528F178CD03B813B69054D3E20FB6A04A713B1B169AB5F0369810FC5CD967A866B70A2D87514605794930BE9B7B5D49D08F5CB08D84FC5BEECB994DD2A5FC10AEEB4CCE1ED251FD4815A9D06EEA6A77F18C56ACABB1F7C478DD52218B7BBF9770A51C3AC96402548A1BB22DD0C29F422330057179EC2D91E1AEE27BBCB1EB4671324265E28676090896E332535BA2E9169C54113F19A45229C2448078AE08F7213F347244527EF726855CA054AB7A1ED1A40A285A60D04C00CCFA8FD318CF182705DDB888DF652669B3224CF54ED6D00DB5121331B40C7E35B1945925E5AC1503A999899317A2370020D81873330327DB7CC1D430E32C6570FCCFFEEA90A871A4B04CA9A8065AB20A6459F1806221A5482802F14A68893B111607C4E085A5A6410C027C1BF5FA33ADECCEA576211F272CC942ACACC879DE398477780A26E7545ED693EDD66AD24772B0C0C4B033A73ED94E1D02C384D1A288A78F5D8838B5AC0B37855DC59B987EC9B0F5C6868B0319F3372024FC15B247144F925D3EDC453FB8551EC1CA734A37A3972B8CC9B84A217A4D0BC0D08C815359ABA59A07EED998133587E062CF735A8521A18AC2059E3F142F12CCC5AE024CA33B8E4A285BE5A3CED460C2C3675B956294EF2C61C96A88441665D56B2029688EFE4540DC35ABF5957BF9477F58277851719547510F22D01A13D0047C5B88372A654C732AFF395DBF825829A06E6AE6923296218AB9AD33D0B07A0B6A3F3B3DD8428A1F82B72A921853D94C5D400494494F0D168148B5392F386B16C6056D6B8D767B7F4D918A9384C855C8B4DE8450EB6CCC2DF18528CBA0A3552D5704683AA3AAF52751FD5A17801A76A58A75038533C1576E8668B5D290CEB20CB1FF91317D9B418311C11E617D10B4B0E9055AEF5CC6EBF8B91F0244ED9854FDE463E55112A238038750CD016735613B9C461996B90A3B4D4B0D8B8825EC087FB138AB868A04CB3C9249B5B118F81ED06B99E6B521F206A32A4013E876C6C80B36AF39AC197A21B78B5F2713AA45C74335F32676C2AC2902710EC42AF5984687A672E0466831C7B436300B30351795B22B3A939347A7B744831BAFE78C3BF09A8224A958568F75D71F822941DFD64D8969ADA5B73C847272B5914491BBCE5F0C09EF501BABD95190F93201AB413E833F7924B696D5C2B62272245932A6D9C35E1670AA3BC57CD41B8D61A52AAA6D80BB774F922ECB41CD58469C4DCAC698F6730890237602064B148650C17E6EDCB3713382FE755E46F3ABE0C6C45184AE5C5C861E35A0E2BAB2B4E8A4775584351170152512C7E99E96F285B916605B53B91D6C501F51A93129483046CB2DFA16D4A540ED2B6AB18BBDFBA2B201BAA20297AB1FE7133757C07DC041971AA4219157D77628F31AAF520650A2250ADAD1524E58268D21CC62E2469C8425E86359DBCAA066ACB466E06947A1C144E53EE0DC930432C806C493B2BB25757B1E2C775A27A896EC25B58E8A78D4DAC91815A1F3D90635EA25AE13CEFA77877DC0C2F0678ADE307243B438C54AA69DDA539B8A41C1B3C98C77FCEB2C0BC2025A272D927943EA338CDF32A0DEC8B187DF5C"+ "2F1E2CA7BD72AF847CAC38CDEFC4D345909D7517543EDF32E2FC491BA05EB5C3"+ "D892E0948544D020ECE56F00C6495B9BD1C469AC0F134002C864D10CF61C6C7887890276A7546310EA077741D83428F22B60E2A40ED8BBF5D9227893BB0B7417DF4380323426EBC9744EF35A1BB6DAB1181FF0B677E8C9B6574360994F96EB87C3524E15E468283169D90D8A994BED0DA9CD778CA239CA6C225390221FD408A3EAE541A032D714F3D078DD1EC722CA83FFFC92A416F46FC1E8710C3F8E9CD452C016F483985A5C1D951FE2B03C4AC2C9A0BA71F6FFA4B29EDE76DF69CAD594D10618B94E783FDDAF11CC801ED5036FEB4A70D071079944B7D9F1FFBB98F123DC34A52A40DE57E079A1F8B180DA9E6ECAA47A111F3C054DD9563D7E74B95AC65AB453AFE0BF0DCEC5578FE6FBC2D9EB91932799EA64B2DFA95B2C9C6E931FFB0A5A499DDC42C6D6B563D5A6C1F34DBCB36F57E6C10C69FEADC70A3D01F5F3715F6BEA0D3E2E7AC2B04E01CA80A2E187BB7D9EFA5EE257B9C3CAEA3947D785CA56A76D0B6692A69418935DFC36FD9B26758F5E1043D5C2A6D5CC1556AED592BBCD652F4024BF324910F8CAA2415171F3EC8B1CC494F2B519CBD2B13D317672DE60F0A3D400E6079F85C3CDAD0DA5A17F2EFACCEC03D3C2E85B063AC3CB29F2FC4850705B8D472E35DC12D06A7D021DB302CA883DC7D4A57B7A9E1D023960D4EA5E39C9D5BF328B8E4AC0AFC3A6C990F598F373059AD28EDBEB90412472C637877EA82E7105CAF467A15CE961695F2FF04C750E277E78C1CE9E490A5C45168936BCF79261B95FB20AFCCDCE790E0AEDDE204DA4F6734700FC1CE05239F4EAD0A496FD8F733049EF9B6ECC54D5C635B56D6FCAC7ABA1BA32D707D888B820CE3E794C2562798210D726C6C5DB3D22A214F3F2A1F2477FCB77377AF41D0F3F0AC635E7BF0674CEBB672EA93B9504EFA8C5AAFB7458E41A1D85C28841877FF2F444005CD1C1E4227573E8151184786489C7CC39549F8A29B9C4D68BE38F24081FE239C5D0E14AD5E361720836A72A99CE88E47BF3E6A1A9EB57CDC3724CB4789FB88C135CA9CCD94CCC9BD73EE8A796BFEA36E894ED27B0791E8AD23FCDA5AE7C167D5774AA00468D97BD0FD2B9BD78D01E1E39A66994A369212D95D7886BE00F2B7A1F187C85B873F33CB7D3E4C23993F7BC9FF7ABBC6FD9D5C54F06E5F31B79DA992BC0F2AA79A36BD00483CA0ADD2DCB2BE36C27DC913668F030128E82281E5E66E87C75DCE3FC60431CE6A19C6C45E19A2D84430E945034B7DA6C506EF009493AF22F701088E01C20ACB21972FF28D284D2A57A69D108161938F028011BA985E2B925532B52EED878646579183EB0EF3FB57BE1E05C1BD2B68D0E4A9CE104E0ECDFB30268E6C007812FD88498D97795447E7F88CE8A3BCD161B1E34DCF77413B778F88A3AD66F247F7664D0F6AF31EC1DADC00FEC09D6AC63FC8A802C3752484B5B43555381043FDD30EF5D7DCF4F5B68EB89F08A0FE5ECB65D6D9418CA5283CA398D3129CD3B44EDF5E3568C953CE2B66A28B473CAB0C43918A05C5A4358D1392604495D06B395187CAA6B36050B436B218485466CB6643EDA7AC66AAEEA758715A22A8849066879CC966DA7E0F7E843A1E234920AECF2A4FED2F78C69A6C103A5A535D77976B1E40BB0D75D6E370212017734F2B0BE9F3F87C5583634EA6A998D8FE9CE5E5905F2A6ABB9D635425D06731C1227EB634A3603D081CB1A7C2B0C1685942EFD6F65992DD39AA67CD954FA0310A0F5866AC6121E27E349D5C2ADF37A672C1DFB021A855000F0C0C29489926B997930B20DF641120EF8605CFD9482EEA9344918AEC689F580A94508F318E63EADC3EB7486B9FCD7A92BA16F5E02CF78EF73F528FAE3A43D451C58261A82B3735E09D4F4679CE112803505006CBAE6EAF641A6F66B0C51EE90095735D87AD88C3D19CD7888248FCC3872939605F205976EF6C6AA52F8316E9DEF841594373FCD177F0E24D8975653A29F738FA8F0D457F2D72B7C00A4B7A66FB080F705A8BD599314354E65842AF598B8A2A40F6CB3208762AC6CD467D06A2E987A7AF72E355BEA297DA87BC9250EBE8D8F85CF292200A21D93475BA46BE78C17C1605C10B97659859F08F980114955E68361E180C98015AA46A776070C4F2B328A903A70AD226742F0022279179FD2530390F190E3008F7202C83A6A15866DF848C8A150D12287451DEC8EE7F04C1C3121E4BBA687D7"+ "5087E3B0C90BF601DD6501E071270EFF8683621E9F5D67A7A668E50C4F460A75"+ ]++decapVectors :: [DecapVector]+decapVectors =+ [ DecapVector+ "ML-KEM-512"+ 76+ "B5589FAD1B1ED2EA0F5869628F8777567062E7C4A06C619635B90C44BC3B0B19ABA7D01A90594CFB8AAFD823329D502A08A5CB10A7827D6B9687634BB1E3805AB22370683871E78A2D5AC9CA582C7B01C1306309BEB9C2F9761552265D6EB67DC635B9025ACF510C6CE8D1801CB81E47434ADE02C00318C4BBE349F751C75A80225F807C7FB976F5BB5478F8056773795DB90F8A762EDBCB1CD2FA53F69A5C44B726E5B5926F92366351ADCBFACF4CD9492B216638F4959E3635A3918BCD9AA59242667DD886BA9B53061421AC3206A68349357A0DB7B1B40C2B6BB41676FAA77E05B3B8C57C549CAA2BA36B6540AA060129B7F5A7C59DB17E5A9708D6C88AC4B3C8B6DB1234A75319931AD3B6608031BA1A45772DBA8C39C13C24CBA719382CF6231AE064C127B782C5781AEB16CB7C7767EE608D37986CEF629FF5796F9ECAB3A0CCC8F4E0687F32BE1426A198B00CBBD2A99616268B5A3B700A1990A75158B668EE9637F3F7032DA19B7F7185B5B70DA163792A79238F8AB3755AB4A387CDB6A7905C17CBC259AEF9B534D03860CDFCB8C4349465B8847C741B0D522AF5D976E2D70CB7350A150A1217481C73CC5CDD6393AAC9BB6C68770819001AF7455EA2687C637AC43AB3812981A511BACBF23DCCD3252D1860A143B4939523C92C94A213BCED384D79555FECE80176F739C955C8283668CC1208B8B08FCA3027BB7494BFFA6B2DF317BD36408FA0580D010DD4376EE6D91A0BFB70AD3C086D24A9991A53C5FCABF04647A7D777EE428F749C309782C2B83049B0961889012478981F16F83B6048C0E2B1C67F739B4D48256AC23341984EC19B7286CC301A7C205AF19B4DD5C3FF59A00C75048992BBA539369295BBB4FC5245F2598B25B935A445EB6CB9305408FEC37BBF8C3E74BAA1BED452A4D77607BA45DEC85F93A30A4E33960893AE6165955141284C335D90C4799E279E1A997A8C93B3696377F0941382B41104B84A0C95882B006435272609023E705415B697B2AD9AC5C16AB5CC343EB293334A298CDCE4BF4E680D92E3CBF9E38684522CD496A237CA72F453AF6CA07EC754A67B21437BE714F2361291B7B76FCC0B740ACB3C1C71EA20149FE5C27948A09852C2A3EA773FA41FA70804AE74A344E7474BCC909A2893A97A89A9D755E7495A9DF0B292259B5B904FFD8249AA5590027B91DEA4209A156994A4AF3F6143FA599869C2C293C9446A431C439A92A94CC1C5977232A47EDAF110202CCB5C881AB81C29D30CBA381C316EA8016CD912FE7C8A3515CC6C6223BD988768D562FE72578C26896D5B03E1D920B574295E88BC3DF657EE000D53B58A28773264730EC50A70EBCC42506939F185B566700CB8D01253F52287CCC02036008F43CAE36BB71ACA892690A548F832314CBA13B118E6608CE584932B58284EF38970589508DA06079679697B8BD8B2345B332CA0115EDFAB5A7F9A0356A3C3F102092F705FBD69016FF9A13B0CA21854A0ABC53D00189FC57B2494E7977BEB352B24BB8F338FD30A4E70C4379851A295A16D8AD0100DB46ED772A99799198D9739E1F543F140324243767D71070BA303177498B029608F894160787F3DE62CE65B74278B5FFA24A65B89A59097A1FF437ECA2C431CC70CA0418141C36574D730CB64A234DB88F826112F694FFCABAD91A93DCF9BBCBD6071EE0C6A500BB6C78A1D5D362A45046700C09CDE0360C9A9754C4B5C26FABB0F420ABA265460F7729B625C5B6B7ACF97766B9032F4A429CC76C8B03C7168F7037E8101830C8B838B5350D2A9CC575AA88C3A673A81FEE3CA79436D2499C280C1A52E0C5086B5C3C209410214AD4315704545B74B98C13AC3131234212FF344C3514C876A96006D5DFA01BED741BDD76260C026C29E78138851319BC71F12A66B2D362066A1C4B2028303B91445AAB80426AE9465AA25859B28F1402179CA3D1C073CE876AEF5B25F964149506C15CAB4698C91AC94ABFBE69DACF06323E9A6E151CA6C019F49078E60C41837B1763BD87EBCE040A2D7A3D4964839E4AB07D121ED3A2E23F75AB6086161603C20B7BFC063CBD00A7392E26D7F697C17BA0C62231A3E34A72C0BC958093C62471B7D1A8B2F7520A0581E809048B4798C0F1A0A2083306D6C42233202F2101C1DB61A58086B24558B17D5BEFC7542572B81E9880D3F164D708BC3C357546DDDC39C71C3CEEEAB482ED67E61CFDEFAAEDD3027B700DE0C72DD84D43F2B81F309EC9D9E0069DFB3705C76B01E9A8DB4B66437487DCAE2E3319165983C937CD3E95C39A9"+ "5EB39C88DB720B7C6B1DBE08047717F870671DB16328EF96F8EE9827AD694CC9BC662E697D2D30103851239332CE9B2A3653BD98C3FAAB41DC5445FA844E05E4D189D4B8E6E3FECA09A81975AA0585E1B179808678F22944108AF19FCFBF1C58BA76FDA508449EA6B3312F8184D1E568C1A4B1C188B04E654978EE9E7A10299C45EEEBAD0A24884E12BBB05F0B9575ED5C8BACDEC1AB16A59EBBE2B7A677559140C9CB00E18B4B5A0C51A4A92A61537DAA2360525D699DE4F30DB0B4D841B708B92E38E1E07F3A2EC124E6C82DF062B110F48262BA95A50956B7430EED0041B295FB84C38BB2B96CF1C85E88D3BD4DD5AF8E1ABA963FCB9D9004F5CCB7FA15F2D8781BE9CC19B77EA6F397372F9B6C9B375342530F4FD4E3B934E52B60C20FAB1DAFED0E6C72FAD8783B62852000A24BB81273BBF1C25EDD879058C5EEF67E3A0DA14E937564A508E9475FF3B25B5687046684760DEC75132D14FA471D54F4BE9ACE2AE9088678F44104230582ED466FBD8BA53150EBDAF126F417183689A24539358B00C4EABFCCEC170C87975B3DFB9FF6267E81B44C93CC9451A3F9F7B913AFDE043748C5F7F36FAB63C801B0BC5042C3D44D836770E9058E04F3571B04C0C87D38B17DD1F7ECEDA701B0532CF35A096F8237083834AED2B2ABA4DF1CFDCAEAABE69DBCFB9B4838C8EA1A9A3CA8FFC75AD9B9C0366517D782F412946539CDF0D02304AD60BEABFF9D9BE1F75ED39BBF2F10FFE061F2828DBCE798DE6685A3F3F52477B361B1A42DB1C0922E8D2D545A6AD54F0CAC86F6AFAF7B381DBCA54330F13BC19F7B698603F844170B90658A9A82AC7D582AE798DB6EDA9087E1E7EFF6CB903558A289F988D5CB5D2C52DB0ECA5898E3AC15B136640A38ACC68F1A7BE53C54F0E1EE66556663E204BA8399633348EF38C7B1FF500AC53958D103B984E540732459E860BA5F065B466FAED85DA80716FFF2F57371D9F9D09EBFE3A088B4C2C44E1F03C73A97830269A41419D2BFA48FCC16523745E67C38E4BE7344C4AAEFE9BD667CF4741E5F16607680378DF0D2EB60DB6DB6AA66A53B63A03E52CA"+ "1EB644EBFD75877D0CA481E1A3B95E7C461B81E0E3DF5A42699775C7BDA3B004"+ , DecapVector+ "ML-KEM-768"+ 86+ "859384DFC0C207A6466CCBBC507C26DFECA413E06E46E30CF18528577B76C108C31E1B6EFA8B41044A4FB2152D3FF391BC6443B57060AE655B0EFCBABC889032ECC986D2A144153C8189505C37C6BB1471E5EA8C1521770D732E88A1A3B0624E5A2A42661C1743677591296B81447169635160F80D61F31C420A5D35F5ADFEA37E4254BE39741E197056082C1BD9390A030B2A63FC37DFA561AE091DD9DA11649BB7406464838BC23C872423B7515E3444FB862BA7D51BECABCF9B458F302905CFF983D00578AD89686721432221782BA7A134E052BBA7879A36654C081E4C3B615A1C764C128AFBD79EF0D816C7E5AF135A51448C14CF3C6172E022741B889DA485C208899190BB2CB8090BD2B7F8213213E090B27A9C3C56BF87024BD2E3001CEA031BE20E959BC1A75281B8A3639F487C12F40786E5458B14747D965E0E294947D6A487D44938D50829D0723A910806B9376E76480141960757CF37E2AC03B4A6767A68381A69F6B45C274C62B800403C1C992D0B84D336560D6B4BFED4A38EDCCA3C61699D31CB6BB88C10909CC6081D1BC9A549F604373576F2B5B14D453EEC5C7673B5A39FFA8B6A223972019F0EE7300BC82DA77414E4413648B68FC8E7B51DC6CBF083098C1A2F2CC680E52A9E7A8964EACB2DF18B8DA1D603317298B797BC17524EA03376BA3B410FF1B1B23526D3EB7A8255380793C5D938861E4983D4D23CAD3A34BB256434DCC315C40C8D0A9D2132A595B78DFBB655F9F2361A379047706400E5CFA1654AC779841C0767AC8C1DD27254599137D9F3355328CD34E581E1E1BDD64C082EAA29A7F71610B65766882CA804106CE0BC5E20630A44BC5FF87DB491C2CF7746B3150BA03ABD907096D8004FA20384FB154F553659BBC451C09C0D42A43035994CC24B4F798BC7C961C6C64C92F2C63E87149D89E98221A9337F0A973226557056952BBC05ABBA46C7EC6CE1B2721A0487687C16CFF6B50CD5BD59443CA765353848AA42323EBED3776C014D60D8A6B6D6B31DE26F11C15822991249F76C3130735D2C22D5D4B4550A3428004349073E3F296A286C3A0D6572C383B3FD277439C85A03B42B584C97B5779A2D675B6DE024949438AA8700AEEBC9FFCA10C8B327CA6B917B899D3ED817C4543CDF288B489C2EF3AA2F95F310B0688E3E81AA83005571812A5267781AC21C92F7C41F359DE74877C19124FA462B3A181BD885A3C3534F16D99AFEA4BB6A434CE177C99020C880EB368FD211261172F0BB7AE4270FC9AA6DFDB5985A1C069F1815E9277BEE24CDB1ECBA0CF2C3C48100F540A24E4050F9B376868098805136E673150C957B73EA3E7BBAB076B459D2157E9E501734500BD2B574B9C563A7CA8AD83B6CFBEC76727B1D3C61C289958FC8805AE6E4455A5257E8903854E35112B7CD46185EA16339DC0838E118971CBA5A78025F86749433A38AB79B58E6E6885003470E208735177F82D738C3F7A52D848B742A75D115A528F4915078632FBB639145668B504C30ABB96929A75FEC5DEAE90973B2431D685F74C98D36DB263BBB6C73D71974993E13B50242167110262A8EB406C987481EAACD44069D53C2B967003DD2DCB979D81A97F62351344BC40A6B203083918C40AA459EA2E20892514749556294534CB3E3B948F1C0F7A941FDE49874C609973A4E845B112456974B3AAF9877B3159B7ED0C4C423889C8A8B1C5B33409A8C6B26F573A6668A5FE0C74AD0CA28304D1C7A498A93BC08921D3943042F986438ABCC61A358C6482839A524845B04FFCA16CB741F7DB5B0AFD7B9874B6B53D156B3E036B3D41E963B27AC0C87B411436471591001635E6440903A0B38302A92C8777F264DE5290DD1E511BC7C0A68E3266882CEDD30ACA8928596B7AFFB28B94D8711E31A8C9E5AC2A9567B55630D5FF8627E52386408A1BB3BC2D6B90A7C33C04AAA60E9046E10D89B41FBB8E3E26CBC27C1C062411436CC22D637DD021CB93B6972CB7AEE5C98AD64006B0CD070B0AD99D467158C85852B88666BA1DE29CFA3047396A5BDDC76B1AFABA62D622E17F846E17487A167CA9B45070309526EE08791647CEE81B7AFF21CA97C7D09A7B0A2880DB0110461F168A8F5BAFA045864373217B05ED1223FD0B00A1D7810E341C15CE69567E8265110918B857F2546BEB5761DE3849FDE971E2CC649C363C0EF296E2DC94B07499869851BA3032248E4A628294D626C6F38C42530A07952D6B44B650112F77C8DF5C5939102090C8188DB61F3713E93C4A22A61C91FEA79EA06BB1344A5CCD63812F540A8907FFCF62FBDD7344A744095BC5DF65047DCF8362A38BFF92B5B6098893AE96286F4199F857DC5351B962204807B9421602A0F00222A82B6F4785DAF303354F9C097CB74A0723CBFDA0553C4B0B4454E1ED11C6926396BE9AD5CD16011D333A1A880E123934243542776A0848B0CCC442BCB117C24982C42D10510683E7D534AC47C6631ACB187039D098B7DE551B868A28FFC42112253AEF0F86F4C358C778017B30C0C3510254E6ABF50664FB54C197BACBB790C32A935783093C88927B65C5C696C695CAB4A845B262219C639C4F07DCF663914094B1DB58DD04105D7C6BC32B6392D481BF954ABACAC7570CB30BDF265E955AC1DF3A356AB59C89B0CC6DB7ECF4185F81BA9AF7BA8CA47BF69A278681C1C8216CABB4887A366204118764669209D502BE63BC8F21429E7434B381B55C95C53DA864E333C13E7818644788A381215A4954CF3A038DAE27CA4684B130268436CC22FA100EBD843D6D9A20DD0B8C4BC8D56899FD3E4272CA1301AC9429986977A9A4910A785990629DFFC7EF2093AAA646741825E53AC6C43B962B5624A4D31A5E054436A2B7B880460A52062E4C5C9B2E72225DC2355D44DD6B22A18CA6A05094BF8D3B2A0C58C13777BBE448FD0182B876A596E595AD4E79DB6F0C1B4E1435507802E2B539EDC25820B23E7127655E11B34A5AAFF25C11A7B617ED7083650C9030462A79AB77BE63780A45F8CA96DCB15358E56AEDC31360B2030010282CA67CDE5169993FA808EE475BDA0AF49C216AF9506330C9B0384BD2C3C51B814CFF09B3C530CA754E67966C22E44F68650B7117F78235D8425929816526B0A46F647A85A7929B49D677A38418368091590FD1276EE9629E63944DB4770B7C22E30B2CFFBA5B6D46A9A1217B7FA44B54B8266F07554B3EA970CC30D1E92C03F41CE2A2093F58A831A4353E02393CB8969A582D55545EC39A61104BA2E5BF082D0D9573988CC26089B222718D29BCFB365365BD9529C55AABA5635850809B305614091247E8BB47315503A07337ED40CA79D3C8B8D240D3AF38C8A7A65ED18047B92600FF09EB509820F0E"+ "F8FFAE03CA26E156AF3AA562E6F84159596D3A985AE42DA4E45171BA7C8A1CFEE2549EFDDC414B27FF697920BFC8A006BE1651B0ECD18F91C91F1DF9E6332178C3BA45407099C0C5C223E9C88D4BED9742D8A3E1251C6AF4B03CF79F2671D8746E7C86EB270F0C54B3362F691387539F5E7B16680AB51C1F93D098D40CB8698827AE9EB1A85F52199589D561A0FB1140FAD888AB0B03FCE03A9D1D1069F0FFE274362F3F1B656E402E3075EFC29CFDCD3D3A4163F74CE6B1566CF4626D14544A2A33DDF5860612AF5D337C6B55A7EAE01422CA01882BAA591F9EDDA6976E2BFDF945277B4C145CDBEA5F22B636DD9761F8EDC518C33223FD46D771E39756B26DFD35798D3AA071F90D2250220D4869C59B3CAD020B1E786EB8C856D2AB33FFA1605FB61CB91C64DE695FCC1D61BA06941B5336CB67E648B4197AF2B297F4EED2DF996E213D63E0BFB54A86C47B286A9D5059580DAD4007E4DC6FB944A53717F325D94EB6A514C13EFC445F8E4EBFFDE3D7B00B955AEBCE826E4E9568C5E4EB69E0F3BE52C7B441CB7F09EE441BC01058183ACCDE0A9473881BF09C992A86C6511319EE5AF76FA80079DE11EA7C8B9D0947C308345A6586A8EDA5D00AA7A26346C3E2C90C1709E52EE7CA292E89E1C22B5447449AE508BA0FE2C801E3A0F6D12E22C93BD98D19CE199AA055D209431FDDC46BEAB5DB20FAD6B53B93E1E723785EBC29D486FB240D6A4813683E2E71FF05B4F53EB57C068C2C35F0B0DDABD8E17B48F8477747D1D97AFF5434D1F09CE1DD325DC2BB5F79523DD4185DD38D3629B6E83F76257BBBDCD6963B1EDFFBDF4456C6A6D2641F035D08A7EE3A7F1C046CD3BE550AB93853FC689913E0E136A83BA15F948D8F229CFD6E12D19C2FDDC9F17DFDE4B28A7EAC09043F1D4F37DDD4923151E837D7C9C9FC4DC9DE1576686B50F6934474402D53439D45A015BA6558C3032571E1B3263BC7E6CAD549D1050FDD49F9B5CB2024CF543B6FB706B636106B54E938F8EA664D9CBD75FF7042ED9D81546BDA255D177D7335A7DF06B5D0EDC98C6678318F7131ED7564D5344DF54802241144502EEDCB057BE3CA1D3FF8C65FA7487521DC9D4DB49265DFF8158B4CE2A1893399BCB4BBFCD2C3F7D9C2286EEDCD76518026D6586F10A79E02477EFD8FF93931B01210C6AD0F75D23A62F68044200597DD277CEBAB9B5C21C74C5CD74B63A8B6D59789F5C50CACE6676C93990DB2F7B4F39E00F455AC002C12367D7BFD0CB0327EF077C45382F0454255B2E11A942254608BC41B890172BE38E3EDD20ED028C37006AA6ED5907C78614748E53B7EE3242A7C1000552EDDB67A5436FE8808D53699E9BB3C190F5B966A1FDBB1DA68E8FC630D841C63F5FBE7161BAA0670BF970C3E617AAAC12C83FF3C1D19761B7421685F1F6C9C93EB35900E54F10744D90F4D444E06C45E2A9FAEF1A96F0B752D4DF1DF86A9503D8D99CDF8A62622184A919E19DF8230E7017D5F411A89DA304B67E63CC3E3D901F998F563CA9BD4C9BE"+ "34CFAE7F2CA3B0B9C3E06AFEED554C053F6E51D875D3BD3FF0EDA2086EE79F3A"+ , DecapVector+ "ML-KEM-1024"+ 96+ "36BB5D4B256DCFCB14028B1EFD5A74E5606617950DA40B2408CB2C0462A222D4373A7A290FCB2997DAACAE414DD8C3AA7B38773C79C949059CF15B12FA8C522C538FDDE905E798A3E7D5A6F27C93330B8243A12983E5C83880971827B01C2A03B4FABE4B15071AD124A5A00C69D98AFD6BC63F305693B50040595995E99AD375BBD58989CDD2AA92E95532EA88336C5DD9F6C4A1B254FBCC4921B96DE593C3134A2219B7A192E87309BBC6EF95499D594B30A4661F809D03297B6BCB2EA894B724083115B0C6E6C2AF2D239AB69574EA95A746CB23A0DC429EF42A79234CAD8B9E85597BEEC196E85870CA1415CAC40213B64CA34A8247595C89FBC19AD73B715AB3990B7E9691C9E5C767D0780E4E3026DC172494BB4ED87311D20896571608B065224B2C08D3B754837C0B8AD8CDAC7775A8158A0FA93872C972F4048DE1E9C9A8E33D59397E8DE188251A6286BA2CE6F258D1F8A99E78C464F576FCE80A999C6475EC03BC597014DA6084F2C97EB88795F28CDB27437FB11F67977BE1F865722B85C3371D4534CDDD396BD6B12298F4287A7CAA87542AFADCC70A222FC1830182831BD1319D980BA03937B2CEE6C2417A766E841609D8417E9C9D8DFCBC956217A5EC02DEBA6BA6717A0C71009AA663DD4CAFD5FB5142EA14FB52C72FAB33B0070FB38C9C9670323DEC00D9EB9F02F64BF4E65D86C2C216799093040F6FFB80FDF6BDCC649F1F39BCD2C613221B528803A75E16BB5B8B067EBA7765738AF0647DA1026DEA6717CBE4116BB28DFF91C224243C1A72350165C30184AD901A9425538D255B72FD834C8E0090CB0289C2F79EF7997CC2AA5928D453198BC392F6524A88B5AE67B96A1C1782267F044A0AA0D8A011332CB8D34E1C6CC149A4A6A3B99621A308818322FD336EDFB867E4495EC9D92B4BC70F7B170A0D28B3EC365E5DA02C7E50373B3059248A439E0A7A0B3A656863785CA749E7419A9B1595EE28B7DF838F1F2BAF02A065B1242C17DA7974C5332BD7724C77644E88549FB572BA8964979979B5A7901CF05823B4B985E4B3D531C21620689F036646D528DBCCB4645B72EC920FD0F5426614BC469664A48A9EE0B5C7D9A52164E3163C299441E11DD246731180C9BD70219171712C7BB371B213FC976236043F9367569561CC00F033D473809B3C49FA3725F111907015C6214C63347C4B8672980DD856E30B55BDB7B3D22554242393AB54314715CEBB43AB224A911B090DF1CC5E6950504ED063BAB0A0D72A5C0FA8A1824962DE9CB685745659ECADBDC3B335627DDB5959FED1B66108AF4F5548F151ACBB3336A94C59A0A6B28D94B9359AC7C766CD7B9015C3B99EF45280470C37E985CCE280ACABE67AFC54151239125845B94FA26656C4CAD63B0494416464F4C7FE5936BC284E9AE388FA844EF71227D1AA0E872BC6FDE92F894C26E885087523CFB56C7440222150A62BB27C067173025BA2953999BB3746094D546A6AE03BAF8630FB3BC6F57239E7A35C20691FBE1611E6C46EFC25B77029ABAA947D81C286C4816E24F35D883B4D315A529A593421BCC5765B34EF6C4D416C3CD95443F9BA2FA248659EE77AA201987F6A5EBDA78D307559A874963B1927198B3524518ACD6618F2D25AFA4B11C4B396C9111A19398D55079EF62054A812C37FE81DB4565F67038B9EDA7A4B63B640252701844917E8CC7B3B1EDE433D0356A2C0EA7E2DF3C237498231A65E8E87874BD78487EBC1CE800C7D1B56476C220A8BCFBA568BA20B773C993A98302C084B2BA25297049841355A16FF098C420A2220837BA6C89B44356B810576760316254646B0A5A41754A18AF17D998541400CAC0DC830AD9616F4F17840A9CE2D69B3BBE8CF9D60B01B00AC7EC8B1E08B3D886A804B4A4F2FE30CEE898E42109F3A16C91FCA5D1C328B90F00DA0996F2365BE54927D93F83F8F319D8FFC375A5126356A94C6B6C3DEC98CC9F2CF0089B3A6B878C73BBEF04A4442B1B0FBE273A994BFB6E88EF627B6FAE415C37691C9D30E10B45A83E02AC3F16A5B1249E8B15C0304AB6F3482477C44A6FA8B5DB797FCBA271FC1544CA333A6144E72B02E929CAB8FB6011EB66399B8AF76CB188AAC3EE43B992482C4585934A5075428FB9966ABA67B414114E379A05C7F81C351F6E18EEB8C3042E4A8DAD0CED9A28CC492A2573590B15A914CD100BA119939610C6CD77D73FCB5BBBC50609152B3946E2D5969BDA6586A7B26BD4284F41269DEC3C124FC3C73A0A369818D9A491931607AFBC85B54410444F9A426C9491099CB4AD40EDCAB36F6924764403ADE37288FEAC93C2544F16ACB41F0355E2491FD948FB8774A62C0AB9FD3BDF48948C94206E386BDF8D463FAE547B580B1D9531D48E11D1DF6539A7911B5577F7D8B55B17547306CCDF4224950776A2C90ADCE2673983C3FF286260FC6B08138CDC93CA3297873AA9443C7165021D767D3C132B1A66B413B8BD3849E8F6547665601C776A558209FFCFC84EF1415DCD6091F12C790F1CFA2C5CC9C170BB7D0072AE997FF443D853B021B1999E1686AC6580EA6A0A946A2B417C17EEF967A51CBCAB67992A16A8C12623CA327AC7FAC5E4B616844E28F43BA45AD9252C695C4EE6707D60C10085917A3D5CE9D46CC9EA52A7759827FD64571B7B4630554946796B308AB65680FB23091515683D1CB815221B294FA403D79124A9989EB7402F9990712E91D78C89AE45B33647A8FC11C5770AB738A3573E2C32E485A07DD0A009DF27B4456834776ADE5C454E5C70D00CC455115169E8C5D9AC35844E5126E784596B191CACC1C8FF6C2C0E371A81726930808CAF3680CA194D0156BB1662BE1928D3B3A432F102EE9C8BA950C3CD05B394320C568474D78DC7E75862A155CC73CF2A093C87A96653FC1B1288FB130CAA41500541E34725E612C475740079BEA8EF227BF1269B9F0F86E2CA929C3D611CE52552859C7CFCA9BB0F125E2CA1BC1CC67EFF49E27EA6B087700B8723C9DC0175A424D1578458D0BCE007C5858361279B0C76E62A03E1977B880552396320976040D9155DBE7A4AE2AAE1A312B010CC430B310CDA08EBBE690E0A102D24BC508B2A31F61171EFC3C80E0038C20469436776C65656F189A972544529CCCE3D917EEEC4912F779E3435A890000042468F93A3D1F104780D26FFAAB53AA33BEA1A3103C3A51A86C6A7E178BB65C8145F6A41EFB6DCA856FE4847AC574476DD768271031E2E8A70220CAACA92EC136120DE09FFA31C034C0A243EB48762286B1B2218ED2B34C551572B6BF741A9D0D8A31BC1678067A3FD13995CB559325B29C39A78C7798C3806682D3B0C810F5740D438A1861C8B5F11343B9B21A11694BB1049DCC0C872B2575911DE0E8997BD19F2DD7B7D2AC501098CC54444ABCF1C266721122A6ACAC20B17DC5CA3F9B5089136E6EDB6061382E96D08B27D869003B8794BA83A50B1E2E72910699BF02513B35A1630F0BC2E8A666087787DD3B0B10529AF8499F6314039B827786B71CA5DC706DD0610DD220EAF024D577CE14F80649B86EF5F839ECF23F5C595223A294DD9BC0CB7BB59F039082E68B57FBB325911C2D8CAC1C694CDCE01D9E890561B43FEA7BB5E2E00C6D3A650D9838801810BDF5013A8B95E9D76B329C1F2A0A40C4964FC52823E55B2AF6412136314937F98150CA4A2679444773B66CC679519815746B74728A0C9B8346187106AA2185669023A1983A38580505D04E4847B2AA2A4127125013DB519F141B5FE640B06935C2B673819A665D9039848803128204C4B223533C9A3F12C9553C8A4161490FCC5B51CA2B6A81385169C38C78ADA8BA7C42877138A50FBFC5863E4967DE01593597CFC7110A0B9C82506C9FDC7BC4D4D6B245751CCEE2119BE688EA0985C5647294A3B9ABD07296473B70578FBDB0C4C5F16EBE034375818A63E408141B57129093FAD761C54A99133111707656811A2351319C04B99259946FE197B7E89283659623BA26185F887680943C1CB88341420C6C32CE2B5B6304FA99BD35A2D7CAB3E39060056BB4D756A78050835F301AC355BA7A412227027DFD2C681440AF9C81664A9C4DC5955F4BBC2479888A7536A9199199AC6710B495949CA82BB9B1A2B06823BB1A79C6162CFB6A8B8A8893152928A169B9801765DEB2AB890BA2D32C5C4409912C86C12C08972CB3A042BBA1B76675987279067C288FAB89DDD93A75368888A40C2C9A20B60A0FCEE8495DABB859B797C7B763CDB9CEB405213B7BB45F134CB6CB106CD7252EA069FEB250138018FC4A5071473CA81C6A495A46AEF27210D556FABC5D12281CA6D03BEC1E62E762B589499CC3DEB96E2A1119663DB91A6A308E2360A02413CCF01948541C54D5589829053E8513A916F58E7318073CC06A308A53FCBA7D197E3A2DED7516D56C84BA0B51B7FFA0E5D77342681EB856686B73D7942C37D8E2C8"+ "4F18ED0D9FF40079EC3A6D67F903ADEE4D5CAF5B4194D969CA222537EECD10ABFF3CE7CDA292B2F632F65A1C73E8CE44B6D7890E123987A8C08FF23F7345313294FB6B3E8E382EAFB352958637A6A01E0839DD1218F1BCA9167894383BE7F651AB0C0E74CA9B19DFA157DB8656E5C4367DED62F7B4432F43F5BA02F394271D79F1AC3DC4B1CAED50723730B337EC50303182ED297AA3908CC28EEA64E80819CB1509FE54302E3E03EBC39BDF0C00A755650AF3610AF7834FFFFEF5E6A2F7B66D68634E5B1B76F2482652B785DC8051494BCE4F41867BAA07EE88C9680E19139EB9BC5128FD5A348E01EE2A7178B3B29BD0C6D6096A7DC3059DA039E2AEB64D5DDE1154BFDFCFAEE6CB6CCDDA5039C64EC3ED3A2CD58FC4F7A4228995CB240B1BA31CBEA59BF6483E100C3816BADAE7414F0E744451FF1FA0912D5C0A2472B1B7C0ABDFB175424DB48AB6F7BE4DB871788AF54960D244035D0E3FFA2FDBCF143C22001D4D7D3CE3BA848562C9CEB024D5DFB8F07689E72B0F2AD31D05890BF96F46002E204EDA434511DAA17385B81E0C513CC36F9D53A6ADB36C1B23C2E37CAC31E5997E1AC3102D4FC2C3A5B3BA3C9772541FCFFF57AE162F5F55FE21DC47A66DA2AB98233DE8EE3058635FE5F3315E7BFBA87E8DD2280930240945E972CC29CCA19060439D317C146DFF758FB183D66542ED00F2841C203095E9020D0F10F416C0B7E641F4B02A9F71F599E42678C85CEACE3EF586EC1B4C4C62D21F2CC75CFA7F58A525CB6A55868EAA31043596E2985D52F32938B304678A89C72D3B91AEB6943A4B34541DA42E1060F6DC03AB7D6AA90B2EC7A4D8B2E1FCE969A8614765F8212A08BCEAFFF50A48C2A522E1A41FC2C2CC89827830D34AD556DBCAB5E69121C22DD035437FC6AD0DD95C0E22916D5284E833CB84286DDFC47DCBE9758B980D78E4DC294A3CE5A1175F7ECBE2D103801089CA744F2BBCEE7212F64B0CCFF10D4E9643027B942134FF2B50DB9084D66857C4DF62B2C6919D9DE84D51170EFC7942B3871FD536D26310FF39AE099E0FA562FB13B0418FFDFD398657259A822A458E3688B42D492117F16E291F913566CFD6CCB097C0C45C2E5416B7D3707E93FEC60037937B3930D1BB6A6CCA5418C1D26083B3286D00EB0AED6C2B7E77028FC70896789AB29F424D91D393F1F63ED8E0D02D3A19F16B80481CAD1809BF92DB925D11651E7DC1B0499E1853865B708D79446166A1C25EDE6FFCD277689B298DA836BF52D4A885C6E8F4320AD031FFBD676378093C0F854236733B002E55636D44A5DB9ECD7358CA24A3EF7638EF03B4B5D8E79DEF61221FDE892987860DFCED702D4CA1536348FCCD99FA58C1031D934CDE4593538474B2EBC8248BD4C4792B2B63B16A152970007574969F34CACB80DC978240C9755903E9F7951ACA608B682FA60A90A8282B3E54862B60CEC8EEB737FBC31B6511052856083CAD928D2E41F226C89A3DE0FC36797714471216EA0FF8A3DC1EEE64E24AFC46A16A818B4FB212410975034F5486A4E84078F59DAD65E5D1EBD37021EEDE2B459041D4250468459D535CA974683CD4584B62B45756EF0D2C0E0F2AF7B946597DCB86129279078C4FEA6563EF1E378BEF179D65B1C8B0432660AF5AFB461478B9A9CBA980925E8035CAFE83C3DBAD397F2BDFAAA08BA05C623115CDEA9A62129D271F237D33CF1266B38F122CD7A365956B9A08D611646CBDA81AD64CF096344F9B905386E7A231526D4C5DE966263B7C8299FC7AC8DAE2CC188AC205F60A14AA108C394260A4630FCCD65AA9F56FB26425B02C6319C305B042EB9E10DBA41221299BD6DD4E02CE5E4085DB6439EF118BB45B6D4C086AF359A3C4B79E9B8BCB93CAB682EB050C745380B65E8775ACDD27F97E934A059B3D663E35BEDC977FBCBBECB3D10C2D8AE2CD8B1940968FF7CE054EA7EEBD0E81E65A372CCD133E098867EDE82DE7C58D99E908B416584645710E2B013A849FC15C209931918CBFB0FCC35D87CA63315719CF8DE8025682F123B6A4288EC654438C6D75522A7E0B80F59BBB44AB8DFCB5372597590AB7644DD1ED74162F5FF389B417981FE46B27396508B28965BC564F8B45D68DDEED1CBB0B3A50EAD14087F3B9E5D8ABD007BC38258D127EC34413DFC4F41EE0B549E106F77C1696DBF04F31A2DCA4691E71D8BDECF4F2721D98E1C8"+ "5D1350C0937B6222662F3D0A6194B8F1B60D4AEAF94FB8426C91078B6C12041D"+ ]++ekCheckVectors :: [KeyCheckVector]+ekCheckVectors =+ [ KeyCheckVector+ "ML-KEM-512"+ 116+ "8B0C8F08A734CF1801A9CB0BF1538EA8AB5754D3A618D4531AE041FB054AA46A1E70DB84D63B722BC69CB4785CEBD791745B6C85367289447E3D266374F30F517B5512C02FFDA63B519CB361F1BFF4738A38A0AB824A232CF332D717169A947FFDA532DBE9BDFB0372146056B97A066568C5AF83554BD818BB6C206695C3ADF29950023321A215DC98A531BC310BD39C22A56ECB047B53E62D6EA837DEB77974492CCD2314B6B21B11E170C4583992A3974A921420B9A033B953C69B651E46A310813891FAB0D420B147E7933C316DEFB8BBD3A9BA3883182B4827721644CA969D5563372DE3C5B9224EBDE4344BA7146198489DC20B6C079036DA8C1020918BF293B3AC0EA9F646B8003D516C394F36A6ECFC92027A92F7EB6369C280D284BAACA3838C03840CB1BEFEACCE0F04454959CE459C62E9790BCAB7CDEB37644841AE66186E9D415FFC548A34E61B75E2A204204803159AC57A50AE209A513337A7420DFAC60D47655FCF239FCB3C9111C6552C273E42E39C8CC180A14927697A3673687556A9A4640ACA90C4C66F00654A13AC7A5C25DEB2427B651B4376C6DD506258E255C7BB8D8B312EAF8552C74598D7D4B7FE388FEE92A12E3381450B948BABB0D8C351538188E7ACACDDFA38A26205C71C4CF72C06A50C83CEA12918E03E95781D03033C22717898C63BFD8A9BA3AC312E61799AE30211777D3FFC33A06ACEAA171180580BE8458A1DE556D8F027D4F95867D6C151404269C215E8C0A42D3A2829A92BF737785B6BC1620578BE53803D8B2DB2A626335561120339761214FE90267CF803C63C27350561E785CC02990B3FD7ABE193B3F1037FED871929F649BCF40D3B0C4161499DEB8324A9375D9A262D65617868D73F519CB59BE36B1B24B14C31BCBF08571A777224C45BC0B73279AA9B6B740A351845D86676C0B333FDA586BC89895DB9113ECB58FBD73540A6A8DD14978915A1D7A3164A378512DBC0B5535152CC0BC5D667470BC6946250E07162135555F0298EAB570188C1C1857A55499807FD53191775BFBB9ACFD38563FDFC2D64666569F94DC2B1BAD1D82824729C8D16292176E97AE6D8E30FE1BF83F44762D689613D105183EE6AA7933B"+ True+ , KeyCheckVector+ "ML-KEM-512"+ 117+ "024DC3A47081953A8C5691A1FAD560C7653820B4413B266882F6A2D7229765B2972347330112BF65E82F4A8C7837272B426A4E3305A8774621872B50B270C9C2314A631144E851A410E083F843575469899D102BD911108B90C475210D9EC35C86502F47D97BB0F65DD2EC0482D5000BC462C32C1B3BA769C6D407E8739850F9814035A8114AA405C086FC738EDA7A8B63CAC736E1B88C0A8F42A6A4F795A628C339C6902C73F6359952BDC3344AD8633C3BC7BE540CBAA2753EF10301ACC8059739649A7A6A23DB341845878243137B2B7DB13B5F866544BAE8C22B972AB812624E4B0648201AAFCB4652197891086C7CB0B99BE95D488AC0F74274ED63870B405A64C338400810C324536B9CA3689A81F9A553BA1450E0C10E3503935A7616D07BA9C87AA269F8526DDC41F732119FA941064C55AC7319A514BE51A587DBA0A5256A53A4868B34B80426A8A8316928A6A57EE7EA2B706C5E5ED37EF7B68D5BE75726480EA9A01C5FD8439949B85AE374486472C9A7970A84B08E997440613EB2E3AEF03B400AB06DB0B27D019576AF6984724479E78A79BBF14494016308D1713DFA9677F94C562C684ECC7CB9137D24F0AE06FA563BE66BFB648FB3AAB64203BB12CA3FBE0AC8A2272C6B26B9378B7A1693A4BFEBA351D18CBE3A838E299E0BB6C77DD13CA9791DE2D989FC61B6A0E99733C72A18575555FC7F99221127BC8272505E5B489FAB663E1A848C497BC3C6EBC8D5C23E2AA45DFF832F75894AEA99135EFC65FE5B2A1FCBC291A39040F00F01569780F2435028114C59964903247421B3CB72C85484340FCC404AA368F405CF7AB9913F81073B585793A1AB1DEA39F452BFAD241DC808367C26075409343B6C8FEB08A9BDD83FAD348FAD5B4A1E7B7EF87376DB7076992C481B0BB6CCC63C2E82CF65701503BC3B22B4BCC1F186909140B1CB19CCB863253BC2A9BA9567F53CB4F7A8A4406CC7176CC68A334BA892F816BE2DF672E4CB6628DA62F961396A1B3F50C42F764C33109A34D2E94DBD1CC43A8284AFAA79199334E54080E238A333D307655A91C07252A8224F1E457298E15EE0966B5C7C6C4824A02F117E6807B25F4F4D993F9B45B086B3D43AE092"+ False+ , KeyCheckVector+ "ML-KEM-768"+ 136+ "0616CAC3A41E537C5B0AE0A6113215A0AA8D2B12855107232D532C4A494F0661C3AC0A417DB950F0F38B47912EFF36615F2311CB3438B825426EE20D2624759AE95D54155EB11CB41AC81EAD028C1EBAA6297A208E9C0418F3954BC981E84080967145A67922764964BFA413C359C398F12889C0429477647B53BE55268E40EA76FAE5AA92720083EC7117C2B14D4A986D9948AF818E67144148DCC85F5704C3924B6640365F5076EFB1AEEC692DD0C57EE2493DF0CB3AC7341CAD495B71553487A868D7E3A26EBA81B93C1867840E7D6064D77B20F08303E3533466446BEC697996B2CA973720E50A3B09DAAC1CD06D5B73077CD803181692ABF80A8FFC415F5089CEAB51BDD60375955FA46CAF282B9E240B3FAA998B5E84A9F5D935214C5DE1A97BF45619CF632ED9EA0ED2E09C6151AA211AA381B57F1537B8802C41F3E5085A4A6D191386A6F6A6506105021C878C4803728228071C3B11E571DB15CF709518C79286D835B462EB7CCBB00E420142DD44C7CEB201DF83343BA28357B9313FFBC21523182C6443C2DAB19F0903C14CC3816787A15C06076A8535536B15E44A00B8B76A0BB2597959C503B1B7753837C9A15A415636F1012404C61E2979F2E3B26CC8294DE779BDFB3C985970F0D64343F0104061C23844B0B81B7F26C895587164A59469087B80F373A882002C8C73B901D61E529682ABC86F87A33E6AF985C0C43ADFA0364F8290C2A923CF7B60D072AF8C94495F439520EB81E4576485002415079AB705CB4C5B3F652B605F915DABA81D71EC667B03C0A7F77F3C7AB188641F3A334C3E8030977BB52FD4C54F39BAA4153E49C832F836940EB3CAEAE0BDAE634241718671D43E57CA24CBE6081ED28FCB180A419297138A4E90A49269A502A940B3F2CB5EFE1261F17B376F839805951413E3100C84C17A576F731C6FFDDA02C69652FFA40E324804C916CF3D511905626A1AD4CF25B52CAC5B0F919BBF19DC37431365F15C6B523540F81321F8A9C02109CAFBC4B7E0C19FC89B9080AAC952306F6632866EFA5CB0D0C53C49921F5996516A7BD95149890A4FD7BB7382033DFD2B5B5D010D859CC79C9A68E41855744832CE2B5916F301B7A5A627E3858B9512CBE4AFBD9049A229972520673EFC9ADBEA700334725F7CC256FC0B74EAAF0CF1BAD802BFD57A737B52748C009F46DB4A74611DA8F8A549D6BAAEC6AC51A14422C1834D199E30D210FDE062402B7ED2A46DE6A54075D1C15E8082A11CCE4FE19289A85A309A371F620398E169C75B7E4405623A5901A7F94DE7B1C240713688E36ED81C971BCB50A5E17B053B37A2D79E724CBBC2EB856FB66B997CC94D8B17C4A9729338022E1636E63BA7DD66BA288793AC72BF1EF36CB9439B19F9A2861340DB711909EBA6DD4C2219347CD68435C6B634101C4F52CA08DC3C939D509FC5382DD5512210609D51246F2D832DEC00BDF4C1AD89B45333E14403AB5A2C51CCC353427EEC746284B8ACB963AC765290417E7CF9CE9849940D1189AD55A12E3C90DE718F1FA0621F7389360B2F5B6923F841B0D58A9A5CA0CF8AF4BE7BB689E2039141FA7E67938630B55BD65C0630F1405E6C760D890EC64752DBA36B23F483770BA563B368A916CFDA68FDA5D35991B632AF5C8D8F2CACA112F926B6"+ True+ , KeyCheckVector+ "ML-KEM-768"+ 137+ "026DBCA9319A20DCBD8AA0AE721344710493091C20DE43273A821043B0CA062A37AF3A4DBE33B503D01375BA11F06A4F10C7A961746F68456B013449F3E416D0E63A40740A1123C5C1051FFDC8AC8BF9C5EA1246CA16A9F7FC1C8E750CD0C40F18584A473A0719591F1040C2D7DB72ADD962AC278A96B4836C1B6F2C33B6ACC30E4865487B1CB2EA609CEF60CD7715715ECC1D0A541A7286962C26ADBAD87F216C5C6073C579153371A767172784F5EABAE70773102B0728D422B1A8B51E09AC56F8A11F4B40F70B9D7E7380130C977563C6F50BA07641C3633363D102A11D3956CB363ECDFAC9E82467715319DE73BD82931803B09B9E52315F3B9E90D634E7B43A097B020F20613FE42AD941A88E7461D831CCE093CDB88331E8283CC0E18BE0C40F89523EBF09AD35C5C4C44708D4E028E6269BBD7721C8F5C2DFC5570F6CB9721B6D9BC303D7C525646539A809200BF3CC30F902EAC04471FB79744CB82BEACC31D97EEC934CAA13BD607717C811241E4379762464713630D4FCADC9E890A6A96450A786B5849108602951960AC9DA84F10A97B04B66E7304CF234804EDAAB8C9355BF05AF2C6346A694C87A90013779904E183D8CF7BDA6950B58FA4A78195BCC7659BB040E6A642741B90E4E27292340A7224B2CD6D53F21311A497516030606EB206EE9625FB0DB621BA10EFC604F2F696D11E667EFB24319193E16C45E317BA0E7120A0012CDD33A62356135F3E18AF837CA868294E7A92F2670492C0996538360B375294E47313AF78418AA262C96CC2986332F7107AFD50C53443C97161858363BB821625BA209DD40AAC3BC48F4C8069A2535040436F799421DA205F5B729F31A7E1469883D07BAC1B62A6CBC7811646B51134CF27B307FA3A61CF9CC0B6393C0859F91ABC734EC9AA573375674134DE65BFEDB39F5E546CBF0CD98FC5C16B36408EC058A5142035ABAE46594BB2A5FB1BACFB648515B6672AD04A88239BCFE4C4DCAB768AB9294CDA37716E62FC08C4E0C1ABA58565919E19B17B795F3A37A3475A499131FEE57B7AAF8B7FAE64207D4B2E32632E4632920278EDB864D61D4B1E8C62A2BD42DC980475E1CCF9061C64B573BAD7BA2F799C5A37054EF3586BE94056A49712CE94743C3236937053974C9B8E274F6D7A0B0D427F565A37C9AACB0242E7C2C86D01630C8627373EA951325476C1A617A845749582E93C724B4C032B8106FE4F101D99BAA5E4B9F3DFA8205279C2927834B6412B73C84E9F3A5E398AA5F0456DE89A10F381D840C19EF92B4FFD96DA90566C51AB5E8EB7ADC5444A76B59A61CBFDE1B889E744D1E389B90EB5BF1888E7C42ACCF28C5C34A63E0E36E64E99636B07AE6255F1BF980A887CD65F2467EAB30071533F4D0B1F454811C3303176A9E8D4140F57B538D89B794D7C069050637F9A7ED6BADB2A910064B21CDDA813FD4CF2CDCCDF58CBAD33C3777A930EB9AA4A2BB1CD22C7B3BA17B6905A8FE96875DE00C34757202581823250133915ACC659559215E5490263975225F29AC8ACA24CC6888A4D77927687A69344A57AC819E7780A7ABA782BA0AF0886826D20040B537719C87D99AA4B4601ACFA7CFA61BBA07ECC9C3A75CE3B01E0A163F1B8ED4D9AD2AE714111A25290CDD167D42F6F56B767583F78CAC"+ False+ , KeyCheckVector+ "ML-KEM-1024"+ 156+ "521CA0F404B0DC436D02EC4D4C783A9FE60A027716754CC83A478EF9B292C71B1B4B78038B582944AA857FEC42ECA10F1D42CF97C6790F4B26152C5328279CC5A0AC9C369DFEB82A79E498398B54357C5B6B7AA3935A5BF456865C93669E0721ACC1C1233BC75ADB9A2845BEFAD66045D93A9A1A22024688DA726B38BB84D8C246A7E67F909A5329715212EA2F8D3A4631645887E0A32DB14957A39DBDB550395352944B7D86368BD4B41F11BBABDA9360B5F7160D8A123439B6E327CE6A55BDF08ACDFA1B71C382ABCFE9209B8C960CD95C7EE8B1A9A98CAFA4242FF25D957A82EFAA926E99372203381F801E0FA1898855A1A228C533F295E342122574936CE4925DB87EEEC11ED7306B0340A44B37C1C819ACDC474ADC004FC6CB2C578A2B089CC4894BC53BC4830F3B54D73AA2C10CA7E46C322AF962E3AA6E45287538B920EDFA71356B0BAD9A0D1E0855701CA4C0E23F66AB3C7A4615A5076848D564A9A154A00958CDC9152D61A9915484E65154C4D33BF29285A2632AE16C81040B2C3618718F807FA0821DA33A7B165379EC3757ECC21FB9A08DDAD77A125476E6266D42E20EFBD99488192DD735A196A5B2D310CD10E3CF711B441C642887CBCF5133B056EB8A1E5089540BA2E4404F25B71FD787BACC2BCA73E0B9DE76CEE2664AECF5499A687C7C167F08836A15B832E69B66A1509725720543A123311159063137F25711250016AE983684112A0E5B0AC03AC28468606ACC5CDE2450AF086E386149C76015CA6ABC228948D8029B224B63A0A485518753E9271B74DB35EB080B839A174326A57E332C24F3506EE748D1C56835799EECE13C16E53D3A3966A17548526B0D9132C22788687C8BCF3288784B31539D615344ABCE04258B6F218E48A78A4732B91F75455024ACD237A3B2F143E447533786889693AC3B96492D58765ECA406317085DC659BF893B7C11CE4655B2452A389CD912BB98A63DFA37DFBB71BD58A5CD6A6094114101940CAB2B90EF215A0A356098AB06EF068002E165BC21ABAA68B58891445D434139746CB54402972B12DC75B53C471ED6951498DA464CC4BFA76B2C5E8165D6D1CC76C3506261052E411B1DB34BDB6CBBC4A8CB0EFB980691AF6508CA8AE97518EA0E689747D2CA89FDD44F017C49FDCC6DFB10465C8412B175697F052940909BE126C5D35C0E642B6D251B7D526C11407962DEDC06C8DC138DE5233C5C976265A8DA7B3683C4467497592E8BCC741449AAAC41050BAD867C36872C2119A4C7DB34067A943F0C951168252EA0FB5FC6D769F940BFFDE7C46DAB183E161E02046CC708CE817046EA8063F469148D2336432B36EFC266077AA88C85A26606AAF7EB5AB6F3B64ED95FAC5C3C6DA380D0DC93A30C437A14816C25499B29132D492C3792C1880865B3670857B1758A97B14A29B569844775444C915707E2870B79CA8695759719A34D73062F51560FFF19684F1048FF9730E2669443EA8635F303214A629B71A04E6042B06584E6386F3B396304535BB8356240353EC8EA8158C01565154BAEA1509DE657625C2731A01BF741C5242BA7B909BDD6D47175681B92898AC8872F013852546A982081895729A369F40A70073B7A59BFEE3460982307BDB163D3A05664A91961B12EB4248C190B8D980392BA0ACCD9705B4D3BA54535426F10099FA8198537ADB66656B2D2843BE15F42B93FB4E521280031B27798AFAB68EA084EC0767FB5C76F3A129D15E985AA18C899A9532DF21D3366024D9C214B338DA5891BCAF287C4940E5B3837F66A4AEFD3C4A8406532FB8FB1CB8E45691BC1E960C976501FD532AB164D20A3368B0780C71A0E68C76A59F6CFC7F6663798B102C0A9FFA90C18E33121D127F6467E49471681B186A752B63B593C02F85F18384BB8F6087E59AE3EF879178534DE3361AC4A03448486BAFC394D369F450C234208377833A5B9768E13EB4C855C7893020A8091A8A9709A13693CF44CB842A747274273989C28FE3CC0753573BFFC565BA3BE02D7348FE98CC9BC616AFAC99308805ED0A1480888168790F4FB5417F27BB7F96EC963611D2CB468A98C508140CF5267C90A21E07BC471064EFAA4476D0B95C9A16EBDC34965B3B219DC759AF65392CA3CD486A2B9744AAF5BB33CB727E5BBA176CC4FC28398CD1EF5EEE6429066EDF3FBA01694C6AFF0F246418712A4BE53E1C13A75"+ True+ , KeyCheckVector+ "ML-KEM-1024"+ 159+ "020D58896B0BC35C2D70DA8992C424899A8BA9414A1A8319A70659FCC6A3F2B794E84B753B0C8822ABCE063106F4A8706358A0A422AF8FF287EAC05A3D8AA84339996763A8F7A9833D2980F19754D18C37E8C7B46425BF8CC1A75CBB7E1DF5AC70E2056AD37F79F4AB83F87E5E32AA2E36CCA999AA9A4C82E81CC94B4AB471785661349F0FD832297A3CAB38CEFDB45FB020495AD1BC1108186BF4B31D8354422B21AE719809762132E076E5A5B692F40C704ACF1D3904D690AF79F217EA38A8545281F3309CD6EC565A5921FDEC938853A63EBA0A1E8A68187CB5A3D639D35439CE5A80985BBB9C7A9762EC14D3EB5501380D7C298F9FA177EB5A510B04D0CA7A19C763ABE4494A72A5BFC985BC403739DD8A9A13681DD39BCC93E965E4A09E44B97690C36D0D58B4F02126E4A0A07E6AC10E58C24B554ED89AAEB15C49FF53C16433835DD72FB4F69458D1BC99330891F1BE9AB172242412FF819F82377914B269B7120137375BE39B9D32112326EA5CFB5A071B479A73E0532B643B53E58ACF1A987473A3AB017A66DA9AC5946097679BBADAB1A0827783E07F7BC198903215A6777FA3B77DC264CE37636AAA48A5F4208C03636E71C7746083BB7D403035390DD78118FDFC2F7DF282287C96AFE8774A20CAA6DB69104038EDA590C648C69511811E9C28D52C3540721428336EE3D7732A287E51109BA8589227864F5C27C5DD07556B33CB4842340E5C5B027464A74734F088A591EBBC3CAA0899359E75C395A6A57670B04BB38619BBB82A27C85F063CB286160CA46C372604B794D4202EA16F8A757E9DD8A54194C3384850D58213C7533CADA33DD6704F082B4EDFA67FAFA2CAB0C89897FC4C8F2255B839CD653C4F03A728F0D25CF2B99AFDD0419C515DBDBCCA0FC6947B0A0C936AB1F2F290DCE5474183AD4DC85218EA13AFC5B671926842A15D7C367D04EA97890B371297491D21720DE148F9780866E1414EC6844EDA97AE951AFB0AB07FE61B1B856209DB6DD522792C03332645A63C460EC595AC5C49B1DE1980DDE8B4BD5505FFA7CC4749C42F0694BF3ABE4CC6ACF8A7235766513319915B58522487C7F9EB9C2754863A40036E28332265C2A4FC420D95AF87063C244233D5C27CA5B0261669386B624B5AFC992CAB773EC948B8957079B3B3FD27C16CF969005B27B4C438B7269FE317024814AE56862D18362FA47A0035339A485528540BCD27A288489C80AB890BC3992BE39249E507CCC92A250EE631297AB19E6C39AC446EA5D54C675368BD5BC4002463D8CB9A294C3E427A2940BA5F00B97FB2B1714CB6880F5B6A13F868FB936E23FC4B7EB563E2F1C5FB421A449715A1F8CA0D3C6AF726CAA5D93C255BBCB7E60843A54EBA470B32D19E3088C17AE5BEC5408ED9758639C2AE50FCC6C5471B638210F8BB1509691B312082DDA85C91B31517D78D4CF96CD94335494A66B8EC3F9B2478DC264A1B7462DEEA1C44CC0D357C00604ABCC2437404771640D01FA1A9A73E900EB0B7CC8CB3A5EF621EA0118796AC7907612F339654CB519B50A04B6C38A1AA7464E34A8E9B9029A4C1687149B0959070CCF4744A03951B4AB9F6B88D85487E2929338CC6CAC39651EF7C9B802657938B7C368C2B6B0539EC0C152EC95AB702662C21685748A968C68C150A96A9050BF3C597ADE0699930C9780287D8811A68A74AEEC8C335D4666C64AEB257C508A45E32D7A6CF86448B81486296CFEE29C3CA7995FB9B55A8D62C6BD3BE1C267627F4754DFCC1A7818051BB0448219794B87089B8969C72B01FFC168703CD11992383552E14197AB1C99AAF848231579EDDC305D4942B27957700ED5AD44797E0B5953AE06EB25CB0CDC94C6B67A423CC6700D1604E362FB35105A8E6A00ED6587A759B98E37BD358C9FFCC4465F9C5D945A93954550A2B45A6CA90A3593E07163D7A8AADA9848924B135C4911224B52FB5C2823F499F98D0238811B86B55A0D444A2B14B885A0B23F510B5B80280830B5DC705122595578FE076FF859B5FE0878E7B17932A3B99AC58D9A86812AB2006073B8D80BB940931CCFBA269D319D68CC6B1FB5F6EBC288AD208379364EFA73384A8A441C1759BCA9D90016024EB2316D50E8042CB74F81320D089F6D08008007B7031A02BF773BB1C620A73B3A1579FFDBCAEF6652400008B53F560F13D5DCC0B716895116434F2EE28"+ False+ ]++dkCheckVectors :: [KeyCheckVector]+dkCheckVectors =+ [ KeyCheckVector+ "ML-KEM-512"+ 108+ "6933455E456332A43690482EE948557981B259004078A024D7A56F45065E238918A9BC0B1817A3BA2451A5EC8EA3D2BE78528CCE1CACACA6027CB409464A74D6DB110C175095C32750E247000B4712E223C9C60455E97D4CF2CA49E85778274C24819A7F55A33E9536DC6C4E56B73429F7BDDDD71FDAA10DF9C33AD9E4656D945DF13A74F7166EFB4B0BED081D7A8A96440A6C9BC8A18E4792B25468ED10764A6C48A2EA61C786B44B06046A1A5C6AE1745B510F8974772EB30455E409E908BE563564F9AC04D79C535705C679038018983786780BB67A5246378B5E71BF00CCA66B4A98CB6A67D29B26C343C67C47A788617395E0B5A6EAC09264830BF811C87C1327F6766A96158E9813E6BA5BBD30592935834191751DB50FF66BC1A807041207144A21242B3222D1B9C37629CE4E717E442394AD15B07597C25EC096E70738D1671AED7768A059A1503144F8CB6A27A667BFEBB458D4970DA2B375D021161A9A868146555AABBF3B5FF8EA671E842F041BB48A1B2AF564166545CD03A7B473B4304449B90F934605B2407B80019DA440A81C64B47649D276B98A43CCE361560B671E7AFBA6F084BC9979718D8A4993CAAED1C6CE1494B264090992298EDDA7B2EFD181EAA76123A333DED9017157AADD9A421BC8173A62B6AAB030FB00A4CCA61ECD6B7D6095AD56FCA21C4A7538D15B652A824576663A95AE0255BB2293CFAADB6019404F052749AAB545A855316F6C865930A44F5B0F6850B0BCB5A7865A8147ACAF4E92BE72437658011B8739543C0C9A637BA99CFC48B8CBC55BB63978142BD079B8CB1808B9FBA4A41B5ED9F05B74658D92C1AAF83639F6C385B5E3396DE92CBF5AC7A65B3FE216B9F3AA183A05C693833312FC0103C216BBAC6444D75B4F50054490140164CCFF96CCE0483FC750CCA75303D657317DB387538A1EC9BB5D0C3615FCEC760DC5BB655CC2FC876D76FB838DF00CCCCBBC48E246F5287149E67AEDF5614E233A37D6BEA5235A95050AB8E8A5D9067860127A8D473BEE0AA4032761F75A5618894ACC7CB4D43C51C34009AAA2940829250323902CB4C21827C5CB034C05569677B92121303EBE773A58F83A794BA4642149F5F5762923695456BF1B27A6D4375B47F618D61643D4F59FEFC750C90ACE66108D95217FABE10397F4ADFC6145BE81775683175DD4381EF640938171B2478703E026EA021D47334506A86348676C4BA1ABE7D84DCBB268A3A2BF17897ED5ACC36E699862093DD25A20453342C9AB4947BCC41111A7FEB60302B0209E16925EFAA61C79570C918ADA33B65887039F62AF4414BF80530651A03085B249F721AAE2410FCA8C484E7B641AF7251ABC6F8AB67F28BA82FF6C51F85395BE5927A2434C7E512206F9923F1817DB3690E2315760CBA546C397136127F0B40E6ED267FC57703D7A5A3339C4E0977E9BC79E95D888DD131770762E5AE63224A9941699292DF8C4E27B41238C7D5B652E1E72890D1A6164CC3EF74A3AC26632B05986A00A533820B97B0458BA2185AEE8AA04C6B7ED2B8D62553652984BDF27022870652E8787BAC8ABEA70BF748343605CB6050143D49694A9459165FAA3ED20806D647BEB3C3962F49D0265AC05864749A63C90D303AF31C0E441077A1187B25C8054A35D03DC2AEC86C8E7CB10661319280C85826BA92AF37F20F11833A8616C64195327CEC6C651DF0B78E4B624E2EA38C499A257CA9C39D93C01C46864A32F7F60026A28C42ECC00C41C064F4C26C4F1AC13487B3CA68A082CB8F8E64C662A6445F853BF79C529035EEFCB685F83A4C2047C93895B7197CAA9A22A57020CE5698B48303A48C0B42AA302030412961381E25C2BF3866447E407A7E659FB54BD81C86E990415D635B9B3F56A5B9229FDE42DC5F6B25A08189FC8278E9C7240D555BA4A1E4BD75278B120021BB850AAC2DF6A00EED00A2F938A3BB014BC31A371A199D01A62ED05520A8C76E5E855D81B023ABC8298F82A39A2441792C5016329CAE0021B1325D32AA6EAFC6310758CBD284B7691C6D58BA97C418D88B0C3A8A909C1837917266B2B71A99E192E33671A58693D2898BC10512C1CE724F7457F2BEC9AE4E8CF1131538DE75E94C5BB1ABABBCA0058F8C3400DC478B4756377F997891AB35DD68D63F42ECD930A0AEB44C9353CBB09F5BCFD3D90171F38506F35F792D6BE40B9EEABF876C7D44CD0C5309448EC6FFFE28D5843BAF9D8CDB84BB499DE1F20B4627059BC0895E8302139A293DF352A9F8F7C1D3D7D581E12E15D5096F2535FA36A510B5B74"+ True+ , KeyCheckVector+ "ML-KEM-512"+ 106+ "7E1587F4F113783B6ED920916A17762B997D0975BC7623B1228000FD178A4B28144463C4C838C9B3D960B1355DBF512018094663C42D7922B8CD04C7FB44695E5B9E2DE46DC0C836AF62756DA16EDBD022A8581C0C655AE491A0D5176023C58AED4B5D6E7A43F3E33D7BC17A39766A0535A355BBBE0D7905BDF01EE0F7015BA1326DC093D4635BDE5155EA6831EF564752507C40476922A5A40516949998B971688A21405C6987BA15F897572AAA641B8064710C9251A9FA203FB1EBC1E8A6C6D6A46291C4165796C7E7B37D31B198F1FB6CA9121B9FB6B2DA6CA973F2C115DCAC598530996524DB29A585040DF5A407C126075252209657B693F718A0F57E2971A567CB25CD932AAB92B233C53BFAEC8738E4B3FB835240937B61F89B81C3250FBB056E4213E822B16323A164D38B7A727388CCA7026C1D519647FD038495823C600563270528FE94336114A3ECD1BF2A3CBA2927312CC5BEB644A0E4E2B16C24658830412ED43C2974341BF8086CB316B681CAB26A90D0EAAD4D3147C8BA47FFF1CA57B96C8D9C019692A6BB6698311C886E2694682B7801F89ECAB9B53DEB779D98033F214895B37C48307A809A528138A174CB3C7763C730840AD1D69FF6961339A04A88146AE64304B19147F6E27FBF7B5AFA0098C674034AFCC4C2443B75ABCA6D643CD72240B7D5C78D18A2AF9608587C43683BA9F9237A3660BC0F288167D6AB5BC27B9925975A50BB7EB7CEAA9B7EEE97A227A09C32D96CBCBB4E0834671E0C076FE5C9C8812E7123ADFAC49F8AC1C3426413AD5202E09C5FDB98582A5C1D0D50A4AC3A32E6B4361BC42D7D3A4417C0B8D4993DE51BBA1FE58B11B05644812232A7CBD786141E39190D6AB59EA79D0B09677654567BC257AE2B476D412ACCB640B87537C067BCC2052EDE550387A34039F910A9776FDB01B44A5A949CF6C3805C351782CDFB3A5D4504093599C0796833C3D55FEA9042E1A24183285B8AC460EC4C17F7C7A79B9B15D458952BBC8D7508C22ACAA0E5CA8E0F01BD2778725487460293050C22A0812C2D19D1AC9E9313F1B82F95B2191444B2DB66B571A2AD2DB831FFB151AE8147820C42A255CE6A762833A5147C4969AD18C2490928775C092379C1A6AC999BF051AE201E05120F8B82756952B6B8F347B520AE37C218EF9A630169885173C731919468C12C7E2BB34B1797A86789BCF11FAF45BFA120741B64916DC0600D067FC172403158B2F8616F876A9E603147B50971BE00460B480E68AB7697B1CEEFEA53840825F31592A9C1AB579CA4E6071D4DE8A62BB9C068E3BE30C01E6EA7A5CC3387C3C6939FF592F8F9ADF776ACA74318D33935FC831628C414E0112AF7139AFA079E7C704E444599F4906B3B07034638B01C5C89B522249B04153540839A8A01EAEBAB55DA64A4DBCE12F5A71BE4AE7C37B2EED09830041497A577F039CA04CC41EC70C6B4720AA5216D2BD292714510666B5AD7C047A1665973010A728A0498D2469AB89B26B53BEC845426DCA7246A7E8B74C429C574584B442D9A7D8CEA4F1B9959E722BF79A14D75E0203594624518A66CB8BFE161268A288252BBCD7F448B7ABA1ADA59529E12C790C263ADE51A5EE7647E76B4283933093B40EBDBB29C8B7F7C527761E1C11B260CE56A2E985CBBEC3422B76BA2CD4484F5973589081AB5BA001B75841D4730C6A19EF77AA2547A997AC10E310B0FD31A8D5A04B07C43330CA183D2886422DB6C2C502FE202462541CA30F522C66C3000ED83932588D4208A3FA9938BF8A7B204257D145737E94AC6D167DBA344DD0476EB717C55A31C9ABB6C03F82969032A49B01DDAA59DAD883527326CCDEA9FD776827C4A84F2E5B6184C8C23222392574CC5D59F3BE81C8B62386AB095550C13521BBFAED914C2D69F707A5DE6EA174E215C8DA76444BC2ED4686256D80E04FCAFD9DBACF9C904BDF347C00B58581120F557AB4BD3ACAE4169C5AB1854C13294051817564D1DFA367B55B1A1004511198775996A0AEC34FA6155B191A9B1489AB8E99DF5C35B24AB711E051A6B147574A951A67615D36C790CA88641D63E252426F6F162436C0B3EAC8BB952462513454040CD5C3B607F05D0A8362B52AB078CA26DECC06F74B39B5049A55DECA588C86386515EBE513A4FDA34949AA74CD7D801A5A0F65A36DEA426D8BC4A188321FDD70DA24920E847D0A871197C91A41421A544F4E158636F19AE7A0C21B61B1E96B4F80D7BBF3F46F13714AA6095BF961808E27395C6E5B280F285D96BDA1D977112493164A14FBA1368E0"+ False+ , KeyCheckVector+ "ML-KEM-768"+ 126+ "F629720F4168EEBC9772E040CF5885DF694EDFF1A2291133470163774BB709D93331139A60A812C70338C74970B90C38D5C16CA25245BA0847BE1CCFB6F97907971EF3DA8188EA5B765A8F6DA13CE85CB6183CC4C9A9C39EA1085D69C88AFB2CD4011114C47C67CCB73D863FC1C45D0DAC2CC48142F595578BB99D4881587A31C2FF03613AC97F76A90ABEE451C4A471E9E5820BB9A340E378AC3B8084C00BFD309BEFE14279DA8C38F540F94431175AAAA1671E29A0CA7B491DB6DB7A72933F921019CF5B8D0ED64A0803C264EAB0224362F32B4C0B12451919C1C95378BDA6B55225ABFA6C43DA0978CB830978226E68375D40BA94DA10302F888FDE8173CE21B080A86028007D2DB743FB4773ABE679515221C48A78B73014BAF5384601425E165680979C65871C7D3B3A78478E1C772A792C1F1891700068BF50C000AB1804D3D76619F600B24770EE356A807A2ACC07004EC9428443AB3345B7C2E36AA134533A727272B097F21CAF2619CD569C3BD5B431282B7ED2874FEAC4B585326421BC12B8623D724A7F9516A4217458F3E365EC42CD772A0534158FD5247F1BC92685B9AD60B9672F07005769CE4A61773293BD48298CE54397C2435E0DEA4DF39B9AADEBC98437617C0B8664DC87748170E240188068B765E3B9693CAAB1500AEE64BAA023701D5C647A844E1477A665514047537F9741B270BB46914534F84303ACAC915434C4E7B6921D6071875B358D56BABD49501766B863F929D045A71A168B9DE16F23996CA1C1CA630CB59B692DE30A0B1E975429AB136EB020437B9300E696AD868F5A92092278606C31168CF21C549033C1445996887306934EBA76519D021C67D2B893734A873403E9B4A91C7839C5320FBB6BABF551C0B410B8F84CCE1D0C0411B6A9765662E7CB4E5F9378D8E4515D847A950A551422A6779529436C49C97B6A703B9582C06A6B31387A94BA568A31BF6268E4134D1671C42A03949EF10A6380B07A60556343CB61087DBC5505FD9512A814334D46915DB7B858B51F2EF80DE446C5E2EBA473967ABA6627C3F066F3353B7DDBBE9553546849400C63BD3BF7A515700DDC874629525DA56A25355352117B8E25494F62383CD8A5AEA34522917A62E4B04A128A7234A33094448AD99782367810D8B7758E3985E0E3BC62203E367726337A9DA1909C4F1690634AC199407B65B9CD2F83C71B688FF3242929A107FA1383F141569A32C631C097D459381AD730258474CBD87889280DAF5A6E9CF45ED2E78E85B052551745085A02664046ED87667331C2FCBB474CB6289BC85A3E0AAD72A949C9458AE2B9586880BDCA52306CA483B3E711FB3409D4AA7CFCEB7CA212191B378865AA7360C629C7E375689BA16E70176189A3B0120EDA33617EE83C65C2347D8B6696EA3B93C49470570E6FC3275A59A58221753EEAAB3DE687A395C7F68824CEB61CBD2C9C1D641113574258A6C5A37BABB039C1C3042FC1796593E43F19948ECB8ABBF6D56DE00B18FF1B94E3AB9AB560AC5F96A7FBF267CF8B64A6B87DFF8409B6AC26FA06718A302BA81504329A3975E456FA86BCA8BA665BE899C8261CD7094904D97F77B8BBFB63BECFE38F582842D1C304D2094601901EC45B32AB391E263282A6C86802E1BDE88C2624823202460EF0E94F782C9CA4542248540217B462CC4671C6793DB3B83D225791BA89700152B34833969701814E502A206C9230740AA1B35B6DBC9132BCBA84EB7AB9AB06FF0512ECF87821A68C4A2CC3936BA945D6848F685ECD718222BCA9BD4B2754D1AB00B70E7ECB4F6E18C38D214F8FCC28E379846FF268BE758558CB56FDD9A2C3A64AF5B1ACCAFC1E3FC09C1F259E00D36F1F4A8C0B473FAE069AA6DCCD54E5503EC79278D8AA28A43262D80C34F6087790250B471ECA8A31F59B2D5FF59699A41CE8A72BA59368473B9A94B4078C178B0A30018036B4235B2737217D18CBC76AC99F62DB8141888C7393C608015DD9249F82B119EFF1545F355C028CAC0D04B60BD702079151150125FD68439C599F61B1A2EA75733733334BA15E6554A7D68CA83C122C4AE999264296EB60047F481AC84A1D6D128CD9B607BCF3A0FFACC78F7B889B73C6BD186CE776408DF89C2AAA6F78929EC378CC2B5C6D40773AFA21BC45DB2EED041EE0C6849AB677A8C6BEB3806B52C527A13B7D065C82F9D02FB861802B36BD3AE11B6298B5A197B3F5F85A5BCC55FB398DE11165B8CACEE26608E17C538A6BC40742CB2141CC1E15527AFC54AA4B16854A7057F1CC54CC0EB795A110FC59B5C53AB6638C999234D3E113B5D83EFDE9164AE604FA97869F5335EAB2288BC650ECF99A16C461F9146A5D047D41C76286B91DBED5BA6F63B036EA6B6F23C5B0EC193F32869307193590281FC39D6112BD41CA5372847EFCA3C47808C249CC9FA6E062362AAD8D2861B2F72509D827239B1966E7011AEA546537C27EC86390B4A126389B7F556FFF575433B70ED7F57EBACB079BF16D46E42492D98C56C7466184B22879A1BC0714B37A3CA7439948BA920B9CA0D5F8CF62FA0827D3055817BE0317B4F7A29925137268002AE577C6C8C15415F22B2E43C4345121F76C01FB9C159E623FB766AD2591412D9A2880A75D8F356904044E066C4A39381462A2A6A1AC8C4AE6B7942B413730BD32344971928DE4C634CE262C77983722AA03A15308D042A3F4A824B48710D511070331A7CC1517C6024B22A35C2103928EE2AAEE7238F123177647BB5CBB708927CD01E52A21B282CEA82F161520AC8697D637491A48236DF100EFC7A86E41B518B0B8F306B8B82581259B3E4966625C499B25498A1F546A64D286BC1BC88200A9F2241FB907C4CC962844347E58533171A090AC7C86A2D48E908C75CE0A25A28B81C9BB5423390AC9A1447FDB6CF4D68CF7FAB10F416BA3823141A088D0C8ADF928984DD0A5D495A96D51A2925C09B02B9DA2959BCBCB3C629B177A382C75F3A1D6D0A50220B82DF815AF75C9CE4A875E723C0931908C639E5892C06C4526A2103A9EF569B2250608624A8758CFE8D3A3F2DA6B64759F279750C7F199A2303F684A826B72BCBB9A8CC78280D93AC1009B58413085104156D2DA65AA8B669E6466E3F4CDB907BF75761417411268E4AD94610ABCEA560E564D65509B48945CAE43333C4B08363C22DA8477510A895B5B0443F839C67B4220A3C080E29052D8BE57C31858DA73A281508F6615BFCA2AA591A6DB290562FFA88EF2AD586F6FE0BA6643106CC1090EC080571F6AA33F728780246BFE2E3074BAD30297D6125220FF1E66C629986EA940BEE294B4FFC62908F6CAE073F555DF3658E58EAF77AC1D1737B49BE4B697E88F76D4326179BCC5120FE458C51C"+ True+ , KeyCheckVector+ "ML-KEM-768"+ 128+ "43A825CEC4A9FAD91BE6F1A402A2C6AE197595064374C62B7BEB6D8DC9C9B69C386AFC6D1615BD7F9461D7C8516B5248F7FBC8FF153EBA1A938C2CA4526A4B98A3C267D758DBEABA0741B42908149148BC442C8096BB12064B031089C551603604298DD786529843A098A67FD2B4315F265F794A2508984538C319F5E7AFB83CB69DA043D2031BD2DC2D940C899A10897033587746933D5C312A171DA98BCBCB777A22F300333A2A61535B40819706240173941BDAA3866DC03C3238465505A7C57CAA5A1685656265EC2B1DE6946C4A8B299A77511D5A1FF2475349FC2DF4E6765939B7258C2CA2C7213AC357A9E72F7E50032AC59621C10945E1B04376C5A46B84B2E7C838B20B0B3324072B0780C3814E59AF30BC9C22121F7519AE9BB2B8820602953A61FF3111151143D37C640E020549985ED863B202D7C0C7D067B5C184960574E7C8991832539BD34A68A816E2E8AE8F8BB596BCBC7B044D6EF16D05915AF12BBABAA069EC177174333780140D8949AD85B0C6CAD7679421A146F601F9A5AA56212A6B97AE81264CB8B167277347C6619B89360CE6297F5C938AFE702AA1097A70267057C69AA2F6662473447112A9AD22B15B5C6A90D28754434A07F16B79EABD129298AAA5AF6F913EFFABCF2E1886433357939053377C0DE4FA167468CE7AA7CD4700C2BEE14744A432FEC92A6D84304A78C697E1660A848D00B722EB202EB0E623B17C68E797B732C491800BC78D126EFE34207A9251D6B6316A457C1BF55AA0CC78B9D30C0C4382D901881E7C52B7E85D12830971848D95C36D9DB18FADB0C8946051C12BA02CBCC557170531F99FDDA5B7199B41D3074746B968EBEC9F2DC07D1E3B9ADF6B050F148D19030C04705D999130A9F36B8E5C4B66ACA585C052B21396E0C03D49A7A661A8C77A29508BD32D4A515D9C515DC259B5D965349608A1DD942837C09677D055160C6E3C8B700BF8155FE22186730E6A8AA97A7788DA5523A98C8283C61331052E6D80660E8821C8C4C19A7888C37663413C05B9B11C63455E8AE51DECA93710040001475E1168CA63434451A46DD946CCD8589FDC4A5CA161064435528BFA2617A8669CD9C2F393B26CDA492B1337B90484B1610A6E3CACE044106CBC5C5798B7178CC3D62B166D6722F6AC9720E918D0E0275A63172BC76ADAC957896653D7BB3380AC0EF99786478CB457FC80A28C8656691D92A3AC68F721182CA376E5B471B1C9D6667EE6470721940E63E49C0F41216702B3327924E70A03BFB5B278B0675DEC9057577581D8B0A2A2379C9861C59578F3969152D689E9AB3673F66F1E38CE6EA938D59C405B5C6A330589D6559022D373E9284F57A727749A44B1277F7FE45200CD0B31843B16D759B997973A8AA603DC61D3C62850C78EA40C1FA3DCCD6C84915FA987DBE838D5EC018EC5B527FC2ADAC0753453CA90B62E91071BC5026751718253B0BC84D3C74FBC71012224A59776DADB39E3A12D0DFB2F83D125AA0754990790BEF598CAEC547B845E16779399E827FC6B1756C29F29E91E572A3F05F53A81314C77BCA26C042350F75B42B9A629C1C45328BE769A50851A8EB2263A1425C11E59ADA0472F34526696053843BCA6288A9C9AFACC9225A08FE598C1CA31C7C7B5B3C04D8F562263666B723AB1B47927CDB7AFC704B422911952F48069F04ADBAC60867A0A8C683F5AA30D7D7CAD28A845811552E17C34BD297F19B9BAAD72A45858924EB95F3D29CD95269C55D7560A41B85A1204681049CAD04F08025C5DBC2D9D72725D67CC99413A898751011287F2C74485A6A7159878F66040CF859BABD901A21A57D4FB865BE940F75295CBE3283906C45C2C388B3A5C47BC56C7D56C4DB6ADFE8065EC7AC6781CBDA9BB1A89D211C6C68B9A4202033B109FC99B3DF191EBB523ACB34247379B5935CE4A8A56961149B7C918C2CB801B148817151A317370A7809A33068C89B4B1AC831103391AF0BA3463D2A45EE35ABE4A06753C8D2B862ED1B04047A21F8829421AE773EDFA753032271BE67DC4FCC8EC73338F295C8A67C3D5B2A5A36C6E1F65386341CB7042B504EB3B957944438A05A36CA9A1E40C0D4600F5543A15B99DBA91222FE5769F07318315B58B5CB1DB6596D93C8B6B704B12C1BBBAE6145329B92E3A966B2C7C38643A0A2C38D4824642AB90C654C57CFA96819B3AF81034F9CC6582F477E8A58342A36A6AC776FFF183FEB03D6B34CEE9211F96A77E05FCBAA0783E2F95301AB99D8356411CB6196BA2BFE6476B7719B50F2AB2BC3C92E4FA0A77121D3967CFAF9542084249557399A8EB8E7D79315F82788A614E59807C5BC1A3DD8C09980C2FB33295A31B4728EA218C3929347B73C35A5A2686B4584715017CABD0E5A35C8C87D5332C937583F4242A385A7FB67AAA77815137AC991B4656D8D93619071B57836A928326144027483799396238984C73D966055861AE19F76424999E213885D9D967C0CC828A307376F41DE38B7B57DA66355677D0FCBC4BB15E35413D7C3C35B1A9A389CBC4120415014135B135641F060B14C8946456A603181227121C5925C3A0C70210FB03FDACA7DBB73AF14AB794800A7D2BB27D8B2DCB9CB9C356A0C4E38B815CA3910036DC9748045B403968486EBA5FC2A67947B92D4F247D6DC686DC28C2E76648BD89201E273121C6B286C9B3B686406EB69FC5C458E8C7AFDB00C873E4AEDD307904600EA46981816C5B87B10260E165358147B3827C380A8AA849310B9389CAF9145711BC22E74BAF280D56991D346A1645A12E37FACC66C12700DC39CB8621B6C5C001A6181B42B2527173FFD0A91CD15719839F2703A68DC82A7BA083A051834D393F18F01C897248C1649897BBA2BEB8B0DEE66A39AB4164483041B7333EFC9D54352BB4659CCE581D447C77263C5E4DF74F5727B8392ABF2503739C5C98CD5C3CC1E562F41A1FE0664E07C17A2636959A51B17EB1919D1942CFCAA3A32A554D499B1205A60676033BCBCA8056978F98293A683E80E06D6E261F6185C8861312C642C04BD93962912EAF56672A019163DC91F0D844ADF88212F51F43601EB99B043BAAA31A9279489BB0B629A137187317802F571391C88B2E5E757A17D63EB0374F40B9A918D742A9851420B65355E3C441E048A9D51857B64E411AA217575ED8957A9F0247A7F071289BBAEDC70D6CC24E1F273646C0B9832B6DC4861638CCA434D6341C8645F5E561187AB9A478B65BF5A3EE96D7846D34AC145DB03B70BEF620D782AEF1A9D2983B3BB086D37A486C5C020B73C2544B1EA91CE5CD5AA22348DCCAF9B62F405960918CB2C22EDF7DF58216FD7F1903AEB2B10CFEFFE0ED3EF73B980E6D0C500E9898103BDE8C9D5B"+ False+ , KeyCheckVector+ "ML-KEM-1024"+ 148+ "31C48008D52C5365473A1512E7813452784A12410BAA3313EC0C0B4769C6E52CC31776074B2282E035B611AA1AC8DA6DFE63C0A1941B473644162288E53A5EC9C59037999DB3E8C6C7857CA57C42C7501182740E7C041A7404620A9C76F0F611D010703E720CB1004C20E27274B44C0CC3967C8566BB07442A40CD670C0C6869AECD22A57DA32059C19256CABF02F16AF25A3E2C876CB575C12B13871A37B61A00176885B08B158C6B86C02A15539DC264A951C9FAFB2F26DC1C65F3040EAABBA5186A2315944E4B7AE93489E896BCF8F677FB7AA24A180BFACCC19F80678EFC64F8638280B64E1FCBBC4E2672475C5B4F4689E2E6A0B16B262214BE03138364341E774466C5F52C14A5645CDBC5691C9C4D7329C596089EA8B10128BC01E8AB31A56708CA7185FAC4E8E68EF8B8096E6351E19AB2FB2A7E0EB35282198B8BCB3783C05F71B65FE7F389AC166D3E09AB767A14B3575153801A7B38C3F1B085FB28777D326954A66FDE72CCAD43C92E971E2365563F987669E84CDFD43EF9312C5458BEDF9C5EDC757F01365144BA2F4DB954FB2A6168B46F534A0277CC00390B81AB07164A14A830B69ADDB5011D37BE8F7332CBB5AC3E1C042E4373E56B6180459785F08F6311405348990FE907C6F9213321CE23591ECA58180384CEB5B876FC46BE59C559F77AAC21D22BF384901430CB5FF24D4CC2B4AAA0B9A95A4F2EFA60D396040D4315C83911220947A9E0CA7D2C081CC6810370602FE38B1AA9711A87B4331B7F0D4576D1F20A16F0B6EDE5767356419739777C5380FA0411C1052581F412E0C050AB9393D997BEA1E51AE6E32539795E66C46D9C5A04A5B56A8EF808D8AB8CA7D82609E727CC1029FD8002E4E551A7CB05DB27841BDC3B634922B1000D4395AD76039A80745EE72A4A6FA12BEDE61A35D58AE7752B3924CAD762351E598C10354574532583C07B82B4BFACF88C18D12674EB1F3F28CF25DA43BB6541CA6691579651828372C90067E379A982E623BFD062087138D31957F05136ADE1173E0390A62B07EC061918E34089543B9D1C4170F40B1A562C6A11402CA00A3715A41697BF5D8C2E570C5F6AC89D04698F9241CD135C2FA69771A3B9CF1BB793D1E6C945C97C7665658753C31A00778F72C89E30BC2AF746AD23AC21B40E299B57B5D44E55335EDC0A3D1EA42D8C7855166B4CD0229F6CB2A4F9435F412137AA904BCD0ACCEA09B8ED10C9922594BD756E70873B1AC52A575C1210D0A4CD80AB11B7200BC373F6C208D725605207C04D8492F8C9C5E6C1CFF023AE8BA547BE1743582044E5D0C4F1F20534A0A62717AD60D91A6C7A03BCE2721B7503E75581A23BBD8F1848C00AB5509466F144B8FFD5C5EC1370EC8B7C960755A645C4CF23BF4CE9A19E3134D2E5830706910DA7B0CD9AB92D145F2591CA2DC982E6E66716B7754A82BB335709A5249F88BAA71C895BFF403340CBCF70768CAA563F5E9348F4B3C6F29B4C77141FC5D52AA532287876590B15AAFDA8289B9C8FD4EBB0EF8944BB99A0CF83661D78727854B0950A29715A4AA8732B65A08E6BC81F123358CBC3C2E4E93FD2307895597D6D66B16ED139E8B2B260B63E14720BDEA963F8B537A85A1A455BB176D310F9855F6AB019A4D919EB1453C8CA1028260C64513584108DF2D56B6451172B674270744328CACE2E167359FB594106630C887973B35776EC50FD030641312A623908DEC6C81FA573D13C475981ABC5D30F77D4B74999209F4A565C1233E3CB702897379D39B8B68A43C4199D7F64889281BCDFA47ECDB1312D981CB9AB8AAD764AB6EB88216CAE97D93D75E3CF1E654A21209246023B42376344D909C651BA92A4A9CF1CB03A522632B41EC27C8F180CD05A230A2F60A12110C8D832372EF1A1A3BA1769FB49888728BD080DC2E476971756A4E8AAD5C18AA807D0D708714CAB2BF4272C458C20DAE8B089A60945B4B5C5697F31BC4C4DFB1494830323D962C8520A6F7C6C11F8613BE83C22B362A352B47B2B0838392A2DC4745068211E238DC385211BF048ECB2478946C9C2B837CBD0604FB241FDC0445008013DA6480B826EF484AF939C4A399101D380CB55C17E64B081C201D016261407A6C95C570E7BB08C79BA4BBBB92260388342789E7D7784A0E80E602A7EFD97CF79202F38E8A1C0595DD59C97B7B559D344517FD472C8B673FE5012CA935E6DD59E70614ED7C86244132F8771AD2D1A9A884570F40AB82317BBFCC5879844A6C0C757236C94E8D6923906C065B675B54053176B4C6AD2AAB8A5AEC323CCD68A5AF8DB0AE94BB4D431CA1EF198851BBFF64B4EF96C6D0EA045F82C9B7D48337D2A3F1EA7185C2881A4E267DFF8407748C03FCAC286F302A16173128396D76A9ACB217F29353AA87001EA9663D8F48B1E4B0D9AD6CBF81A6B600B763B75493577B85694AF8FA2729323CF5341961F423770217F34466A9115929447AECFD741E39C6796C5A132E831AF3CC2B70A434524C764C9C783EA1D4F269BA091575139C092D91BC46C4402CA0C65B29588B8A34E7C5744FB38189279A507CC9A810588F8166F95833183618AE2470BB25AE12B330AECB4F296B11E8C0EE2E34F12E2ACD0C891CC136D033A17E218A7FFAA7D96807267B083C1CAA8E1D162283AAFD98C5AC1657883CCCDA04A9AC8428B5EA6BFA4B04336D6A6D222B79714CC9696C8CD62B286C04F7AF69F0FDB7ADFC999DA613BC8F6A156161C2E135E7656938EF7C80489B0DFB7AEE086C9E66446EF1173569A8A7E23A0B82B92BC50250CF577236199322B783D6B8C5AB82C9C378716B231C4850ACFCC131F03CA51055A0030CFDCA874CA08BB6E3462FE517DD5F282EC0C5DAEC64414C4534C264D68DB5C2085CA25644059B2AFFCA64BB1D8762A293D42B79E4A9BAAE8D73E1012C114702D2BD19629D646B5300B0500167F3C9B1A7145FCD95BC712BBEC42731D321D7DC8C4C361C271DB1863759B147BA14A524690625823DB8772CA0E2A45AB7C734A1FF0951156A305D70B458AB422A6557AB423641266464A88799987B72572E60A79D0732C84049683310EE475B695556C4A65615651734F56A90D25178D604CD058230E3338FA94003FFC48AE600D2660212D648FA8062B38182006030857882B26E62BF8A23CD5D275B5517F4EA736A298402F155B2141A3AA8A74E5F69678185EB4334FBC380980E850A1399AFA0C4EDB172EFE326BCAFAA3A0DCC07575455C084442E2B5D0105C39B234B409A5BDA7C33C2B5D2994B7CFB90CFA23BA2A65C28E255F0BA3728671AB9E410642D1812D0969C88C721D867E36B8B4808CBD1898B05B563FFC23A306136100975AED0A0DBC1AA8E823C7FC402821F65E31972C3FDCCC96F55B16C3AD48BC4A9CAA05B4C0CB7E863A47296D3C727D7B3B90491364430CCC4756A2203150958770FE354E0F8803D1A64E75A309F7EA01260345AA1C3E5FFB863BAA4B0E5768F20B7D906766F8B86ADD6066CC485FA5C9141D16039E5562E433B511092C1D18C50F7186642152C6986100BA728C136DC28B237E2C8347B28E3FB07C0AE48A42F955476340BD5CBC145860CA6449C597920575AC49C0317F2926BEB2BA7396BD73864B86490E1EBC8A03475A9FF082A7F967F509C8F8BB0BFD4C98174B0B36D06CAC5789919C02B3C0162AD4C4DF639207416BBB5071E5F692668B5D40AC2131988535C68E68C227810C0B68E1B72103B61D98A32A7A05AFE1BB737627955A2276921703A5B0FDE38E415C4E752784FCBBC432A53BC4FC0DB52A46C3D0C3014041C870C00041C183D21D5BF277ADBCAA73D6241CEB018E5550A8E24FA5CAC9310B6735EB5D245239F35B06E803697186680B6AA8BA1730B4CBA23E105565F37B2B16070C3B0E77A965CF577F3C3A82A5192C4C3BB52373C54034B872604ED4A833F9E18E389A4BCCCC6C5991CB49F097771CABD8AA6D53020875080E91D198F93C2EFD276C86586A9E53AC1AA119851C06BE12127AAC3D662AB5325B729F38CAC505A3D0E135CD9A38A8069A7A2A76C5B39B30C7AAACD666EB061093B5C41E720B580A2914552AF509AE2035329D28050C5016AA06AE5EC62E17D84C9024494049816AB32DCF204015A092AEF05857794861175B19A72C9687B6E9EA385963BE3E6A197627359728BF47D4CEDA32779EC51B2F6BC8287241C7B39474805B5AD13DB6C47627AA40F91CCB6CD062DDC58FEA6A22EC98A65F73A0054B3E211ACA6904BC01A9A04B042AC128910B793448A98D19EC1A48DCC451B00F66629AF4DBC2C24101F9480C390226F7B184FF307BBA63B6A1D5B1FA4281AF4C358052CDFDE37B07B2C7627429859EFC72C31BFCA310E4A63E25B10BD4A57CA6A2611E141A8C0CFDE7A700EA3B849017A99F79273DC7FC7702D269197D22112996A9AAAD6803C7CF2AC8132D84A9B5016EFD34D64543E136993D98D12FDA"+ True+ , KeyCheckVector+ "ML-KEM-1024"+ 146+ "B94C8F189B7CEC12B6981C64BE0509E6A2C6F687445732862C984690FACF0AD6817C863F4AA45F9D87C0ED432EB1318354A60BA8EA823E9838FC1B08A9D7C5569976BC624378882CAD589586C95B27C6448819AD8F59A10B60B645BA69C9479F604747D2DA20827B6255AC01CDEB71F5B157729872CCB5A1DDDC715D0768B104AFBF279024366E92954FF45A06DE32C87818033267611B2B9F20A2246DF979FDE2C595929910803A2EB8B4193B029FD51594015A2CB48DEDE90D8913B118F50C9CE4AD1A6B3FDA8B6EECE6673EFB3F7B229CAC7B5B23E86ABCBC2E58E5CACE626FFE040D2EC83A4D457EE014B780727474E51B93A71FEF1545D3C913384A6375E62AC54404809C2BA3245678014D636A435AE50FEB45124AF16518851F2B121FBB982170E5AB454B53AD8A9D3802A400D13A8A724A8A1B9F6F49C30D059695D520BA73A370FAB8F042A3714AC400714F66328E0CD473E15B71AA62773637BDC4B29BBA1A07EE15BC73839B7F08B9EBD4CB36D2427E3CCC92266F19EBA7B6737FCC58848BF0AF174613A4316D7933651FB39625496C26CA9705E48BA105046E8A226DB181543A0C26B455482618460680F522C92F3549F0B87A14915D0FF15C6BA5862C556122311A4DDB07B5248C3ED995F4AC80D015B3E6A71C97226BA61C189A460D6D368F587825EC51C50669269B8B8872B9933FC1BAB5472D90AC8D36E453242826F7A7319FC8C262795947C2810104B98023AC84E854BA59A3319C6EE6905E36D0C3B0B41DFB393CE378046FE292097C528EBB72B91B20C637C531D393BFC43D54692AE4D281E9E637137463A0701D286AA89ED464499A1458913C586A2FAB90506B8097E061C0B710B68EBCACD8353C7CCC0AD37A0EEF32A4CD1232B69AAFCE9AA5620B953BE52EC7594D68D57976EB1B38CC677885840A463C72C3CF17445AE4202AEFFA4A7EE0275134162C4BC3A82BAC03864D39C9406DF72EE60218D83C0E0EDC067A1592D93A8FA32B84CA85BC51D15BFFBA17F7D740143A068879362DE00669A0CF7EF70C604299DC3CB52F023E56F544E3E35CFC9739A64A9392C0AD1D927D62F09D96418F3D926E227C26E02CB1E2C10B89E373A1D515D39C18F259212E5C14A7678A3A761F18E488619C46D2A21590270FE511ACB7C3BB45F9C699E63596DA08827458142791893070D6A130182430665A3152F62328DC4C2294C874F293331604298675A4C2C31F79ADEF67C85329082DDBB222DA4DD0C11B6FD8C9234ACAF4D40679C82BD0479EEA17CCFFD209409C5C2AF148A87C006B65B39436BA5CC6440812ABF7B08364764BEA966982979EB55684D1433E71A471FFF73AD423554514C144CC4377BA07197CA72EA6CC69E215488CB91C7980EF74CBB5E780931A34CDE964DBF21E3BC266556281897530361BB3F8B82CB5415FFFA1CF8E32B12E2124DCE774D6D692E0F803CA357536101CA5944D4E1A063679B0937ABE6BA20A8CCC75FDC03AD4BA2F61522CB8F565B05232E2FC104AE09EA156898E8B2D3D0A7FFBF195E9D1A623E73AF87181B4D932AB259B7D2B3814E721D499950D184E56F109CB1106E9E1062888B6D3185DC30AAA952946C81B6FB83B507773877F7C84EF142C235241E2901D72CC8A6F481A19721A9E877BE36369B4F614E598698083906061074D6055DF372510C543458CCBD9F10881D209176396EDDCB8118B3E5A22A989A56C67912721B290868A75F6D820802A0FCBF00AEE135139D8A1868745FDC22D06F05748D323875187FF298E2D37240AD93EF7A8A3CB462665209DE6F1990828830A65446E2A4C30958BE3AA2D4F192D673A8E99497084544122A56BE4F6BF6BD72E98894E5B696A6A19CCA2C1C79FA976384C3670B8465C2ACBDB46C1B5881F5BFA3480675F83F5535606CB393934A6DBA30469282A3127B5B77BD7709350091FFF453CF8F76370F847ABA2B78131093FE4B699D342BE220E085458E150B081B81FCDD35CDFAC605FBC4F9757C24D7AC542E90EA4E9255B5B29F78479AB36AF1BD691D005AB401269BC9395C9986ED9A2017E3818B1F87DF9846F1EA1A18EB681A0E561F9F379214A6CE44334CDC5C4D489436A5850B272499D32165D10CC8B6577965C56BB55034181BCD7C81488BB9FB1DC8D13C1536870BFD4868F136CBC490B62C79741B273ABA970742D7012AA2165C297A844AB4C27414360B5CDF0E29D31B5CD802CA3B8149B00DA23F7E213A736017C2B9D8846C12E4A5DB12B57E83168D8D6BEF82CBA020116758CB06BF694FA186000458BDA4358C537ADBFAC277FEA5C02E1C8B79998AF6C4F1A6CA47085022CA7A6CAAB0780254FD8C0CB78223D86B8656E56231E9A28F0612747E2075E25B602895FE7C498197A62541420CF839C31B347F478BCED325336BCAEBEE986D52C7AA5B5CA56723F1641CA3480A040D1267A407FE3E92D2C8023DE00846AD79B0AA89CC1F99965E88DBDE06B66415E36D2120E2400F435C3145B57EB28C62C248FEE401525FB3BE6491E02D4156EA38C819AC59D55CAAE14A7D1D3B9E2CB57B7BB58704A8C2D239E622A0D15A589916517768387E21408E0E81A01216A5D4A636686321B919E7019A1697B886E81A376391EDAD6C5E8044F6BAC63A84B02104221F19AA8E3468579A0A156897FEFA66A11A37FDF46088EA586AF8804A9854884597A7327A733315E020693DBC62BF4A33E1DEB74E91437EA30670E7430B7E1596611A10C3595C3AA14F6337B55CA5D0B953068D52EAEB7415BF18E88D7905945CCF3140A75263A25D931F802702EEB40A944BA6AF03FA2509ACFE753EC60CDB0A845CA5131EAD4401E2C8D0A32306FFCC7DD189FA367AB6A4A9A6DD1806BA70AA4440BC80B99D0C91C7931871B436F94B39B1D04448D18085C5AA946E642CE04143670BF7F8514A0797C6FA71BE4BB1416A0B3248075251CB1A5B79E36862AF4947C76D82CE8363B89139808C4CBD14B3C4F1700160B61E5648FB954C25009BACFA265F1A0564673AF000120D93A2D4E211CAACB1D69713D4E183268803163C859A71B92D9117FF807089F47940C57955E3343506762B56272AEE182C719A5478444F29513C75AB44E3C4F0658111473CAB33933C5E2999BD01C5DCB71F1F82F6CA57B9EB6376C388C20064944E3AA803574FD816C119A312140117BA492A3D3204541335C082CAD440F0B22BC55A65E2B852E3EB1042ED046E8A9B755773425C920D2F4260CB29D1DB76A879B36283BCFA8598F799302DA5C14C5AC03230A338EE3C8497899A92CBE645B3D9E91499F441FAA5211D87162E76262BD795409559B294A0453F028A3C737962AA6B9969D8B45A934C7CF22FC2EF51C661541202085554BBB5AA2B207D161A75B7831000B8A532C76C1D01FF3A359212BBF79137B42839A217B009702192A92470F4B3DA4A9397E51CEB6F883C5168DF122076CA783C0C2BBBF572680E5268CFBA41E8024EC58A133A8934C46553302753F39AE7E4B9311A984CDC68360293D94D8A6D857318D5B257291685C5068311839A489485C64AFB7430376108502279FD47042B7216E8F7616EBB408CE4C3876548AFDA6B31596C21EF096B92782D5D745885106C3770E1E04C1DA2A027912137CBA3FFD2632B808AD0BFA8F7D2C9AB4E24245C8131AC5167684190583AD527A25BD8369FD747F29528C61287EEED97D6CCBBF791AC8A90AB9A5E19F57482A89E3A9DE9C00CCBB9884C80F7CC7680B654FE1A3CA5A160431A1C42CA445AF7A4D60CCA7F6A27A7B919C6D70B030025D86341943897296D61B0C8414A696C95F70337C54BDD04C08A3FBC5F9835B9A080496A1308E03A88E76973C85A9DA0C0340B95C29228DE1FB3D4CD96A9204848A92A3C912B0BD8817F6212242D06CFE803E91B656661A51A6A1841DB00F1F8A02518A971F61199F93712C306CB944C52B469B04E76389E71543896A410979009A2A9F09CC58C0868294A72EA8A0FF96A43F543E5E96AA1934545F755382859F962615B8B546D5609B59157DB4CA4D5FB87366E03ED93A2D3E223590A3B9F03C95327A35D65C5B957546D3F0745A31A3AA012719340F13992D8BACB18153A83D59295BC497EFC5C875C4542B430285F54FE245558538B15CFC03B69889E2EA1B1E11CA89D165C9ECC1278C2D5E878C82C13D5AFB7C2DEA9D71437924D97D698180C0B626E683A2AD3BCA5652C9E3021ADF322BBDCB6D9C762D0A12270505A811E3C27FB16A9493047755163DF828E0C8B0CAB4B32EF17C784B0BC1A1922967112122792B20A2917A7FC46016556611416675AB9C5AFF696F80D48B659ACCE23FF815C80547A9762CC0D85C3F2501DD291BA56692474A2E13B3EEB40A73D7BA5E222A09506341F00EAA4E856209AC3525E24569BEA949B03B755A36972911F56028D4EA3F21B1537B186287A191FA400689D618C2C2C20CC091EF65"+ False+ ]