crypton 2.1.10 → 2.2.0
raw patch · 65 files changed
+14452/−894 lines, 65 filesdep +unixdep ~bytestringPVP ok
version bump matches the API change (PVP)
Dependencies added: unix
Dependency ranges changed: bytestring
API changes (from Hackage documentation)
- Crypto.System.CPU: AESNI :: ProcessorOption
- Crypto.System.CPU: PCLMUL :: ProcessorOption
- Crypto.System.CPU: RDRAND :: ProcessorOption
- Crypto.System.CPU: instance Data.Data.Data Crypto.System.CPU.ProcessorOption
- Crypto.System.CPU: instance GHC.Enum.Enum Crypto.System.CPU.ProcessorOption
+ Crypto.Random: EntropyShort :: Int -> Int -> EntropyError
+ Crypto.Random: EntropySourceLost :: String -> EntropyError
+ Crypto.Random: NoEntropySource :: EntropyError
+ Crypto.Random: data EntropyError
+ Crypto.Random.Entropy: EntropyShort :: Int -> Int -> EntropyError
+ Crypto.Random.Entropy: EntropySourceLost :: String -> EntropyError
+ Crypto.Random.Entropy: NoEntropySource :: EntropyError
+ Crypto.Random.Entropy: data EntropyError
+ Crypto.Random.Entropy.Unsafe: EntropyShort :: Int -> Int -> EntropyError
+ Crypto.Random.Entropy.Unsafe: EntropySourceLost :: String -> EntropyError
+ Crypto.Random.Entropy.Unsafe: NoEntropySource :: EntropyError
+ Crypto.Random.Entropy.Unsafe: data EntropyError
+ Crypto.System.CPU: hasAESAcceleration :: Bool
+ Crypto.System.CPU: hasGHASHAcceleration :: Bool
+ Crypto.System.CPU: instance GHC.Classes.Ord Crypto.System.CPU.ProcessorOption
+ Crypto.System.CPU: pattern ADX :: ProcessorOption
+ Crypto.System.CPU: pattern AESNI :: ProcessorOption
+ Crypto.System.CPU: pattern ARMAES :: ProcessorOption
+ Crypto.System.CPU: pattern ARMPMULL :: ProcessorOption
+ Crypto.System.CPU: pattern ARMSHA1 :: ProcessorOption
+ Crypto.System.CPU: pattern ARMSHA2 :: ProcessorOption
+ Crypto.System.CPU: pattern ARMSHA512 :: ProcessorOption
+ Crypto.System.CPU: pattern AVX :: ProcessorOption
+ Crypto.System.CPU: pattern AVX2 :: ProcessorOption
+ Crypto.System.CPU: pattern MOVBE :: ProcessorOption
+ Crypto.System.CPU: pattern NEON :: ProcessorOption
+ Crypto.System.CPU: pattern PCLMUL :: ProcessorOption
+ Crypto.System.CPU: pattern PPCAES :: ProcessorOption
+ Crypto.System.CPU: pattern PPCVPMSUM :: ProcessorOption
+ Crypto.System.CPU: pattern SHANI :: ProcessorOption
+ Crypto.System.CPU: pattern SSSE3 :: ProcessorOption
+ Crypto.System.CPU: pattern VAES :: ProcessorOption
+ Crypto.System.CPU: pattern VAES512 :: ProcessorOption
Files
- CHANGELOG.md +42/−0
- Crypto/PubKey/RSA/PKCS15.hs +23/−0
- Crypto/Random.hs +139/−12
- Crypto/Random/Entropy.hs +25/−4
- Crypto/Random/Entropy/Backend.hs +18/−6
- Crypto/Random/Entropy/RDRand.hs +0/−39
- Crypto/Random/Entropy/Source.hs +24/−0
- Crypto/Random/Entropy/SysRandom.hs +50/−0
- Crypto/Random/Entropy/Unix.hs +1/−1
- Crypto/Random/Entropy/Unsafe.hs +8/−6
- Crypto/Random/Entropy/Windows.hs +5/−2
- Crypto/Random/SysDRG.hs +82/−0
- Crypto/Random/Types.hs +7/−1
- Crypto/System/CPU.hs +185/−41
- Crypto/Tutorial.hs +262/−47
- README.md +40/−19
- cbits/aes/generic.c +141/−370
- cbits/aes/generic.h +32/−0
- cbits/aes/gf.c +24/−98
- cbits/aes/ppc8.c +287/−0
- cbits/aes/ppc8.h +29/−0
- cbits/asm/README.md +3/−1
- cbits/asm/aesp8-ppc-linux64le.S +3658/−0
- cbits/asm/aesp8-ppc.pl +3801/−0
- cbits/asm/generate.sh +23/−0
- cbits/asm/ghashp8-ppc-linux64le.S +575/−0
- cbits/asm/ghashp8-ppc.pl +664/−0
- cbits/asm/ppc-xlate.pl +352/−0
- cbits/bearssl/LICENSE +21/−0
- cbits/bearssl/README.md +57/−0
- cbits/bearssl/VERSION +1/−0
- cbits/bearssl/aes_ct64.c +398/−0
- cbits/bearssl/aes_ct64_dec.c +159/−0
- cbits/bearssl/aes_ct64_enc.c +115/−0
- cbits/bearssl/dec32le.c +38/−0
- cbits/bearssl/ghash_ctmul64.c +154/−0
- cbits/bearssl/import.sh +34/−0
- cbits/bearssl/inner.h +92/−0
- cbits/crypton_aes.c +283/−32
- cbits/crypton_armv8_target.h +13/−1
- cbits/crypton_cpu.c +187/−10
- cbits/crypton_cpu.h +40/−1
- cbits/crypton_rdrand.c +0/−126
- cbits/crypton_sysdrg.c +388/−0
- cbits/crypton_sysrandom.c +119/−0
- cbits/tests/bearssl_diff.c +432/−0
- cbits/tests/ct/README +24/−20
- cbits/tests/ct/ct_aes.c +11/−0
- cbits/tests/ct/known.txt +0/−9
- cbits/tests/ct/run.sh +84/−27
- cbits/tests/endian/endian.c +130/−0
- cbits/tests/endian/run.sh +14/−2
- cbits/tests/endian/vectors.txt +129/−0
- cbits/tests/fuzz/run.sh +4/−1
- cbits/tests/ppc8_diff.c +254/−0
- cbits/tests/sysdrg/run.sh +22/−0
- cbits/tests/sysdrg/sysdrg.c +140/−0
- crypton.cabal +120/−16
- tests/EntropySpec.hs +58/−0
- tests/PubKey/RSASpec.hs +4/−0
- tests/RuntimeSpec.hs +67/−2
- tests/SysDRGSpec.hs +100/−0
- tests/forkprocess/ForkProcess.hs +81/−0
- tests/tutorial/extract.awk +53/−0
- tests/tutorial/run.sh +126/−0
CHANGELOG.md view
@@ -1,5 +1,47 @@ # CHANGELOG for crypton +## 2.2.0++`ProcessorOption` is no longer a sum of constructors. It is a number with+pattern synonyms for the names it knows, so it has `Ord` now and has lost+`Enum` and `Data`, and a `case` over the names needs a catch-all. `RDRAND`+is gone from it, and with it the `support_rdrand` flag. An AArch64 machine+reports `ARMAES` and `ARMPMULL` where it used to report `AESNI` and `PCLMUL`.++Two more things change under a caller without the type checker saying so:+`MonadRandom IO` draws from a per-thread generator rather than reading the+system on every call, and a missing or failing entropy source raises+`EntropyError` rather than an `ErrorCall` or an `IOException`.++* feat: take entropy from the kernel without a file descriptor+ [#300](https://github.com/kazu-yamamoto/crypton/pull/300)+* feat: a ChaCha20 generator per operating system thread+ [#300](https://github.com/kazu-yamamoto/crypton/pull/300)+* feat: MonadRandom IO goes through the per-thread generator+ [#300](https://github.com/kazu-yamamoto/crypton/pull/300)+* fix: say that the system gave no entropy, rather than error+ [#300](https://github.com/kazu-yamamoto/crypton/pull/300)+* fix: stop using RDRAND at all, since it cannot add anything+ [#300](https://github.com/kazu-yamamoto/crypton/pull/300)+* feat: name every processor feature crypton dispatches on, not three of them+ [#302](https://github.com/kazu-yamamoto/crypton/pull/302)+* fix: an AArch64 machine no longer reports AESNI and PCLMUL+ [#302](https://github.com/kazu-yamamoto/crypton/pull/302)+* chore(rsa): deprecate `sign`, `signSafer` and `verify` in favour of them+ [#305](https://github.com/kazu-yamamoto/crypton/pull/305)+* fix: compute the portable AES and GHASH without a table+ [#308](https://github.com/kazu-yamamoto/crypton/pull/308)+* feat: use the AES instructions on 32-bit ARM+ [#309](https://github.com/kazu-yamamoto/crypton/pull/309)+* test: skip the accelerated constant-time driver where the processor has no instructions+ [#310](https://github.com/kazu-yamamoto/crypton/pull/310)+* feat: use the POWER8 AES and GHASH instructions on ppc64le+ [#311](https://github.com/kazu-yamamoto/crypton/pull/311)+* test: the big-endian harness now asks about AES+ [#312](https://github.com/kazu-yamamoto/crypton/pull/312)+* doc(tutorial): compile the examples, and teach the things people come for+ [#313](https://github.com/kazu-yamamoto/crypton/pull/313)+ ## 2.1.10 * build: ask the processor for AES-NI on every x86 system, not four of them
Crypto/PubKey/RSA/PKCS15.hs view
@@ -512,6 +512,12 @@ -- | sign message using private key, a hash and its ASN1 description --+-- __Deprecated.__ Use 'signDigest', or 'signDigestInfo' where this would+-- have been given 'Nothing'. The @Maybe hashAlg@ argument says both \"hash+-- it with this\" and \"do not hash it\", and down the first path the value+-- inside the 'Just' is never read: it is there to fix the type, which+-- leaves a caller that has only a type with nothing to pass.+-- -- The blinder is optional and 'Nothing' is accepted, but see t'Blinder' for -- what it covers and when leaving it out is a decision rather than a default. -- 'signSafer' generates one for you.@@ -527,8 +533,14 @@ -- ^ message to sign -> Either Error ByteString sign blinder hashDescr pk m = dp blinder pk `fmap` makeSignature hashDescr (private_size pk) m+{-# DEPRECATED+ sign+ "Use signDigest, or signDigestInfo when the message is already a DigestInfo"+ #-} -- | sign message using the private key and by automatically generating a blinder.+--+-- __Deprecated.__ Use 'signSaferDigest', or 'signSaferDigestInfo'. signSafer :: (HashAlgorithmASN1 hashAlg, MonadRandom m) => Maybe hashAlg@@ -541,9 +553,16 @@ signSafer hashAlg pk m = do blinder <- generateBlinder (private_n pk) return (sign (Just blinder) hashAlg pk m)+{-# DEPRECATED+ signSafer+ "Use signSaferDigest, or signSaferDigestInfo when the message is already a DigestInfo"+ #-} -- | verify message with the signed message --+-- __Deprecated.__ Use 'verifyDigest', or 'verifyDigestInfo' where this+-- would have been given 'Nothing'.+-- -- Following RFC 8017, the signature is rejected unless it is exactly as long -- as the modulus (section 8.2.2, step 1) and its integer representative is -- below the modulus (section 5.2.2, step 1). Verification works by@@ -562,6 +581,10 @@ -> Bool verify hashAlg pk m sm = verifyEncoded pk (makeSignature hashAlg (public_size pk) m) sm+{-# DEPRECATED+ verify+ "Use verifyDigest, or verifyDigestInfo when the message is already a DigestInfo"+ #-} -- | The two checks of RFC 8017 and the comparison, shared by the three -- verification entry points. The expected encoding is a thunk and is
Crypto/Random.hs view
@@ -7,35 +7,159 @@ -- Maintainer : Vincent Hanquez <vincent@snarc.org> -- Stability : stable -- Portability : good+--+-- Random bytes, drawn either from the system or from a generator whose+-- seed you hold.+--+-- == Drawing from the system+--+-- 'getRandomBytes' in 'IO' is the ordinary way to get bytes nobody can+-- predict. The length is in bytes and the type is any 'ByteArray', so the+-- result is usually pinned down by where it goes:+--+-- > import Crypto.Random (getRandomBytes)+-- > import Data.ByteString (ByteString)+-- >+-- > nonce <- getRandomBytes 12 :: IO ByteString+--+-- Everything in this library that needs randomness takes it the same way,+-- through a @MonadRandom m =>@ constraint, so running it in 'IO' is the+-- whole of choosing this source:+--+-- > import qualified Crypto.PubKey.RSA as RSA+-- >+-- > (pub, priv) <- RSA.generate 256 0x10001+--+-- Where the operating system offers @getrandom(2)@ or @getentropy(3)@ the+-- bytes come from a ChaCha20 generator belonging to the operating system+-- thread the call runs on, seeded from the system and reseeded as it goes;+-- see "Crypto.Random.SysDRG" for what that is and what it is not. Where+-- it does not -- Windows, for now -- they come from the system on every+-- call, as they always did. Either way the caller writes the same line.+--+-- Any Haskell thread may draw, and none has to say which generator it+-- wants:+--+-- > import Control.Concurrent (forkIO)+-- >+-- > mapM_ (\_ -> forkIO (getRandomBytes 32 >>= use)) [1 .. 64 :: Int]+--+-- The generator belongs to the operating system thread that happens to be+-- carrying the Haskell thread when the call is made, and is found through+-- that thread's own storage rather than by anything the caller passes. A+-- @forkIO@ thread moves between capabilities, so two draws from one Haskell+-- thread may be answered by two generators; they are seeded independently+-- of each other, so it makes no difference which one answers.+--+-- == Drawing from a generator you hold+--+-- A generator of your own is the way to draw many times from one seed.+-- The system is asked once, when the generator is made, and not again.+-- That is how @tls@ does a connection: 'seedNew' when it opens, and+-- everything the handshake needs afterwards from the generator, whose+-- state the connection carries from one draw to the next.+--+-- > import Crypto.Random (ChaChaDRG, drgNewSeed, seedNew, withDRG)+-- >+-- > data Connection = Connection { connRNG :: ChaChaDRG }+-- >+-- > newConnection :: IO Connection+-- > newConnection = do+-- > seed <- seedNew -- the only draw from the system+-- > return (Connection (drgNewSeed seed))+-- >+-- > -- every later draw advances the connection's own generator+-- > connRandom :: Int -> Connection -> (ByteString, Connection)+-- > connRandom n conn =+-- > let (bytes, rng') = withDRG (connRNG conn) (getRandomBytes n)+-- > in (bytes, conn{connRNG = rng'})+--+-- Inside 'withDRG' the same 'getRandomBytes' resolves to the instance for+-- 'MonadPseudoRandom' rather than the one for 'IO', so it touches neither+-- the system nor the per-thread generator: it advances the generator it was+-- given and hands it back. Keep that generator and the bytes are+-- reproducible from the seed, which is what makes a test repeatable -- and+-- what makes a generator unfit for keys unless its seed came from the+-- system.+--+-- 'drgNew' is 'seedNew' and 'drgNewSeed' in one step; 'drgNewTest' takes+-- four numbers instead of a seed, for a test that must give the same answer+-- twice.+--+-- == Drawing from a monad stacked on IO+--+-- There are two instances, one for 'IO' and one for 'MonadPseudoRandom',+-- and none for the transformers, so a @ReaderT env IO@ or a @StateT s IO@+-- reaches the system through 'liftIO':+--+-- > import Control.Monad.IO.Class (liftIO)+-- > import Control.Monad.Trans.Reader (ReaderT)+-- >+-- > newKey :: ReaderT env IO ByteString+-- > newKey = liftIO (getRandomBytes 32)+--+-- What that draws is what the first section describes, no more and no less:+-- the 'IO' instance, the generator of the operating system thread the call+-- lands on, seeded from the system and reseeded as it goes. 'liftIO' says+-- which monad the draw happens in and nothing about where the bytes come+-- from.+--+-- Writing an instance for the stack instead would make the choice invisible+-- at the call site, and the class cannot tell a strong source from a weak+-- one -- see 'MonadRandom'.+--+-- == When the system will not give any+--+-- Drawing randomness is the one thing here with nothing to fall back on,+-- so the failure is an exception rather than a value: there is no sensible+-- 'Maybe' to return and no partial answer worth having. Since 2.2 it is an+-- 'EntropyError' and not a @Control.Exception.ErrorCall@ or an+-- @Control.Exception.IOException@, which is worth knowing if you were+-- catching one of those:+--+-- > import Control.Exception (catch)+-- > import Crypto.Random (EntropyError (..), getRandomBytes)+-- >+-- > key <- getRandomBytes 32 `catch` \e -> case e of+-- > NoEntropySource -> fail "this system offers no randomness at all"+-- > EntropyShort w g -> fail (show w ++ " bytes wanted, " ++ show g ++ " arrived")+-- > EntropySourceLost s -> fail ("the source " ++ s ++ " went away")+--+-- Catching it at all is a decision rather than a default. A program that+-- cannot get randomness cannot make a key, and stopping is usually the+-- honest thing; the reason to catch is to say so in the program's own+-- terms rather than in crypton's. module Crypto.Random (- -- * Deterministic instances- ChaChaDRG,- SystemDRG,- Seed,+ -- * Drawing from the system+ MonadRandom (..), - -- * Seed+ -- * Drawing from a generator you hold+ Seed, seedNew, seedFromInteger, seedToInteger, seedFromBinary,-- -- * Deterministic Random class- getSystemDRG,- drgNew, drgNewSeed,+ drgNew, drgNewTest, withDRG, withRandomBytes,+ MonadPseudoRandom, DRG (..), - -- * Random abstraction- MonadRandom (..),- MonadPseudoRandom,+ -- * The generators+ ChaChaDRG,+ SystemDRG,+ getSystemDRG,++ -- * When the system will not give any+ EntropyError (..), ) where import Crypto.Error import Crypto.Internal.Imports import Crypto.Random.ChaChaDRG+import Crypto.Random.Entropy (EntropyError (..)) import Crypto.Random.SystemDRG import Crypto.Random.Types import Data.ByteArray (ByteArray, ByteArrayAccess, ScrubbedBytes)@@ -50,6 +174,9 @@ import Foreign.Ptr (Ptr, castPtr) #endif +-- | The material a deterministic generator is built from. Two generators+-- made from one seed produce the same bytes, which is what 'drgNewSeed' is+-- for and why a seed kept anywhere is as good as the keys drawn from it. newtype Seed = Seed ScrubbedBytes deriving (ByteArrayAccess)
Crypto/Random/Entropy.hs view
@@ -6,16 +6,37 @@ -- Portability : Good module Crypto.Random.Entropy ( getEntropy,+ EntropyError (..), ) where import Crypto.Internal.ByteArray (ByteArray) import qualified Crypto.Internal.ByteArray as B-import Data.Maybe (catMaybes)+import System.IO.Unsafe (unsafeInterleaveIO, unsafePerformIO) import Crypto.Random.Entropy.Unsafe +-- | The backends this system has, worked out once and no further than+-- needed.+--+-- Opening one is not free: for a device file it means opening and closing+-- @\/dev\/random@ or @\/dev\/urandom@ just to learn that it is there. This+-- used to be done on every call, and for the whole list before any backend+-- was asked for a byte, so a program paid for both devices even when the+-- first backend answered everything.+--+-- Once, now, and lazily: 'replenish' stops as soon as the buffer is full,+-- which leaves the tail of this list unforced, so a system where the first+-- backend answers never opens a device at all.+{-# NOINLINE openedBackends #-}+openedBackends :: [EntropyBackend]+openedBackends = unsafePerformIO (openAsNeeded supportedBackends)+ where+ openAsNeeded [] = return []+ openAsNeeded (o : os) = do+ m <- o+ rest <- unsafeInterleaveIO (openAsNeeded os)+ return $ maybe rest (: rest) m+ -- | Get some entropy from the system source of entropy getEntropy :: ByteArray byteArray => Int -> IO byteArray-getEntropy n = do- backends <- catMaybes `fmap` sequence supportedBackends- B.alloc n (replenish n backends)+getEntropy n = B.alloc n (replenish n openedBackends)
Crypto/Random/Entropy/Backend.hs view
@@ -11,27 +11,39 @@ ( EntropyBackend , supportedBackends , gatherBackend+ , EntropyError(..) ) where import Foreign.Ptr import Data.Proxy import Data.Word (Word8) import Crypto.Random.Entropy.Source-#ifdef SUPPORT_RDRAND-import Crypto.Random.Entropy.RDRand-#endif #ifdef WINDOWS import Crypto.Random.Entropy.Windows #else+import Crypto.Random.Entropy.SysRandom import Crypto.Random.Entropy.Unix #endif --- | All supported backends +-- | All supported backends, best first.+--+-- The system call comes before everything else: it is the kernel's own+-- generator, it needs no descriptor, and it is what the rest of the world+-- reaches for now. The device files are what is left when the system has+-- no such call.+--+-- RDRAND is deliberately not here, though it used to be first on x86. A+-- list like this one is a list of alternatives, and whichever answers+-- first decides the bytes on its own -- which is the one thing #298 says+-- RDRAND should not do. Nor does it contribute anywhere else: every+-- system this builds for already feeds the instruction into the pool the+-- call above draws from, so asking it again adds nothing. See the note in+-- @cbits\/crypton_sysdrg.c@. supportedBackends :: [IO (Maybe EntropyBackend)] supportedBackends = [-#ifdef SUPPORT_RDRAND- openBackend (Proxy :: Proxy RDRand),+#ifndef WINDOWS+ openBackend (Proxy :: Proxy SysRandom), #endif #ifdef WINDOWS openBackend (Proxy :: Proxy WinCryptoAPI)
− Crypto/Random/Entropy/RDRand.hs
@@ -1,39 +0,0 @@-{-# LANGUAGE ForeignFunctionInterface #-}---- |--- Module : Crypto.Random.Entropy.RDRand--- License : BSD-style--- Maintainer : Vincent Hanquez <vincent@snarc.org>--- Stability : experimental--- Portability : Good-module Crypto.Random.Entropy.RDRand (- RDRand,-) where--import Crypto.Random.Entropy.Source-import Data.Word (Word8)-import Foreign.C.Types-import Foreign.Ptr--foreign import ccall unsafe "crypton_cpu_has_rdrand"- c_cpu_has_rdrand :: IO CInt--foreign import ccall unsafe "crypton_get_rand_bytes"- c_get_rand_bytes :: Ptr Word8 -> CInt -> IO CInt---- | Fake handle to Intel RDRand entropy CPU instruction-data RDRand = RDRand--instance EntropySource RDRand where- entropyOpen = rdrandGrab- entropyGather _ = rdrandGetBytes- entropyClose _ = return ()--rdrandGrab :: IO (Maybe RDRand)-rdrandGrab = supported `fmap` c_cpu_has_rdrand- where- supported 0 = Nothing- supported _ = Just RDRand--rdrandGetBytes :: Ptr Word8 -> Int -> IO Int-rdrandGetBytes ptr sz = fromIntegral `fmap` c_get_rand_bytes ptr (fromIntegral sz)
Crypto/Random/Entropy/Source.hs view
@@ -6,8 +6,32 @@ -- Portability : Good module Crypto.Random.Entropy.Source where +import Control.Exception (Exception) import Data.Word (Word8) import Foreign.Ptr++-- | The system would not give the entropy it was asked for.+--+-- This is the one failure in the library with nothing to fall back on and+-- nothing sensible to return: a key drawn from bytes that are not random+-- is worse than no key. It used to be reported with 'error' and 'fail',+-- which left a caller no way to tell it from a bug in the library, and no+-- way to say anything useful about it.+data EntropyError+ = -- | The system offers no source of entropy at all. On Unix that+ -- means no @getrandom(2)@, no @getentropy(3)@ and no @\/dev@ to read+ -- from; a container built from nothing would look like this.+ NoEntropySource+ | -- | The sources between them gave fewer bytes than were asked for,+ -- three times over. The two numbers are how many were wanted and+ -- how many arrived.+ EntropyShort Int Int+ | -- | A source that could be opened once could not be opened again.+ -- The string names it.+ EntropySourceLost String+ deriving (Show, Eq)++instance Exception EntropyError -- | A handle to an entropy maker, either a system capability -- or a hardware generator.
+ Crypto/Random/Entropy/SysRandom.hs view
@@ -0,0 +1,50 @@+{-# LANGUAGE ForeignFunctionInterface #-}++-- |+-- Module : Crypto.Random.Entropy.SysRandom+-- License : BSD-style+-- Maintainer : Kazu Yamamoto <kazu@iij.ad.jp>+-- Stability : experimental+-- Portability : Unix+--+-- The kernel's own generator, reached without a file descriptor:+-- @getrandom(2)@ on Linux and FreeBSD, @getentropy(3)@ where that is what+-- the system has.+--+-- This is the source to prefer over reading @\/dev\/urandom@. It needs no+-- path and no descriptor, so it still answers where @\/dev@ is not mounted+-- or not populated, and it cannot be defeated by a full descriptor table.+module Crypto.Random.Entropy.SysRandom (+ SysRandom,+) where++import Crypto.Random.Entropy.Source+import Data.Word (Word8)+import Foreign.C.Types+import Foreign.Ptr++-- Both are @safe@ rather than @unsafe@: at early boot, before the kernel+-- pool is initialised, these calls block, and an unsafe call that blocks+-- holds the capability it runs on.+foreign import ccall safe "crypton_sysrandom_available"+ c_sysrandom_available :: IO CInt++foreign import ccall safe "crypton_sysrandom_bytes"+ c_sysrandom_bytes :: Ptr Word8 -> CInt -> IO CInt++-- | The system call, where there is one.+data SysRandom = SysRandom++instance EntropySource SysRandom where+ -- Asked at run time and not only at compile time: a binary built where+ -- the header declares the call can still run on a kernel that answers+ -- ENOSYS.+ entropyOpen = available `fmap` c_sysrandom_available+ where+ available 0 = Nothing+ available _ = Just SysRandom++ entropyGather _ ptr n =+ fromIntegral `fmap` c_sysrandom_bytes ptr (fromIntegral n)++ entropyClose _ = return ()
Crypto/Random/Entropy/Unix.hs view
@@ -61,7 +61,7 @@ withDev filepath f = openDev filepath >>= \h -> case h of- Nothing -> error ("device " ++ filepath ++ " cannot be grabbed")+ Nothing -> E.throwIO (EntropySourceLost filepath) Just fd -> f fd `E.finally` closeDev fd closeDev :: H -> IO ()
Crypto/Random/Entropy/Unsafe.hs view
@@ -9,25 +9,27 @@ module Crypto.Random.Entropy.Backend, ) where +import Control.Exception (throwIO)+ import Crypto.Random.Entropy.Backend import Data.Word (Word8) import Foreign.Ptr (Ptr, plusPtr) -- | Refill the entropy in a buffer ----- Call each entropy backend in turn until the buffer has--- been replenished.+-- Call each entropy backend in turn until the buffer has been replenished. ----- If the buffer cannot be refill after 3 loopings, this will raise--- an User Error exception+-- Throws 'EntropyError': 'NoEntropySource' when there is no backend at all,+-- and 'EntropyShort' when three passes over the backends still leave the+-- buffer unfilled. replenish :: Int -> [EntropyBackend] -> Ptr Word8 -> IO ()-replenish _ [] _ = fail "crypton: random: cannot get any source of entropy on this system"+replenish _ [] _ = throwIO NoEntropySource replenish poolSize backends ptr = loop 0 backends ptr poolSize where loop :: Int -> [EntropyBackend] -> Ptr Word8 -> Int -> IO () loop _ _ _ 0 = return () loop retry [] p n- | retry == 3 = error "crypton: random: cannot fully replenish"+ | retry == 3 = throwIO $ EntropyShort poolSize (poolSize - n) | otherwise = loop (retry + 1) backends p n loop retry (b : bs) p n = do r <- gatherBackend b p n
Crypto/Random/Entropy/Windows.hs view
@@ -23,6 +23,7 @@ import Foreign.Storable (peek) import System.Win32.Types (getLastError) +import Control.Exception (throwIO) import Crypto.Random.Entropy.Source @@ -38,7 +39,8 @@ case mctx of Nothing -> do lastError <- getLastError- fail $ "cannot re-grab win crypto api: error " ++ show lastError+ throwIO $ EntropySourceLost+ ("the Windows crypto API: error " ++ show lastError) Just ctx -> do r <- cryptGenRandom ctx ptr n cryptReleaseCtx ctx@@ -100,4 +102,5 @@ then return () else do lastError <- getLastError- fail $ "cryptReleaseCtx: error " ++ show lastError+ throwIO $ EntropySourceLost+ ("cryptReleaseCtx: error " ++ show lastError)
+ Crypto/Random/SysDRG.hs view
@@ -0,0 +1,82 @@+{-# LANGUAGE ForeignFunctionInterface #-}++-- |+-- Module : Crypto.Random.SysDRG+-- License : BSD-style+-- Maintainer : Kazu Yamamoto <kazu@iij.ad.jp>+-- Stability : experimental+-- Portability : Unix, Windows+--+-- The generator behind the 'Crypto.Random.MonadRandom' instance for 'IO'.+--+-- A ChaCha20 generator per operating system thread, seeded from a+-- process-wide generator, seeded in turn from the system entropy pool.+-- The state and the reseeding live in+-- @cbits\/crypton_sysdrg.c@, because a @forkIO@ thread is not an operating+-- system thread -- it moves between capabilities -- so state held against+-- one would be shared by threads running at the same time.+--+-- == What it is, and what it is not+--+-- It is not a DRBG of NIST SP 800-90A. That standard names three --+-- @Hash_DRBG@, @HMAC_DRBG@ and @CTR_DRBG@ -- and none of them is built on a+-- stream cipher. The name here is the older and looser sense of the word.+--+-- What it is is the shape @arc4random@ on OpenBSD and @get_random_bytes@ in+-- the Linux kernel both have, and the one asked for in+-- <https://github.com/kazu-yamamoto/crypton/issues/298>. That shape is+-- common; it is not specified anywhere, so the parts a standard would have+-- fixed were chosen here instead:+--+-- * ChaCha20, rather than AES in counter mode.+-- * SHA-512 over the system's bytes, rather than a derivation function a+-- standard would have named.+-- * A mebibyte per thread, and a mebibyte of issued seed for the process+-- generator, as the points at which to reseed. Those numbers are a+-- choice, not a result.+--+-- The pieces underneath are specified: ChaCha20 is RFC 8439, SHA-512 is+-- FIPS 180-4. The way they are put together is not.+--+-- == What a compromised state gives away+--+-- Each draw ends by taking the next forty bytes of keystream as the key and+-- nonce and dropping the ones that made the output, so a state read after a+-- draw is not the state that produced it. ChaCha20 does not run backwards+-- and the key that would be needed is gone, which is backtracking+-- resistance in the terms of SP 800-90A. Without it, a key stands until+-- the next reseed and anyone holding the state can wind the counter back+-- over everything issued since -- a mebibyte of output that was meant to be+-- secret. It is what @arc4random@ does, and for the same reason.+--+-- Prediction resistance is what the reseeding gives. A reseed draws from+-- the system again, so a state that has been read does not determine what+-- comes after one.+module Crypto.Random.SysDRG (+ sysDRGBytes,+) where++import Data.Word (Word8)+import Foreign.C.Types (CInt (..))+import Foreign.Ptr (Ptr)++import Crypto.Internal.ByteArray (ByteArray)+import qualified Crypto.Internal.ByteArray as B++-- Safe, not unsafe: seeding can reach a getrandom(2) that blocks until the+-- kernel pool is ready, and an unsafe call that blocks holds the capability+-- it runs on.+foreign import ccall safe "crypton_sysdrg_bytes"+ c_sysdrg_bytes :: Ptr Word8 -> CInt -> IO CInt++-- | Draw bytes, or 'Nothing' if the generator cannot be seeded.+--+-- It cannot be seeded where the system has no @getrandom(2)@ or+-- @getentropy(3)@; the caller falls back to+-- 'Crypto.Random.Entropy.getEntropy' there, which is the path this system+-- had before the generator existed.+sysDRGBytes :: ByteArray byteArray => Int -> IO (Maybe byteArray)+sysDRGBytes n = do+ (got, out) <- B.allocRet n $ \ptr ->+ fromIntegral `fmap` c_sysdrg_bytes ptr (fromIntegral n)+ return $ if got == n then Just out else Nothing
Crypto/Random/Types.hs view
@@ -13,6 +13,7 @@ import Crypto.Internal.ByteArray import Crypto.Random.Entropy+import Crypto.Random.SysDRG (sysDRGBytes) -- | A monad constraint that allows to generate random bytes --@@ -35,8 +36,13 @@ -- | Generate N bytes of randomness from a DRG randomBytesGenerate :: ByteArray byteArray => Int -> gen -> (byteArray, gen) +-- | Through the generator of 'Crypto.Random.SysDRG': a ChaCha20 generator+-- per operating system thread, reseeded from the system. Where that+-- cannot be seeded -- a system with no @getrandom(2)@ or @getentropy(3)@,+-- which includes Windows for now -- this is the system entropy source+-- directly, as it was before the generator existed. instance MonadRandom IO where- getRandomBytes = getEntropy+ getRandomBytes n = sysDRGBytes n >>= maybe (getEntropy n) return -- | A simple Monad class very similar to a State Monad -- with the state being a DRG.
Crypto/System/CPU.hs view
@@ -1,6 +1,6 @@ {-# LANGUAGE CPP #-}-{-# LANGUAGE DeriveDataTypeable #-} {-# LANGUAGE ForeignFunctionInterface #-}+{-# LANGUAGE PatternSynonyms #-} -- | -- Module : Crypto.System.CPU@@ -11,57 +11,201 @@ -- -- Gives information about crypton runtime environment. module Crypto.System.CPU (- ProcessorOption (..),+ -- The names are bundled with the type rather than listed as+ -- `pattern' exports, so an importer writes ProcessorOption (..), or+ -- names the ones it wants, as it would for a type with constructors.+ ProcessorOption (+ -- x86+ AESNI,+ PCLMUL,+ SSSE3,+ AVX,+ AVX2,+ SHANI,+ MOVBE,+ ADX,+ VAES,+ VAES512,+ -- AArch64+ NEON,+ ARMAES,+ ARMPMULL,+ ARMSHA1,+ ARMSHA2,+ ARMSHA512,+ -- PowerISA+ PPCAES,+ PPCVPMSUM+ ), processorOptions,-) where -import Data.Data-import Data.List (findIndices)-#ifdef SUPPORT_RDRAND-import Data.Maybe (isJust)-#endif-import Data.Word (Word8)-import Foreign.Ptr-import Foreign.Storable+ -- * Questions that do not name an architecture+ hasAESAcceleration,+ hasGHASHAcceleration,+) where +import Control.Monad (filterM)+import Data.List (sort)+import Data.Word (Word16)+import Foreign.C.Types (CInt (..), CUInt (..)) import Crypto.Internal.Compat -#ifdef SUPPORT_RDRAND-import Crypto.Random.Entropy.RDRand-import Crypto.Random.Entropy.Source-#endif+-- | A processor feature crypton looked for, and dispatches on where it+-- finds it.+--+-- This is a number with names rather than a sum of constructors, and the+-- names are pattern synonyms with no @COMPLETE@ pragma, so a @case@ over+-- them needs a catch-all and a feature named in a later release breaks+-- nothing that compiled against this one. The same reason 'Show' is+-- written out below: a program built against an older crypton still says+-- something useful about a value from a newer one.+--+-- The names are the processor's, not the operation's. 'AESNI' is x86's+-- and 'ARMAES' is AArch64's, and a machine reports only the ones it has;+-- ask 'hasAESAcceleration' if the question is whether AES is fast here.+--+-- They are bundled with the type in the export list, so @ProcessorOption+-- (..)@ brings in all of them and naming one brings in that one, as for a+-- type with constructors. The constructor underneath is not exported:+-- these values say what the processor was found to have, and a caller has+-- nothing to build.+newtype ProcessorOption = ProcessorOption Word16+ deriving (Eq, Ord) --- | CPU options impacting cryptography implementation and library performance.-data ProcessorOption- = -- | Support for AES instructions, with flag @support_aesni@- AESNI- | -- | Support for CLMUL instructions, with flag @support_pclmuldq@- PCLMUL- | -- | Support for RDRAND instruction, with flag @support_rdrand@- RDRAND- deriving (Show, Eq, Enum, Data)+-- | Support for AES instructions, with flag @support_aesni@.+pattern AESNI :: ProcessorOption+pattern AESNI = ProcessorOption 0 +-- | Support for CLMUL instructions, with flag @support_pclmuldq@.+pattern PCLMUL :: ProcessorOption+pattern PCLMUL = ProcessorOption 1++-- | Supplemental SSE3.+pattern SSSE3 :: ProcessorOption+pattern SSSE3 = ProcessorOption 3++-- | AVX, and an operating system that saves its registers.+pattern AVX :: ProcessorOption+pattern AVX = ProcessorOption 4++-- | AVX2, and an operating system that saves its registers.+pattern AVX2 :: ProcessorOption+pattern AVX2 = ProcessorOption 5++-- | The SHA extensions, @sha1rnds4@ and @sha256rnds2@ and their neighbours.+pattern SHANI :: ProcessorOption+pattern SHANI = ProcessorOption 6++-- | The byte-swapping load.+pattern MOVBE :: ProcessorOption+pattern MOVBE = ProcessorOption 7++-- | @MULX@, @ADCX@ and @ADOX@: the two independent carry chains.+pattern ADX :: ProcessorOption+pattern ADX = ProcessorOption 8++-- | The AES and carry-less multiply instructions in their 256-bit form.+pattern VAES :: ProcessorOption+pattern VAES = ProcessorOption 9++-- | The same pair in their 512-bit form.+pattern VAES512 :: ProcessorOption+pattern VAES512 = ProcessorOption 10++-- | Advanced SIMD, which is not optional on AArch64.+pattern NEON :: ProcessorOption+pattern NEON = ProcessorOption 11++-- | The ARMv8 AES instructions.+pattern ARMAES :: ProcessorOption+pattern ARMAES = ProcessorOption 12++-- | @PMULL@, the ARMv8 carry-less multiply.+pattern ARMPMULL :: ProcessorOption+pattern ARMPMULL = ProcessorOption 13++-- | The ARMv8 SHA-1 instructions.+pattern ARMSHA1 :: ProcessorOption+pattern ARMSHA1 = ProcessorOption 14++-- | The ARMv8 SHA-256 instructions.+pattern ARMSHA2 :: ProcessorOption+pattern ARMSHA2 = ProcessorOption 15++-- | The ARMv8.2 SHA-512 instructions, which are optional where SHA-256's+-- are not.+pattern ARMSHA512 :: ProcessorOption+pattern ARMSHA512 = ProcessorOption 16++-- | Support for the PowerISA 2.07 vector AES instructions, which POWER8 was+-- the first to implement.+pattern PPCAES :: ProcessorOption+pattern PPCAES = ProcessorOption 17++-- | Support for @vpmsumd@, the vector carry-less multiply that came with+-- them, which is what makes GHASH fast.+pattern PPCVPMSUM :: ProcessorOption+pattern PPCVPMSUM = ProcessorOption 18++-- | Named where the name is known, numbered where it is not, so that a+-- binary built against an older crypton can still print a value a newer one+-- produced.+instance Show ProcessorOption where+ show AESNI = "AESNI"+ show PCLMUL = "PCLMUL"+ show SSSE3 = "SSSE3"+ show AVX = "AVX"+ show AVX2 = "AVX2"+ show SHANI = "SHANI"+ show MOVBE = "MOVBE"+ show ADX = "ADX"+ show VAES = "VAES"+ show VAES512 = "VAES512"+ show NEON = "NEON"+ show ARMAES = "ARMAES"+ show ARMPMULL = "ARMPMULL"+ show ARMSHA1 = "ARMSHA1"+ show ARMSHA2 = "ARMSHA2"+ show ARMSHA512 = "ARMSHA512"+ show PPCAES = "PPCAES"+ show PPCVPMSUM = "PPCVPMSUM"+ show (ProcessorOption n) = "ProcessorOption " ++ show n+ -- | Options which have been enabled at compile time and are supported by the -- current CPU.+--+-- Sorted, and without repeats. A machine reports the names of its own+-- architecture only: an AArch64 processor with AES says 'ARMAES', not+-- 'AESNI', which it does not have. processorOptions :: [ProcessorOption]-processorOptions = unsafeDoIO $ do- p <- crypton_aes_cpu_init- options <- traverse (getOption p) aesOptions- rdrand <- hasRDRand- return (decodeOptions options ++ [RDRAND | rdrand])+processorOptions = unsafeDoIO (sort <$> filterM askC allOptions) where- aesOptions = [AESNI .. PCLMUL]- getOption p = peekElemOff p . fromEnum- decodeOptions = map toEnum . findIndices (> 0)+ -- 2 is not asked for and has no name: it was RDRAND, which crypton no+ -- longer dispatches on. The number is left out rather than reused, so+ -- that the others keep the values they had.+ allOptions = [ProcessorOption n | n <- [0 .. 18], n /= 2]+ askC (ProcessorOption n) =+ (/= 0) <$> crypton_cpu_option (fromIntegral n) {-# NOINLINE processorOptions #-} -hasRDRand :: IO Bool-#ifdef SUPPORT_RDRAND-hasRDRand = fmap isJust getRDRand- where getRDRand = entropyOpen :: IO (Maybe RDRand)-#else-hasRDRand = return False-#endif+-- | Is there hardware AES on this machine?+--+-- The instructions have different names on different architectures, and a+-- caller that wants to know whether AES-GCM will be fast wants this rather+-- than either name.+hasAESAcceleration :: Bool+hasAESAcceleration =+ any+ (`elem` processorOptions)+ [AESNI, ARMAES, PPCAES] -foreign import ccall unsafe "crypton_aes_cpu_init"- crypton_aes_cpu_init :: IO (Ptr Word8)+-- | Is there a hardware carry-less multiply, which is what GHASH, and so+-- AES-GCM, spends its time in once AES itself is fast?+hasGHASHAcceleration :: Bool+hasGHASHAcceleration =+ any+ (`elem` processorOptions)+ [PCLMUL, ARMPMULL, PPCVPMSUM]++foreign import ccall unsafe "crypton_cpu_option"+ crypton_cpu_option :: CUInt -> IO CInt
Crypto/Tutorial.hs view
@@ -1,4 +1,7 @@ -- | Examples of how to use @crypton@.+--+-- Every code block here is extracted and compiled against this version of+-- the library by @tests\/tutorial\/run.sh@, so what is written below builds. module Crypto.Tutorial ( -- * API design -- $api_design@@ -6,9 +9,21 @@ -- * Hash algorithms -- $hash_algorithms - -- * Symmetric block ciphers- -- $symmetric_block_ciphers+ -- * Authenticated encryption+ -- $authenticated_encryption + -- * Comparing secrets+ -- $comparing_secrets++ -- * Password storage+ -- $password_storage++ -- * Key derivation+ -- $key_derivation++ -- * Digital signatures+ -- $digital_signatures+ -- * Combining primitives -- $combining_primitives ) where@@ -31,6 +46,14 @@ -- Error conditions are returned with data type 'Crypto.Error.CryptoFailable'. -- Functions in module "Crypto.Error" can convert those values to runtime -- exceptions, 'Maybe' or 'Either' values.+--+-- Types that hold a secret do not print it. A private key's 'Show' renders+-- whatever is public and @\<secret\>@ or @\<scrubbed-bytes\>@ for the rest,+-- because 'Show' is what @print@, @error@, an exception and a failing test+-- all reach for, and a key arriving in a log that way is an accident nobody+-- asked for. "Crypto.Debug" is how one is printed when printing it is what+-- was meant. Several of those types keep their bytes in+-- 'Data.ByteArray.ScrubbedBytes', which is wiped when it is collected. -- $hash_algorithms --@@ -90,72 +113,258 @@ -- > hashMutableUpdate ctx ("dog" :: ByteString) -- > hashMutableFinalize ctx >>= print --- $symmetric_block_ciphers+-- $authenticated_encryption --+-- Encrypting hides a message; it does not stop anyone changing it. Under a+-- counter or stream mode, flipping a bit of the ciphertext flips the same+-- bit of the plaintext, and the receiver has no way to tell. An AEAD mode+-- binds the message, and anything else named as associated data, to a short+-- authentication tag, and decrypting something that does not match that tag+-- returns nothing at all.+--+-- That is the mode to use. The unauthenticated ones are in this library+-- for protocols that authenticate separately, not as a starting point.+-- -- > {-# LANGUAGE OverloadedStrings #-}--- > {-# LANGUAGE ScopedTypeVariables #-}--- > {-# LANGUAGE GADTs #-} -- > -- > import Crypto.Cipher.AES (AES256)--- > import Crypto.Cipher.Types (BlockCipher(..), Cipher(..), nullIV, KeySizeSpecifier(..), IV, makeIV)--- > import Crypto.Error (CryptoFailable(..), CryptoError(..))+-- > import Crypto.Cipher.Types+-- > ( AEADMode (AEAD_GCM)+-- > , AuthTag+-- > , BlockCipher (aeadInit)+-- > , Cipher (cipherInit, cipherKeySize)+-- > , KeySizeSpecifier (..)+-- > , aeadSimpleDecrypt+-- > , aeadSimpleEncrypt+-- > )+-- > import Crypto.Error (CryptoError, eitherCryptoError)+-- > import qualified Crypto.Random.Types as CRT -- >+-- > import Data.ByteArray (ByteArray, ByteArrayAccess, ScrubbedBytes)+-- > import Data.ByteString (ByteString)+-- >+-- > -- | A key of the length the cipher asks for, rather than a length+-- > -- written out here. ScrubbedBytes rather than ByteString, so that it+-- > -- is wiped when it is collected and does not print.+-- > genSecretKey :: (Cipher c, CRT.MonadRandom m) => c -> m ScrubbedBytes+-- > genSecretKey c = CRT.getRandomBytes (longest (cipherKeySize c))+-- > where+-- > longest (KeySizeFixed n) = n+-- > longest (KeySizeRange _ n) = n+-- > longest (KeySizeEnum ns) = maximum ns+-- >+-- > -- | A fresh nonce for every message. GCM must never see one twice+-- > -- under the same key: a repeat does not just expose those two messages,+-- > -- it hands over the key that authenticates all of them. Twelve random+-- > -- bytes, sent along with the ciphertext.+-- > genNonce :: CRT.MonadRandom m => m ByteString+-- > genNonce = CRT.getRandomBytes 12+-- >+-- > -- | Encrypt and authenticate. The associated data is authenticated but+-- > -- not encrypted: it is for what the receiver can already see and must+-- > -- not have had altered, such as a header or an address.+-- > encrypt+-- > :: (ByteArray key, ByteArrayAccess nonce, ByteArrayAccess aad, ByteArray ba)+-- > => key -> nonce -> aad -> ba -> Either CryptoError (AuthTag, ba)+-- > encrypt key nonce aad plaintext = do+-- > cipher <- eitherCryptoError (cipherInit key) :: Either CryptoError AES256+-- > aead <- eitherCryptoError (aeadInit AEAD_GCM cipher nonce)+-- > return (aeadSimpleEncrypt aead aad plaintext 16)+-- >+-- > -- | And back, with two different failures. Left is this code used+-- > -- wrongly -- a key of the wrong length, a mode the cipher has not got.+-- > -- Nothing is a message that is not the one that was sent; it carries no+-- > -- plaintext and says nothing about which byte was wrong, both of which+-- > -- are the point.+-- > decrypt+-- > :: (ByteArray key, ByteArrayAccess nonce, ByteArrayAccess aad, ByteArray ba)+-- > => key -> nonce -> aad -> AuthTag -> ba -> Either CryptoError (Maybe ba)+-- > decrypt key nonce aad tag ciphertext = do+-- > cipher <- eitherCryptoError (cipherInit key) :: Either CryptoError AES256+-- > aead <- eitherCryptoError (aeadInit AEAD_GCM cipher nonce)+-- > return (aeadSimpleDecrypt aead aad ciphertext tag)+-- >+-- > exampleAES256GCM :: ByteString -> IO ()+-- > exampleAES256GCM msg = do+-- > key <- genSecretKey (undefined :: AES256)+-- > nonce <- genNonce+-- > let aad = "to: alice" :: ByteString+-- > case encrypt key nonce aad msg of+-- > Left err -> error (show err)+-- > Right (tag, ciphertext) -> do+-- > putStrLn $ "ciphertext: " ++ show ciphertext+-- > putStrLn $ " tag: " ++ show tag+-- > putStrLn $ " recovered: "+-- > ++ show (decrypt key nonce aad tag ciphertext)+-- > -- The same bytes and the same tag, with one thing changed+-- > -- that was never encrypted: Right Nothing.+-- > putStrLn $ "redirected: "+-- > ++ show (decrypt key nonce ("to: eve" :: ByteString) tag ciphertext)+--+-- The two functions above work for any cipher that has an AEAD mode; what+-- changes is the mode given to 'Crypto.Cipher.Types.aeadInit'.+-- "Crypto.Cipher.ChaChaPoly1305" is the one to prefer on a machine with no+-- AES instructions, and "Crypto.Cipher.AESGCMSIV" is the one that survives+-- a repeated nonce, at the price of needing the whole message before it can+-- begin.++-- $comparing_secrets+--+-- Comparing two byte strings with '==' stops at the first byte that+-- differs, so how long it takes says where that byte was. Against an+-- authentication tag that is the whole secret: someone who can send a guess+-- and time the answer finds the first byte in a few hundred tries, then the+-- second, and has a tag that was supposed to cost 2^128 in a few thousand.+--+-- crypton's own authentication types already compare in constant time, so+-- for those there is nothing to do: 'Crypto.MAC.HMAC.HMAC',+-- 'Crypto.MAC.CMAC.CMAC', 'Crypto.MAC.Poly1305.Auth' and+-- 'Crypto.Cipher.Types.AuthTag' have an 'Eq' that looks at every byte+-- whatever it finds. What needs care is a tag that arrives as bytes, and+-- 'Data.ByteArray.constEq' is the comparison for it.+--+-- > {-# LANGUAGE OverloadedStrings #-}+-- >+-- > import Crypto.Hash.Algorithms (SHA256)+-- > import Crypto.MAC.HMAC (HMAC, hmac)+-- >+-- > import qualified Data.ByteArray as BA+-- > import Data.ByteString (ByteString)+-- >+-- > -- | The tag came off the wire as bytes, so it is compared as bytes.+-- > authentic :: ByteString -> ByteString -> ByteString -> Bool+-- > authentic key message tag =+-- > BA.constEq tag (hmac key message :: HMAC SHA256)+-- >+-- > -- | Once it has been parsed into the library's own type, (==) is+-- > -- already the constant-time comparison.+-- > authentic' :: ByteString -> ByteString -> HMAC SHA256 -> Bool+-- > authentic' key message tag = tag == hmac key message++-- $password_storage+--+-- A password is not a key. It is short and it is guessable, and whoever+-- takes the database can try every likely one without being watched. What+-- answers that is a function that is deliberately expensive to compute, and+-- crypton has four: "Crypto.KDF.BCrypt", "Crypto.KDF.Scrypt",+-- "Crypto.KDF.Argon2" and "Crypto.KDF.PBKDF2". A plain hash is not one of+-- them, however many times it is applied by hand.+--+-- bcrypt leaves the least to get wrong, because the record it returns+-- carries the salt and the cost inside it:+--+-- > import Crypto.KDF.BCrypt (hashPassword, validatePassword)+-- >+-- > import Data.ByteString (ByteString)+-- >+-- > -- | What goes in the database. The salt is drawn inside and ends up in+-- > -- the result, so two accounts with the same password do not look alike.+-- > register :: ByteString -> IO ByteString+-- > register password = hashPassword 12 password+-- >+-- > -- | And what is checked against it. The cost comes out of the stored+-- > -- record, so raising it for new accounts leaves the old ones working.+-- > login :: ByteString -> ByteString -> Bool+-- > login password stored = validatePassword password stored+--+-- Argon2 is the stronger choice and the one to pick for something new: it+-- asks for memory as well as time, which is what takes the advantage away+-- from the hardware that bcrypt's small working set leaves room for.+-- Nothing is encoded for the caller, though -- the salt and the options are+-- theirs to store, and without all three the hash cannot be recomputed when+-- the user comes back.+--+-- > import Crypto.Error (CryptoFailable)+-- > import qualified Crypto.KDF.Argon2 as Argon2 -- > import qualified Crypto.Random.Types as CRT -- >--- > import Data.ByteArray (ByteArray) -- > import Data.ByteString (ByteString) -- >--- > -- | Not required, but most general implementation--- > data Key c a where--- > Key :: (BlockCipher c, ByteArray a) => a -> Key c a+-- > -- | Argon2id, which is the variant to prefer: it resists both a machine+-- > -- built to guess and a process watching the cache.+-- > options :: Argon2.Options+-- > options = Argon2.defaultOptions{Argon2.variant = Argon2.Argon2id} -- >--- > -- | Generates a string of bytes (key) of a specific length for a given block cipher--- > genSecretKey :: forall m c a. (CRT.MonadRandom m, BlockCipher c, ByteArray a) => c -> Int -> m (Key c a)--- > genSecretKey _ = fmap Key . CRT.getRandomBytes+-- > newSalt :: CRT.MonadRandom m => m ByteString+-- > newSalt = CRT.getRandomBytes 16 -- >--- > -- | Generate a random initialization vector for a given block cipher--- > genRandomIV :: forall m c. (CRT.MonadRandom m, BlockCipher c) => c -> m (Maybe (IV c))--- > genRandomIV _ = do--- > bytes :: ByteString <- CRT.getRandomBytes $ blockSize (undefined :: c)--- > return $ makeIV bytes+-- > derive :: ByteString -> ByteString -> CryptoFailable ByteString+-- > derive salt password = Argon2.hash options password salt 32++-- $key_derivation+--+-- HKDF turns one secret into as many keys as a protocol needs. It is for+-- material that is already unguessable -- what comes out of a+-- Diffie-Hellman, or a key already agreed -- and it is deliberately cheap,+-- which is exactly what makes it the wrong thing for a password. Those go+-- to the section above.+--+-- The info string is what keeps the outputs independent: the same secret+-- with a different info gives an unrelated key, so each use of a secret+-- names itself there.+--+-- > {-# LANGUAGE OverloadedStrings #-} -- >--- > -- | Initialize a block cipher--- > initCipher :: (BlockCipher c, ByteArray a) => Key c a -> Either CryptoError c--- > initCipher (Key k) = case cipherInit k of--- > CryptoFailed e -> Left e--- > CryptoPassed a -> Right a+-- > import Crypto.Hash.Algorithms (SHA256)+-- > import qualified Crypto.KDF.HKDF as HKDF -- >--- > encrypt :: (BlockCipher c, ByteArray a) => Key c a -> IV c -> a -> Either CryptoError a--- > encrypt secretKey initIV msg =--- > case initCipher secretKey of--- > Left e -> Left e--- > Right c -> Right $ ctrCombine c initIV msg+-- > import Data.ByteArray (ScrubbedBytes)+-- > import Data.ByteString (ByteString) -- >--- > decrypt :: (BlockCipher c, ByteArray a) => Key c a -> IV c -> a -> Either CryptoError a--- > decrypt = encrypt+-- > -- | One shared secret in, two unrelated keys out.+-- > directionKeys :: ByteString -> ByteString -> (ScrubbedBytes, ScrubbedBytes)+-- > directionKeys salt shared = (keyFor "client write", keyFor "server write")+-- > where+-- > prk = HKDF.extract salt shared :: HKDF.PRK SHA256+-- > keyFor info = HKDF.expand prk (info :: ByteString) 32++-- $digital_signatures+--+-- Ed25519 is the one to reach for. The keys are thirty-two bytes, there is+-- nothing to choose and nothing to encode, and signing needs no randomness,+-- so it cannot be ruined by a bad source of it. "Crypto.PubKey.Ed448" is+-- the same shape at a larger size, "Crypto.PubKey.ECDSA" and+-- "Crypto.PubKey.RSA.PSS" are there for protocols that ask for them, and+-- "Crypto.PubKey.MLDSA" is the post-quantum one.+--+-- > {-# LANGUAGE OverloadedStrings #-} -- >--- > exampleAES256 :: ByteString -> IO ()--- > exampleAES256 msg = do--- > -- secret key needs 256 bits (32 * 8)--- > secretKey <- genSecretKey (undefined :: AES256) 32--- > mInitIV <- genRandomIV (undefined :: AES256)--- > case mInitIV of--- > Nothing -> error "Failed to generate and initialization vector."--- > Just initIV -> do--- > let encryptedMsg = encrypt secretKey initIV msg--- > decryptedMsg = decrypt secretKey initIV =<< encryptedMsg--- > case (,) <$> encryptedMsg <*> decryptedMsg of--- > Left err -> error $ show err--- > Right (eMsg, dMsg) -> do--- > putStrLn $ "Original Message: " ++ show msg--- > putStrLn $ "Message after encryption: " ++ show eMsg--- > putStrLn $ "Message after decryption: " ++ show dMsg+-- > import Crypto.Error (throwCryptoError)+-- > import qualified Crypto.PubKey.Ed25519 as Ed25519+-- >+-- > import qualified Data.ByteArray as BA+-- > import Data.ByteString (ByteString)+-- >+-- > exampleEd25519 :: ByteString -> IO ()+-- > exampleEd25519 msg = do+-- > sk <- Ed25519.generateSecretKey+-- > let pk = Ed25519.toPublic sk+-- > sig = Ed25519.sign sk pk msg+-- > print (Ed25519.verify pk msg sig)+-- > -- The same signature against a message one byte longer: False.+-- > print (Ed25519.verify pk (msg <> "!") sig)+-- >+-- > -- | A secret key on its way to storage. Printing one does not reveal+-- > -- it -- see the first section -- so this is the way out.+-- > store :: Ed25519.SecretKey -> ByteString+-- > store = BA.convert+-- >+-- > -- | And the way back in, which is checked, because the bytes read from+-- > -- a file may be anything at all.+-- > load :: ByteString -> Ed25519.SecretKey+-- > load = throwCryptoError . Ed25519.secretKey -- $combining_primitives -- -- This example shows how to use Curve25519, XSalsa and Poly1305 primitives to -- emulate NaCl's @crypto_box@ construct. --+-- It is here to show how the pieces fit together, not as something to+-- deploy. An authenticated encryption scheme assembled by hand is the kind+-- of code that is wrong in ways no test notices; for actual use,+-- "Crypto.Cipher.ChaChaPoly1305" does this job with the mistakes already+-- made.+-- -- > import qualified Data.ByteArray as BA -- > import Data.ByteString (ByteString) -- > import qualified Data.ByteString as B@@ -167,6 +376,9 @@ -- > -- > -- | Build a @crypto_box@ packet encrypting the specified content with a -- > -- 192-bit nonce, receiver public key and sender private key.+-- > crypto_box+-- > :: ByteString -> ByteString -> X25519.PublicKey -> X25519.SecretKey+-- > -> ByteString -- > crypto_box content nonce pk sk = BA.convert tag `B.append` c -- > where -- > zero = B.replicate 16 0@@ -181,6 +393,9 @@ -- > -- > -- | Try to open a @crypto_box@ packet and recover the content using the -- > -- 192-bit nonce, sender public key and receiver private key.+-- > crypto_box_open+-- > :: ByteString -> ByteString -> X25519.PublicKey -> X25519.SecretKey+-- > -> Maybe ByteString -- > crypto_box_open packet nonce pk sk -- > | B.length packet < 16 = Nothing -- > | BA.constEq tag' tag = Just content
README.md view
@@ -14,33 +14,49 @@ Side channels ------------- +### AES+ AES is where this matters most, and which implementation runs is decided at runtime from what the processor has. -On x86-64 with AES-NI and carry-less multiply, and on AArch64 with the ARMv8-cryptographic extension, AES and GHASH are instructions rather than tables.-crypton's AES and AES-GCM then make no branch and no memory access that-depends on the key or on the data: the secrets stay in vector registers and-never reach one a branch can test, which the generated code is checked-against. Every x86-64 part since about 2010 and every AArch64 part in-ordinary use has these.+On x86-64 with AES-NI and carry-less multiply, on AArch64 and on 32-bit ARM+with the ARMv8 cryptographic extension, and on ppc64le with the vector+instructions PowerISA 2.07 brought to POWER8, AES and GHASH are instructions+rather than software. crypton's AES and AES-GCM then make no branch and no+memory access that depends on the key or on the data: the secrets stay in+vector registers and never reach one a branch can test. Every x86-64 part+since about 2010 and every AArch64 part in ordinary use has these, and the+ppc64le path is little-endian only. -Where neither is present crypton falls back to a table-driven AES, which-indexes a 256-byte substitution table with data derived from the key and the-input. **That is not constant time**, and on a machine where an attacker can-observe the cache it is open to a timing attack. The fallback exists so that-the library builds and runs everywhere; it is not meant for a setting where-that matters.+Where none of them is there, crypton computes AES and GHASH without a table.+The portable AES is bitsliced -- the S-box is boolean algebra over four+blocks held across eight 64-bit words -- and the GF(2^128) multiply is built+from shifts, masks and integer multiplies. Neither looks anything up, so+neither derives an address from a secret, and **the portable path is constant+time as well**. The code is BearSSL's, under MIT, in `cbits/bearssl`. -`Crypto.System.CPU.processorOptions` says which is in use. `AESNI` in that-list means the instruction path, and `PCLMUL` that GHASH has its instruction-too; without `AESNI` it is the tables. The list also reports `RDRAND`, which-is unrelated to this.+The constant-time harness in `cbits/tests/ct` runs the same driver against+the portable implementation and against each architecture's instructions, and+every one of them is required to report nothing at all. +So the answer does not depend on the machine any more, and what is left to+ask is whether AES is fast on it. `hasAESAcceleration` and+`hasGHASHAcceleration` answer that without naming an architecture.+`processorOptions` says what was found, in the processor's own names --+`AESNI` and `PCLMUL` on x86, `ARMAES` and `ARMPMULL` on AArch64 and 32-bit+ARM, `PPCAES` and `PPCVPMSUM` on ppc64le -- along with everything else it was+asked about, which is unrelated to this.+ ghci> import Crypto.System.CPU- ghci> processorOptions- [AESNI,PCLMUL]+ ghci> hasAESAcceleration+ True+ ghci> processorOptions -- on an x86-64 machine+ [AESNI,PCLMUL,SSSE3,AVX,AVX2,SHANI,MOVBE,ADX,VAES]+ ghci> processorOptions -- and on an AArch64 one+ [NEON,ARMAES,ARMPMULL,ARMSHA1,ARMSHA2,ARMSHA512] +### RSA+ RSA is the other place to know about, and there the choice is the caller's. The private key operations in `Crypto.PubKey.RSA.PKCS15`, `.OAEP` and `.PSS` take a `Maybe Blinder`, and `Nothing` is no harder to write than the safe@@ -252,3 +268,8 @@ ask for it, not because it is a good choice for anything new. The algorithms that nothing should ask for any more -- MD5, 3DES, RC4, CBC mode -- are left out.++ML-KEM and ML-DSA are left out too, though 2.1.8 brought both and they are+fast -- `mlkem-native` and `mldsa-native`, from the PQ Code Package. No+comparison is reported here: the implementations to measure against are+still developing.
cbits/aes/generic.c view
@@ -27,420 +27,191 @@ * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF * SUCH DAMAGE. *- * AES implementation+ * AES, portable, for machines whose processor has no AES instructions or+ * whose build was not given them.+ *+ * This was a table-driven implementation: an S-box indexed by a byte of the+ * state, in every round and in the key expansion, which is what made it fast+ * and what made it variable-time. It is now BearSSL's bitsliced aes_ct64,+ * which holds four blocks interleaved across eight 64-bit words and computes+ * the S-box as boolean algebra -- no table, and so no address derived from a+ * secret. See cbits/bearssl/README.md.+ *+ * What is here is only the glue. Nothing in this file branches or indexes on+ * the key or the data. */ #include <stdint.h>-#include <stdlib.h>+#include <string.h> #include <crypton_aes.h>-#include <crypton_bitfn.h>--static uint8_t sbox[256] = {- 0x63, 0x7c, 0x77, 0x7b, 0xf2, 0x6b, 0x6f, 0xc5, 0x30, 0x01, 0x67, 0x2b, 0xfe,- 0xd7, 0xab, 0x76, 0xca, 0x82, 0xc9, 0x7d, 0xfa, 0x59, 0x47, 0xf0, 0xad, 0xd4,- 0xa2, 0xaf, 0x9c, 0xa4, 0x72, 0xc0, 0xb7, 0xfd, 0x93, 0x26, 0x36, 0x3f, 0xf7,- 0xcc, 0x34, 0xa5, 0xe5, 0xf1, 0x71, 0xd8, 0x31, 0x15, 0x04, 0xc7, 0x23, 0xc3,- 0x18, 0x96, 0x05, 0x9a, 0x07, 0x12, 0x80, 0xe2, 0xeb, 0x27, 0xb2, 0x75, 0x09,- 0x83, 0x2c, 0x1a, 0x1b, 0x6e, 0x5a, 0xa0, 0x52, 0x3b, 0xd6, 0xb3, 0x29, 0xe3,- 0x2f, 0x84, 0x53, 0xd1, 0x00, 0xed, 0x20, 0xfc, 0xb1, 0x5b, 0x6a, 0xcb, 0xbe,- 0x39, 0x4a, 0x4c, 0x58, 0xcf, 0xd0, 0xef, 0xaa, 0xfb, 0x43, 0x4d, 0x33, 0x85,- 0x45, 0xf9, 0x02, 0x7f, 0x50, 0x3c, 0x9f, 0xa8, 0x51, 0xa3, 0x40, 0x8f, 0x92,- 0x9d, 0x38, 0xf5, 0xbc, 0xb6, 0xda, 0x21, 0x10, 0xff, 0xf3, 0xd2, 0xcd, 0x0c,- 0x13, 0xec, 0x5f, 0x97, 0x44, 0x17, 0xc4, 0xa7, 0x7e, 0x3d, 0x64, 0x5d, 0x19,- 0x73, 0x60, 0x81, 0x4f, 0xdc, 0x22, 0x2a, 0x90, 0x88, 0x46, 0xee, 0xb8, 0x14,- 0xde, 0x5e, 0x0b, 0xdb, 0xe0, 0x32, 0x3a, 0x0a, 0x49, 0x06, 0x24, 0x5c, 0xc2,- 0xd3, 0xac, 0x62, 0x91, 0x95, 0xe4, 0x79, 0xe7, 0xc8, 0x37, 0x6d, 0x8d, 0xd5,- 0x4e, 0xa9, 0x6c, 0x56, 0xf4, 0xea, 0x65, 0x7a, 0xae, 0x08, 0xba, 0x78, 0x25,- 0x2e, 0x1c, 0xa6, 0xb4, 0xc6, 0xe8, 0xdd, 0x74, 0x1f, 0x4b, 0xbd, 0x8b, 0x8a,- 0x70, 0x3e, 0xb5, 0x66, 0x48, 0x03, 0xf6, 0x0e, 0x61, 0x35, 0x57, 0xb9, 0x86,- 0xc1, 0x1d, 0x9e, 0xe1, 0xf8, 0x98, 0x11, 0x69, 0xd9, 0x8e, 0x94, 0x9b, 0x1e,- 0x87, 0xe9, 0xce, 0x55, 0x28, 0xdf, 0x8c, 0xa1, 0x89, 0x0d, 0xbf, 0xe6, 0x42,- 0x68, 0x41, 0x99, 0x2d, 0x0f, 0xb0, 0x54, 0xbb, 0x16-};--static uint8_t rsbox[256] = {- 0x52, 0x09, 0x6a, 0xd5, 0x30, 0x36, 0xa5, 0x38, 0xbf, 0x40, 0xa3, 0x9e, 0x81,- 0xf3, 0xd7, 0xfb, 0x7c, 0xe3, 0x39, 0x82, 0x9b, 0x2f, 0xff, 0x87, 0x34, 0x8e,- 0x43, 0x44, 0xc4, 0xde, 0xe9, 0xcb, 0x54, 0x7b, 0x94, 0x32, 0xa6, 0xc2, 0x23,- 0x3d, 0xee, 0x4c, 0x95, 0x0b, 0x42, 0xfa, 0xc3, 0x4e, 0x08, 0x2e, 0xa1, 0x66,- 0x28, 0xd9, 0x24, 0xb2, 0x76, 0x5b, 0xa2, 0x49, 0x6d, 0x8b, 0xd1, 0x25, 0x72,- 0xf8, 0xf6, 0x64, 0x86, 0x68, 0x98, 0x16, 0xd4, 0xa4, 0x5c, 0xcc, 0x5d, 0x65,- 0xb6, 0x92, 0x6c, 0x70, 0x48, 0x50, 0xfd, 0xed, 0xb9, 0xda, 0x5e, 0x15, 0x46,- 0x57, 0xa7, 0x8d, 0x9d, 0x84, 0x90, 0xd8, 0xab, 0x00, 0x8c, 0xbc, 0xd3, 0x0a,- 0xf7, 0xe4, 0x58, 0x05, 0xb8, 0xb3, 0x45, 0x06, 0xd0, 0x2c, 0x1e, 0x8f, 0xca,- 0x3f, 0x0f, 0x02, 0xc1, 0xaf, 0xbd, 0x03, 0x01, 0x13, 0x8a, 0x6b, 0x3a, 0x91,- 0x11, 0x41, 0x4f, 0x67, 0xdc, 0xea, 0x97, 0xf2, 0xcf, 0xce, 0xf0, 0xb4, 0xe6,- 0x73, 0x96, 0xac, 0x74, 0x22, 0xe7, 0xad, 0x35, 0x85, 0xe2, 0xf9, 0x37, 0xe8,- 0x1c, 0x75, 0xdf, 0x6e, 0x47, 0xf1, 0x1a, 0x71, 0x1d, 0x29, 0xc5, 0x89, 0x6f,- 0xb7, 0x62, 0x0e, 0xaa, 0x18, 0xbe, 0x1b, 0xfc, 0x56, 0x3e, 0x4b, 0xc6, 0xd2,- 0x79, 0x20, 0x9a, 0xdb, 0xc0, 0xfe, 0x78, 0xcd, 0x5a, 0xf4, 0x1f, 0xdd, 0xa8,- 0x33, 0x88, 0x07, 0xc7, 0x31, 0xb1, 0x12, 0x10, 0x59, 0x27, 0x80, 0xec, 0x5f,- 0x60, 0x51, 0x7f, 0xa9, 0x19, 0xb5, 0x4a, 0x0d, 0x2d, 0xe5, 0x7a, 0x9f, 0x93,- 0xc9, 0x9c, 0xef, 0xa0, 0xe0, 0x3b, 0x4d, 0xae, 0x2a, 0xf5, 0xb0, 0xc8, 0xeb,- 0xbb, 0x3c, 0x83, 0x53, 0x99, 0x61, 0x17, 0x2b, 0x04, 0x7e, 0xba, 0x77, 0xd6,- 0x26, 0xe1, 0x69, 0x14, 0x63, 0x55, 0x21, 0x0c, 0x7d-};--static uint8_t Rcon[] = {- 0x8d, 0x01, 0x02, 0x04, 0x08, 0x10, 0x20, 0x40, 0x80, 0x1b, 0x36, 0x6c, 0xd8,- 0xab, 0x4d, 0x9a, 0x2f, 0x5e, 0xbc, 0x63, 0xc6, 0x97, 0x35, 0x6a, 0xd4, 0xb3,- 0x7d, 0xfa, 0xef, 0xc5, 0x91, 0x39, 0x72, 0xe4, 0xd3, 0xbd, 0x61, 0xc2, 0x9f,- 0x25, 0x4a, 0x94, 0x33, 0x66, 0xcc, 0x83, 0x1d, 0x3a, 0x74, 0xe8, 0xcb,-};+#include "bearssl/inner.h"+#include "aes/block128.h"+#include "aes/generic.h" -#define G(a,b,c,d,e,f) { a,b,c,d,e,f }-static uint8_t gmtab[256][6] =-{- G(0x00, 0x00, 0x00, 0x00, 0x00, 0x00), G(0x02, 0x03, 0x09, 0x0b, 0x0d, 0x0e),- G(0x04, 0x06, 0x12, 0x16, 0x1a, 0x1c), G(0x06, 0x05, 0x1b, 0x1d, 0x17, 0x12),- G(0x08, 0x0c, 0x24, 0x2c, 0x34, 0x38), G(0x0a, 0x0f, 0x2d, 0x27, 0x39, 0x36),- G(0x0c, 0x0a, 0x36, 0x3a, 0x2e, 0x24), G(0x0e, 0x09, 0x3f, 0x31, 0x23, 0x2a),- G(0x10, 0x18, 0x48, 0x58, 0x68, 0x70), G(0x12, 0x1b, 0x41, 0x53, 0x65, 0x7e),- G(0x14, 0x1e, 0x5a, 0x4e, 0x72, 0x6c), G(0x16, 0x1d, 0x53, 0x45, 0x7f, 0x62),- G(0x18, 0x14, 0x6c, 0x74, 0x5c, 0x48), G(0x1a, 0x17, 0x65, 0x7f, 0x51, 0x46),- G(0x1c, 0x12, 0x7e, 0x62, 0x46, 0x54), G(0x1e, 0x11, 0x77, 0x69, 0x4b, 0x5a),- G(0x20, 0x30, 0x90, 0xb0, 0xd0, 0xe0), G(0x22, 0x33, 0x99, 0xbb, 0xdd, 0xee),- G(0x24, 0x36, 0x82, 0xa6, 0xca, 0xfc), G(0x26, 0x35, 0x8b, 0xad, 0xc7, 0xf2),- G(0x28, 0x3c, 0xb4, 0x9c, 0xe4, 0xd8), G(0x2a, 0x3f, 0xbd, 0x97, 0xe9, 0xd6),- G(0x2c, 0x3a, 0xa6, 0x8a, 0xfe, 0xc4), G(0x2e, 0x39, 0xaf, 0x81, 0xf3, 0xca),- G(0x30, 0x28, 0xd8, 0xe8, 0xb8, 0x90), G(0x32, 0x2b, 0xd1, 0xe3, 0xb5, 0x9e),- G(0x34, 0x2e, 0xca, 0xfe, 0xa2, 0x8c), G(0x36, 0x2d, 0xc3, 0xf5, 0xaf, 0x82),- G(0x38, 0x24, 0xfc, 0xc4, 0x8c, 0xa8), G(0x3a, 0x27, 0xf5, 0xcf, 0x81, 0xa6),- G(0x3c, 0x22, 0xee, 0xd2, 0x96, 0xb4), G(0x3e, 0x21, 0xe7, 0xd9, 0x9b, 0xba),- G(0x40, 0x60, 0x3b, 0x7b, 0xbb, 0xdb), G(0x42, 0x63, 0x32, 0x70, 0xb6, 0xd5),- G(0x44, 0x66, 0x29, 0x6d, 0xa1, 0xc7), G(0x46, 0x65, 0x20, 0x66, 0xac, 0xc9),- G(0x48, 0x6c, 0x1f, 0x57, 0x8f, 0xe3), G(0x4a, 0x6f, 0x16, 0x5c, 0x82, 0xed),- G(0x4c, 0x6a, 0x0d, 0x41, 0x95, 0xff), G(0x4e, 0x69, 0x04, 0x4a, 0x98, 0xf1),- G(0x50, 0x78, 0x73, 0x23, 0xd3, 0xab), G(0x52, 0x7b, 0x7a, 0x28, 0xde, 0xa5),- G(0x54, 0x7e, 0x61, 0x35, 0xc9, 0xb7), G(0x56, 0x7d, 0x68, 0x3e, 0xc4, 0xb9),- G(0x58, 0x74, 0x57, 0x0f, 0xe7, 0x93), G(0x5a, 0x77, 0x5e, 0x04, 0xea, 0x9d),- G(0x5c, 0x72, 0x45, 0x19, 0xfd, 0x8f), G(0x5e, 0x71, 0x4c, 0x12, 0xf0, 0x81),- G(0x60, 0x50, 0xab, 0xcb, 0x6b, 0x3b), G(0x62, 0x53, 0xa2, 0xc0, 0x66, 0x35),- G(0x64, 0x56, 0xb9, 0xdd, 0x71, 0x27), G(0x66, 0x55, 0xb0, 0xd6, 0x7c, 0x29),- G(0x68, 0x5c, 0x8f, 0xe7, 0x5f, 0x03), G(0x6a, 0x5f, 0x86, 0xec, 0x52, 0x0d),- G(0x6c, 0x5a, 0x9d, 0xf1, 0x45, 0x1f), G(0x6e, 0x59, 0x94, 0xfa, 0x48, 0x11),- G(0x70, 0x48, 0xe3, 0x93, 0x03, 0x4b), G(0x72, 0x4b, 0xea, 0x98, 0x0e, 0x45),- G(0x74, 0x4e, 0xf1, 0x85, 0x19, 0x57), G(0x76, 0x4d, 0xf8, 0x8e, 0x14, 0x59),- G(0x78, 0x44, 0xc7, 0xbf, 0x37, 0x73), G(0x7a, 0x47, 0xce, 0xb4, 0x3a, 0x7d),- G(0x7c, 0x42, 0xd5, 0xa9, 0x2d, 0x6f), G(0x7e, 0x41, 0xdc, 0xa2, 0x20, 0x61),- G(0x80, 0xc0, 0x76, 0xf6, 0x6d, 0xad), G(0x82, 0xc3, 0x7f, 0xfd, 0x60, 0xa3),- G(0x84, 0xc6, 0x64, 0xe0, 0x77, 0xb1), G(0x86, 0xc5, 0x6d, 0xeb, 0x7a, 0xbf),- G(0x88, 0xcc, 0x52, 0xda, 0x59, 0x95), G(0x8a, 0xcf, 0x5b, 0xd1, 0x54, 0x9b),- G(0x8c, 0xca, 0x40, 0xcc, 0x43, 0x89), G(0x8e, 0xc9, 0x49, 0xc7, 0x4e, 0x87),- G(0x90, 0xd8, 0x3e, 0xae, 0x05, 0xdd), G(0x92, 0xdb, 0x37, 0xa5, 0x08, 0xd3),- G(0x94, 0xde, 0x2c, 0xb8, 0x1f, 0xc1), G(0x96, 0xdd, 0x25, 0xb3, 0x12, 0xcf),- G(0x98, 0xd4, 0x1a, 0x82, 0x31, 0xe5), G(0x9a, 0xd7, 0x13, 0x89, 0x3c, 0xeb),- G(0x9c, 0xd2, 0x08, 0x94, 0x2b, 0xf9), G(0x9e, 0xd1, 0x01, 0x9f, 0x26, 0xf7),- G(0xa0, 0xf0, 0xe6, 0x46, 0xbd, 0x4d), G(0xa2, 0xf3, 0xef, 0x4d, 0xb0, 0x43),- G(0xa4, 0xf6, 0xf4, 0x50, 0xa7, 0x51), G(0xa6, 0xf5, 0xfd, 0x5b, 0xaa, 0x5f),- G(0xa8, 0xfc, 0xc2, 0x6a, 0x89, 0x75), G(0xaa, 0xff, 0xcb, 0x61, 0x84, 0x7b),- G(0xac, 0xfa, 0xd0, 0x7c, 0x93, 0x69), G(0xae, 0xf9, 0xd9, 0x77, 0x9e, 0x67),- G(0xb0, 0xe8, 0xae, 0x1e, 0xd5, 0x3d), G(0xb2, 0xeb, 0xa7, 0x15, 0xd8, 0x33),- G(0xb4, 0xee, 0xbc, 0x08, 0xcf, 0x21), G(0xb6, 0xed, 0xb5, 0x03, 0xc2, 0x2f),- G(0xb8, 0xe4, 0x8a, 0x32, 0xe1, 0x05), G(0xba, 0xe7, 0x83, 0x39, 0xec, 0x0b),- G(0xbc, 0xe2, 0x98, 0x24, 0xfb, 0x19), G(0xbe, 0xe1, 0x91, 0x2f, 0xf6, 0x17),- G(0xc0, 0xa0, 0x4d, 0x8d, 0xd6, 0x76), G(0xc2, 0xa3, 0x44, 0x86, 0xdb, 0x78),- G(0xc4, 0xa6, 0x5f, 0x9b, 0xcc, 0x6a), G(0xc6, 0xa5, 0x56, 0x90, 0xc1, 0x64),- G(0xc8, 0xac, 0x69, 0xa1, 0xe2, 0x4e), G(0xca, 0xaf, 0x60, 0xaa, 0xef, 0x40),- G(0xcc, 0xaa, 0x7b, 0xb7, 0xf8, 0x52), G(0xce, 0xa9, 0x72, 0xbc, 0xf5, 0x5c),- G(0xd0, 0xb8, 0x05, 0xd5, 0xbe, 0x06), G(0xd2, 0xbb, 0x0c, 0xde, 0xb3, 0x08),- G(0xd4, 0xbe, 0x17, 0xc3, 0xa4, 0x1a), G(0xd6, 0xbd, 0x1e, 0xc8, 0xa9, 0x14),- G(0xd8, 0xb4, 0x21, 0xf9, 0x8a, 0x3e), G(0xda, 0xb7, 0x28, 0xf2, 0x87, 0x30),- G(0xdc, 0xb2, 0x33, 0xef, 0x90, 0x22), G(0xde, 0xb1, 0x3a, 0xe4, 0x9d, 0x2c),- G(0xe0, 0x90, 0xdd, 0x3d, 0x06, 0x96), G(0xe2, 0x93, 0xd4, 0x36, 0x0b, 0x98),- G(0xe4, 0x96, 0xcf, 0x2b, 0x1c, 0x8a), G(0xe6, 0x95, 0xc6, 0x20, 0x11, 0x84),- G(0xe8, 0x9c, 0xf9, 0x11, 0x32, 0xae), G(0xea, 0x9f, 0xf0, 0x1a, 0x3f, 0xa0),- G(0xec, 0x9a, 0xeb, 0x07, 0x28, 0xb2), G(0xee, 0x99, 0xe2, 0x0c, 0x25, 0xbc),- G(0xf0, 0x88, 0x95, 0x65, 0x6e, 0xe6), G(0xf2, 0x8b, 0x9c, 0x6e, 0x63, 0xe8),- G(0xf4, 0x8e, 0x87, 0x73, 0x74, 0xfa), G(0xf6, 0x8d, 0x8e, 0x78, 0x79, 0xf4),- G(0xf8, 0x84, 0xb1, 0x49, 0x5a, 0xde), G(0xfa, 0x87, 0xb8, 0x42, 0x57, 0xd0),- G(0xfc, 0x82, 0xa3, 0x5f, 0x40, 0xc2), G(0xfe, 0x81, 0xaa, 0x54, 0x4d, 0xcc),- G(0x1b, 0x9b, 0xec, 0xf7, 0xda, 0x41), G(0x19, 0x98, 0xe5, 0xfc, 0xd7, 0x4f),- G(0x1f, 0x9d, 0xfe, 0xe1, 0xc0, 0x5d), G(0x1d, 0x9e, 0xf7, 0xea, 0xcd, 0x53),- G(0x13, 0x97, 0xc8, 0xdb, 0xee, 0x79), G(0x11, 0x94, 0xc1, 0xd0, 0xe3, 0x77),- G(0x17, 0x91, 0xda, 0xcd, 0xf4, 0x65), G(0x15, 0x92, 0xd3, 0xc6, 0xf9, 0x6b),- G(0x0b, 0x83, 0xa4, 0xaf, 0xb2, 0x31), G(0x09, 0x80, 0xad, 0xa4, 0xbf, 0x3f),- G(0x0f, 0x85, 0xb6, 0xb9, 0xa8, 0x2d), G(0x0d, 0x86, 0xbf, 0xb2, 0xa5, 0x23),- G(0x03, 0x8f, 0x80, 0x83, 0x86, 0x09), G(0x01, 0x8c, 0x89, 0x88, 0x8b, 0x07),- G(0x07, 0x89, 0x92, 0x95, 0x9c, 0x15), G(0x05, 0x8a, 0x9b, 0x9e, 0x91, 0x1b),- G(0x3b, 0xab, 0x7c, 0x47, 0x0a, 0xa1), G(0x39, 0xa8, 0x75, 0x4c, 0x07, 0xaf),- G(0x3f, 0xad, 0x6e, 0x51, 0x10, 0xbd), G(0x3d, 0xae, 0x67, 0x5a, 0x1d, 0xb3),- G(0x33, 0xa7, 0x58, 0x6b, 0x3e, 0x99), G(0x31, 0xa4, 0x51, 0x60, 0x33, 0x97),- G(0x37, 0xa1, 0x4a, 0x7d, 0x24, 0x85), G(0x35, 0xa2, 0x43, 0x76, 0x29, 0x8b),- G(0x2b, 0xb3, 0x34, 0x1f, 0x62, 0xd1), G(0x29, 0xb0, 0x3d, 0x14, 0x6f, 0xdf),- G(0x2f, 0xb5, 0x26, 0x09, 0x78, 0xcd), G(0x2d, 0xb6, 0x2f, 0x02, 0x75, 0xc3),- G(0x23, 0xbf, 0x10, 0x33, 0x56, 0xe9), G(0x21, 0xbc, 0x19, 0x38, 0x5b, 0xe7),- G(0x27, 0xb9, 0x02, 0x25, 0x4c, 0xf5), G(0x25, 0xba, 0x0b, 0x2e, 0x41, 0xfb),- G(0x5b, 0xfb, 0xd7, 0x8c, 0x61, 0x9a), G(0x59, 0xf8, 0xde, 0x87, 0x6c, 0x94),- G(0x5f, 0xfd, 0xc5, 0x9a, 0x7b, 0x86), G(0x5d, 0xfe, 0xcc, 0x91, 0x76, 0x88),- G(0x53, 0xf7, 0xf3, 0xa0, 0x55, 0xa2), G(0x51, 0xf4, 0xfa, 0xab, 0x58, 0xac),- G(0x57, 0xf1, 0xe1, 0xb6, 0x4f, 0xbe), G(0x55, 0xf2, 0xe8, 0xbd, 0x42, 0xb0),- G(0x4b, 0xe3, 0x9f, 0xd4, 0x09, 0xea), G(0x49, 0xe0, 0x96, 0xdf, 0x04, 0xe4),- G(0x4f, 0xe5, 0x8d, 0xc2, 0x13, 0xf6), G(0x4d, 0xe6, 0x84, 0xc9, 0x1e, 0xf8),- G(0x43, 0xef, 0xbb, 0xf8, 0x3d, 0xd2), G(0x41, 0xec, 0xb2, 0xf3, 0x30, 0xdc),- G(0x47, 0xe9, 0xa9, 0xee, 0x27, 0xce), G(0x45, 0xea, 0xa0, 0xe5, 0x2a, 0xc0),- G(0x7b, 0xcb, 0x47, 0x3c, 0xb1, 0x7a), G(0x79, 0xc8, 0x4e, 0x37, 0xbc, 0x74),- G(0x7f, 0xcd, 0x55, 0x2a, 0xab, 0x66), G(0x7d, 0xce, 0x5c, 0x21, 0xa6, 0x68),- G(0x73, 0xc7, 0x63, 0x10, 0x85, 0x42), G(0x71, 0xc4, 0x6a, 0x1b, 0x88, 0x4c),- G(0x77, 0xc1, 0x71, 0x06, 0x9f, 0x5e), G(0x75, 0xc2, 0x78, 0x0d, 0x92, 0x50),- G(0x6b, 0xd3, 0x0f, 0x64, 0xd9, 0x0a), G(0x69, 0xd0, 0x06, 0x6f, 0xd4, 0x04),- G(0x6f, 0xd5, 0x1d, 0x72, 0xc3, 0x16), G(0x6d, 0xd6, 0x14, 0x79, 0xce, 0x18),- G(0x63, 0xdf, 0x2b, 0x48, 0xed, 0x32), G(0x61, 0xdc, 0x22, 0x43, 0xe0, 0x3c),- G(0x67, 0xd9, 0x39, 0x5e, 0xf7, 0x2e), G(0x65, 0xda, 0x30, 0x55, 0xfa, 0x20),- G(0x9b, 0x5b, 0x9a, 0x01, 0xb7, 0xec), G(0x99, 0x58, 0x93, 0x0a, 0xba, 0xe2),- G(0x9f, 0x5d, 0x88, 0x17, 0xad, 0xf0), G(0x9d, 0x5e, 0x81, 0x1c, 0xa0, 0xfe),- G(0x93, 0x57, 0xbe, 0x2d, 0x83, 0xd4), G(0x91, 0x54, 0xb7, 0x26, 0x8e, 0xda),- G(0x97, 0x51, 0xac, 0x3b, 0x99, 0xc8), G(0x95, 0x52, 0xa5, 0x30, 0x94, 0xc6),- G(0x8b, 0x43, 0xd2, 0x59, 0xdf, 0x9c), G(0x89, 0x40, 0xdb, 0x52, 0xd2, 0x92),- G(0x8f, 0x45, 0xc0, 0x4f, 0xc5, 0x80), G(0x8d, 0x46, 0xc9, 0x44, 0xc8, 0x8e),- G(0x83, 0x4f, 0xf6, 0x75, 0xeb, 0xa4), G(0x81, 0x4c, 0xff, 0x7e, 0xe6, 0xaa),- G(0x87, 0x49, 0xe4, 0x63, 0xf1, 0xb8), G(0x85, 0x4a, 0xed, 0x68, 0xfc, 0xb6),- G(0xbb, 0x6b, 0x0a, 0xb1, 0x67, 0x0c), G(0xb9, 0x68, 0x03, 0xba, 0x6a, 0x02),- G(0xbf, 0x6d, 0x18, 0xa7, 0x7d, 0x10), G(0xbd, 0x6e, 0x11, 0xac, 0x70, 0x1e),- G(0xb3, 0x67, 0x2e, 0x9d, 0x53, 0x34), G(0xb1, 0x64, 0x27, 0x96, 0x5e, 0x3a),- G(0xb7, 0x61, 0x3c, 0x8b, 0x49, 0x28), G(0xb5, 0x62, 0x35, 0x80, 0x44, 0x26),- G(0xab, 0x73, 0x42, 0xe9, 0x0f, 0x7c), G(0xa9, 0x70, 0x4b, 0xe2, 0x02, 0x72),- G(0xaf, 0x75, 0x50, 0xff, 0x15, 0x60), G(0xad, 0x76, 0x59, 0xf4, 0x18, 0x6e),- G(0xa3, 0x7f, 0x66, 0xc5, 0x3b, 0x44), G(0xa1, 0x7c, 0x6f, 0xce, 0x36, 0x4a),- G(0xa7, 0x79, 0x74, 0xd3, 0x21, 0x58), G(0xa5, 0x7a, 0x7d, 0xd8, 0x2c, 0x56),- G(0xdb, 0x3b, 0xa1, 0x7a, 0x0c, 0x37), G(0xd9, 0x38, 0xa8, 0x71, 0x01, 0x39),- G(0xdf, 0x3d, 0xb3, 0x6c, 0x16, 0x2b), G(0xdd, 0x3e, 0xba, 0x67, 0x1b, 0x25),- G(0xd3, 0x37, 0x85, 0x56, 0x38, 0x0f), G(0xd1, 0x34, 0x8c, 0x5d, 0x35, 0x01),- G(0xd7, 0x31, 0x97, 0x40, 0x22, 0x13), G(0xd5, 0x32, 0x9e, 0x4b, 0x2f, 0x1d),- G(0xcb, 0x23, 0xe9, 0x22, 0x64, 0x47), G(0xc9, 0x20, 0xe0, 0x29, 0x69, 0x49),- G(0xcf, 0x25, 0xfb, 0x34, 0x7e, 0x5b), G(0xcd, 0x26, 0xf2, 0x3f, 0x73, 0x55),- G(0xc3, 0x2f, 0xcd, 0x0e, 0x50, 0x7f), G(0xc1, 0x2c, 0xc4, 0x05, 0x5d, 0x71),- G(0xc7, 0x29, 0xdf, 0x18, 0x4a, 0x63), G(0xc5, 0x2a, 0xd6, 0x13, 0x47, 0x6d),- G(0xfb, 0x0b, 0x31, 0xca, 0xdc, 0xd7), G(0xf9, 0x08, 0x38, 0xc1, 0xd1, 0xd9),- G(0xff, 0x0d, 0x23, 0xdc, 0xc6, 0xcb), G(0xfd, 0x0e, 0x2a, 0xd7, 0xcb, 0xc5),- G(0xf3, 0x07, 0x15, 0xe6, 0xe8, 0xef), G(0xf1, 0x04, 0x1c, 0xed, 0xe5, 0xe1),- G(0xf7, 0x01, 0x07, 0xf0, 0xf2, 0xf3), G(0xf5, 0x02, 0x0e, 0xfb, 0xff, 0xfd),- G(0xeb, 0x13, 0x79, 0x92, 0xb4, 0xa7), G(0xe9, 0x10, 0x70, 0x99, 0xb9, 0xa9),- G(0xef, 0x15, 0x6b, 0x84, 0xae, 0xbb), G(0xed, 0x16, 0x62, 0x8f, 0xa3, 0xb5),- G(0xe3, 0x1f, 0x5d, 0xbe, 0x80, 0x9f), G(0xe1, 0x1c, 0x54, 0xb5, 0x8d, 0x91),- G(0xe7, 0x19, 0x4f, 0xa8, 0x9a, 0x83), G(0xe5, 0x1a, 0x46, 0xa3, 0x97, 0x8d),-};-#undef G+/*+ * The schedule is kept in its compressed form, 30 words of it, inside the+ * aes_key the caller already has -- comfortably inside the 448 bytes that+ * held the round keys before. br_aes_ct64_skey_expand blows it up to 120+ * words on the stack once per call, which is how BearSSL's own CTR and CBC+ * use it.+ *+ * It travels through memcpy rather than a cast because aes_key is all+ * uint8_t and so carries no alignment of its own, whatever the allocator+ * happens to give it.+ */+#define COMP_SKEY_WORDS 30 -static void expand_key(uint8_t *expandedKey, uint8_t *key, int size, size_t expandedKeySize)+void crypton_aes_generic_schedule(aes_sched *sched, const aes_key *key) {- int csz;- int i;- uint8_t t[4] = { 0 };-- for (i = 0; i < size; i++)- expandedKey[i] = key[i];- csz = size;+ uint64_t comp_skey[COMP_SKEY_WORDS]; - i = 1;- while (csz < expandedKeySize) {- t[0] = expandedKey[(csz - 4) + 0];- t[1] = expandedKey[(csz - 4) + 1];- t[2] = expandedKey[(csz - 4) + 2];- t[3] = expandedKey[(csz - 4) + 3];+ memcpy(comp_skey, key->data, sizeof comp_skey);+ sched->nbr = key->nbr;+ br_aes_ct64_skey_expand(sched->sk_exp, sched->nbr, comp_skey);+} - if (csz % size == 0) {- uint8_t tmp;+/*+ * Up to four blocks through the bitsliced core at once. Fewer than four is+ * the same work as four -- the lanes are there whether anything is in them+ * -- so a caller with four to offer gets them for what one used to cost.+ */+static void pass(uint8_t *output, const uint8_t *input, unsigned n,+ const aes_sched *sched, int decrypt)+{+ uint32_t w[16];+ uint64_t q[8];+ unsigned i; - tmp = t[0];- t[0] = sbox[t[1]] ^ Rcon[i++ % sizeof(Rcon)];- t[1] = sbox[t[2]];- t[2] = sbox[t[3]];- t[3] = sbox[tmp];- }+ memset(w, 0, sizeof w);+ for (i = 0; i < n; i++) {+ w[4 * i] = br_dec32le(input + 16 * i);+ w[4 * i + 1] = br_dec32le(input + 16 * i + 4);+ w[4 * i + 2] = br_dec32le(input + 16 * i + 8);+ w[4 * i + 3] = br_dec32le(input + 16 * i + 12);+ } - if (size == 32 && ((csz % size) == 16)) {- t[0] = sbox[t[0]];- t[1] = sbox[t[1]];- t[2] = sbox[t[2]];- t[3] = sbox[t[3]];- }+ for (i = 0; i < 4; i++)+ br_aes_ct64_interleave_in(&q[i], &q[i + 4], w + 4 * i);+ br_aes_ct64_ortho(q);+ if (decrypt)+ br_aes_ct64_bitslice_decrypt(sched->nbr, sched->sk_exp, q);+ else+ br_aes_ct64_bitslice_encrypt(sched->nbr, sched->sk_exp, q);+ br_aes_ct64_ortho(q);+ for (i = 0; i < 4; i++)+ br_aes_ct64_interleave_out(w + 4 * i, q[i], q[i + 4]); - expandedKey[csz] = expandedKey[csz - size] ^ t[0]; csz++;- expandedKey[csz] = expandedKey[csz - size] ^ t[1]; csz++;- expandedKey[csz] = expandedKey[csz - size] ^ t[2]; csz++;- expandedKey[csz] = expandedKey[csz - size] ^ t[3]; csz++;+ for (i = 0; i < n; i++) {+ br_enc32le(output + 16 * i, w[4 * i]);+ br_enc32le(output + 16 * i + 4, w[4 * i + 1]);+ br_enc32le(output + 16 * i + 8, w[4 * i + 2]);+ br_enc32le(output + 16 * i + 12, w[4 * i + 3]); } } -static void shift_rows(uint8_t *state)+void crypton_aes_generic_blocks(uint8_t *output, const uint8_t *input,+ uint32_t nb_blocks, const aes_sched *sched,+ int decrypt) {- uint32_t *s32;- int i;-- for (i = 0; i < 16; i++)- state[i] = sbox[state[i]];- s32 = (uint32_t *) state;- s32[1] = rol32_be(s32[1], 8);- s32[2] = rol32_be(s32[2], 16);- s32[3] = rol32_be(s32[3], 24);+ while (nb_blocks >= 4) {+ pass(output, input, 4, sched, decrypt);+ output += 64;+ input += 64;+ nb_blocks -= 4;+ }+ if (nb_blocks > 0)+ pass(output, input, (unsigned) nb_blocks, sched, decrypt); } -static void add_round_key(uint8_t *state, uint8_t *rk)+static void one(aes_block *output, aes_key *key, aes_block *input, int decrypt) {- uint32_t *s32, *r32;+ aes_sched sched; - s32 = (uint32_t *) state;- r32 = (uint32_t *) rk;- s32[0] ^= r32[0];- s32[1] ^= r32[1];- s32[2] ^= r32[2];- s32[3] ^= r32[3];+ crypton_aes_generic_schedule(&sched, key);+ crypton_aes_generic_blocks((uint8_t *) output, (const uint8_t *) input,+ 1, &sched, decrypt); } -#define gm1(a) (a)-#define gm2(a) gmtab[a][0]-#define gm3(a) gmtab[a][1]-#define gm9(a) gmtab[a][2]-#define gm11(a) gmtab[a][3]-#define gm13(a) gmtab[a][4]-#define gm14(a) gmtab[a][5]--static void mix_columns(uint8_t *state)+void crypton_aes_generic_encrypt_block(aes_block *output, aes_key *key, aes_block *input) {- int i;- uint8_t cpy[4];-- for (i = 0; i < 4; i++) {- cpy[0] = state[0 * 4 + i];- cpy[1] = state[1 * 4 + i];- cpy[2] = state[2 * 4 + i];- cpy[3] = state[3 * 4 + i];- state[i] = gm2(cpy[0]) ^ gm1(cpy[3]) ^ gm1(cpy[2]) ^ gm3(cpy[1]);- state[4+i] = gm2(cpy[1]) ^ gm1(cpy[0]) ^ gm1(cpy[3]) ^ gm3(cpy[2]);- state[8+i] = gm2(cpy[2]) ^ gm1(cpy[1]) ^ gm1(cpy[0]) ^ gm3(cpy[3]);- state[12+i] = gm2(cpy[3]) ^ gm1(cpy[2]) ^ gm1(cpy[1]) ^ gm3(cpy[0]);- }+ one(output, key, input, 0); } -static void create_round_key(uint8_t *expandedKey, uint8_t *rk)+void crypton_aes_generic_decrypt_block(aes_block *output, aes_key *key, aes_block *input) {- int i,j;- for (i = 0; i < 4; i++)- for (j = 0; j < 4; j++)- rk[i + j * 4] = expandedKey[i * 4 + j];+ one(output, key, input, 1); } -static void aes_main(aes_key *key, uint8_t *state)+void crypton_aes_generic_init(aes_key *key, uint8_t *origkey, uint8_t size) {- int i = 0;- uint32_t rk[4];- uint8_t *rkptr = (uint8_t *) rk;-- create_round_key(key->data, rkptr);- add_round_key(state, rkptr);-- for (i = 1; i < key->nbr; i++) {- create_round_key(key->data + 16 * i, rkptr);- shift_rows(state);- mix_columns(state);- add_round_key(state, rkptr);- }-- create_round_key(key->data + 16 * key->nbr, rkptr);- shift_rows(state);- add_round_key(state, rkptr);-}+ uint64_t comp_skey[COMP_SKEY_WORDS];+ unsigned nbr; -static void shift_rows_inv(uint8_t *state)-{- uint32_t *s32;- int i;+ /* 0 for a key length that is not 16, 24 or 32; the old code returned+ * without touching the key in that case and so does this */+ nbr = br_aes_ct64_keysched(comp_skey, origkey, size);+ if (nbr == 0)+ return; - s32 = (uint32_t *) state;- s32[1] = ror32_be(s32[1], 8);- s32[2] = ror32_be(s32[2], 16);- s32[3] = ror32_be(s32[3], 24);- for (i = 0; i < 16; i++)- state[i] = rsbox[state[i]];+ key->nbr = (uint8_t) nbr;+ memcpy(key->data, comp_skey, sizeof comp_skey); } -static void mix_columns_inv(uint8_t *state)+/*+ * CTR, four counter blocks at a time. The counter itself is serial, but+ * nothing about it depends on the keystream, so the four blocks it will+ * reach next can be written down before any of them is encrypted.+ */+static void ctr(uint8_t *output, aes_key *key, aes_block *iv,+ uint8_t *input, uint32_t len, int c32) {- int i;- uint8_t cpy[4];+ aes_sched sched;+ aes_block counter;+ uint8_t ks[64];+ uint32_t nb_blocks = len / 16;+ uint32_t tail = len % 16;+ uint32_t i; - for (i = 0; i < 4; i++) {- cpy[0] = state[0 * 4 + i];- cpy[1] = state[1 * 4 + i];- cpy[2] = state[2 * 4 + i];- cpy[3] = state[3 * 4 + i];- state[i] = gm14(cpy[0]) ^ gm9(cpy[3]) ^ gm13(cpy[2]) ^ gm11(cpy[1]);- state[4+i] = gm14(cpy[1]) ^ gm9(cpy[0]) ^ gm13(cpy[3]) ^ gm11(cpy[2]);- state[8+i] = gm14(cpy[2]) ^ gm9(cpy[1]) ^ gm13(cpy[0]) ^ gm11(cpy[3]);- state[12+i] = gm14(cpy[3]) ^ gm9(cpy[2]) ^ gm13(cpy[1]) ^ gm11(cpy[0]);- }-}+ crypton_aes_generic_schedule(&sched, key);+ block128_copy(&counter, iv); -static void aes_main_inv(aes_key *key, uint8_t *state)-{- int i = 0;- uint32_t rk[4];- uint8_t *rkptr = (uint8_t *) rk;+ while (nb_blocks > 0) {+ uint32_t n = nb_blocks < 4 ? nb_blocks : 4; - create_round_key(key->data + 16 * key->nbr, rkptr);- add_round_key(state, rkptr);+ for (i = 0; i < n; i++) {+ block128_copy((block128 *) (ks + 16 * i), &counter);+ if (c32)+ block128_inc32_le(&counter);+ else+ block128_inc_be(&counter);+ }+ crypton_aes_generic_blocks(ks, ks, n, &sched, 0);+ for (i = 0; i < n * 16; i++)+ output[i] = ks[i] ^ input[i]; - for (i = key->nbr - 1; i > 0; i--) {- create_round_key(key->data + 16 * i, rkptr);- shift_rows_inv(state);- add_round_key(state, rkptr);- mix_columns_inv(state);+ output += n * 16;+ input += n * 16;+ nb_blocks -= n; } - create_round_key(key->data, rkptr);- shift_rows_inv(state);- add_round_key(state, rkptr);-}--/* Set the block values, for the block:- * a0,0 a0,1 a0,2 a0,3- * a1,0 a1,1 a1,2 a1,3 -> a0,0 a1,0 a2,0 a3,0 a0,1 a1,1 ... a2,3 a3,3- * a2,0 a2,1 a2,2 a2,3- * a3,0 a3,1 a3,2 a3,3- */-#define swap_block(t, f) \- t[0] = f[0]; t[4] = f[1]; t[8] = f[2]; t[12] = f[3]; \- t[1] = f[4]; t[5] = f[5]; t[9] = f[6]; t[13] = f[7]; \- t[2] = f[8]; t[6] = f[9]; t[10] = f[10]; t[14] = f[11]; \- t[3] = f[12]; t[7] = f[13]; t[11] = f[14]; t[15] = f[15]--void crypton_aes_generic_encrypt_block(aes_block *output, aes_key *key, aes_block *input)-{- uint32_t block[4];- uint8_t *iptr, *optr, *bptr;-- iptr = (uint8_t *) input;- optr = (uint8_t *) output;- bptr = (uint8_t *) block;- swap_block(bptr, iptr);- aes_main(key, bptr);- swap_block(optr, bptr);+ if (tail != 0) {+ block128_copy((block128 *) ks, &counter);+ crypton_aes_generic_blocks(ks, ks, 1, &sched, 0);+ for (i = 0; i < tail; i++)+ output[i] = ks[i] ^ input[i];+ } } -void crypton_aes_generic_decrypt_block(aes_block *output, aes_key *key, aes_block *input)+void crypton_aes_bitsliced_encrypt_ctr(uint8_t *output, aes_key *key,+ aes_block *iv, uint8_t *input,+ uint32_t len) {- uint32_t block[4];- uint8_t *iptr, *optr, *bptr;-- iptr = (uint8_t *) input;- optr = (uint8_t *) output;- bptr = (uint8_t *) block;- swap_block(bptr, iptr);- aes_main_inv(key, bptr);- swap_block(optr, bptr);+ ctr(output, key, iv, input, len, 0); } -void crypton_aes_generic_init(aes_key *key, uint8_t *origkey, uint8_t size)+void crypton_aes_bitsliced_encrypt_c32(uint8_t *output, aes_key *key,+ aes_block *iv, uint8_t *input,+ uint32_t len) {- int esz;-- switch (size) {- case 16: key->nbr = 10; esz = 176; break;- case 24: key->nbr = 12; esz = 208; break;- case 32: key->nbr = 14; esz = 240; break;- default: return;- }- expand_key(key->data, origkey, size, esz);- return;+ ctr(output, key, iv, input, len, 1); }
cbits/aes/generic.h view
@@ -32,3 +32,35 @@ void crypton_aes_generic_encrypt_block(aes_block *output, aes_key *key, aes_block *input); void crypton_aes_generic_decrypt_block(aes_block *output, aes_key *key, aes_block *input); void crypton_aes_generic_init(aes_key *key, uint8_t *origkey, uint8_t size);++/*+ * The bitsliced core takes four blocks at a time, and the schedule it reads+ * is the expanded one rather than the compressed form the aes_key holds.+ * Expanding costs about what a block costs, so a mode with more than one+ * block to do expands once, here, and hands the result to every group.+ */+typedef struct {+ uint64_t sk_exp[120];+ unsigned nbr;+} aes_sched;++void crypton_aes_generic_schedule(aes_sched *sched, const aes_key *key);++/* nb_blocks of them, four at a pass; decrypt selects the direction */+void crypton_aes_generic_blocks(uint8_t *output, const uint8_t *input,+ uint32_t nb_blocks, const aes_sched *sched,+ int decrypt);++/*+ * CTR with the two counters crypton uses. These are not the generic+ * entries: those go through the branch table for the block itself and so+ * run on accelerated machines too, where the key holds a different+ * schedule. crypton_aes.c installs these only when nothing was+ * accelerated.+ */+void crypton_aes_bitsliced_encrypt_ctr(uint8_t *output, aes_key *key,+ aes_block *iv, uint8_t *input,+ uint32_t len);+void crypton_aes_bitsliced_encrypt_c32(uint8_t *output, aes_key *key,+ aes_block *iv, uint8_t *input,+ uint32_t len);
cbits/aes/gf.c view
@@ -33,6 +33,7 @@ #include <crypton_cpu.h> #include <aes/gf.h> #include <aes/x86ni.h>+#include "bearssl/inner.h" /* inplace GFMUL for xts mode */ void crypton_aes_generic_gf_mulx(block128 *a)@@ -45,118 +46,43 @@ /*- * GF multiplication with Shoup's method and 4-bit table.+ * GHASH, without a table. *- * We precompute the products of H with all 4-bit polynomials and store them in- * a 'table_4bit' array. To avoid unnecessary byte swapping, the 16 blocks are- * written to the table with qwords already converted to CPU order. Table- * indices use the reflected bit ordering, i.e. polynomials X^0, X^1, X^2, X^3- * map to bit positions 3, 2, 1, 0 respectively.+ * This was Shoup's method: the products of H with all sixteen 4-bit+ * polynomials, precomputed, and thirty-two lookups a block at indices taken+ * from the accumulator -- which is to say, thirty-two addresses derived from+ * a secret. It is now BearSSL's ghash_ctmul64, which builds the GF(2^128)+ * multiply out of shifts, masks and integer multiplies and looks nothing up.+ * See cbits/bearssl/README.md. *- * To multiply an arbitrary block with H, the input block is decomposed in 4-bit- * segments. We get the final result after 32 table lookups and additions, one- * for each segment, interleaving multiplication by P(X)=X^4.+ * The table_4bit the interface names is sixteen blocks wide because the+ * table needed it. This keeps H in the first and leaves the rest alone; the+ * PMULL and PCLMUL implementations keep their own powers of H in that same+ * space, so the width stays as it is. */ -/* convert block128 qwords between BE and CPU order */-static inline void block128_cpu_swap_be(block128 *a, const block128 *b)-{- a->q[1] = cpu_to_be64(b->q[1]);- a->q[0] = cpu_to_be64(b->q[0]);-}--/* multiplication by P(X)=X, assuming qwords already in CPU order */-static inline void cpu_gf_mulx(block128 *a, const block128 *b)-{- uint64_t v0 = b->q[0];- uint64_t v1 = b->q[1];- a->q[1] = v1 >> 1 | v0 << 63;- a->q[0] = v0 >> 1 ^ ((0-(v1 & 1)) & 0xe100000000000000ULL);-}--static const uint64_t r4_0[] =- { 0x0000000000000000ULL, 0x1c20000000000000ULL- , 0x3840000000000000ULL, 0x2460000000000000ULL- , 0x7080000000000000ULL, 0x6ca0000000000000ULL- , 0x48c0000000000000ULL, 0x54e0000000000000ULL- , 0xe100000000000000ULL, 0xfd20000000000000ULL- , 0xd940000000000000ULL, 0xc560000000000000ULL- , 0x9180000000000000ULL, 0x8da0000000000000ULL- , 0xa9c0000000000000ULL, 0xb5e0000000000000ULL- };--/* multiplication by P(X)=X^4, assuming qwords already in CPU order */-static inline void cpu_gf_mulx4(block128 *a, const block128 *b)-{- uint64_t v0 = b->q[0];- uint64_t v1 = b->q[1];- a->q[1] = v1 >> 4 | v0 << 60;- a->q[0] = v0 >> 4 ^ r4_0[v1 & 0xf];-}--/* initialize the 4-bit table given H */+/* remember H */ void crypton_aes_generic_hinit(table_4bit htable, const block128 *h) {- block128 v, *p;- int i, j;-- /* multiplication by 0 is 0 */- block128_zero(&htable[0]);-- /* at index 8=2^3 we have H.X^0 = H */- i = 8;- block128_cpu_swap_be(&htable[i], h); /* in CPU order */- p = &htable[i];-- /* for other powers of 2, repeat multiplication by P(X)=X */- for (i = 4; i > 0; i >>= 1)- {- cpu_gf_mulx(&htable[i], p);- p = &htable[i];- }-- /* remaining elements are linear combinations */- for (i = 2; i < 16; i <<= 1) {- p = &htable[i];- v = *p;- for (j = 1; j < i; j++) {- p[j] = v;- block128_xor_aligned(&p[j], &htable[j]);- }- }+ block128_copy(&htable[0], h); } -/* multiply a block with H */+/*+ * br_ghash_ctmul64 computes y = (y ^ x) * H for each block x it is given, so+ * a block of zeros is the bare multiply this entry is asked for.+ */ void crypton_aes_generic_gf_mul(block128 *a, const table_4bit htable) {- block128 b;- int i;- block128_zero(&b);- for (i = 15; i >= 0; i--)- {- uint8_t v = a->b[i];- block128_xor_aligned(&b, &htable[v & 0xf]); /* high bits (reflected) */- cpu_gf_mulx4(&b, &b);- block128_xor_aligned(&b, &htable[v >> 4]); /* low bits (reflected) */- if (i > 0)- cpu_gf_mulx4(&b, &b);- else- block128_cpu_swap_be(a, &b); /* restore BE order when done */- }+ static const uint8_t zero[16] = { 0 };++ br_ghash_ctmul64(a, &htable[0], zero, sizeof zero); } /*- * Four GHASH steps at once. The generic table-driven multiply has no cheaper- * way to do this than one block at a time; the point of the entry is that the- * PMULL and PCLMUL versions can fold the four products into one reduction, so- * the GCM loops hand over four blocks whenever they have them.+ * Four GHASH steps at once, which here is one call rather than four: the+ * loop inside br_ghash_ctmul64 takes the blocks as they come. */ void crypton_aes_generic_gf_mul4(block128 *a, const block128 *blocks, const table_4bit htable) {- int i;-- for (i = 0; i < 4; i++) {- block128_xor(a, &blocks[i]);- crypton_aes_generic_gf_mul(a, htable);- }+ br_ghash_ctmul64(a, &htable[0], blocks, 4 * sizeof(block128)); }
+ cbits/aes/ppc8.c view
@@ -0,0 +1,287 @@+/*+ * AES and GHASH using the PowerISA 2.07 vector instructions, first+ * implemented by POWER8.+ *+ * The instructions themselves come from CRYPTOGAMS, assembled from+ * cbits/asm/aesp8-ppc-*.S and cbits/asm/ghashp8-ppc-*.S; what is here is the+ * glue that puts them behind crypton's branch table. Nothing in this file+ * branches or indexes on a key or on data.+ *+ * == Where the key schedule lives+ *+ * The assembly takes OpenSSL's AES_KEY -- sixty round-key words and a round+ * count, 244 bytes -- and its set_decrypt_key writes a second, complete one+ * rather than sharing ends with the first the way cbits/aes/armv8.c does.+ * Two of those are 488 bytes and aes_key.data is 448, so both do not fit.+ *+ * So the forward schedule is kept there, with the key the caller gave after+ * it, and the inverse is built on the stack by the operations that need it.+ * Those are all bulk -- ECB, CBC and XTS decryption -- so it is one key+ * schedule per call rather than per block. Single-block decryption pays for+ * one too, and is reached by nothing that runs in a loop: OCB and CCM drive+ * the block function in the encrypting direction.+ */++#include <stdint.h>+#include <string.h>+#include <crypton_aes.h>+#include <crypton_bitfn.h>+#include "aes/block128.h"+#include "aes/gf.h"+#include "aes/ppc8.h"++/* OpenSSL's AES_KEY, which is what the assembly was written against */+typedef struct {+ unsigned int rd_key[60];+ int rounds;+} p8_key;++int crypton_aes_p8_set_encrypt_key(const unsigned char *, int, p8_key *);+int crypton_aes_p8_set_decrypt_key(const unsigned char *, int, p8_key *);+void crypton_aes_p8_encrypt(const unsigned char *, unsigned char *, const p8_key *);+void crypton_aes_p8_decrypt(const unsigned char *, unsigned char *, const p8_key *);+void crypton_aes_p8_cbc_encrypt(const unsigned char *, unsigned char *, size_t,+ const p8_key *, unsigned char *, int);+void crypton_aes_p8_ctr32_encrypt_blocks(const unsigned char *, unsigned char *,+ size_t, const p8_key *,+ const unsigned char *);+void crypton_aes_p8_xts_encrypt(const unsigned char *, unsigned char *, size_t,+ const p8_key *, const p8_key *,+ const unsigned char *);+void crypton_aes_p8_xts_decrypt(const unsigned char *, unsigned char *, size_t,+ const p8_key *, const p8_key *,+ const unsigned char *);+/* void * rather than uint64_t *: the assembly loads these with lvx_u and so+ * does not want the alignment a uint64_t pointer would promise, and+ * block128 is packed. */+void crypton_gcm_init_p8(void *Htable, const void *H);+void crypton_gcm_gmult_p8(void *Xi, const void *Htable);+void crypton_gcm_ghash_p8(void *Xi, const void *Htable,+ const unsigned char *inp, size_t len);++#define FORWARD(k) ((p8_key *) (void *) (k)->data)+#define USERKEY(k) ((uint8_t *) (k)->data + sizeof(p8_key))+#define USERLEN(k) (*((uint8_t *) (k)->data + sizeof(p8_key) + 32))++static void inverse(p8_key *dk, aes_key *key)+{+ crypton_aes_p8_set_decrypt_key(USERKEY(key), USERLEN(key) * 8, dk);+}++void crypton_aes_ppc8_init(aes_key *key, uint8_t *origkey, uint8_t size)+{+ if (size != 16 && size != 24 && size != 32)+ return;+ crypton_aes_p8_set_encrypt_key(origkey, size * 8, FORWARD(key));+ memcpy(USERKEY(key), origkey, size);+ USERLEN(key) = size;+}++void crypton_aes_ppc8_encrypt_block(aes_block *output, aes_key *key, aes_block *input)+{+ crypton_aes_p8_encrypt((const unsigned char *) input,+ (unsigned char *) output, FORWARD(key));+}++void crypton_aes_ppc8_decrypt_block(aes_block *output, aes_key *key, aes_block *input)+{+ p8_key dk;++ inverse(&dk, key);+ crypton_aes_p8_decrypt((const unsigned char *) input,+ (unsigned char *) output, &dk);+ memset(&dk, 0, sizeof dk);+}++/* The assembly has no ECB entry: it is the block function in a loop, which+ * is what the generic implementation does too. */+void crypton_aes_ppc8_encrypt_ecb(aes_block *output, aes_key *key, aes_block *input,+ uint32_t nb_blocks)+{+ for (; nb_blocks-- > 0; input++, output++)+ crypton_aes_p8_encrypt((const unsigned char *) input,+ (unsigned char *) output, FORWARD(key));+}++void crypton_aes_ppc8_decrypt_ecb(aes_block *output, aes_key *key, aes_block *input,+ uint32_t nb_blocks)+{+ p8_key dk;++ inverse(&dk, key);+ for (; nb_blocks-- > 0; input++, output++)+ crypton_aes_p8_decrypt((const unsigned char *) input,+ (unsigned char *) output, &dk);+ memset(&dk, 0, sizeof dk);+}++void crypton_aes_ppc8_encrypt_cbc(aes_block *output, aes_key *key, aes_block *iv,+ aes_block *input, uint32_t nb_blocks)+{+ uint8_t ivbuf[16];++ /* a copy, because the assembly writes the last block back through this+ * pointer and the callers of this entry do not expect their IV touched */+ memcpy(ivbuf, iv, 16);+ crypton_aes_p8_cbc_encrypt((const unsigned char *) input,+ (unsigned char *) output,+ (size_t) nb_blocks * 16, FORWARD(key), ivbuf, 1);+}++void crypton_aes_ppc8_decrypt_cbc(aes_block *output, aes_key *key, aes_block *iv,+ aes_block *input, uint32_t nb_blocks)+{+ p8_key dk;+ uint8_t ivbuf[16];++ inverse(&dk, key);+ memcpy(ivbuf, iv, 16);+ crypton_aes_p8_cbc_encrypt((const unsigned char *) input,+ (unsigned char *) output,+ (size_t) nb_blocks * 16, &dk, ivbuf, 0);+ memset(&dk, 0, sizeof dk);+}++/*+ * CTR, which is where the two counters have to be reconciled.+ *+ * crypton counts with block128_inc_be, over the whole 128 bits; the assembly+ * counts over the low 32 only. They agree until that word wraps, so the+ * work is handed over in runs that stop there, and the carry into the upper+ * bits is done here. This is the same arrangement OpenSSL makes for the+ * same reason.+ */+void crypton_aes_ppc8_encrypt_ctr(uint8_t *output, aes_key *key, aes_block *iv,+ uint8_t *input, uint32_t len)+{+ block128 ctr;+ uint32_t nb_blocks = len / 16;+ uint32_t tail = len % 16;+ uint32_t i;++ block128_copy(&ctr, iv);++ while (nb_blocks > 0) {+ uint64_t low = (uint64_t) be32_to_cpu(ctr.d[3]);+ uint64_t room = 0x100000000ULL - low; /* blocks before it wraps */+ uint32_t n;++ /* room is 2^32 when the word is at zero, which does not fit the+ * type the comparison would narrow it to -- and a zero n here+ * would not advance */+ if (room > (uint64_t) nb_blocks)+ room = (uint64_t) nb_blocks;+ n = (uint32_t) room;++ crypton_aes_p8_ctr32_encrypt_blocks((const unsigned char *) input,+ (unsigned char *) output,+ n, FORWARD(key),+ (const unsigned char *) &ctr);+ /* the assembly does not write the counter back */+ for (i = 0; i < n; i++)+ block128_inc_be(&ctr);++ output += (size_t) n * 16;+ input += (size_t) n * 16;+ nb_blocks -= n;+ }++ if (tail != 0) {+ block128 ks;++ crypton_aes_p8_encrypt((const unsigned char *) &ctr,+ (unsigned char *) &ks, FORWARD(key));+ for (i = 0; i < tail; i++)+ output[i] = ks.b[i] ^ input[i];+ }+}++/*+ * XTS. The assembly encrypts the tweak with its second key, unless that key+ * is NULL, in which case it takes the tweak already encrypted -- which is+ * what this needs, because crypton's entry also carries a starting point and+ * the tweak has to be advanced that many doublings before any block is+ * enciphered.+ */+static void xts_tweak(block128 *tweak, aes_key *k2, aes_block *dataunit,+ uint32_t spoint)+{+ block128_copy(tweak, dataunit);+ crypton_aes_p8_encrypt((const unsigned char *) tweak,+ (unsigned char *) tweak, FORWARD(k2));+ while (spoint-- > 0)+ crypton_aes_generic_gf_mulx(tweak);+}++void crypton_aes_ppc8_encrypt_xts(aes_block *output, aes_key *k1, aes_key *k2,+ aes_block *dataunit, uint32_t spoint,+ aes_block *input, uint32_t nb_blocks)+{+ block128 tweak;++ xts_tweak(&tweak, k2, dataunit, spoint);+ crypton_aes_p8_xts_encrypt((const unsigned char *) input,+ (unsigned char *) output,+ (size_t) nb_blocks * 16, FORWARD(k1), NULL,+ (const unsigned char *) &tweak);+}++void crypton_aes_ppc8_decrypt_xts(aes_block *output, aes_key *k1, aes_key *k2,+ aes_block *dataunit, uint32_t spoint,+ aes_block *input, uint32_t nb_blocks)+{+ block128 tweak;+ p8_key dk;++ /* the tweak is enciphered with k2 forwards either way */+ xts_tweak(&tweak, k2, dataunit, spoint);+ inverse(&dk, k1);+ crypton_aes_p8_xts_decrypt((const unsigned char *) input,+ (unsigned char *) output,+ (size_t) nb_blocks * 16, &dk, NULL,+ (const unsigned char *) &tweak);+ memset(&dk, 0, sizeof dk);+}++/*+ * GHASH. The table the assembly builds is 192 bytes, which fits the sixteen+ * blocks the interface names.+ *+ * The two arguments do not take the same convention, which is worth saying+ * because it does not look like an accident until it is checked. H arrives+ * here as the sixteen bytes GCM defines, and gcm_init_p8 wants those as a+ * pair of host-order words -- so they are swapped on a little-endian machine+ * and left alone on a big-endian one, which be64_to_cpu does. The+ * accumulator is the byte string throughout, in and out, and is passed as it+ * stands. OpenSSL makes the same pair of choices.+ *+ * Checked rather than assumed: against crypton's own GHASH, over two+ * different H and with a non-zero accumulator, so that neither a symmetric+ * value nor a zero start could hide a wrong one.+ */+void crypton_aes_ppc8_hinit(table_4bit htable, const block128 *h)+{+ uint64_t H[2];++ memcpy(H, h, sizeof H);+ H[0] = be64_to_cpu(H[0]);+ H[1] = be64_to_cpu(H[1]);+ crypton_gcm_init_p8(htable, H);+}++void crypton_aes_ppc8_gf_mul(block128 *a, const table_4bit htable)+{+ crypton_gcm_gmult_p8(a, htable);+}++void crypton_aes_ppc8_gf_mul4(block128 *a, const block128 *blocks,+ const table_4bit htable)+{+ crypton_gcm_ghash_p8(a, htable, (const unsigned char *) blocks,+ 4 * sizeof(block128));+}++int crypton_aes_ppc8_available(void)+{+ return (crypton_ppc_features() & CRYPTON_PPC_VCRYPTO) != 0;+}
+ cbits/aes/ppc8.h view
@@ -0,0 +1,29 @@+/*+ * AES and GHASH on PowerISA 2.07, as implemented by cbits/aes/ppc8.c over+ * the CRYPTOGAMS assembly. crypton_aes.c installs these when+ * crypton_aes_ppc8_available says the processor has the instructions.+ */+#ifndef CRYPTON_AES_PPC8_H+#define CRYPTON_AES_PPC8_H++#include "crypton_aes.h"+#include "aes/gf.h"+#include "crypton_cpu.h"++int crypton_aes_ppc8_available(void);++void crypton_aes_ppc8_init(aes_key *key, uint8_t *origkey, uint8_t size);+void crypton_aes_ppc8_encrypt_block(aes_block *output, aes_key *key, aes_block *input);+void crypton_aes_ppc8_decrypt_block(aes_block *output, aes_key *key, aes_block *input);+void crypton_aes_ppc8_encrypt_ecb(aes_block *output, aes_key *key, aes_block *input, uint32_t nb_blocks);+void crypton_aes_ppc8_decrypt_ecb(aes_block *output, aes_key *key, aes_block *input, uint32_t nb_blocks);+void crypton_aes_ppc8_encrypt_cbc(aes_block *output, aes_key *key, aes_block *iv, aes_block *input, uint32_t nb_blocks);+void crypton_aes_ppc8_decrypt_cbc(aes_block *output, aes_key *key, aes_block *iv, aes_block *input, uint32_t nb_blocks);+void crypton_aes_ppc8_encrypt_ctr(uint8_t *output, aes_key *key, aes_block *iv, uint8_t *input, uint32_t len);+void crypton_aes_ppc8_encrypt_xts(aes_block *output, aes_key *k1, aes_key *k2, aes_block *dataunit, uint32_t spoint, aes_block *input, uint32_t nb_blocks);+void crypton_aes_ppc8_decrypt_xts(aes_block *output, aes_key *k1, aes_key *k2, aes_block *dataunit, uint32_t spoint, aes_block *input, uint32_t nb_blocks);+void crypton_aes_ppc8_hinit(table_4bit htable, const block128 *h);+void crypton_aes_ppc8_gf_mul(block128 *a, const table_4bit htable);+void crypton_aes_ppc8_gf_mul4(block128 *a, const block128 *blocks, const table_4bit htable);++#endif
cbits/asm/README.md view
@@ -2,7 +2,7 @@ ## What is here -Two modules from [CRYPTOGAMS](https://github.com/dot-asm/cryptogams), by Andy+Modules from [CRYPTOGAMS](https://github.com/dot-asm/cryptogams), by Andy Polyakov, checked in unmodified together with the translators they need: | generator | what it is |@@ -17,6 +17,8 @@ | `sha1-armv8.pl` | SHA-1 for AArch64 | | `sha512-armv8.pl` | SHA-256 for AArch64 (the generator emits SHA-512 or SHA-256 according to the name it is given, and only the latter is wanted) | | `keccak1600-armv8.pl` | Keccak for AArch64 |+| `aesp8-ppc.pl` | AES for PowerISA 2.07, which POWER8 was the first to implement |+| `ghashp8-ppc.pl` | GHASH for the same, over `vpmsumd` | `x86_64-xlate.pl`, `arm-xlate.pl` and `arm_arch.h` are the machinery those modules use. `generate.sh` runs the generators to produce the `.S` files, which are
+ cbits/asm/aesp8-ppc-linux64le.S view
@@ -0,0 +1,3658 @@+.machine "any"++.abiversion 2+.text++.align 7+rcon:+.byte 0x00,0x00,0x00,0x01,0x00,0x00,0x00,0x01,0x00,0x00,0x00,0x01,0x00,0x00,0x00,0x01+.byte 0x00,0x00,0x00,0x1b,0x00,0x00,0x00,0x1b,0x00,0x00,0x00,0x1b,0x00,0x00,0x00,0x1b+.byte 0x0c,0x0f,0x0e,0x0d,0x0c,0x0f,0x0e,0x0d,0x0c,0x0f,0x0e,0x0d,0x0c,0x0f,0x0e,0x0d+.byte 0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00+.Lconsts:+ mflr 0+ bcl 20,31,$+4+ mflr 6+ addi 6,6,-0x48+ mtlr 0+ blr +.long 0+.byte 0,12,0x14,0,0,0,0,0+.byte 65,69,83,32,102,111,114,32,80,111,119,101,114,73,83,65,32,50,46,48,55,44,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0+.align 2++.globl crypton_aes_p8_set_encrypt_key+.type crypton_aes_p8_set_encrypt_key,@function+.align 5+crypton_aes_p8_set_encrypt_key:+.localentry crypton_aes_p8_set_encrypt_key,0++.Lset_encrypt_key:+ mflr 11+ std 11,16(1)++ li 6,-1+ cmpldi 3,0+ beq- .Lenc_key_abort+ cmpldi 5,0+ beq- .Lenc_key_abort+ li 6,-2+ cmpwi 4,128+ blt- .Lenc_key_abort+ cmpwi 4,256+ bgt- .Lenc_key_abort+ andi. 0,4,0x3f+ bne- .Lenc_key_abort++ lis 0,0xfff0+ li 12,-1+ or 0,0,0++ bl .Lconsts+ mtlr 11++ neg 9,3+ lvx 1,0,3+ addi 3,3,15+ lvsr 3,0,9+ li 8,0x20+ cmpwi 4,192+ lvx 2,0,3+ vspltisb 5,0x0f+ lvx 4,0,6+ vxor 3,3,5+ lvx 5,8,6+ addi 6,6,0x10+ vperm 1,1,2,3+ li 7,8+ vxor 0,0,0+ mtctr 7++ lvsl 8,0,5+ vspltisb 9,-1+ lvx 10,0,5+ vperm 9,9,0,8++ blt .Loop128+ addi 3,3,8+ beq .L192+ addi 3,3,8+ b .L256++.align 4+.Loop128:+ vperm 3,1,1,5+ vsldoi 6,0,1,12+ vperm 11,1,1,8+ vsel 7,10,11,9+ vor 10,11,11+ .long 0x10632509+ stvx 7,0,5+ addi 5,5,16++ vxor 1,1,6+ vsldoi 6,0,6,12+ vxor 1,1,6+ vsldoi 6,0,6,12+ vxor 1,1,6+ vadduwm 4,4,4+ vxor 1,1,3+ bdnz .Loop128++ lvx 4,0,6++ vperm 3,1,1,5+ vsldoi 6,0,1,12+ vperm 11,1,1,8+ vsel 7,10,11,9+ vor 10,11,11+ .long 0x10632509+ stvx 7,0,5+ addi 5,5,16++ vxor 1,1,6+ vsldoi 6,0,6,12+ vxor 1,1,6+ vsldoi 6,0,6,12+ vxor 1,1,6+ vadduwm 4,4,4+ vxor 1,1,3++ vperm 3,1,1,5+ vsldoi 6,0,1,12+ vperm 11,1,1,8+ vsel 7,10,11,9+ vor 10,11,11+ .long 0x10632509+ stvx 7,0,5+ addi 5,5,16++ vxor 1,1,6+ vsldoi 6,0,6,12+ vxor 1,1,6+ vsldoi 6,0,6,12+ vxor 1,1,6+ vxor 1,1,3+ vperm 11,1,1,8+ vsel 7,10,11,9+ vor 10,11,11+ stvx 7,0,5++ addi 3,5,15+ addi 5,5,0x50++ li 8,10+ b .Ldone++.align 4+.L192:+ lvx 6,0,3+ li 7,4+ vperm 11,1,1,8+ vsel 7,10,11,9+ vor 10,11,11+ stvx 7,0,5+ addi 5,5,16+ vperm 2,2,6,3+ vspltisb 3,8+ mtctr 7+ vsububm 5,5,3++.Loop192:+ vperm 3,2,2,5+ vsldoi 6,0,1,12+ .long 0x10632509++ vxor 1,1,6+ vsldoi 6,0,6,12+ vxor 1,1,6+ vsldoi 6,0,6,12+ vxor 1,1,6++ vsldoi 7,0,2,8+ vspltw 6,1,3+ vxor 6,6,2+ vsldoi 2,0,2,12+ vadduwm 4,4,4+ vxor 2,2,6+ vxor 1,1,3+ vxor 2,2,3+ vsldoi 7,7,1,8++ vperm 3,2,2,5+ vsldoi 6,0,1,12+ vperm 11,7,7,8+ vsel 7,10,11,9+ vor 10,11,11+ .long 0x10632509+ stvx 7,0,5+ addi 5,5,16++ vsldoi 7,1,2,8+ vxor 1,1,6+ vsldoi 6,0,6,12+ vperm 11,7,7,8+ vsel 7,10,11,9+ vor 10,11,11+ vxor 1,1,6+ vsldoi 6,0,6,12+ vxor 1,1,6+ stvx 7,0,5+ addi 5,5,16++ vspltw 6,1,3+ vxor 6,6,2+ vsldoi 2,0,2,12+ vadduwm 4,4,4+ vxor 2,2,6+ vxor 1,1,3+ vxor 2,2,3+ vperm 11,1,1,8+ vsel 7,10,11,9+ vor 10,11,11+ stvx 7,0,5+ addi 3,5,15+ addi 5,5,16+ bdnz .Loop192++ li 8,12+ addi 5,5,0x20+ b .Ldone++.align 4+.L256:+ lvx 6,0,3+ li 7,7+ li 8,14+ vperm 11,1,1,8+ vsel 7,10,11,9+ vor 10,11,11+ stvx 7,0,5+ addi 5,5,16+ vperm 2,2,6,3+ mtctr 7++.Loop256:+ vperm 3,2,2,5+ vsldoi 6,0,1,12+ vperm 11,2,2,8+ vsel 7,10,11,9+ vor 10,11,11+ .long 0x10632509+ stvx 7,0,5+ addi 5,5,16++ vxor 1,1,6+ vsldoi 6,0,6,12+ vxor 1,1,6+ vsldoi 6,0,6,12+ vxor 1,1,6+ vadduwm 4,4,4+ vxor 1,1,3+ vperm 11,1,1,8+ vsel 7,10,11,9+ vor 10,11,11+ stvx 7,0,5+ addi 3,5,15+ addi 5,5,16+ bdz .Ldone++ vspltw 3,1,3+ vsldoi 6,0,2,12+ .long 0x106305C8++ vxor 2,2,6+ vsldoi 6,0,6,12+ vxor 2,2,6+ vsldoi 6,0,6,12+ vxor 2,2,6++ vxor 2,2,3+ b .Loop256++.align 4+.Ldone:+ lvx 2,0,3+ vsel 2,10,2,9+ stvx 2,0,3+ li 6,0+ or 12,12,12+ stw 8,0(5)++.Lenc_key_abort:+ mr 3,6+ blr +.long 0+.byte 0,12,0x14,1,0,0,3,0+.long 0+.size crypton_aes_p8_set_encrypt_key,.-crypton_aes_p8_set_encrypt_key++.globl crypton_aes_p8_set_decrypt_key+.type crypton_aes_p8_set_decrypt_key,@function+.align 5+crypton_aes_p8_set_decrypt_key:+.localentry crypton_aes_p8_set_decrypt_key,0++ stdu 1,-64(1)+ mflr 10+ std 10,64+16(1)+ bl .Lset_encrypt_key+ mtlr 10++ cmpwi 3,0+ bne- .Ldec_key_abort++ slwi 7,8,4+ subi 3,5,240+ srwi 8,8,1+ add 5,3,7+ mtctr 8++.Ldeckey:+ lwz 0, 0(3)+ lwz 6, 4(3)+ lwz 7, 8(3)+ lwz 8, 12(3)+ addi 3,3,16+ lwz 9, 0(5)+ lwz 10,4(5)+ lwz 11,8(5)+ lwz 12,12(5)+ stw 0, 0(5)+ stw 6, 4(5)+ stw 7, 8(5)+ stw 8, 12(5)+ subi 5,5,16+ stw 9, -16(3)+ stw 10,-12(3)+ stw 11,-8(3)+ stw 12,-4(3)+ bdnz .Ldeckey++ xor 3,3,3+.Ldec_key_abort:+ addi 1,1,64+ blr +.long 0+.byte 0,12,4,1,0x80,0,3,0+.long 0+.size crypton_aes_p8_set_decrypt_key,.-crypton_aes_p8_set_decrypt_key+.globl crypton_aes_p8_encrypt+.type crypton_aes_p8_encrypt,@function+.align 5+crypton_aes_p8_encrypt:+.localentry crypton_aes_p8_encrypt,0++ lwz 6,240(5)+ lis 0,0xfc00+ li 12,-1+ li 7,15+ or 0,0,0++ lvx 0,0,3+ neg 11,4+ lvx 1,7,3+ lvsl 2,0,3+ vspltisb 4,0x0f+ lvsr 3,0,11+ vxor 2,2,4+ li 7,16+ vperm 0,0,1,2+ lvx 1,0,5+ lvsr 5,0,5+ srwi 6,6,1+ lvx 2,7,5+ addi 7,7,16+ subi 6,6,1+ vperm 1,2,1,5++ vxor 0,0,1+ lvx 1,7,5+ addi 7,7,16+ mtctr 6++.Loop_enc:+ vperm 2,1,2,5+ .long 0x10001508+ lvx 2,7,5+ addi 7,7,16+ vperm 1,2,1,5+ .long 0x10000D08+ lvx 1,7,5+ addi 7,7,16+ bdnz .Loop_enc++ vperm 2,1,2,5+ .long 0x10001508+ lvx 2,7,5+ vperm 1,2,1,5+ .long 0x10000D09++ vspltisb 2,-1+ vxor 1,1,1+ li 7,15+ vperm 2,2,1,3+ vxor 3,3,4+ lvx 1,0,4+ vperm 0,0,0,3+ vsel 1,1,0,2+ lvx 4,7,4+ stvx 1,0,4+ vsel 0,0,4,2+ stvx 0,7,4++ or 12,12,12+ blr +.long 0+.byte 0,12,0x14,0,0,0,3,0+.long 0+.size crypton_aes_p8_encrypt,.-crypton_aes_p8_encrypt+.globl crypton_aes_p8_decrypt+.type crypton_aes_p8_decrypt,@function+.align 5+crypton_aes_p8_decrypt:+.localentry crypton_aes_p8_decrypt,0++ lwz 6,240(5)+ lis 0,0xfc00+ li 12,-1+ li 7,15+ or 0,0,0++ lvx 0,0,3+ neg 11,4+ lvx 1,7,3+ lvsl 2,0,3+ vspltisb 4,0x0f+ lvsr 3,0,11+ vxor 2,2,4+ li 7,16+ vperm 0,0,1,2+ lvx 1,0,5+ lvsr 5,0,5+ srwi 6,6,1+ lvx 2,7,5+ addi 7,7,16+ subi 6,6,1+ vperm 1,2,1,5++ vxor 0,0,1+ lvx 1,7,5+ addi 7,7,16+ mtctr 6++.Loop_dec:+ vperm 2,1,2,5+ .long 0x10001548+ lvx 2,7,5+ addi 7,7,16+ vperm 1,2,1,5+ .long 0x10000D48+ lvx 1,7,5+ addi 7,7,16+ bdnz .Loop_dec++ vperm 2,1,2,5+ .long 0x10001548+ lvx 2,7,5+ vperm 1,2,1,5+ .long 0x10000D49++ vspltisb 2,-1+ vxor 1,1,1+ li 7,15+ vperm 2,2,1,3+ vxor 3,3,4+ lvx 1,0,4+ vperm 0,0,0,3+ vsel 1,1,0,2+ lvx 4,7,4+ stvx 1,0,4+ vsel 0,0,4,2+ stvx 0,7,4++ or 12,12,12+ blr +.long 0+.byte 0,12,0x14,0,0,0,3,0+.long 0+.size crypton_aes_p8_decrypt,.-crypton_aes_p8_decrypt+.globl crypton_aes_p8_cbc_encrypt+.type crypton_aes_p8_cbc_encrypt,@function+.align 5+crypton_aes_p8_cbc_encrypt:+.localentry crypton_aes_p8_cbc_encrypt,0++ cmpldi 5,16+ .long 0x4dc00020++ cmpwi 8,0+ lis 0,0xffe0+ li 12,-1+ or 0,0,0++ li 10,15+ vxor 0,0,0+ vspltisb 3,0x0f++ lvx 4,0,7+ lvsl 6,0,7+ lvx 5,10,7+ vxor 6,6,3+ vperm 4,4,5,6++ neg 11,3+ lvsr 10,0,6+ lwz 9,240(6)++ lvsr 6,0,11+ lvx 5,0,3+ addi 3,3,15+ vxor 6,6,3++ lvsl 8,0,4+ vspltisb 9,-1+ lvx 7,0,4+ vperm 9,9,0,8+ vxor 8,8,3++ srwi 9,9,1+ li 10,16+ subi 9,9,1+ beq .Lcbc_dec++.Lcbc_enc:+ vor 2,5,5+ lvx 5,0,3+ addi 3,3,16+ mtctr 9+ subi 5,5,16++ lvx 0,0,6+ vperm 2,2,5,6+ lvx 1,10,6+ addi 10,10,16+ vperm 0,1,0,10+ vxor 2,2,0+ lvx 0,10,6+ addi 10,10,16+ vxor 2,2,4++.Loop_cbc_enc:+ vperm 1,0,1,10+ .long 0x10420D08+ lvx 1,10,6+ addi 10,10,16+ vperm 0,1,0,10+ .long 0x10420508+ lvx 0,10,6+ addi 10,10,16+ bdnz .Loop_cbc_enc++ vperm 1,0,1,10+ .long 0x10420D08+ lvx 1,10,6+ li 10,16+ vperm 0,1,0,10+ .long 0x10820509+ cmpldi 5,16++ vperm 3,4,4,8+ vsel 2,7,3,9+ vor 7,3,3+ stvx 2,0,4+ addi 4,4,16+ bge .Lcbc_enc++ b .Lcbc_done++.align 4+.Lcbc_dec:+ cmpldi 5,128+ bge _aesp8_cbc_decrypt8x+ vor 3,5,5+ lvx 5,0,3+ addi 3,3,16+ mtctr 9+ subi 5,5,16++ lvx 0,0,6+ vperm 3,3,5,6+ lvx 1,10,6+ addi 10,10,16+ vperm 0,1,0,10+ vxor 2,3,0+ lvx 0,10,6+ addi 10,10,16++.Loop_cbc_dec:+ vperm 1,0,1,10+ .long 0x10420D48+ lvx 1,10,6+ addi 10,10,16+ vperm 0,1,0,10+ .long 0x10420548+ lvx 0,10,6+ addi 10,10,16+ bdnz .Loop_cbc_dec++ vperm 1,0,1,10+ .long 0x10420D48+ lvx 1,10,6+ li 10,16+ vperm 0,1,0,10+ .long 0x10420549+ cmpldi 5,16++ vxor 2,2,4+ vor 4,3,3+ vperm 3,2,2,8+ vsel 2,7,3,9+ vor 7,3,3+ stvx 2,0,4+ addi 4,4,16+ bge .Lcbc_dec++.Lcbc_done:+ addi 4,4,-1+ lvx 2,0,4+ vsel 2,7,2,9+ stvx 2,0,4++ neg 8,7+ li 10,15+ vxor 0,0,0+ vspltisb 9,-1+ vspltisb 3,0x0f+ lvsr 8,0,8+ vperm 9,9,0,8+ vxor 8,8,3+ lvx 7,0,7+ vperm 4,4,4,8+ vsel 2,7,4,9+ lvx 5,10,7+ stvx 2,0,7+ vsel 2,4,5,9+ stvx 2,10,7++ or 12,12,12+ blr +.long 0+.byte 0,12,0x14,0,0,0,6,0+.long 0+.align 5+_aesp8_cbc_decrypt8x:+ stdu 1,-448(1)+ li 10,207+ li 11,223+ stvx 20,10,1+ addi 10,10,32+ stvx 21,11,1+ addi 11,11,32+ stvx 22,10,1+ addi 10,10,32+ stvx 23,11,1+ addi 11,11,32+ stvx 24,10,1+ addi 10,10,32+ stvx 25,11,1+ addi 11,11,32+ stvx 26,10,1+ addi 10,10,32+ stvx 27,11,1+ addi 11,11,32+ stvx 28,10,1+ addi 10,10,32+ stvx 29,11,1+ addi 11,11,32+ stvx 30,10,1+ stvx 31,11,1+ li 0,-1+ stw 12,396(1)+ li 8,0x10+ std 26,400(1)+ li 26,0x20+ std 27,408(1)+ li 27,0x30+ std 28,416(1)+ li 28,0x40+ std 29,424(1)+ li 29,0x50+ std 30,432(1)+ li 30,0x60+ std 31,440(1)+ li 31,0x70+ or 0,0,0++ subi 9,9,3+ subi 5,5,128++ lvx 23,0,6+ lvx 30,8,6+ addi 6,6,0x20+ lvx 31,0,6+ vperm 23,30,23,10+ addi 11,1,64+15+ mtctr 9++.Load_cbc_dec_key:+ vperm 24,31,30,10+ lvx 30,8,6+ addi 6,6,0x20+ stvx 24,0,11+ vperm 25,30,31,10+ lvx 31,0,6+ stvx 25,8,11+ addi 11,11,0x20+ bdnz .Load_cbc_dec_key++ lvx 26,8,6+ vperm 24,31,30,10+ lvx 27,26,6+ stvx 24,0,11+ vperm 25,26,31,10+ lvx 28,27,6+ stvx 25,8,11+ addi 11,1,64+15+ vperm 26,27,26,10+ lvx 29,28,6+ vperm 27,28,27,10+ lvx 30,29,6+ vperm 28,29,28,10+ lvx 31,30,6+ vperm 29,30,29,10+ lvx 14,31,6+ vperm 30,31,30,10+ lvx 24,0,11+ vperm 31,14,31,10+ lvx 25,8,11++++ subi 3,3,15++ li 10,8+ .long 0x7C001E99+ lvsl 6,0,10+ vspltisb 3,0x0f+ .long 0x7C281E99+ vxor 6,6,3+ .long 0x7C5A1E99+ vperm 0,0,0,6+ .long 0x7C7B1E99+ vperm 1,1,1,6+ .long 0x7D5C1E99+ vperm 2,2,2,6+ vxor 14,0,23+ .long 0x7D7D1E99+ vperm 3,3,3,6+ vxor 15,1,23+ .long 0x7D9E1E99+ vperm 10,10,10,6+ vxor 16,2,23+ .long 0x7DBF1E99+ addi 3,3,0x80+ vperm 11,11,11,6+ vxor 17,3,23+ vperm 12,12,12,6+ vxor 18,10,23+ vperm 13,13,13,6+ vxor 19,11,23+ vxor 20,12,23+ vxor 21,13,23++ mtctr 9+ b .Loop_cbc_dec8x+.align 5+.Loop_cbc_dec8x:+ .long 0x11CEC548+ .long 0x11EFC548+ .long 0x1210C548+ .long 0x1231C548+ .long 0x1252C548+ .long 0x1273C548+ .long 0x1294C548+ .long 0x12B5C548+ lvx 24,26,11+ addi 11,11,0x20++ .long 0x11CECD48+ .long 0x11EFCD48+ .long 0x1210CD48+ .long 0x1231CD48+ .long 0x1252CD48+ .long 0x1273CD48+ .long 0x1294CD48+ .long 0x12B5CD48+ lvx 25,8,11+ bdnz .Loop_cbc_dec8x++ subic 5,5,128+ .long 0x11CEC548+ .long 0x11EFC548+ .long 0x1210C548+ .long 0x1231C548+ .long 0x1252C548+ .long 0x1273C548+ .long 0x1294C548+ .long 0x12B5C548++ subfe. 0,0,0+ .long 0x11CECD48+ .long 0x11EFCD48+ .long 0x1210CD48+ .long 0x1231CD48+ .long 0x1252CD48+ .long 0x1273CD48+ .long 0x1294CD48+ .long 0x12B5CD48++ and 0,0,5+ .long 0x11CED548+ .long 0x11EFD548+ .long 0x1210D548+ .long 0x1231D548+ .long 0x1252D548+ .long 0x1273D548+ .long 0x1294D548+ .long 0x12B5D548++ add 3,3,0++++ .long 0x11CEDD48+ .long 0x11EFDD48+ .long 0x1210DD48+ .long 0x1231DD48+ .long 0x1252DD48+ .long 0x1273DD48+ .long 0x1294DD48+ .long 0x12B5DD48++ addi 11,1,64+15+ .long 0x11CEE548+ .long 0x11EFE548+ .long 0x1210E548+ .long 0x1231E548+ .long 0x1252E548+ .long 0x1273E548+ .long 0x1294E548+ .long 0x12B5E548+ lvx 24,0,11++ .long 0x11CEED48+ .long 0x11EFED48+ .long 0x1210ED48+ .long 0x1231ED48+ .long 0x1252ED48+ .long 0x1273ED48+ .long 0x1294ED48+ .long 0x12B5ED48+ lvx 25,8,11++ .long 0x11CEF548+ vxor 4,4,31+ .long 0x11EFF548+ vxor 0,0,31+ .long 0x1210F548+ vxor 1,1,31+ .long 0x1231F548+ vxor 2,2,31+ .long 0x1252F548+ vxor 3,3,31+ .long 0x1273F548+ vxor 10,10,31+ .long 0x1294F548+ vxor 11,11,31+ .long 0x12B5F548+ vxor 12,12,31++ .long 0x11CE2549+ .long 0x11EF0549+ .long 0x7C001E99+ .long 0x12100D49+ .long 0x7C281E99+ .long 0x12311549+ vperm 0,0,0,6+ .long 0x7C5A1E99+ .long 0x12521D49+ vperm 1,1,1,6+ .long 0x7C7B1E99+ .long 0x12735549+ vperm 2,2,2,6+ .long 0x7D5C1E99+ .long 0x12945D49+ vperm 3,3,3,6+ .long 0x7D7D1E99+ .long 0x12B56549+ vperm 10,10,10,6+ .long 0x7D9E1E99+ vor 4,13,13+ vperm 11,11,11,6+ .long 0x7DBF1E99+ addi 3,3,0x80++ vperm 14,14,14,6+ vperm 15,15,15,6+ .long 0x7DC02799+ vperm 12,12,12,6+ vxor 14,0,23+ vperm 16,16,16,6+ .long 0x7DE82799+ vperm 13,13,13,6+ vxor 15,1,23+ vperm 17,17,17,6+ .long 0x7E1A2799+ vxor 16,2,23+ vperm 18,18,18,6+ .long 0x7E3B2799+ vxor 17,3,23+ vperm 19,19,19,6+ .long 0x7E5C2799+ vxor 18,10,23+ vperm 20,20,20,6+ .long 0x7E7D2799+ vxor 19,11,23+ vperm 21,21,21,6+ .long 0x7E9E2799+ vxor 20,12,23+ .long 0x7EBF2799+ addi 4,4,0x80+ vxor 21,13,23++ mtctr 9+ beq .Loop_cbc_dec8x++ addic. 5,5,128+ beq .Lcbc_dec8x_done+ nop + nop ++.Loop_cbc_dec8x_tail:+ .long 0x11EFC548+ .long 0x1210C548+ .long 0x1231C548+ .long 0x1252C548+ .long 0x1273C548+ .long 0x1294C548+ .long 0x12B5C548+ lvx 24,26,11+ addi 11,11,0x20++ .long 0x11EFCD48+ .long 0x1210CD48+ .long 0x1231CD48+ .long 0x1252CD48+ .long 0x1273CD48+ .long 0x1294CD48+ .long 0x12B5CD48+ lvx 25,8,11+ bdnz .Loop_cbc_dec8x_tail++ .long 0x11EFC548+ .long 0x1210C548+ .long 0x1231C548+ .long 0x1252C548+ .long 0x1273C548+ .long 0x1294C548+ .long 0x12B5C548++ .long 0x11EFCD48+ .long 0x1210CD48+ .long 0x1231CD48+ .long 0x1252CD48+ .long 0x1273CD48+ .long 0x1294CD48+ .long 0x12B5CD48++ .long 0x11EFD548+ .long 0x1210D548+ .long 0x1231D548+ .long 0x1252D548+ .long 0x1273D548+ .long 0x1294D548+ .long 0x12B5D548++ .long 0x11EFDD48+ .long 0x1210DD48+ .long 0x1231DD48+ .long 0x1252DD48+ .long 0x1273DD48+ .long 0x1294DD48+ .long 0x12B5DD48++ .long 0x11EFE548+ .long 0x1210E548+ .long 0x1231E548+ .long 0x1252E548+ .long 0x1273E548+ .long 0x1294E548+ .long 0x12B5E548++ .long 0x11EFED48+ .long 0x1210ED48+ .long 0x1231ED48+ .long 0x1252ED48+ .long 0x1273ED48+ .long 0x1294ED48+ .long 0x12B5ED48++ .long 0x11EFF548+ vxor 4,4,31+ .long 0x1210F548+ vxor 1,1,31+ .long 0x1231F548+ vxor 2,2,31+ .long 0x1252F548+ vxor 3,3,31+ .long 0x1273F548+ vxor 10,10,31+ .long 0x1294F548+ vxor 11,11,31+ .long 0x12B5F548+ vxor 12,12,31++ cmplwi 5,32+ blt .Lcbc_dec8x_one+ nop + beq .Lcbc_dec8x_two+ cmplwi 5,64+ blt .Lcbc_dec8x_three+ nop + beq .Lcbc_dec8x_four+ cmplwi 5,96+ blt .Lcbc_dec8x_five+ nop + beq .Lcbc_dec8x_six++.Lcbc_dec8x_seven:+ .long 0x11EF2549+ .long 0x12100D49+ .long 0x12311549+ .long 0x12521D49+ .long 0x12735549+ .long 0x12945D49+ .long 0x12B56549+ vor 4,13,13++ vperm 15,15,15,6+ vperm 16,16,16,6+ .long 0x7DE02799+ vperm 17,17,17,6+ .long 0x7E082799+ vperm 18,18,18,6+ .long 0x7E3A2799+ vperm 19,19,19,6+ .long 0x7E5B2799+ vperm 20,20,20,6+ .long 0x7E7C2799+ vperm 21,21,21,6+ .long 0x7E9D2799+ .long 0x7EBE2799+ addi 4,4,0x70+ b .Lcbc_dec8x_done++.align 5+.Lcbc_dec8x_six:+ .long 0x12102549+ .long 0x12311549+ .long 0x12521D49+ .long 0x12735549+ .long 0x12945D49+ .long 0x12B56549+ vor 4,13,13++ vperm 16,16,16,6+ vperm 17,17,17,6+ .long 0x7E002799+ vperm 18,18,18,6+ .long 0x7E282799+ vperm 19,19,19,6+ .long 0x7E5A2799+ vperm 20,20,20,6+ .long 0x7E7B2799+ vperm 21,21,21,6+ .long 0x7E9C2799+ .long 0x7EBD2799+ addi 4,4,0x60+ b .Lcbc_dec8x_done++.align 5+.Lcbc_dec8x_five:+ .long 0x12312549+ .long 0x12521D49+ .long 0x12735549+ .long 0x12945D49+ .long 0x12B56549+ vor 4,13,13++ vperm 17,17,17,6+ vperm 18,18,18,6+ .long 0x7E202799+ vperm 19,19,19,6+ .long 0x7E482799+ vperm 20,20,20,6+ .long 0x7E7A2799+ vperm 21,21,21,6+ .long 0x7E9B2799+ .long 0x7EBC2799+ addi 4,4,0x50+ b .Lcbc_dec8x_done++.align 5+.Lcbc_dec8x_four:+ .long 0x12522549+ .long 0x12735549+ .long 0x12945D49+ .long 0x12B56549+ vor 4,13,13++ vperm 18,18,18,6+ vperm 19,19,19,6+ .long 0x7E402799+ vperm 20,20,20,6+ .long 0x7E682799+ vperm 21,21,21,6+ .long 0x7E9A2799+ .long 0x7EBB2799+ addi 4,4,0x40+ b .Lcbc_dec8x_done++.align 5+.Lcbc_dec8x_three:+ .long 0x12732549+ .long 0x12945D49+ .long 0x12B56549+ vor 4,13,13++ vperm 19,19,19,6+ vperm 20,20,20,6+ .long 0x7E602799+ vperm 21,21,21,6+ .long 0x7E882799+ .long 0x7EBA2799+ addi 4,4,0x30+ b .Lcbc_dec8x_done++.align 5+.Lcbc_dec8x_two:+ .long 0x12942549+ .long 0x12B56549+ vor 4,13,13++ vperm 20,20,20,6+ vperm 21,21,21,6+ .long 0x7E802799+ .long 0x7EA82799+ addi 4,4,0x20+ b .Lcbc_dec8x_done++.align 5+.Lcbc_dec8x_one:+ .long 0x12B52549+ vor 4,13,13++ vperm 21,21,21,6+ .long 0x7EA02799+ addi 4,4,0x10++.Lcbc_dec8x_done:+ vperm 4,4,4,6+ .long 0x7C803F99++ li 10,79+ li 11,95+ stvx 6,10,1+ addi 10,10,32+ stvx 6,11,1+ addi 11,11,32+ stvx 6,10,1+ addi 10,10,32+ stvx 6,11,1+ addi 11,11,32+ stvx 6,10,1+ addi 10,10,32+ stvx 6,11,1+ addi 11,11,32+ stvx 6,10,1+ addi 10,10,32+ stvx 6,11,1+ addi 11,11,32++ or 12,12,12+ lvx 20,10,1+ addi 10,10,32+ lvx 21,11,1+ addi 11,11,32+ lvx 22,10,1+ addi 10,10,32+ lvx 23,11,1+ addi 11,11,32+ lvx 24,10,1+ addi 10,10,32+ lvx 25,11,1+ addi 11,11,32+ lvx 26,10,1+ addi 10,10,32+ lvx 27,11,1+ addi 11,11,32+ lvx 28,10,1+ addi 10,10,32+ lvx 29,11,1+ addi 11,11,32+ lvx 30,10,1+ lvx 31,11,1+ ld 26,400(1)+ ld 27,408(1)+ ld 28,416(1)+ ld 29,424(1)+ ld 30,432(1)+ ld 31,440(1)+ addi 1,1,448+ blr +.long 0+.byte 0,12,0x04,0,0x80,6,6,0+.long 0+.size crypton_aes_p8_cbc_encrypt,.-crypton_aes_p8_cbc_encrypt+.globl crypton_aes_p8_ctr32_encrypt_blocks+.type crypton_aes_p8_ctr32_encrypt_blocks,@function+.align 5+crypton_aes_p8_ctr32_encrypt_blocks:+.localentry crypton_aes_p8_ctr32_encrypt_blocks,0++ cmpldi 5,1+ .long 0x4dc00020++ lis 0,0xfff0+ li 12,-1+ or 0,0,0++ li 10,15+ vxor 0,0,0+ vspltisb 3,0x0f++ lvx 4,0,7+ lvsl 6,0,7+ lvx 5,10,7+ vspltisb 11,1+ vxor 6,6,3+ vperm 4,4,5,6+ vsldoi 11,0,11,1++ neg 11,3+ lvsr 10,0,6+ lwz 9,240(6)++ lvsr 6,0,11+ lvx 5,0,3+ addi 3,3,15+ vxor 6,6,3++ srwi 9,9,1+ li 10,16+ subi 9,9,1++ cmpldi 5,8+ bge _aesp8_ctr32_encrypt8x++ lvsl 8,0,4+ vspltisb 9,-1+ lvx 7,0,4+ vperm 9,9,0,8+ vxor 8,8,3++ lvx 0,0,6+ mtctr 9+ lvx 1,10,6+ addi 10,10,16+ vperm 0,1,0,10+ vxor 2,4,0+ lvx 0,10,6+ addi 10,10,16+ b .Loop_ctr32_enc++.align 5+.Loop_ctr32_enc:+ vperm 1,0,1,10+ .long 0x10420D08+ lvx 1,10,6+ addi 10,10,16+ vperm 0,1,0,10+ .long 0x10420508+ lvx 0,10,6+ addi 10,10,16+ bdnz .Loop_ctr32_enc++ vadduwm 4,4,11+ vor 3,5,5+ lvx 5,0,3+ addi 3,3,16+ subic. 5,5,1++ vperm 1,0,1,10+ .long 0x10420D08+ lvx 1,10,6+ vperm 3,3,5,6+ li 10,16+ vperm 1,1,0,10+ lvx 0,0,6+ vxor 3,3,1+ .long 0x10421D09++ lvx 1,10,6+ addi 10,10,16+ vperm 2,2,2,8+ vsel 3,7,2,9+ mtctr 9+ vperm 0,1,0,10+ vor 7,2,2+ vxor 2,4,0+ lvx 0,10,6+ addi 10,10,16+ stvx 3,0,4+ addi 4,4,16+ bne .Loop_ctr32_enc++ addi 4,4,-1+ lvx 2,0,4+ vsel 2,7,2,9+ stvx 2,0,4++ or 12,12,12+ blr +.long 0+.byte 0,12,0x14,0,0,0,6,0+.long 0+.align 5+_aesp8_ctr32_encrypt8x:+ stdu 1,-448(1)+ li 10,207+ li 11,223+ stvx 20,10,1+ addi 10,10,32+ stvx 21,11,1+ addi 11,11,32+ stvx 22,10,1+ addi 10,10,32+ stvx 23,11,1+ addi 11,11,32+ stvx 24,10,1+ addi 10,10,32+ stvx 25,11,1+ addi 11,11,32+ stvx 26,10,1+ addi 10,10,32+ stvx 27,11,1+ addi 11,11,32+ stvx 28,10,1+ addi 10,10,32+ stvx 29,11,1+ addi 11,11,32+ stvx 30,10,1+ stvx 31,11,1+ li 0,-1+ stw 12,396(1)+ li 8,0x10+ std 26,400(1)+ li 26,0x20+ std 27,408(1)+ li 27,0x30+ std 28,416(1)+ li 28,0x40+ std 29,424(1)+ li 29,0x50+ std 30,432(1)+ li 30,0x60+ std 31,440(1)+ li 31,0x70+ or 0,0,0++ subi 9,9,3++ lvx 23,0,6+ lvx 30,8,6+ addi 6,6,0x20+ lvx 31,0,6+ vperm 23,30,23,10+ addi 11,1,64+15+ mtctr 9++.Load_ctr32_enc_key:+ vperm 24,31,30,10+ lvx 30,8,6+ addi 6,6,0x20+ stvx 24,0,11+ vperm 25,30,31,10+ lvx 31,0,6+ stvx 25,8,11+ addi 11,11,0x20+ bdnz .Load_ctr32_enc_key++ lvx 26,8,6+ vperm 24,31,30,10+ lvx 27,26,6+ stvx 24,0,11+ vperm 25,26,31,10+ lvx 28,27,6+ stvx 25,8,11+ addi 11,1,64+15+ vperm 26,27,26,10+ lvx 29,28,6+ vperm 27,28,27,10+ lvx 30,29,6+ vperm 28,29,28,10+ lvx 31,30,6+ vperm 29,30,29,10+ lvx 15,31,6+ vperm 30,31,30,10+ lvx 24,0,11+ vperm 31,15,31,10+ lvx 25,8,11++ vadduwm 7,11,11+ subi 3,3,15+ sldi 5,5,4++ vadduwm 16,4,11+ vadduwm 17,4,7+ vxor 15,4,23+ li 10,8+ vadduwm 18,16,7+ vxor 16,16,23+ lvsl 6,0,10+ vadduwm 19,17,7+ vxor 17,17,23+ vspltisb 3,0x0f+ vadduwm 20,18,7+ vxor 18,18,23+ vxor 6,6,3+ vadduwm 21,19,7+ vxor 19,19,23+ vadduwm 22,20,7+ vxor 20,20,23+ vadduwm 4,21,7+ vxor 21,21,23+ vxor 22,22,23++ mtctr 9+ b .Loop_ctr32_enc8x+.align 5+.Loop_ctr32_enc8x:+ .long 0x11EFC508+ .long 0x1210C508+ .long 0x1231C508+ .long 0x1252C508+ .long 0x1273C508+ .long 0x1294C508+ .long 0x12B5C508+ .long 0x12D6C508+.Loop_ctr32_enc8x_middle:+ lvx 24,26,11+ addi 11,11,0x20++ .long 0x11EFCD08+ .long 0x1210CD08+ .long 0x1231CD08+ .long 0x1252CD08+ .long 0x1273CD08+ .long 0x1294CD08+ .long 0x12B5CD08+ .long 0x12D6CD08+ lvx 25,8,11+ bdnz .Loop_ctr32_enc8x++ subic 11,5,256+ .long 0x11EFC508+ .long 0x1210C508+ .long 0x1231C508+ .long 0x1252C508+ .long 0x1273C508+ .long 0x1294C508+ .long 0x12B5C508+ .long 0x12D6C508++ subfe 0,0,0+ .long 0x11EFCD08+ .long 0x1210CD08+ .long 0x1231CD08+ .long 0x1252CD08+ .long 0x1273CD08+ .long 0x1294CD08+ .long 0x12B5CD08+ .long 0x12D6CD08++ and 0,0,11+ addi 11,1,64+15+ .long 0x11EFD508+ .long 0x1210D508+ .long 0x1231D508+ .long 0x1252D508+ .long 0x1273D508+ .long 0x1294D508+ .long 0x12B5D508+ .long 0x12D6D508+ lvx 24,0,11++ subic 5,5,129+ .long 0x11EFDD08+ addi 5,5,1+ .long 0x1210DD08+ .long 0x1231DD08+ .long 0x1252DD08+ .long 0x1273DD08+ .long 0x1294DD08+ .long 0x12B5DD08+ .long 0x12D6DD08+ lvx 25,8,11++ .long 0x11EFE508+ .long 0x7C001E99+ .long 0x1210E508+ .long 0x7C281E99+ .long 0x1231E508+ .long 0x7C5A1E99+ .long 0x1252E508+ .long 0x7C7B1E99+ .long 0x1273E508+ .long 0x7D5C1E99+ .long 0x1294E508+ .long 0x7D9D1E99+ .long 0x12B5E508+ .long 0x7DBE1E99+ .long 0x12D6E508+ .long 0x7DDF1E99+ addi 3,3,0x80++ .long 0x11EFED08+ vperm 0,0,0,6+ .long 0x1210ED08+ vperm 1,1,1,6+ .long 0x1231ED08+ vperm 2,2,2,6+ .long 0x1252ED08+ vperm 3,3,3,6+ .long 0x1273ED08+ vperm 10,10,10,6+ .long 0x1294ED08+ vperm 12,12,12,6+ .long 0x12B5ED08+ vperm 13,13,13,6+ .long 0x12D6ED08+ vperm 14,14,14,6++ add 3,3,0++++ subfe. 0,0,0+ .long 0x11EFF508+ vxor 0,0,31+ .long 0x1210F508+ vxor 1,1,31+ .long 0x1231F508+ vxor 2,2,31+ .long 0x1252F508+ vxor 3,3,31+ .long 0x1273F508+ vxor 10,10,31+ .long 0x1294F508+ vxor 12,12,31+ .long 0x12B5F508+ vxor 13,13,31+ .long 0x12D6F508+ vxor 14,14,31++ bne .Lctr32_enc8x_break++ .long 0x100F0509+ .long 0x10300D09+ vadduwm 16,4,11+ .long 0x10511509+ vadduwm 17,4,7+ vxor 15,4,23+ .long 0x10721D09+ vadduwm 18,16,7+ vxor 16,16,23+ .long 0x11535509+ vadduwm 19,17,7+ vxor 17,17,23+ .long 0x11946509+ vadduwm 20,18,7+ vxor 18,18,23+ .long 0x11B56D09+ vadduwm 21,19,7+ vxor 19,19,23+ .long 0x11D67509+ vadduwm 22,20,7+ vxor 20,20,23+ vperm 0,0,0,6+ vadduwm 4,21,7+ vxor 21,21,23+ vperm 1,1,1,6+ vxor 22,22,23+ mtctr 9++ .long 0x11EFC508+ .long 0x7C002799+ vperm 2,2,2,6+ .long 0x1210C508+ .long 0x7C282799+ vperm 3,3,3,6+ .long 0x1231C508+ .long 0x7C5A2799+ vperm 10,10,10,6+ .long 0x1252C508+ .long 0x7C7B2799+ vperm 12,12,12,6+ .long 0x1273C508+ .long 0x7D5C2799+ vperm 13,13,13,6+ .long 0x1294C508+ .long 0x7D9D2799+ vperm 14,14,14,6+ .long 0x12B5C508+ .long 0x7DBE2799+ .long 0x12D6C508+ .long 0x7DDF2799+ addi 4,4,0x80++ b .Loop_ctr32_enc8x_middle++.align 5+.Lctr32_enc8x_break:+ cmpwi 5,-0x60+ blt .Lctr32_enc8x_one+ nop + beq .Lctr32_enc8x_two+ cmpwi 5,-0x40+ blt .Lctr32_enc8x_three+ nop + beq .Lctr32_enc8x_four+ cmpwi 5,-0x20+ blt .Lctr32_enc8x_five+ nop + beq .Lctr32_enc8x_six+ cmpwi 5,0x00+ blt .Lctr32_enc8x_seven++.Lctr32_enc8x_eight:+ .long 0x11EF0509+ .long 0x12100D09+ .long 0x12311509+ .long 0x12521D09+ .long 0x12735509+ .long 0x12946509+ .long 0x12B56D09+ .long 0x12D67509++ vperm 15,15,15,6+ vperm 16,16,16,6+ .long 0x7DE02799+ vperm 17,17,17,6+ .long 0x7E082799+ vperm 18,18,18,6+ .long 0x7E3A2799+ vperm 19,19,19,6+ .long 0x7E5B2799+ vperm 20,20,20,6+ .long 0x7E7C2799+ vperm 21,21,21,6+ .long 0x7E9D2799+ vperm 22,22,22,6+ .long 0x7EBE2799+ .long 0x7EDF2799+ addi 4,4,0x80+ b .Lctr32_enc8x_done++.align 5+.Lctr32_enc8x_seven:+ .long 0x11EF0D09+ .long 0x12101509+ .long 0x12311D09+ .long 0x12525509+ .long 0x12736509+ .long 0x12946D09+ .long 0x12B57509++ vperm 15,15,15,6+ vperm 16,16,16,6+ .long 0x7DE02799+ vperm 17,17,17,6+ .long 0x7E082799+ vperm 18,18,18,6+ .long 0x7E3A2799+ vperm 19,19,19,6+ .long 0x7E5B2799+ vperm 20,20,20,6+ .long 0x7E7C2799+ vperm 21,21,21,6+ .long 0x7E9D2799+ .long 0x7EBE2799+ addi 4,4,0x70+ b .Lctr32_enc8x_done++.align 5+.Lctr32_enc8x_six:+ .long 0x11EF1509+ .long 0x12101D09+ .long 0x12315509+ .long 0x12526509+ .long 0x12736D09+ .long 0x12947509++ vperm 15,15,15,6+ vperm 16,16,16,6+ .long 0x7DE02799+ vperm 17,17,17,6+ .long 0x7E082799+ vperm 18,18,18,6+ .long 0x7E3A2799+ vperm 19,19,19,6+ .long 0x7E5B2799+ vperm 20,20,20,6+ .long 0x7E7C2799+ .long 0x7E9D2799+ addi 4,4,0x60+ b .Lctr32_enc8x_done++.align 5+.Lctr32_enc8x_five:+ .long 0x11EF1D09+ .long 0x12105509+ .long 0x12316509+ .long 0x12526D09+ .long 0x12737509++ vperm 15,15,15,6+ vperm 16,16,16,6+ .long 0x7DE02799+ vperm 17,17,17,6+ .long 0x7E082799+ vperm 18,18,18,6+ .long 0x7E3A2799+ vperm 19,19,19,6+ .long 0x7E5B2799+ .long 0x7E7C2799+ addi 4,4,0x50+ b .Lctr32_enc8x_done++.align 5+.Lctr32_enc8x_four:+ .long 0x11EF5509+ .long 0x12106509+ .long 0x12316D09+ .long 0x12527509++ vperm 15,15,15,6+ vperm 16,16,16,6+ .long 0x7DE02799+ vperm 17,17,17,6+ .long 0x7E082799+ vperm 18,18,18,6+ .long 0x7E3A2799+ .long 0x7E5B2799+ addi 4,4,0x40+ b .Lctr32_enc8x_done++.align 5+.Lctr32_enc8x_three:+ .long 0x11EF6509+ .long 0x12106D09+ .long 0x12317509++ vperm 15,15,15,6+ vperm 16,16,16,6+ .long 0x7DE02799+ vperm 17,17,17,6+ .long 0x7E082799+ .long 0x7E3A2799+ addi 4,4,0x30+ b .Lctr32_enc8x_done++.align 5+.Lctr32_enc8x_two:+ .long 0x11EF6D09+ .long 0x12107509++ vperm 15,15,15,6+ vperm 16,16,16,6+ .long 0x7DE02799+ .long 0x7E082799+ addi 4,4,0x20+ b .Lctr32_enc8x_done++.align 5+.Lctr32_enc8x_one:+ .long 0x11EF7509++ vperm 15,15,15,6+ .long 0x7DE02799+ addi 4,4,0x10++.Lctr32_enc8x_done:+ li 10,79+ li 11,95+ stvx 6,10,1+ addi 10,10,32+ stvx 6,11,1+ addi 11,11,32+ stvx 6,10,1+ addi 10,10,32+ stvx 6,11,1+ addi 11,11,32+ stvx 6,10,1+ addi 10,10,32+ stvx 6,11,1+ addi 11,11,32+ stvx 6,10,1+ addi 10,10,32+ stvx 6,11,1+ addi 11,11,32++ or 12,12,12+ lvx 20,10,1+ addi 10,10,32+ lvx 21,11,1+ addi 11,11,32+ lvx 22,10,1+ addi 10,10,32+ lvx 23,11,1+ addi 11,11,32+ lvx 24,10,1+ addi 10,10,32+ lvx 25,11,1+ addi 11,11,32+ lvx 26,10,1+ addi 10,10,32+ lvx 27,11,1+ addi 11,11,32+ lvx 28,10,1+ addi 10,10,32+ lvx 29,11,1+ addi 11,11,32+ lvx 30,10,1+ lvx 31,11,1+ ld 26,400(1)+ ld 27,408(1)+ ld 28,416(1)+ ld 29,424(1)+ ld 30,432(1)+ ld 31,440(1)+ addi 1,1,448+ blr +.long 0+.byte 0,12,0x04,0,0x80,6,6,0+.long 0+.size crypton_aes_p8_ctr32_encrypt_blocks,.-crypton_aes_p8_ctr32_encrypt_blocks+.globl crypton_aes_p8_xts_encrypt+.type crypton_aes_p8_xts_encrypt,@function+.align 5+crypton_aes_p8_xts_encrypt:+.localentry crypton_aes_p8_xts_encrypt,0++ mr 10,3+ li 3,-1+ cmpldi 5,16+ .long 0x4dc00020++ lis 0,0xfff0+ li 12,-1+ li 11,0+ or 0,0,0++ vspltisb 9,0x07+ lvsl 6,11,11+ vspltisb 11,0x0f+ vxor 6,6,9++ li 3,15+ lvx 8,0,8+ lvsl 5,0,8+ lvx 4,3,8+ vxor 5,5,11+ vperm 8,8,4,5++ neg 11,10+ lvsr 5,0,11+ lvx 2,0,10+ addi 10,10,15+ vxor 5,5,11++ cmpldi 7,0+ beq .Lxts_enc_no_key2++ lvsr 7,0,7+ lwz 9,240(7)+ srwi 9,9,1+ subi 9,9,1+ li 3,16++ lvx 0,0,7+ lvx 1,3,7+ addi 3,3,16+ vperm 0,1,0,7+ vxor 8,8,0+ lvx 0,3,7+ addi 3,3,16+ mtctr 9++.Ltweak_xts_enc:+ vperm 1,0,1,7+ .long 0x11080D08+ lvx 1,3,7+ addi 3,3,16+ vperm 0,1,0,7+ .long 0x11080508+ lvx 0,3,7+ addi 3,3,16+ bdnz .Ltweak_xts_enc++ vperm 1,0,1,7+ .long 0x11080D08+ lvx 1,3,7+ vperm 0,1,0,7+ .long 0x11080509++ li 8,0+ b .Lxts_enc++.Lxts_enc_no_key2:+ li 3,-16+ and 5,5,3+++.Lxts_enc:+ lvx 4,0,10+ addi 10,10,16++ lvsr 7,0,6+ lwz 9,240(6)+ srwi 9,9,1+ subi 9,9,1+ li 3,16++ vslb 10,9,9+ vor 10,10,9+ vspltisb 11,1+ vsldoi 10,10,11,15++ cmpldi 5,96+ bge _aesp8_xts_encrypt6x++ andi. 7,5,15+ subic 0,5,32+ subi 7,7,16+ subfe 0,0,0+ and 0,0,7+ add 10,10,0++ lvx 0,0,6+ lvx 1,3,6+ addi 3,3,16+ vperm 2,2,4,5+ vperm 0,1,0,7+ vxor 2,2,8+ vxor 2,2,0+ lvx 0,3,6+ addi 3,3,16+ mtctr 9+ b .Loop_xts_enc++.align 5+.Loop_xts_enc:+ vperm 1,0,1,7+ .long 0x10420D08+ lvx 1,3,6+ addi 3,3,16+ vperm 0,1,0,7+ .long 0x10420508+ lvx 0,3,6+ addi 3,3,16+ bdnz .Loop_xts_enc++ vperm 1,0,1,7+ .long 0x10420D08+ lvx 1,3,6+ li 3,16+ vperm 0,1,0,7+ vxor 0,0,8+ .long 0x10620509++ vperm 11,3,3,6++ .long 0x7D602799++ addi 4,4,16++ subic. 5,5,16+ beq .Lxts_enc_done++ vor 2,4,4+ lvx 4,0,10+ addi 10,10,16+ lvx 0,0,6+ lvx 1,3,6+ addi 3,3,16++ subic 0,5,32+ subfe 0,0,0+ and 0,0,7+ add 10,10,0++ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vand 11,11,10+ vxor 8,8,11++ vperm 2,2,4,5+ vperm 0,1,0,7+ vxor 2,2,8+ vxor 3,3,0+ vxor 2,2,0+ lvx 0,3,6+ addi 3,3,16++ mtctr 9+ cmpldi 5,16+ bge .Loop_xts_enc++ vxor 3,3,8+ lvsr 5,0,5+ vxor 4,4,4+ vspltisb 11,-1+ vperm 4,4,11,5+ vsel 2,2,3,4++ subi 11,4,17+ subi 4,4,16+ mtctr 5+ li 5,16+.Loop_xts_enc_steal:+ lbzu 0,1(11)+ stb 0,16(11)+ bdnz .Loop_xts_enc_steal++ mtctr 9+ b .Loop_xts_enc++.Lxts_enc_done:+ cmpldi 8,0+ beq .Lxts_enc_ret++ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vand 11,11,10+ vxor 8,8,11++ vperm 8,8,8,6+ .long 0x7D004799++.Lxts_enc_ret:+ or 12,12,12+ li 3,0+ blr +.long 0+.byte 0,12,0x04,0,0x80,6,6,0+.long 0+.size crypton_aes_p8_xts_encrypt,.-crypton_aes_p8_xts_encrypt++.globl crypton_aes_p8_xts_decrypt+.type crypton_aes_p8_xts_decrypt,@function+.align 5+crypton_aes_p8_xts_decrypt:+.localentry crypton_aes_p8_xts_decrypt,0++ mr 10,3+ li 3,-1+ cmpldi 5,16+ .long 0x4dc00020++ lis 0,0xfff8+ li 12,-1+ li 11,0+ or 0,0,0++ andi. 0,5,15+ neg 0,0+ andi. 0,0,16+ sub 5,5,0++ vspltisb 9,0x07+ lvsl 6,11,11+ vspltisb 11,0x0f+ vxor 6,6,9++ li 3,15+ lvx 8,0,8+ lvsl 5,0,8+ lvx 4,3,8+ vxor 5,5,11+ vperm 8,8,4,5++ neg 11,10+ lvsr 5,0,11+ lvx 2,0,10+ addi 10,10,15+ vxor 5,5,11++ cmpldi 7,0+ beq .Lxts_dec_no_key2++ lvsr 7,0,7+ lwz 9,240(7)+ srwi 9,9,1+ subi 9,9,1+ li 3,16++ lvx 0,0,7+ lvx 1,3,7+ addi 3,3,16+ vperm 0,1,0,7+ vxor 8,8,0+ lvx 0,3,7+ addi 3,3,16+ mtctr 9++.Ltweak_xts_dec:+ vperm 1,0,1,7+ .long 0x11080D08+ lvx 1,3,7+ addi 3,3,16+ vperm 0,1,0,7+ .long 0x11080508+ lvx 0,3,7+ addi 3,3,16+ bdnz .Ltweak_xts_dec++ vperm 1,0,1,7+ .long 0x11080D08+ lvx 1,3,7+ vperm 0,1,0,7+ .long 0x11080509++ li 8,0+ b .Lxts_dec++.Lxts_dec_no_key2:+ neg 3,5+ andi. 3,3,15+ add 5,5,3+++.Lxts_dec:+ lvx 4,0,10+ addi 10,10,16++ lvsr 7,0,6+ lwz 9,240(6)+ srwi 9,9,1+ subi 9,9,1+ li 3,16++ vslb 10,9,9+ vor 10,10,9+ vspltisb 11,1+ vsldoi 10,10,11,15++ cmpldi 5,96+ bge _aesp8_xts_decrypt6x++ lvx 0,0,6+ lvx 1,3,6+ addi 3,3,16+ vperm 2,2,4,5+ vperm 0,1,0,7+ vxor 2,2,8+ vxor 2,2,0+ lvx 0,3,6+ addi 3,3,16+ mtctr 9++ cmpldi 5,16+ blt .Ltail_xts_dec+++.align 5+.Loop_xts_dec:+ vperm 1,0,1,7+ .long 0x10420D48+ lvx 1,3,6+ addi 3,3,16+ vperm 0,1,0,7+ .long 0x10420548+ lvx 0,3,6+ addi 3,3,16+ bdnz .Loop_xts_dec++ vperm 1,0,1,7+ .long 0x10420D48+ lvx 1,3,6+ li 3,16+ vperm 0,1,0,7+ vxor 0,0,8+ .long 0x10620549++ vperm 11,3,3,6++ .long 0x7D602799++ addi 4,4,16++ subic. 5,5,16+ beq .Lxts_dec_done++ vor 2,4,4+ lvx 4,0,10+ addi 10,10,16+ lvx 0,0,6+ lvx 1,3,6+ addi 3,3,16++ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vand 11,11,10+ vxor 8,8,11++ vperm 2,2,4,5+ vperm 0,1,0,7+ vxor 2,2,8+ vxor 2,2,0+ lvx 0,3,6+ addi 3,3,16++ mtctr 9+ cmpldi 5,16+ bge .Loop_xts_dec++.Ltail_xts_dec:+ vsrab 11,8,9+ vaddubm 12,8,8+ vsldoi 11,11,11,15+ vand 11,11,10+ vxor 12,12,11++ subi 10,10,16+ add 10,10,5++ vxor 2,2,8+ vxor 2,2,12++.Loop_xts_dec_short:+ vperm 1,0,1,7+ .long 0x10420D48+ lvx 1,3,6+ addi 3,3,16+ vperm 0,1,0,7+ .long 0x10420548+ lvx 0,3,6+ addi 3,3,16+ bdnz .Loop_xts_dec_short++ vperm 1,0,1,7+ .long 0x10420D48+ lvx 1,3,6+ li 3,16+ vperm 0,1,0,7+ vxor 0,0,12+ .long 0x10620549++ vperm 11,3,3,6++ .long 0x7D602799+++ vor 2,4,4+ lvx 4,0,10++ lvx 0,0,6+ lvx 1,3,6+ addi 3,3,16+ vperm 2,2,4,5+ vperm 0,1,0,7++ lvsr 5,0,5+ vxor 4,4,4+ vspltisb 11,-1+ vperm 4,4,11,5+ vsel 2,2,3,4++ vxor 0,0,8+ vxor 2,2,0+ lvx 0,3,6+ addi 3,3,16++ subi 11,4,1+ mtctr 5+ li 5,16+.Loop_xts_dec_steal:+ lbzu 0,1(11)+ stb 0,16(11)+ bdnz .Loop_xts_dec_steal++ mtctr 9+ b .Loop_xts_dec++.Lxts_dec_done:+ cmpldi 8,0+ beq .Lxts_dec_ret++ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vand 11,11,10+ vxor 8,8,11++ vperm 8,8,8,6+ .long 0x7D004799++.Lxts_dec_ret:+ or 12,12,12+ li 3,0+ blr +.long 0+.byte 0,12,0x04,0,0x80,6,6,0+.long 0+.size crypton_aes_p8_xts_decrypt,.-crypton_aes_p8_xts_decrypt+.align 5+_aesp8_xts_encrypt6x:+ stdu 1,-448(1)+ mflr 11+ li 7,207+ li 3,223+ std 11,464(1)+ stvx 20,7,1+ addi 7,7,32+ stvx 21,3,1+ addi 3,3,32+ stvx 22,7,1+ addi 7,7,32+ stvx 23,3,1+ addi 3,3,32+ stvx 24,7,1+ addi 7,7,32+ stvx 25,3,1+ addi 3,3,32+ stvx 26,7,1+ addi 7,7,32+ stvx 27,3,1+ addi 3,3,32+ stvx 28,7,1+ addi 7,7,32+ stvx 29,3,1+ addi 3,3,32+ stvx 30,7,1+ stvx 31,3,1+ li 0,-1+ stw 12,396(1)+ li 3,0x10+ std 26,400(1)+ li 26,0x20+ std 27,408(1)+ li 27,0x30+ std 28,416(1)+ li 28,0x40+ std 29,424(1)+ li 29,0x50+ std 30,432(1)+ li 30,0x60+ std 31,440(1)+ li 31,0x70+ or 0,0,0++ subi 9,9,3++ lvx 23,0,6+ lvx 30,3,6+ addi 6,6,0x20+ lvx 31,0,6+ vperm 23,30,23,7+ addi 7,1,64+15+ mtctr 9++.Load_xts_enc_key:+ vperm 24,31,30,7+ lvx 30,3,6+ addi 6,6,0x20+ stvx 24,0,7+ vperm 25,30,31,7+ lvx 31,0,6+ stvx 25,3,7+ addi 7,7,0x20+ bdnz .Load_xts_enc_key++ lvx 26,3,6+ vperm 24,31,30,7+ lvx 27,26,6+ stvx 24,0,7+ vperm 25,26,31,7+ lvx 28,27,6+ stvx 25,3,7+ addi 7,1,64+15+ vperm 26,27,26,7+ lvx 29,28,6+ vperm 27,28,27,7+ lvx 30,29,6+ vperm 28,29,28,7+ lvx 31,30,6+ vperm 29,30,29,7+ lvx 22,31,6+ vperm 30,31,30,7+ lvx 24,0,7+ vperm 31,22,31,7+ lvx 25,3,7++ vperm 0,2,4,5+ subi 10,10,31+ vxor 17,8,23+ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vand 11,11,10+ vxor 7,0,17+ vxor 8,8,11++ .long 0x7C235699+ vxor 18,8,23+ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vperm 1,1,1,6+ vand 11,11,10+ vxor 12,1,18+ vxor 8,8,11++ .long 0x7C5A5699+ andi. 31,5,15+ vxor 19,8,23+ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vperm 2,2,2,6+ vand 11,11,10+ vxor 13,2,19+ vxor 8,8,11++ .long 0x7C7B5699+ sub 5,5,31+ vxor 20,8,23+ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vperm 3,3,3,6+ vand 11,11,10+ vxor 14,3,20+ vxor 8,8,11++ .long 0x7C9C5699+ subi 5,5,0x60+ vxor 21,8,23+ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vperm 4,4,4,6+ vand 11,11,10+ vxor 15,4,21+ vxor 8,8,11++ .long 0x7CBD5699+ addi 10,10,0x60+ vxor 22,8,23+ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vperm 5,5,5,6+ vand 11,11,10+ vxor 16,5,22+ vxor 8,8,11++ vxor 31,31,23+ mtctr 9+ b .Loop_xts_enc6x++.align 5+.Loop_xts_enc6x:+ .long 0x10E7C508+ .long 0x118CC508+ .long 0x11ADC508+ .long 0x11CEC508+ .long 0x11EFC508+ .long 0x1210C508+ lvx 24,26,7+ addi 7,7,0x20++ .long 0x10E7CD08+ .long 0x118CCD08+ .long 0x11ADCD08+ .long 0x11CECD08+ .long 0x11EFCD08+ .long 0x1210CD08+ lvx 25,3,7+ bdnz .Loop_xts_enc6x++ subic 5,5,96+ vxor 0,17,31+ .long 0x10E7C508+ .long 0x118CC508+ vsrab 11,8,9+ vxor 17,8,23+ vaddubm 8,8,8+ .long 0x11ADC508+ .long 0x11CEC508+ vsldoi 11,11,11,15+ .long 0x11EFC508+ .long 0x1210C508++ subfe. 0,0,0+ vand 11,11,10+ .long 0x10E7CD08+ .long 0x118CCD08+ vxor 8,8,11+ .long 0x11ADCD08+ .long 0x11CECD08+ vxor 1,18,31+ vsrab 11,8,9+ vxor 18,8,23+ .long 0x11EFCD08+ .long 0x1210CD08++ and 0,0,5+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ .long 0x10E7D508+ .long 0x118CD508+ vand 11,11,10+ .long 0x11ADD508+ .long 0x11CED508+ vxor 8,8,11+ .long 0x11EFD508+ .long 0x1210D508++ add 10,10,0++++ vxor 2,19,31+ vsrab 11,8,9+ vxor 19,8,23+ vaddubm 8,8,8+ .long 0x10E7DD08+ .long 0x118CDD08+ vsldoi 11,11,11,15+ .long 0x11ADDD08+ .long 0x11CEDD08+ vand 11,11,10+ .long 0x11EFDD08+ .long 0x1210DD08++ addi 7,1,64+15+ vxor 8,8,11+ .long 0x10E7E508+ .long 0x118CE508+ vxor 3,20,31+ vsrab 11,8,9+ vxor 20,8,23+ .long 0x11ADE508+ .long 0x11CEE508+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ .long 0x11EFE508+ .long 0x1210E508+ lvx 24,0,7+ vand 11,11,10++ .long 0x10E7ED08+ .long 0x118CED08+ vxor 8,8,11+ .long 0x11ADED08+ .long 0x11CEED08+ vxor 4,21,31+ vsrab 11,8,9+ vxor 21,8,23+ .long 0x11EFED08+ .long 0x1210ED08+ lvx 25,3,7+ vaddubm 8,8,8+ vsldoi 11,11,11,15++ .long 0x10E7F508+ .long 0x118CF508+ vand 11,11,10+ .long 0x11ADF508+ .long 0x11CEF508+ vxor 8,8,11+ .long 0x11EFF508+ .long 0x1210F508+ vxor 5,22,31+ vsrab 11,8,9+ vxor 22,8,23++ .long 0x10E70509+ .long 0x7C005699+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ .long 0x118C0D09+ .long 0x7C235699+ .long 0x11AD1509+ vperm 0,0,0,6+ .long 0x7C5A5699+ vand 11,11,10+ .long 0x11CE1D09+ vperm 1,1,1,6+ .long 0x7C7B5699+ .long 0x11EF2509+ vperm 2,2,2,6+ .long 0x7C9C5699+ vxor 8,8,11+ .long 0x11702D09++ vperm 3,3,3,6+ .long 0x7CBD5699+ addi 10,10,0x60+ vperm 4,4,4,6+ vperm 5,5,5,6++ vperm 7,7,7,6+ vperm 12,12,12,6+ .long 0x7CE02799+ vxor 7,0,17+ vperm 13,13,13,6+ .long 0x7D832799+ vxor 12,1,18+ vperm 14,14,14,6+ .long 0x7DBA2799+ vxor 13,2,19+ vperm 15,15,15,6+ .long 0x7DDB2799+ vxor 14,3,20+ vperm 16,11,11,6+ .long 0x7DFC2799+ vxor 15,4,21+ .long 0x7E1D2799++ vxor 16,5,22+ addi 4,4,0x60++ mtctr 9+ beq .Loop_xts_enc6x++ addic. 5,5,0x60+ beq .Lxts_enc6x_zero+ cmpwi 5,0x20+ blt .Lxts_enc6x_one+ nop + beq .Lxts_enc6x_two+ cmpwi 5,0x40+ blt .Lxts_enc6x_three+ nop + beq .Lxts_enc6x_four++.Lxts_enc6x_five:+ vxor 7,1,17+ vxor 12,2,18+ vxor 13,3,19+ vxor 14,4,20+ vxor 15,5,21++ bl _aesp8_xts_enc5x++ vperm 7,7,7,6+ vor 17,22,22+ vperm 12,12,12,6+ .long 0x7CE02799+ vperm 13,13,13,6+ .long 0x7D832799+ vperm 14,14,14,6+ .long 0x7DBA2799+ vxor 11,15,22+ vperm 15,15,15,6+ .long 0x7DDB2799+ .long 0x7DFC2799+ addi 4,4,0x50+ bne .Lxts_enc6x_steal+ b .Lxts_enc6x_done++.align 4+.Lxts_enc6x_four:+ vxor 7,2,17+ vxor 12,3,18+ vxor 13,4,19+ vxor 14,5,20+ vxor 15,15,15++ bl _aesp8_xts_enc5x++ vperm 7,7,7,6+ vor 17,21,21+ vperm 12,12,12,6+ .long 0x7CE02799+ vperm 13,13,13,6+ .long 0x7D832799+ vxor 11,14,21+ vperm 14,14,14,6+ .long 0x7DBA2799+ .long 0x7DDB2799+ addi 4,4,0x40+ bne .Lxts_enc6x_steal+ b .Lxts_enc6x_done++.align 4+.Lxts_enc6x_three:+ vxor 7,3,17+ vxor 12,4,18+ vxor 13,5,19+ vxor 14,14,14+ vxor 15,15,15++ bl _aesp8_xts_enc5x++ vperm 7,7,7,6+ vor 17,20,20+ vperm 12,12,12,6+ .long 0x7CE02799+ vxor 11,13,20+ vperm 13,13,13,6+ .long 0x7D832799+ .long 0x7DBA2799+ addi 4,4,0x30+ bne .Lxts_enc6x_steal+ b .Lxts_enc6x_done++.align 4+.Lxts_enc6x_two:+ vxor 7,4,17+ vxor 12,5,18+ vxor 13,13,13+ vxor 14,14,14+ vxor 15,15,15++ bl _aesp8_xts_enc5x++ vperm 7,7,7,6+ vor 17,19,19+ vxor 11,12,19+ vperm 12,12,12,6+ .long 0x7CE02799+ .long 0x7D832799+ addi 4,4,0x20+ bne .Lxts_enc6x_steal+ b .Lxts_enc6x_done++.align 4+.Lxts_enc6x_one:+ vxor 7,5,17+ nop +.Loop_xts_enc1x:+ .long 0x10E7C508+ lvx 24,26,7+ addi 7,7,0x20++ .long 0x10E7CD08+ lvx 25,3,7+ bdnz .Loop_xts_enc1x++ add 10,10,31+ cmpwi 31,0+ .long 0x10E7C508++ subi 10,10,16+ .long 0x10E7CD08++ lvsr 5,0,31+ .long 0x10E7D508++ .long 0x7C005699+ .long 0x10E7DD08++ addi 7,1,64+15+ .long 0x10E7E508+ lvx 24,0,7++ .long 0x10E7ED08+ lvx 25,3,7+ vxor 17,17,31++ vperm 0,0,0,6+ .long 0x10E7F508++ vperm 0,0,0,5+ .long 0x10E78D09++ vor 17,18,18+ vxor 11,7,18+ vperm 7,7,7,6+ .long 0x7CE02799+ addi 4,4,0x10+ bne .Lxts_enc6x_steal+ b .Lxts_enc6x_done++.align 4+.Lxts_enc6x_zero:+ cmpwi 31,0+ beq .Lxts_enc6x_done++ add 10,10,31+ subi 10,10,16+ .long 0x7C005699+ lvsr 5,0,31+ vperm 0,0,0,6+ vperm 0,0,0,5+ vxor 11,11,17+.Lxts_enc6x_steal:+ vxor 0,0,17+ vxor 7,7,7+ vspltisb 12,-1+ vperm 7,7,12,5+ vsel 7,0,11,7++ subi 30,4,17+ subi 4,4,16+ mtctr 31+.Loop_xts_enc6x_steal:+ lbzu 0,1(30)+ stb 0,16(30)+ bdnz .Loop_xts_enc6x_steal++ li 31,0+ mtctr 9+ b .Loop_xts_enc1x++.align 4+.Lxts_enc6x_done:+ cmpldi 8,0+ beq .Lxts_enc6x_ret++ vxor 8,17,23+ vperm 8,8,8,6+ .long 0x7D004799++.Lxts_enc6x_ret:+ mtlr 11+ li 10,79+ li 11,95+ stvx 9,10,1+ addi 10,10,32+ stvx 9,11,1+ addi 11,11,32+ stvx 9,10,1+ addi 10,10,32+ stvx 9,11,1+ addi 11,11,32+ stvx 9,10,1+ addi 10,10,32+ stvx 9,11,1+ addi 11,11,32+ stvx 9,10,1+ addi 10,10,32+ stvx 9,11,1+ addi 11,11,32++ or 12,12,12+ lvx 20,10,1+ addi 10,10,32+ lvx 21,11,1+ addi 11,11,32+ lvx 22,10,1+ addi 10,10,32+ lvx 23,11,1+ addi 11,11,32+ lvx 24,10,1+ addi 10,10,32+ lvx 25,11,1+ addi 11,11,32+ lvx 26,10,1+ addi 10,10,32+ lvx 27,11,1+ addi 11,11,32+ lvx 28,10,1+ addi 10,10,32+ lvx 29,11,1+ addi 11,11,32+ lvx 30,10,1+ lvx 31,11,1+ ld 26,400(1)+ ld 27,408(1)+ ld 28,416(1)+ ld 29,424(1)+ ld 30,432(1)+ ld 31,440(1)+ addi 1,1,448+ blr +.long 0+.byte 0,12,0x04,1,0x80,6,6,0+.long 0++.align 5+_aesp8_xts_enc5x:+ .long 0x10E7C508+ .long 0x118CC508+ .long 0x11ADC508+ .long 0x11CEC508+ .long 0x11EFC508+ lvx 24,26,7+ addi 7,7,0x20++ .long 0x10E7CD08+ .long 0x118CCD08+ .long 0x11ADCD08+ .long 0x11CECD08+ .long 0x11EFCD08+ lvx 25,3,7+ bdnz _aesp8_xts_enc5x++ add 10,10,31+ cmpwi 31,0+ .long 0x10E7C508+ .long 0x118CC508+ .long 0x11ADC508+ .long 0x11CEC508+ .long 0x11EFC508++ subi 10,10,16+ .long 0x10E7CD08+ .long 0x118CCD08+ .long 0x11ADCD08+ .long 0x11CECD08+ .long 0x11EFCD08+ vxor 17,17,31++ .long 0x10E7D508+ lvsr 5,0,31+ .long 0x118CD508+ .long 0x11ADD508+ .long 0x11CED508+ .long 0x11EFD508+ vxor 1,18,31++ .long 0x10E7DD08+ .long 0x7C005699+ .long 0x118CDD08+ .long 0x11ADDD08+ .long 0x11CEDD08+ .long 0x11EFDD08+ vxor 2,19,31++ addi 7,1,64+15+ .long 0x10E7E508+ .long 0x118CE508+ .long 0x11ADE508+ .long 0x11CEE508+ .long 0x11EFE508+ lvx 24,0,7+ vxor 3,20,31++ .long 0x10E7ED08+ vperm 0,0,0,6+ .long 0x118CED08+ .long 0x11ADED08+ .long 0x11CEED08+ .long 0x11EFED08+ lvx 25,3,7+ vxor 4,21,31++ .long 0x10E7F508+ vperm 0,0,0,5+ .long 0x118CF508+ .long 0x11ADF508+ .long 0x11CEF508+ .long 0x11EFF508++ .long 0x10E78D09+ .long 0x118C0D09+ .long 0x11AD1509+ .long 0x11CE1D09+ .long 0x11EF2509+ blr +.long 0+.byte 0,12,0x14,0,0,0,0,0++.align 5+_aesp8_xts_decrypt6x:+ stdu 1,-448(1)+ mflr 11+ li 7,207+ li 3,223+ std 11,464(1)+ stvx 20,7,1+ addi 7,7,32+ stvx 21,3,1+ addi 3,3,32+ stvx 22,7,1+ addi 7,7,32+ stvx 23,3,1+ addi 3,3,32+ stvx 24,7,1+ addi 7,7,32+ stvx 25,3,1+ addi 3,3,32+ stvx 26,7,1+ addi 7,7,32+ stvx 27,3,1+ addi 3,3,32+ stvx 28,7,1+ addi 7,7,32+ stvx 29,3,1+ addi 3,3,32+ stvx 30,7,1+ stvx 31,3,1+ li 0,-1+ stw 12,396(1)+ li 3,0x10+ std 26,400(1)+ li 26,0x20+ std 27,408(1)+ li 27,0x30+ std 28,416(1)+ li 28,0x40+ std 29,424(1)+ li 29,0x50+ std 30,432(1)+ li 30,0x60+ std 31,440(1)+ li 31,0x70+ or 0,0,0++ subi 9,9,3++ lvx 23,0,6+ lvx 30,3,6+ addi 6,6,0x20+ lvx 31,0,6+ vperm 23,30,23,7+ addi 7,1,64+15+ mtctr 9++.Load_xts_dec_key:+ vperm 24,31,30,7+ lvx 30,3,6+ addi 6,6,0x20+ stvx 24,0,7+ vperm 25,30,31,7+ lvx 31,0,6+ stvx 25,3,7+ addi 7,7,0x20+ bdnz .Load_xts_dec_key++ lvx 26,3,6+ vperm 24,31,30,7+ lvx 27,26,6+ stvx 24,0,7+ vperm 25,26,31,7+ lvx 28,27,6+ stvx 25,3,7+ addi 7,1,64+15+ vperm 26,27,26,7+ lvx 29,28,6+ vperm 27,28,27,7+ lvx 30,29,6+ vperm 28,29,28,7+ lvx 31,30,6+ vperm 29,30,29,7+ lvx 22,31,6+ vperm 30,31,30,7+ lvx 24,0,7+ vperm 31,22,31,7+ lvx 25,3,7++ vperm 0,2,4,5+ subi 10,10,31+ vxor 17,8,23+ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vand 11,11,10+ vxor 7,0,17+ vxor 8,8,11++ .long 0x7C235699+ vxor 18,8,23+ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vperm 1,1,1,6+ vand 11,11,10+ vxor 12,1,18+ vxor 8,8,11++ .long 0x7C5A5699+ andi. 31,5,15+ vxor 19,8,23+ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vperm 2,2,2,6+ vand 11,11,10+ vxor 13,2,19+ vxor 8,8,11++ .long 0x7C7B5699+ sub 5,5,31+ vxor 20,8,23+ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vperm 3,3,3,6+ vand 11,11,10+ vxor 14,3,20+ vxor 8,8,11++ .long 0x7C9C5699+ subi 5,5,0x60+ vxor 21,8,23+ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vperm 4,4,4,6+ vand 11,11,10+ vxor 15,4,21+ vxor 8,8,11++ .long 0x7CBD5699+ addi 10,10,0x60+ vxor 22,8,23+ vsrab 11,8,9+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ vperm 5,5,5,6+ vand 11,11,10+ vxor 16,5,22+ vxor 8,8,11++ vxor 31,31,23+ mtctr 9+ b .Loop_xts_dec6x++.align 5+.Loop_xts_dec6x:+ .long 0x10E7C548+ .long 0x118CC548+ .long 0x11ADC548+ .long 0x11CEC548+ .long 0x11EFC548+ .long 0x1210C548+ lvx 24,26,7+ addi 7,7,0x20++ .long 0x10E7CD48+ .long 0x118CCD48+ .long 0x11ADCD48+ .long 0x11CECD48+ .long 0x11EFCD48+ .long 0x1210CD48+ lvx 25,3,7+ bdnz .Loop_xts_dec6x++ subic 5,5,96+ vxor 0,17,31+ .long 0x10E7C548+ .long 0x118CC548+ vsrab 11,8,9+ vxor 17,8,23+ vaddubm 8,8,8+ .long 0x11ADC548+ .long 0x11CEC548+ vsldoi 11,11,11,15+ .long 0x11EFC548+ .long 0x1210C548++ subfe. 0,0,0+ vand 11,11,10+ .long 0x10E7CD48+ .long 0x118CCD48+ vxor 8,8,11+ .long 0x11ADCD48+ .long 0x11CECD48+ vxor 1,18,31+ vsrab 11,8,9+ vxor 18,8,23+ .long 0x11EFCD48+ .long 0x1210CD48++ and 0,0,5+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ .long 0x10E7D548+ .long 0x118CD548+ vand 11,11,10+ .long 0x11ADD548+ .long 0x11CED548+ vxor 8,8,11+ .long 0x11EFD548+ .long 0x1210D548++ add 10,10,0++++ vxor 2,19,31+ vsrab 11,8,9+ vxor 19,8,23+ vaddubm 8,8,8+ .long 0x10E7DD48+ .long 0x118CDD48+ vsldoi 11,11,11,15+ .long 0x11ADDD48+ .long 0x11CEDD48+ vand 11,11,10+ .long 0x11EFDD48+ .long 0x1210DD48++ addi 7,1,64+15+ vxor 8,8,11+ .long 0x10E7E548+ .long 0x118CE548+ vxor 3,20,31+ vsrab 11,8,9+ vxor 20,8,23+ .long 0x11ADE548+ .long 0x11CEE548+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ .long 0x11EFE548+ .long 0x1210E548+ lvx 24,0,7+ vand 11,11,10++ .long 0x10E7ED48+ .long 0x118CED48+ vxor 8,8,11+ .long 0x11ADED48+ .long 0x11CEED48+ vxor 4,21,31+ vsrab 11,8,9+ vxor 21,8,23+ .long 0x11EFED48+ .long 0x1210ED48+ lvx 25,3,7+ vaddubm 8,8,8+ vsldoi 11,11,11,15++ .long 0x10E7F548+ .long 0x118CF548+ vand 11,11,10+ .long 0x11ADF548+ .long 0x11CEF548+ vxor 8,8,11+ .long 0x11EFF548+ .long 0x1210F548+ vxor 5,22,31+ vsrab 11,8,9+ vxor 22,8,23++ .long 0x10E70549+ .long 0x7C005699+ vaddubm 8,8,8+ vsldoi 11,11,11,15+ .long 0x118C0D49+ .long 0x7C235699+ .long 0x11AD1549+ vperm 0,0,0,6+ .long 0x7C5A5699+ vand 11,11,10+ .long 0x11CE1D49+ vperm 1,1,1,6+ .long 0x7C7B5699+ .long 0x11EF2549+ vperm 2,2,2,6+ .long 0x7C9C5699+ vxor 8,8,11+ .long 0x12102D49+ vperm 3,3,3,6+ .long 0x7CBD5699+ addi 10,10,0x60+ vperm 4,4,4,6+ vperm 5,5,5,6++ vperm 7,7,7,6+ vperm 12,12,12,6+ .long 0x7CE02799+ vxor 7,0,17+ vperm 13,13,13,6+ .long 0x7D832799+ vxor 12,1,18+ vperm 14,14,14,6+ .long 0x7DBA2799+ vxor 13,2,19+ vperm 15,15,15,6+ .long 0x7DDB2799+ vxor 14,3,20+ vperm 16,16,16,6+ .long 0x7DFC2799+ vxor 15,4,21+ .long 0x7E1D2799+ vxor 16,5,22+ addi 4,4,0x60++ mtctr 9+ beq .Loop_xts_dec6x++ addic. 5,5,0x60+ beq .Lxts_dec6x_zero+ cmpwi 5,0x20+ blt .Lxts_dec6x_one+ nop + beq .Lxts_dec6x_two+ cmpwi 5,0x40+ blt .Lxts_dec6x_three+ nop + beq .Lxts_dec6x_four++.Lxts_dec6x_five:+ vxor 7,1,17+ vxor 12,2,18+ vxor 13,3,19+ vxor 14,4,20+ vxor 15,5,21++ bl _aesp8_xts_dec5x++ vperm 7,7,7,6+ vor 17,22,22+ vxor 18,8,23+ vperm 12,12,12,6+ .long 0x7CE02799+ vxor 7,0,18+ vperm 13,13,13,6+ .long 0x7D832799+ vperm 14,14,14,6+ .long 0x7DBA2799+ vperm 15,15,15,6+ .long 0x7DDB2799+ .long 0x7DFC2799+ addi 4,4,0x50+ bne .Lxts_dec6x_steal+ b .Lxts_dec6x_done++.align 4+.Lxts_dec6x_four:+ vxor 7,2,17+ vxor 12,3,18+ vxor 13,4,19+ vxor 14,5,20+ vxor 15,15,15++ bl _aesp8_xts_dec5x++ vperm 7,7,7,6+ vor 17,21,21+ vor 18,22,22+ vperm 12,12,12,6+ .long 0x7CE02799+ vxor 7,0,22+ vperm 13,13,13,6+ .long 0x7D832799+ vperm 14,14,14,6+ .long 0x7DBA2799+ .long 0x7DDB2799+ addi 4,4,0x40+ bne .Lxts_dec6x_steal+ b .Lxts_dec6x_done++.align 4+.Lxts_dec6x_three:+ vxor 7,3,17+ vxor 12,4,18+ vxor 13,5,19+ vxor 14,14,14+ vxor 15,15,15++ bl _aesp8_xts_dec5x++ vperm 7,7,7,6+ vor 17,20,20+ vor 18,21,21+ vperm 12,12,12,6+ .long 0x7CE02799+ vxor 7,0,21+ vperm 13,13,13,6+ .long 0x7D832799+ .long 0x7DBA2799+ addi 4,4,0x30+ bne .Lxts_dec6x_steal+ b .Lxts_dec6x_done++.align 4+.Lxts_dec6x_two:+ vxor 7,4,17+ vxor 12,5,18+ vxor 13,13,13+ vxor 14,14,14+ vxor 15,15,15++ bl _aesp8_xts_dec5x++ vperm 7,7,7,6+ vor 17,19,19+ vor 18,20,20+ vperm 12,12,12,6+ .long 0x7CE02799+ vxor 7,0,20+ .long 0x7D832799+ addi 4,4,0x20+ bne .Lxts_dec6x_steal+ b .Lxts_dec6x_done++.align 4+.Lxts_dec6x_one:+ vxor 7,5,17+ nop +.Loop_xts_dec1x:+ .long 0x10E7C548+ lvx 24,26,7+ addi 7,7,0x20++ .long 0x10E7CD48+ lvx 25,3,7+ bdnz .Loop_xts_dec1x++ subi 0,31,1+ .long 0x10E7C548++ andi. 0,0,16+ cmpwi 31,0+ .long 0x10E7CD48++ sub 10,10,0+ .long 0x10E7D548++ .long 0x7C005699+ .long 0x10E7DD48++ addi 7,1,64+15+ .long 0x10E7E548+ lvx 24,0,7++ .long 0x10E7ED48+ lvx 25,3,7+ vxor 17,17,31++ vperm 0,0,0,6+ .long 0x10E7F548++ mtctr 9+ .long 0x10E78D49++ vor 17,18,18+ vor 18,19,19+ vperm 7,7,7,6+ .long 0x7CE02799+ addi 4,4,0x10+ vxor 7,0,19+ bne .Lxts_dec6x_steal+ b .Lxts_dec6x_done++.align 4+.Lxts_dec6x_zero:+ cmpwi 31,0+ beq .Lxts_dec6x_done++ .long 0x7C005699+ vperm 0,0,0,6+ vxor 7,0,18+.Lxts_dec6x_steal:+ .long 0x10E7C548+ lvx 24,26,7+ addi 7,7,0x20++ .long 0x10E7CD48+ lvx 25,3,7+ bdnz .Lxts_dec6x_steal++ add 10,10,31+ .long 0x10E7C548++ cmpwi 31,0+ .long 0x10E7CD48++ .long 0x7C005699+ .long 0x10E7D548++ lvsr 5,0,31+ .long 0x10E7DD48++ addi 7,1,64+15+ .long 0x10E7E548+ lvx 24,0,7++ .long 0x10E7ED48+ lvx 25,3,7+ vxor 18,18,31++ vperm 0,0,0,6+ .long 0x10E7F548++ vperm 0,0,0,5+ .long 0x11679549++ vperm 7,11,11,6+ .long 0x7CE02799+++ vxor 7,7,7+ vspltisb 12,-1+ vperm 7,7,12,5+ vsel 7,0,11,7+ vxor 7,7,17++ subi 30,4,1+ mtctr 31+.Loop_xts_dec6x_steal:+ lbzu 0,1(30)+ stb 0,16(30)+ bdnz .Loop_xts_dec6x_steal++ li 31,0+ mtctr 9+ b .Loop_xts_dec1x++.align 4+.Lxts_dec6x_done:+ cmpldi 8,0+ beq .Lxts_dec6x_ret++ vxor 8,17,23+ vperm 8,8,8,6+ .long 0x7D004799++.Lxts_dec6x_ret:+ mtlr 11+ li 10,79+ li 11,95+ stvx 9,10,1+ addi 10,10,32+ stvx 9,11,1+ addi 11,11,32+ stvx 9,10,1+ addi 10,10,32+ stvx 9,11,1+ addi 11,11,32+ stvx 9,10,1+ addi 10,10,32+ stvx 9,11,1+ addi 11,11,32+ stvx 9,10,1+ addi 10,10,32+ stvx 9,11,1+ addi 11,11,32++ or 12,12,12+ lvx 20,10,1+ addi 10,10,32+ lvx 21,11,1+ addi 11,11,32+ lvx 22,10,1+ addi 10,10,32+ lvx 23,11,1+ addi 11,11,32+ lvx 24,10,1+ addi 10,10,32+ lvx 25,11,1+ addi 11,11,32+ lvx 26,10,1+ addi 10,10,32+ lvx 27,11,1+ addi 11,11,32+ lvx 28,10,1+ addi 10,10,32+ lvx 29,11,1+ addi 11,11,32+ lvx 30,10,1+ lvx 31,11,1+ ld 26,400(1)+ ld 27,408(1)+ ld 28,416(1)+ ld 29,424(1)+ ld 30,432(1)+ ld 31,440(1)+ addi 1,1,448+ blr +.long 0+.byte 0,12,0x04,1,0x80,6,6,0+.long 0++.align 5+_aesp8_xts_dec5x:+ .long 0x10E7C548+ .long 0x118CC548+ .long 0x11ADC548+ .long 0x11CEC548+ .long 0x11EFC548+ lvx 24,26,7+ addi 7,7,0x20++ .long 0x10E7CD48+ .long 0x118CCD48+ .long 0x11ADCD48+ .long 0x11CECD48+ .long 0x11EFCD48+ lvx 25,3,7+ bdnz _aesp8_xts_dec5x++ subi 0,31,1+ .long 0x10E7C548+ .long 0x118CC548+ .long 0x11ADC548+ .long 0x11CEC548+ .long 0x11EFC548++ andi. 0,0,16+ cmpwi 31,0+ .long 0x10E7CD48+ .long 0x118CCD48+ .long 0x11ADCD48+ .long 0x11CECD48+ .long 0x11EFCD48+ vxor 17,17,31++ sub 10,10,0+ .long 0x10E7D548+ .long 0x118CD548+ .long 0x11ADD548+ .long 0x11CED548+ .long 0x11EFD548+ vxor 1,18,31++ .long 0x10E7DD48+ .long 0x7C005699+ .long 0x118CDD48+ .long 0x11ADDD48+ .long 0x11CEDD48+ .long 0x11EFDD48+ vxor 2,19,31++ addi 7,1,64+15+ .long 0x10E7E548+ .long 0x118CE548+ .long 0x11ADE548+ .long 0x11CEE548+ .long 0x11EFE548+ lvx 24,0,7+ vxor 3,20,31++ .long 0x10E7ED48+ vperm 0,0,0,6+ .long 0x118CED48+ .long 0x11ADED48+ .long 0x11CEED48+ .long 0x11EFED48+ lvx 25,3,7+ vxor 4,21,31++ .long 0x10E7F548+ .long 0x118CF548+ .long 0x11ADF548+ .long 0x11CEF548+ .long 0x11EFF548++ .long 0x10E78D49+ .long 0x118C0D49+ .long 0x11AD1549+ .long 0x11CE1D49+ .long 0x11EF2549+ mtctr 9+ blr +.long 0+.byte 0,12,0x14,0,0,0,0,0
+ cbits/asm/aesp8-ppc.pl view
@@ -0,0 +1,3801 @@+#!/usr/bin/env perl+#+# ====================================================================+# Written by Andy Polyakov <appro@openssl.org> for the OpenSSL+# project. The module is, however, dual licensed under OpenSSL and+# CRYPTOGAMS licenses depending on where you obtain it. For further+# details see http://www.openssl.org/~appro/cryptogams/.+# ====================================================================+#+# This module implements support for AES instructions as per PowerISA+# specification version 2.07, first implemented by POWER8 processor.+# The module is endian-agnostic in sense that it supports both big-+# and little-endian cases. Data alignment in parallelizable modes is+# handled with VSX loads and stores, which implies MSR.VSX flag being+# set. It should also be noted that ISA specification doesn't prohibit+# alignment exceptions for these instructions on page boundaries.+# Initially alignment was handled in pure AltiVec/VMX way [when data+# is aligned programmatically, which in turn guarantees exception-+# free execution], but it turned to hamper performance when vcipher+# instructions are interleaved. It's reckoned that eventual+# misalignment penalties at page boundaries are in average lower+# than additional overhead in pure AltiVec approach.+#+# May 2016+#+# Add XTS subroutine, 9x on little- and 12x improvement on big-endian+# systems were measured.+#+######################################################################+# Current large-block performance in cycles per byte processed with+# 128-bit key (less is better).+#+# CBC en-/decrypt CTR XTS+# POWER8[le] 3.96/0.72 0.74 1.1+# POWER8[be] 3.75/0.65 0.66 1.0+# POWER9[le] 4.02/0.86 0.84 1.05+# POWER9[be] 3.99/0.78 0.79 0.97+++$flavour = shift;++if ($flavour =~ /64/) {+ $SIZE_T =8;+ $LRSAVE =2*$SIZE_T;+ $STU ="stdu";+ $POP ="ld";+ $PUSH ="std";+ $UCMP ="cmpld";+ $SHL ="sldi";+} elsif ($flavour =~ /32/) {+ $SIZE_T =4;+ $LRSAVE =$SIZE_T;+ $STU ="stwu";+ $POP ="lwz";+ $PUSH ="stw";+ $UCMP ="cmplw";+ $SHL ="slwi";+} else { die "nonsense $flavour"; }++$LITTLE_ENDIAN = ($flavour=~/le$/) ? $SIZE_T : 0;++$0 =~ m/(.*[\/\\])[^\/\\]+$/; $dir=$1;+( $xlate="${dir}ppc-xlate.pl" and -f $xlate ) or+( $xlate="${dir}../../perlasm/ppc-xlate.pl" and -f $xlate) or+die "can't locate ppc-xlate.pl";++open STDOUT,"| $^X $xlate $flavour ".shift || die "can't call $xlate: $!";++$FRAME=8*$SIZE_T;+$prefix="aes_p8";++$sp="r1";+$vrsave="r12";++#########################################################################+{{{ # Key setup procedures #+my ($inp,$bits,$out,$ptr,$cnt,$rounds)=map("r$_",(3..8));+my ($zero,$in0,$in1,$key,$rcon,$mask,$tmp)=map("v$_",(0..6));+my ($stage,$outperm,$outmask,$outhead,$outtail)=map("v$_",(7..11));++$code.=<<___;+.machine "any"++.text++.align 7+rcon:+.long 0x01000000, 0x01000000, 0x01000000, 0x01000000 ?rev+.long 0x1b000000, 0x1b000000, 0x1b000000, 0x1b000000 ?rev+.long 0x0d0e0f0c, 0x0d0e0f0c, 0x0d0e0f0c, 0x0d0e0f0c ?rev+.long 0,0,0,0 ?asis+Lconsts:+ mflr r0+ bcl 20,31,\$+4+ mflr $ptr #vvvvv "distance between . and rcon+ addi $ptr,$ptr,-0x48+ mtlr r0+ blr+ .long 0+ .byte 0,12,0x14,0,0,0,0,0+.asciz "AES for PowerISA 2.07, CRYPTOGAMS by <appro\@openssl.org>"++.globl .${prefix}_set_encrypt_key+.align 5+.${prefix}_set_encrypt_key:+Lset_encrypt_key:+ mflr r11+ $PUSH r11,$LRSAVE($sp)++ li $ptr,-1+ ${UCMP}i $inp,0+ beq- Lenc_key_abort # if ($inp==0) return -1;+ ${UCMP}i $out,0+ beq- Lenc_key_abort # if ($out==0) return -1;+ li $ptr,-2+ cmpwi $bits,128+ blt- Lenc_key_abort+ cmpwi $bits,256+ bgt- Lenc_key_abort+ andi. r0,$bits,0x3f+ bne- Lenc_key_abort++ lis r0,0xfff0+ mfspr $vrsave,256+ mtspr 256,r0++ bl Lconsts+ mtlr r11++ neg r9,$inp+ lvx $in0,0,$inp+ addi $inp,$inp,15 # 15 is not typo+ lvsr $key,0,r9 # borrow $key+ li r8,0x20+ cmpwi $bits,192+ lvx $in1,0,$inp+ le?vspltisb $mask,0x0f # borrow $mask+ lvx $rcon,0,$ptr+ le?vxor $key,$key,$mask # adjust for byte swap+ lvx $mask,r8,$ptr+ addi $ptr,$ptr,0x10+ vperm $in0,$in0,$in1,$key # align [and byte swap in LE]+ li $cnt,8+ vxor $zero,$zero,$zero+ mtctr $cnt++ ?lvsr $outperm,0,$out+ vspltisb $outmask,-1+ lvx $outhead,0,$out+ ?vperm $outmask,$zero,$outmask,$outperm++ blt Loop128+ addi $inp,$inp,8+ beq L192+ addi $inp,$inp,8+ b L256++.align 4+Loop128:+ vperm $key,$in0,$in0,$mask # rotate-n-splat+ vsldoi $tmp,$zero,$in0,12 # >>32+ vperm $outtail,$in0,$in0,$outperm # rotate+ vsel $stage,$outhead,$outtail,$outmask+ vmr $outhead,$outtail+ vcipherlast $key,$key,$rcon+ stvx $stage,0,$out+ addi $out,$out,16++ vxor $in0,$in0,$tmp+ vsldoi $tmp,$zero,$tmp,12 # >>32+ vxor $in0,$in0,$tmp+ vsldoi $tmp,$zero,$tmp,12 # >>32+ vxor $in0,$in0,$tmp+ vadduwm $rcon,$rcon,$rcon+ vxor $in0,$in0,$key+ bdnz Loop128++ lvx $rcon,0,$ptr # last two round keys++ vperm $key,$in0,$in0,$mask # rotate-n-splat+ vsldoi $tmp,$zero,$in0,12 # >>32+ vperm $outtail,$in0,$in0,$outperm # rotate+ vsel $stage,$outhead,$outtail,$outmask+ vmr $outhead,$outtail+ vcipherlast $key,$key,$rcon+ stvx $stage,0,$out+ addi $out,$out,16++ vxor $in0,$in0,$tmp+ vsldoi $tmp,$zero,$tmp,12 # >>32+ vxor $in0,$in0,$tmp+ vsldoi $tmp,$zero,$tmp,12 # >>32+ vxor $in0,$in0,$tmp+ vadduwm $rcon,$rcon,$rcon+ vxor $in0,$in0,$key++ vperm $key,$in0,$in0,$mask # rotate-n-splat+ vsldoi $tmp,$zero,$in0,12 # >>32+ vperm $outtail,$in0,$in0,$outperm # rotate+ vsel $stage,$outhead,$outtail,$outmask+ vmr $outhead,$outtail+ vcipherlast $key,$key,$rcon+ stvx $stage,0,$out+ addi $out,$out,16++ vxor $in0,$in0,$tmp+ vsldoi $tmp,$zero,$tmp,12 # >>32+ vxor $in0,$in0,$tmp+ vsldoi $tmp,$zero,$tmp,12 # >>32+ vxor $in0,$in0,$tmp+ vxor $in0,$in0,$key+ vperm $outtail,$in0,$in0,$outperm # rotate+ vsel $stage,$outhead,$outtail,$outmask+ vmr $outhead,$outtail+ stvx $stage,0,$out++ addi $inp,$out,15 # 15 is not typo+ addi $out,$out,0x50++ li $rounds,10+ b Ldone++.align 4+L192:+ lvx $tmp,0,$inp+ li $cnt,4+ vperm $outtail,$in0,$in0,$outperm # rotate+ vsel $stage,$outhead,$outtail,$outmask+ vmr $outhead,$outtail+ stvx $stage,0,$out+ addi $out,$out,16+ vperm $in1,$in1,$tmp,$key # align [and byte swap in LE]+ vspltisb $key,8 # borrow $key+ mtctr $cnt+ vsububm $mask,$mask,$key # adjust the mask++Loop192:+ vperm $key,$in1,$in1,$mask # roate-n-splat+ vsldoi $tmp,$zero,$in0,12 # >>32+ vcipherlast $key,$key,$rcon++ vxor $in0,$in0,$tmp+ vsldoi $tmp,$zero,$tmp,12 # >>32+ vxor $in0,$in0,$tmp+ vsldoi $tmp,$zero,$tmp,12 # >>32+ vxor $in0,$in0,$tmp++ vsldoi $stage,$zero,$in1,8+ vspltw $tmp,$in0,3+ vxor $tmp,$tmp,$in1+ vsldoi $in1,$zero,$in1,12 # >>32+ vadduwm $rcon,$rcon,$rcon+ vxor $in1,$in1,$tmp+ vxor $in0,$in0,$key+ vxor $in1,$in1,$key+ vsldoi $stage,$stage,$in0,8++ vperm $key,$in1,$in1,$mask # rotate-n-splat+ vsldoi $tmp,$zero,$in0,12 # >>32+ vperm $outtail,$stage,$stage,$outperm # rotate+ vsel $stage,$outhead,$outtail,$outmask+ vmr $outhead,$outtail+ vcipherlast $key,$key,$rcon+ stvx $stage,0,$out+ addi $out,$out,16++ vsldoi $stage,$in0,$in1,8+ vxor $in0,$in0,$tmp+ vsldoi $tmp,$zero,$tmp,12 # >>32+ vperm $outtail,$stage,$stage,$outperm # rotate+ vsel $stage,$outhead,$outtail,$outmask+ vmr $outhead,$outtail+ vxor $in0,$in0,$tmp+ vsldoi $tmp,$zero,$tmp,12 # >>32+ vxor $in0,$in0,$tmp+ stvx $stage,0,$out+ addi $out,$out,16++ vspltw $tmp,$in0,3+ vxor $tmp,$tmp,$in1+ vsldoi $in1,$zero,$in1,12 # >>32+ vadduwm $rcon,$rcon,$rcon+ vxor $in1,$in1,$tmp+ vxor $in0,$in0,$key+ vxor $in1,$in1,$key+ vperm $outtail,$in0,$in0,$outperm # rotate+ vsel $stage,$outhead,$outtail,$outmask+ vmr $outhead,$outtail+ stvx $stage,0,$out+ addi $inp,$out,15 # 15 is not typo+ addi $out,$out,16+ bdnz Loop192++ li $rounds,12+ addi $out,$out,0x20+ b Ldone++.align 4+L256:+ lvx $tmp,0,$inp+ li $cnt,7+ li $rounds,14+ vperm $outtail,$in0,$in0,$outperm # rotate+ vsel $stage,$outhead,$outtail,$outmask+ vmr $outhead,$outtail+ stvx $stage,0,$out+ addi $out,$out,16+ vperm $in1,$in1,$tmp,$key # align [and byte swap in LE]+ mtctr $cnt++Loop256:+ vperm $key,$in1,$in1,$mask # rotate-n-splat+ vsldoi $tmp,$zero,$in0,12 # >>32+ vperm $outtail,$in1,$in1,$outperm # rotate+ vsel $stage,$outhead,$outtail,$outmask+ vmr $outhead,$outtail+ vcipherlast $key,$key,$rcon+ stvx $stage,0,$out+ addi $out,$out,16++ vxor $in0,$in0,$tmp+ vsldoi $tmp,$zero,$tmp,12 # >>32+ vxor $in0,$in0,$tmp+ vsldoi $tmp,$zero,$tmp,12 # >>32+ vxor $in0,$in0,$tmp+ vadduwm $rcon,$rcon,$rcon+ vxor $in0,$in0,$key+ vperm $outtail,$in0,$in0,$outperm # rotate+ vsel $stage,$outhead,$outtail,$outmask+ vmr $outhead,$outtail+ stvx $stage,0,$out+ addi $inp,$out,15 # 15 is not typo+ addi $out,$out,16+ bdz Ldone++ vspltw $key,$in0,3 # just splat+ vsldoi $tmp,$zero,$in1,12 # >>32+ vsbox $key,$key++ vxor $in1,$in1,$tmp+ vsldoi $tmp,$zero,$tmp,12 # >>32+ vxor $in1,$in1,$tmp+ vsldoi $tmp,$zero,$tmp,12 # >>32+ vxor $in1,$in1,$tmp++ vxor $in1,$in1,$key+ b Loop256++.align 4+Ldone:+ lvx $in1,0,$inp # redundant in aligned case+ vsel $in1,$outhead,$in1,$outmask+ stvx $in1,0,$inp+ li $ptr,0+ mtspr 256,$vrsave+ stw $rounds,0($out)++Lenc_key_abort:+ mr r3,$ptr+ blr+ .long 0+ .byte 0,12,0x14,1,0,0,3,0+ .long 0+.size .${prefix}_set_encrypt_key,.-.${prefix}_set_encrypt_key++.globl .${prefix}_set_decrypt_key+.align 5+.${prefix}_set_decrypt_key:+ $STU $sp,-$FRAME($sp)+ mflr r10+ $PUSH r10,$FRAME+$LRSAVE($sp)+ bl Lset_encrypt_key+ mtlr r10++ cmpwi r3,0+ bne- Ldec_key_abort++ slwi $cnt,$rounds,4+ subi $inp,$out,240 # first round key+ srwi $rounds,$rounds,1+ add $out,$inp,$cnt # last round key+ mtctr $rounds++Ldeckey:+ lwz r0, 0($inp)+ lwz r6, 4($inp)+ lwz r7, 8($inp)+ lwz r8, 12($inp)+ addi $inp,$inp,16+ lwz r9, 0($out)+ lwz r10,4($out)+ lwz r11,8($out)+ lwz r12,12($out)+ stw r0, 0($out)+ stw r6, 4($out)+ stw r7, 8($out)+ stw r8, 12($out)+ subi $out,$out,16+ stw r9, -16($inp)+ stw r10,-12($inp)+ stw r11,-8($inp)+ stw r12,-4($inp)+ bdnz Ldeckey++ xor r3,r3,r3 # return value+Ldec_key_abort:+ addi $sp,$sp,$FRAME+ blr+ .long 0+ .byte 0,12,4,1,0x80,0,3,0+ .long 0+.size .${prefix}_set_decrypt_key,.-.${prefix}_set_decrypt_key+___+}}}+#########################################################################+{{{ # Single block en- and decrypt procedures #+sub gen_block () {+my $dir = shift;+my $n = $dir eq "de" ? "n" : "";+my ($inp,$out,$key,$rounds,$idx)=map("r$_",(3..7));++$code.=<<___;+.globl .${prefix}_${dir}crypt+.align 5+.${prefix}_${dir}crypt:+ lwz $rounds,240($key)+ lis r0,0xfc00+ mfspr $vrsave,256+ li $idx,15 # 15 is not typo+ mtspr 256,r0++ lvx v0,0,$inp+ neg r11,$out+ lvx v1,$idx,$inp+ lvsl v2,0,$inp # inpperm+ le?vspltisb v4,0x0f+ ?lvsl v3,0,r11 # outperm+ le?vxor v2,v2,v4+ li $idx,16+ vperm v0,v0,v1,v2 # align [and byte swap in LE]+ lvx v1,0,$key+ ?lvsl v5,0,$key # keyperm+ srwi $rounds,$rounds,1+ lvx v2,$idx,$key+ addi $idx,$idx,16+ subi $rounds,$rounds,1+ ?vperm v1,v1,v2,v5 # align round key++ vxor v0,v0,v1+ lvx v1,$idx,$key+ addi $idx,$idx,16+ mtctr $rounds++Loop_${dir}c:+ ?vperm v2,v2,v1,v5+ v${n}cipher v0,v0,v2+ lvx v2,$idx,$key+ addi $idx,$idx,16+ ?vperm v1,v1,v2,v5+ v${n}cipher v0,v0,v1+ lvx v1,$idx,$key+ addi $idx,$idx,16+ bdnz Loop_${dir}c++ ?vperm v2,v2,v1,v5+ v${n}cipher v0,v0,v2+ lvx v2,$idx,$key+ ?vperm v1,v1,v2,v5+ v${n}cipherlast v0,v0,v1++ vspltisb v2,-1+ vxor v1,v1,v1+ li $idx,15 # 15 is not typo+ ?vperm v2,v1,v2,v3 # outmask+ le?vxor v3,v3,v4+ lvx v1,0,$out # outhead+ vperm v0,v0,v0,v3 # rotate [and byte swap in LE]+ vsel v1,v1,v0,v2+ lvx v4,$idx,$out+ stvx v1,0,$out+ vsel v0,v0,v4,v2+ stvx v0,$idx,$out++ mtspr 256,$vrsave+ blr+ .long 0+ .byte 0,12,0x14,0,0,0,3,0+ .long 0+.size .${prefix}_${dir}crypt,.-.${prefix}_${dir}crypt+___+}+&gen_block("en");+&gen_block("de");+}}}+#########################################################################+{{{ # CBC en- and decrypt procedures #+my ($inp,$out,$len,$key,$ivp,$enc,$rounds,$idx)=map("r$_",(3..10));+my ($rndkey0,$rndkey1,$inout,$tmp)= map("v$_",(0..3));+my ($ivec,$inptail,$inpperm,$outhead,$outperm,$outmask,$keyperm)=+ map("v$_",(4..10));+$code.=<<___;+.globl .${prefix}_cbc_encrypt+.align 5+.${prefix}_cbc_encrypt:+ ${UCMP}i $len,16+ bltlr-++ cmpwi $enc,0 # test direction+ lis r0,0xffe0+ mfspr $vrsave,256+ mtspr 256,r0++ li $idx,15+ vxor $rndkey0,$rndkey0,$rndkey0+ le?vspltisb $tmp,0x0f++ lvx $ivec,0,$ivp # load [unaligned] iv+ lvsl $inpperm,0,$ivp+ lvx $inptail,$idx,$ivp+ le?vxor $inpperm,$inpperm,$tmp+ vperm $ivec,$ivec,$inptail,$inpperm++ neg r11,$inp+ ?lvsl $keyperm,0,$key # prepare for unaligned key+ lwz $rounds,240($key)++ lvsr $inpperm,0,r11 # prepare for unaligned load+ lvx $inptail,0,$inp+ addi $inp,$inp,15 # 15 is not typo+ le?vxor $inpperm,$inpperm,$tmp++ ?lvsr $outperm,0,$out # prepare for unaligned store+ vspltisb $outmask,-1+ lvx $outhead,0,$out+ ?vperm $outmask,$rndkey0,$outmask,$outperm+ le?vxor $outperm,$outperm,$tmp++ srwi $rounds,$rounds,1+ li $idx,16+ subi $rounds,$rounds,1+ beq Lcbc_dec++Lcbc_enc:+ vmr $inout,$inptail+ lvx $inptail,0,$inp+ addi $inp,$inp,16+ mtctr $rounds+ subi $len,$len,16 # len-=16++ lvx $rndkey0,0,$key+ vperm $inout,$inout,$inptail,$inpperm+ lvx $rndkey1,$idx,$key+ addi $idx,$idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vxor $inout,$inout,$rndkey0+ lvx $rndkey0,$idx,$key+ addi $idx,$idx,16+ vxor $inout,$inout,$ivec++Loop_cbc_enc:+ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vcipher $inout,$inout,$rndkey1+ lvx $rndkey1,$idx,$key+ addi $idx,$idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vcipher $inout,$inout,$rndkey0+ lvx $rndkey0,$idx,$key+ addi $idx,$idx,16+ bdnz Loop_cbc_enc++ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vcipher $inout,$inout,$rndkey1+ lvx $rndkey1,$idx,$key+ li $idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vcipherlast $ivec,$inout,$rndkey0+ ${UCMP}i $len,16++ vperm $tmp,$ivec,$ivec,$outperm+ vsel $inout,$outhead,$tmp,$outmask+ vmr $outhead,$tmp+ stvx $inout,0,$out+ addi $out,$out,16+ bge Lcbc_enc++ b Lcbc_done++.align 4+Lcbc_dec:+ ${UCMP}i $len,128+ bge _aesp8_cbc_decrypt8x+ vmr $tmp,$inptail+ lvx $inptail,0,$inp+ addi $inp,$inp,16+ mtctr $rounds+ subi $len,$len,16 # len-=16++ lvx $rndkey0,0,$key+ vperm $tmp,$tmp,$inptail,$inpperm+ lvx $rndkey1,$idx,$key+ addi $idx,$idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vxor $inout,$tmp,$rndkey0+ lvx $rndkey0,$idx,$key+ addi $idx,$idx,16++Loop_cbc_dec:+ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vncipher $inout,$inout,$rndkey1+ lvx $rndkey1,$idx,$key+ addi $idx,$idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vncipher $inout,$inout,$rndkey0+ lvx $rndkey0,$idx,$key+ addi $idx,$idx,16+ bdnz Loop_cbc_dec++ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vncipher $inout,$inout,$rndkey1+ lvx $rndkey1,$idx,$key+ li $idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vncipherlast $inout,$inout,$rndkey0+ ${UCMP}i $len,16++ vxor $inout,$inout,$ivec+ vmr $ivec,$tmp+ vperm $tmp,$inout,$inout,$outperm+ vsel $inout,$outhead,$tmp,$outmask+ vmr $outhead,$tmp+ stvx $inout,0,$out+ addi $out,$out,16+ bge Lcbc_dec++Lcbc_done:+ addi $out,$out,-1+ lvx $inout,0,$out # redundant in aligned case+ vsel $inout,$outhead,$inout,$outmask+ stvx $inout,0,$out++ neg $enc,$ivp # write [unaligned] iv+ li $idx,15 # 15 is not typo+ vxor $rndkey0,$rndkey0,$rndkey0+ vspltisb $outmask,-1+ le?vspltisb $tmp,0x0f+ ?lvsl $outperm,0,$enc+ ?vperm $outmask,$rndkey0,$outmask,$outperm+ le?vxor $outperm,$outperm,$tmp+ lvx $outhead,0,$ivp+ vperm $ivec,$ivec,$ivec,$outperm+ vsel $inout,$outhead,$ivec,$outmask+ lvx $inptail,$idx,$ivp+ stvx $inout,0,$ivp+ vsel $inout,$ivec,$inptail,$outmask+ stvx $inout,$idx,$ivp++ mtspr 256,$vrsave+ blr+ .long 0+ .byte 0,12,0x14,0,0,0,6,0+ .long 0+___+#########################################################################+{{ # Optimized CBC decrypt procedure #+my $key_="r11";+my ($x00,$x10,$x20,$x30,$x40,$x50,$x60,$x70)=map("r$_",(0,8,26..31));+ $x00=0 if ($flavour =~ /osx/);+my ($in0, $in1, $in2, $in3, $in4, $in5, $in6, $in7 )=map("v$_",(0..3,10..13));+my ($out0,$out1,$out2,$out3,$out4,$out5,$out6,$out7)=map("v$_",(14..21));+my $rndkey0="v23"; # v24-v25 rotating buffer for first found keys+ # v26-v31 last 6 round keys+my ($tmp,$keyperm)=($in3,$in4); # aliases with "caller", redundant assignment++$code.=<<___;+.align 5+_aesp8_cbc_decrypt8x:+ $STU $sp,-`($FRAME+21*16+6*$SIZE_T)`($sp)+ li r10,`$FRAME+8*16+15`+ li r11,`$FRAME+8*16+31`+ stvx v20,r10,$sp # ABI says so+ addi r10,r10,32+ stvx v21,r11,$sp+ addi r11,r11,32+ stvx v22,r10,$sp+ addi r10,r10,32+ stvx v23,r11,$sp+ addi r11,r11,32+ stvx v24,r10,$sp+ addi r10,r10,32+ stvx v25,r11,$sp+ addi r11,r11,32+ stvx v26,r10,$sp+ addi r10,r10,32+ stvx v27,r11,$sp+ addi r11,r11,32+ stvx v28,r10,$sp+ addi r10,r10,32+ stvx v29,r11,$sp+ addi r11,r11,32+ stvx v30,r10,$sp+ stvx v31,r11,$sp+ li r0,-1+ stw $vrsave,`$FRAME+21*16-4`($sp) # save vrsave+ li $x10,0x10+ $PUSH r26,`$FRAME+21*16+0*$SIZE_T`($sp)+ li $x20,0x20+ $PUSH r27,`$FRAME+21*16+1*$SIZE_T`($sp)+ li $x30,0x30+ $PUSH r28,`$FRAME+21*16+2*$SIZE_T`($sp)+ li $x40,0x40+ $PUSH r29,`$FRAME+21*16+3*$SIZE_T`($sp)+ li $x50,0x50+ $PUSH r30,`$FRAME+21*16+4*$SIZE_T`($sp)+ li $x60,0x60+ $PUSH r31,`$FRAME+21*16+5*$SIZE_T`($sp)+ li $x70,0x70+ mtspr 256,r0++ subi $rounds,$rounds,3 # -4 in total+ subi $len,$len,128 # bias++ lvx $rndkey0,$x00,$key # load key schedule+ lvx v30,$x10,$key+ addi $key,$key,0x20+ lvx v31,$x00,$key+ ?vperm $rndkey0,$rndkey0,v30,$keyperm+ addi $key_,$sp,$FRAME+15+ mtctr $rounds++Load_cbc_dec_key:+ ?vperm v24,v30,v31,$keyperm+ lvx v30,$x10,$key+ addi $key,$key,0x20+ stvx v24,$x00,$key_ # off-load round[1]+ ?vperm v25,v31,v30,$keyperm+ lvx v31,$x00,$key+ stvx v25,$x10,$key_ # off-load round[2]+ addi $key_,$key_,0x20+ bdnz Load_cbc_dec_key++ lvx v26,$x10,$key+ ?vperm v24,v30,v31,$keyperm+ lvx v27,$x20,$key+ stvx v24,$x00,$key_ # off-load round[3]+ ?vperm v25,v31,v26,$keyperm+ lvx v28,$x30,$key+ stvx v25,$x10,$key_ # off-load round[4]+ addi $key_,$sp,$FRAME+15 # rewind $key_+ ?vperm v26,v26,v27,$keyperm+ lvx v29,$x40,$key+ ?vperm v27,v27,v28,$keyperm+ lvx v30,$x50,$key+ ?vperm v28,v28,v29,$keyperm+ lvx v31,$x60,$key+ ?vperm v29,v29,v30,$keyperm+ lvx $out0,$x70,$key # borrow $out0+ ?vperm v30,v30,v31,$keyperm+ lvx v24,$x00,$key_ # pre-load round[1]+ ?vperm v31,v31,$out0,$keyperm+ lvx v25,$x10,$key_ # pre-load round[2]++ #lvx $inptail,0,$inp # "caller" already did this+ #addi $inp,$inp,15 # 15 is not typo+ subi $inp,$inp,15 # undo "caller"++ le?li $idx,8+ lvx_u $in0,$x00,$inp # load first 8 "words"+ le?lvsl $inpperm,0,$idx+ le?vspltisb $tmp,0x0f+ lvx_u $in1,$x10,$inp+ le?vxor $inpperm,$inpperm,$tmp # transform for lvx_u/stvx_u+ lvx_u $in2,$x20,$inp+ le?vperm $in0,$in0,$in0,$inpperm+ lvx_u $in3,$x30,$inp+ le?vperm $in1,$in1,$in1,$inpperm+ lvx_u $in4,$x40,$inp+ le?vperm $in2,$in2,$in2,$inpperm+ vxor $out0,$in0,$rndkey0+ lvx_u $in5,$x50,$inp+ le?vperm $in3,$in3,$in3,$inpperm+ vxor $out1,$in1,$rndkey0+ lvx_u $in6,$x60,$inp+ le?vperm $in4,$in4,$in4,$inpperm+ vxor $out2,$in2,$rndkey0+ lvx_u $in7,$x70,$inp+ addi $inp,$inp,0x80+ le?vperm $in5,$in5,$in5,$inpperm+ vxor $out3,$in3,$rndkey0+ le?vperm $in6,$in6,$in6,$inpperm+ vxor $out4,$in4,$rndkey0+ le?vperm $in7,$in7,$in7,$inpperm+ vxor $out5,$in5,$rndkey0+ vxor $out6,$in6,$rndkey0+ vxor $out7,$in7,$rndkey0++ mtctr $rounds+ b Loop_cbc_dec8x+.align 5+Loop_cbc_dec8x:+ vncipher $out0,$out0,v24+ vncipher $out1,$out1,v24+ vncipher $out2,$out2,v24+ vncipher $out3,$out3,v24+ vncipher $out4,$out4,v24+ vncipher $out5,$out5,v24+ vncipher $out6,$out6,v24+ vncipher $out7,$out7,v24+ lvx v24,$x20,$key_ # round[3]+ addi $key_,$key_,0x20++ vncipher $out0,$out0,v25+ vncipher $out1,$out1,v25+ vncipher $out2,$out2,v25+ vncipher $out3,$out3,v25+ vncipher $out4,$out4,v25+ vncipher $out5,$out5,v25+ vncipher $out6,$out6,v25+ vncipher $out7,$out7,v25+ lvx v25,$x10,$key_ # round[4]+ bdnz Loop_cbc_dec8x++ subic $len,$len,128 # $len-=128+ vncipher $out0,$out0,v24+ vncipher $out1,$out1,v24+ vncipher $out2,$out2,v24+ vncipher $out3,$out3,v24+ vncipher $out4,$out4,v24+ vncipher $out5,$out5,v24+ vncipher $out6,$out6,v24+ vncipher $out7,$out7,v24++ subfe. r0,r0,r0 # borrow?-1:0+ vncipher $out0,$out0,v25+ vncipher $out1,$out1,v25+ vncipher $out2,$out2,v25+ vncipher $out3,$out3,v25+ vncipher $out4,$out4,v25+ vncipher $out5,$out5,v25+ vncipher $out6,$out6,v25+ vncipher $out7,$out7,v25++ and r0,r0,$len+ vncipher $out0,$out0,v26+ vncipher $out1,$out1,v26+ vncipher $out2,$out2,v26+ vncipher $out3,$out3,v26+ vncipher $out4,$out4,v26+ vncipher $out5,$out5,v26+ vncipher $out6,$out6,v26+ vncipher $out7,$out7,v26++ add $inp,$inp,r0 # $inp is adjusted in such+ # way that at exit from the+ # loop inX-in7 are loaded+ # with last "words"+ vncipher $out0,$out0,v27+ vncipher $out1,$out1,v27+ vncipher $out2,$out2,v27+ vncipher $out3,$out3,v27+ vncipher $out4,$out4,v27+ vncipher $out5,$out5,v27+ vncipher $out6,$out6,v27+ vncipher $out7,$out7,v27++ addi $key_,$sp,$FRAME+15 # rewind $key_+ vncipher $out0,$out0,v28+ vncipher $out1,$out1,v28+ vncipher $out2,$out2,v28+ vncipher $out3,$out3,v28+ vncipher $out4,$out4,v28+ vncipher $out5,$out5,v28+ vncipher $out6,$out6,v28+ vncipher $out7,$out7,v28+ lvx v24,$x00,$key_ # re-pre-load round[1]++ vncipher $out0,$out0,v29+ vncipher $out1,$out1,v29+ vncipher $out2,$out2,v29+ vncipher $out3,$out3,v29+ vncipher $out4,$out4,v29+ vncipher $out5,$out5,v29+ vncipher $out6,$out6,v29+ vncipher $out7,$out7,v29+ lvx v25,$x10,$key_ # re-pre-load round[2]++ vncipher $out0,$out0,v30+ vxor $ivec,$ivec,v31 # xor with last round key+ vncipher $out1,$out1,v30+ vxor $in0,$in0,v31+ vncipher $out2,$out2,v30+ vxor $in1,$in1,v31+ vncipher $out3,$out3,v30+ vxor $in2,$in2,v31+ vncipher $out4,$out4,v30+ vxor $in3,$in3,v31+ vncipher $out5,$out5,v30+ vxor $in4,$in4,v31+ vncipher $out6,$out6,v30+ vxor $in5,$in5,v31+ vncipher $out7,$out7,v30+ vxor $in6,$in6,v31++ vncipherlast $out0,$out0,$ivec+ vncipherlast $out1,$out1,$in0+ lvx_u $in0,$x00,$inp # load next input block+ vncipherlast $out2,$out2,$in1+ lvx_u $in1,$x10,$inp+ vncipherlast $out3,$out3,$in2+ le?vperm $in0,$in0,$in0,$inpperm+ lvx_u $in2,$x20,$inp+ vncipherlast $out4,$out4,$in3+ le?vperm $in1,$in1,$in1,$inpperm+ lvx_u $in3,$x30,$inp+ vncipherlast $out5,$out5,$in4+ le?vperm $in2,$in2,$in2,$inpperm+ lvx_u $in4,$x40,$inp+ vncipherlast $out6,$out6,$in5+ le?vperm $in3,$in3,$in3,$inpperm+ lvx_u $in5,$x50,$inp+ vncipherlast $out7,$out7,$in6+ le?vperm $in4,$in4,$in4,$inpperm+ lvx_u $in6,$x60,$inp+ vmr $ivec,$in7+ le?vperm $in5,$in5,$in5,$inpperm+ lvx_u $in7,$x70,$inp+ addi $inp,$inp,0x80++ le?vperm $out0,$out0,$out0,$inpperm+ le?vperm $out1,$out1,$out1,$inpperm+ stvx_u $out0,$x00,$out+ le?vperm $in6,$in6,$in6,$inpperm+ vxor $out0,$in0,$rndkey0+ le?vperm $out2,$out2,$out2,$inpperm+ stvx_u $out1,$x10,$out+ le?vperm $in7,$in7,$in7,$inpperm+ vxor $out1,$in1,$rndkey0+ le?vperm $out3,$out3,$out3,$inpperm+ stvx_u $out2,$x20,$out+ vxor $out2,$in2,$rndkey0+ le?vperm $out4,$out4,$out4,$inpperm+ stvx_u $out3,$x30,$out+ vxor $out3,$in3,$rndkey0+ le?vperm $out5,$out5,$out5,$inpperm+ stvx_u $out4,$x40,$out+ vxor $out4,$in4,$rndkey0+ le?vperm $out6,$out6,$out6,$inpperm+ stvx_u $out5,$x50,$out+ vxor $out5,$in5,$rndkey0+ le?vperm $out7,$out7,$out7,$inpperm+ stvx_u $out6,$x60,$out+ vxor $out6,$in6,$rndkey0+ stvx_u $out7,$x70,$out+ addi $out,$out,0x80+ vxor $out7,$in7,$rndkey0++ mtctr $rounds+ beq Loop_cbc_dec8x # did $len-=128 borrow?++ addic. $len,$len,128+ beq Lcbc_dec8x_done+ nop+ nop++Loop_cbc_dec8x_tail: # up to 7 "words" tail...+ vncipher $out1,$out1,v24+ vncipher $out2,$out2,v24+ vncipher $out3,$out3,v24+ vncipher $out4,$out4,v24+ vncipher $out5,$out5,v24+ vncipher $out6,$out6,v24+ vncipher $out7,$out7,v24+ lvx v24,$x20,$key_ # round[3]+ addi $key_,$key_,0x20++ vncipher $out1,$out1,v25+ vncipher $out2,$out2,v25+ vncipher $out3,$out3,v25+ vncipher $out4,$out4,v25+ vncipher $out5,$out5,v25+ vncipher $out6,$out6,v25+ vncipher $out7,$out7,v25+ lvx v25,$x10,$key_ # round[4]+ bdnz Loop_cbc_dec8x_tail++ vncipher $out1,$out1,v24+ vncipher $out2,$out2,v24+ vncipher $out3,$out3,v24+ vncipher $out4,$out4,v24+ vncipher $out5,$out5,v24+ vncipher $out6,$out6,v24+ vncipher $out7,$out7,v24++ vncipher $out1,$out1,v25+ vncipher $out2,$out2,v25+ vncipher $out3,$out3,v25+ vncipher $out4,$out4,v25+ vncipher $out5,$out5,v25+ vncipher $out6,$out6,v25+ vncipher $out7,$out7,v25++ vncipher $out1,$out1,v26+ vncipher $out2,$out2,v26+ vncipher $out3,$out3,v26+ vncipher $out4,$out4,v26+ vncipher $out5,$out5,v26+ vncipher $out6,$out6,v26+ vncipher $out7,$out7,v26++ vncipher $out1,$out1,v27+ vncipher $out2,$out2,v27+ vncipher $out3,$out3,v27+ vncipher $out4,$out4,v27+ vncipher $out5,$out5,v27+ vncipher $out6,$out6,v27+ vncipher $out7,$out7,v27++ vncipher $out1,$out1,v28+ vncipher $out2,$out2,v28+ vncipher $out3,$out3,v28+ vncipher $out4,$out4,v28+ vncipher $out5,$out5,v28+ vncipher $out6,$out6,v28+ vncipher $out7,$out7,v28++ vncipher $out1,$out1,v29+ vncipher $out2,$out2,v29+ vncipher $out3,$out3,v29+ vncipher $out4,$out4,v29+ vncipher $out5,$out5,v29+ vncipher $out6,$out6,v29+ vncipher $out7,$out7,v29++ vncipher $out1,$out1,v30+ vxor $ivec,$ivec,v31 # last round key+ vncipher $out2,$out2,v30+ vxor $in1,$in1,v31+ vncipher $out3,$out3,v30+ vxor $in2,$in2,v31+ vncipher $out4,$out4,v30+ vxor $in3,$in3,v31+ vncipher $out5,$out5,v30+ vxor $in4,$in4,v31+ vncipher $out6,$out6,v30+ vxor $in5,$in5,v31+ vncipher $out7,$out7,v30+ vxor $in6,$in6,v31++ cmplwi $len,32 # switch($len)+ blt Lcbc_dec8x_one+ nop+ beq Lcbc_dec8x_two+ cmplwi $len,64+ blt Lcbc_dec8x_three+ nop+ beq Lcbc_dec8x_four+ cmplwi $len,96+ blt Lcbc_dec8x_five+ nop+ beq Lcbc_dec8x_six++Lcbc_dec8x_seven:+ vncipherlast $out1,$out1,$ivec+ vncipherlast $out2,$out2,$in1+ vncipherlast $out3,$out3,$in2+ vncipherlast $out4,$out4,$in3+ vncipherlast $out5,$out5,$in4+ vncipherlast $out6,$out6,$in5+ vncipherlast $out7,$out7,$in6+ vmr $ivec,$in7++ le?vperm $out1,$out1,$out1,$inpperm+ le?vperm $out2,$out2,$out2,$inpperm+ stvx_u $out1,$x00,$out+ le?vperm $out3,$out3,$out3,$inpperm+ stvx_u $out2,$x10,$out+ le?vperm $out4,$out4,$out4,$inpperm+ stvx_u $out3,$x20,$out+ le?vperm $out5,$out5,$out5,$inpperm+ stvx_u $out4,$x30,$out+ le?vperm $out6,$out6,$out6,$inpperm+ stvx_u $out5,$x40,$out+ le?vperm $out7,$out7,$out7,$inpperm+ stvx_u $out6,$x50,$out+ stvx_u $out7,$x60,$out+ addi $out,$out,0x70+ b Lcbc_dec8x_done++.align 5+Lcbc_dec8x_six:+ vncipherlast $out2,$out2,$ivec+ vncipherlast $out3,$out3,$in2+ vncipherlast $out4,$out4,$in3+ vncipherlast $out5,$out5,$in4+ vncipherlast $out6,$out6,$in5+ vncipherlast $out7,$out7,$in6+ vmr $ivec,$in7++ le?vperm $out2,$out2,$out2,$inpperm+ le?vperm $out3,$out3,$out3,$inpperm+ stvx_u $out2,$x00,$out+ le?vperm $out4,$out4,$out4,$inpperm+ stvx_u $out3,$x10,$out+ le?vperm $out5,$out5,$out5,$inpperm+ stvx_u $out4,$x20,$out+ le?vperm $out6,$out6,$out6,$inpperm+ stvx_u $out5,$x30,$out+ le?vperm $out7,$out7,$out7,$inpperm+ stvx_u $out6,$x40,$out+ stvx_u $out7,$x50,$out+ addi $out,$out,0x60+ b Lcbc_dec8x_done++.align 5+Lcbc_dec8x_five:+ vncipherlast $out3,$out3,$ivec+ vncipherlast $out4,$out4,$in3+ vncipherlast $out5,$out5,$in4+ vncipherlast $out6,$out6,$in5+ vncipherlast $out7,$out7,$in6+ vmr $ivec,$in7++ le?vperm $out3,$out3,$out3,$inpperm+ le?vperm $out4,$out4,$out4,$inpperm+ stvx_u $out3,$x00,$out+ le?vperm $out5,$out5,$out5,$inpperm+ stvx_u $out4,$x10,$out+ le?vperm $out6,$out6,$out6,$inpperm+ stvx_u $out5,$x20,$out+ le?vperm $out7,$out7,$out7,$inpperm+ stvx_u $out6,$x30,$out+ stvx_u $out7,$x40,$out+ addi $out,$out,0x50+ b Lcbc_dec8x_done++.align 5+Lcbc_dec8x_four:+ vncipherlast $out4,$out4,$ivec+ vncipherlast $out5,$out5,$in4+ vncipherlast $out6,$out6,$in5+ vncipherlast $out7,$out7,$in6+ vmr $ivec,$in7++ le?vperm $out4,$out4,$out4,$inpperm+ le?vperm $out5,$out5,$out5,$inpperm+ stvx_u $out4,$x00,$out+ le?vperm $out6,$out6,$out6,$inpperm+ stvx_u $out5,$x10,$out+ le?vperm $out7,$out7,$out7,$inpperm+ stvx_u $out6,$x20,$out+ stvx_u $out7,$x30,$out+ addi $out,$out,0x40+ b Lcbc_dec8x_done++.align 5+Lcbc_dec8x_three:+ vncipherlast $out5,$out5,$ivec+ vncipherlast $out6,$out6,$in5+ vncipherlast $out7,$out7,$in6+ vmr $ivec,$in7++ le?vperm $out5,$out5,$out5,$inpperm+ le?vperm $out6,$out6,$out6,$inpperm+ stvx_u $out5,$x00,$out+ le?vperm $out7,$out7,$out7,$inpperm+ stvx_u $out6,$x10,$out+ stvx_u $out7,$x20,$out+ addi $out,$out,0x30+ b Lcbc_dec8x_done++.align 5+Lcbc_dec8x_two:+ vncipherlast $out6,$out6,$ivec+ vncipherlast $out7,$out7,$in6+ vmr $ivec,$in7++ le?vperm $out6,$out6,$out6,$inpperm+ le?vperm $out7,$out7,$out7,$inpperm+ stvx_u $out6,$x00,$out+ stvx_u $out7,$x10,$out+ addi $out,$out,0x20+ b Lcbc_dec8x_done++.align 5+Lcbc_dec8x_one:+ vncipherlast $out7,$out7,$ivec+ vmr $ivec,$in7++ le?vperm $out7,$out7,$out7,$inpperm+ stvx_u $out7,0,$out+ addi $out,$out,0x10++Lcbc_dec8x_done:+ le?vperm $ivec,$ivec,$ivec,$inpperm+ stvx_u $ivec,0,$ivp # write [unaligned] iv++ li r10,`$FRAME+15`+ li r11,`$FRAME+31`+ stvx $inpperm,r10,$sp # wipe copies of round keys+ addi r10,r10,32+ stvx $inpperm,r11,$sp+ addi r11,r11,32+ stvx $inpperm,r10,$sp+ addi r10,r10,32+ stvx $inpperm,r11,$sp+ addi r11,r11,32+ stvx $inpperm,r10,$sp+ addi r10,r10,32+ stvx $inpperm,r11,$sp+ addi r11,r11,32+ stvx $inpperm,r10,$sp+ addi r10,r10,32+ stvx $inpperm,r11,$sp+ addi r11,r11,32++ mtspr 256,$vrsave+ lvx v20,r10,$sp # ABI says so+ addi r10,r10,32+ lvx v21,r11,$sp+ addi r11,r11,32+ lvx v22,r10,$sp+ addi r10,r10,32+ lvx v23,r11,$sp+ addi r11,r11,32+ lvx v24,r10,$sp+ addi r10,r10,32+ lvx v25,r11,$sp+ addi r11,r11,32+ lvx v26,r10,$sp+ addi r10,r10,32+ lvx v27,r11,$sp+ addi r11,r11,32+ lvx v28,r10,$sp+ addi r10,r10,32+ lvx v29,r11,$sp+ addi r11,r11,32+ lvx v30,r10,$sp+ lvx v31,r11,$sp+ $POP r26,`$FRAME+21*16+0*$SIZE_T`($sp)+ $POP r27,`$FRAME+21*16+1*$SIZE_T`($sp)+ $POP r28,`$FRAME+21*16+2*$SIZE_T`($sp)+ $POP r29,`$FRAME+21*16+3*$SIZE_T`($sp)+ $POP r30,`$FRAME+21*16+4*$SIZE_T`($sp)+ $POP r31,`$FRAME+21*16+5*$SIZE_T`($sp)+ addi $sp,$sp,`$FRAME+21*16+6*$SIZE_T`+ blr+ .long 0+ .byte 0,12,0x04,0,0x80,6,6,0+ .long 0+.size .${prefix}_cbc_encrypt,.-.${prefix}_cbc_encrypt+___+}} }}}++#########################################################################+{{{ # CTR procedure[s] #+my ($inp,$out,$len,$key,$ivp,$x10,$rounds,$idx)=map("r$_",(3..10));+my ($rndkey0,$rndkey1,$inout,$tmp)= map("v$_",(0..3));+my ($ivec,$inptail,$inpperm,$outhead,$outperm,$outmask,$keyperm,$one)=+ map("v$_",(4..11));+my $dat=$tmp;++$code.=<<___;+.globl .${prefix}_ctr32_encrypt_blocks+.align 5+.${prefix}_ctr32_encrypt_blocks:+ ${UCMP}i $len,1+ bltlr-++ lis r0,0xfff0+ mfspr $vrsave,256+ mtspr 256,r0++ li $idx,15+ vxor $rndkey0,$rndkey0,$rndkey0+ le?vspltisb $tmp,0x0f++ lvx $ivec,0,$ivp # load [unaligned] iv+ lvsl $inpperm,0,$ivp+ lvx $inptail,$idx,$ivp+ vspltisb $one,1+ le?vxor $inpperm,$inpperm,$tmp+ vperm $ivec,$ivec,$inptail,$inpperm+ vsldoi $one,$rndkey0,$one,1++ neg r11,$inp+ ?lvsl $keyperm,0,$key # prepare for unaligned key+ lwz $rounds,240($key)++ lvsr $inpperm,0,r11 # prepare for unaligned load+ lvx $inptail,0,$inp+ addi $inp,$inp,15 # 15 is not typo+ le?vxor $inpperm,$inpperm,$tmp++ srwi $rounds,$rounds,1+ li $idx,16+ subi $rounds,$rounds,1++ ${UCMP}i $len,8+ bge _aesp8_ctr32_encrypt8x++ ?lvsr $outperm,0,$out # prepare for unaligned store+ vspltisb $outmask,-1+ lvx $outhead,0,$out+ ?vperm $outmask,$rndkey0,$outmask,$outperm+ le?vxor $outperm,$outperm,$tmp++ lvx $rndkey0,0,$key+ mtctr $rounds+ lvx $rndkey1,$idx,$key+ addi $idx,$idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vxor $inout,$ivec,$rndkey0+ lvx $rndkey0,$idx,$key+ addi $idx,$idx,16+ b Loop_ctr32_enc++.align 5+Loop_ctr32_enc:+ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vcipher $inout,$inout,$rndkey1+ lvx $rndkey1,$idx,$key+ addi $idx,$idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vcipher $inout,$inout,$rndkey0+ lvx $rndkey0,$idx,$key+ addi $idx,$idx,16+ bdnz Loop_ctr32_enc++ vadduwm $ivec,$ivec,$one+ vmr $dat,$inptail+ lvx $inptail,0,$inp+ addi $inp,$inp,16+ subic. $len,$len,1 # blocks--++ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vcipher $inout,$inout,$rndkey1+ lvx $rndkey1,$idx,$key+ vperm $dat,$dat,$inptail,$inpperm+ li $idx,16+ ?vperm $rndkey1,$rndkey0,$rndkey1,$keyperm+ lvx $rndkey0,0,$key+ vxor $dat,$dat,$rndkey1 # last round key+ vcipherlast $inout,$inout,$dat++ lvx $rndkey1,$idx,$key+ addi $idx,$idx,16+ vperm $inout,$inout,$inout,$outperm+ vsel $dat,$outhead,$inout,$outmask+ mtctr $rounds+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vmr $outhead,$inout+ vxor $inout,$ivec,$rndkey0+ lvx $rndkey0,$idx,$key+ addi $idx,$idx,16+ stvx $dat,0,$out+ addi $out,$out,16+ bne Loop_ctr32_enc++ addi $out,$out,-1+ lvx $inout,0,$out # redundant in aligned case+ vsel $inout,$outhead,$inout,$outmask+ stvx $inout,0,$out++ mtspr 256,$vrsave+ blr+ .long 0+ .byte 0,12,0x14,0,0,0,6,0+ .long 0+___+#########################################################################+{{ # Optimized CTR procedure #+my $key_="r11";+my ($x00,$x10,$x20,$x30,$x40,$x50,$x60,$x70)=map("r$_",(0,8,26..31));+ $x00=0 if ($flavour =~ /osx/);+my ($in0, $in1, $in2, $in3, $in4, $in5, $in6, $in7 )=map("v$_",(0..3,10,12..14));+my ($out0,$out1,$out2,$out3,$out4,$out5,$out6,$out7)=map("v$_",(15..22));+my $rndkey0="v23"; # v24-v25 rotating buffer for first found keys+ # v26-v31 last 6 round keys+my ($tmp,$keyperm)=($in3,$in4); # aliases with "caller", redundant assignment+my ($two,$three,$four)=($outhead,$outperm,$outmask);++$code.=<<___;+.align 5+_aesp8_ctr32_encrypt8x:+ $STU $sp,-`($FRAME+21*16+6*$SIZE_T)`($sp)+ li r10,`$FRAME+8*16+15`+ li r11,`$FRAME+8*16+31`+ stvx v20,r10,$sp # ABI says so+ addi r10,r10,32+ stvx v21,r11,$sp+ addi r11,r11,32+ stvx v22,r10,$sp+ addi r10,r10,32+ stvx v23,r11,$sp+ addi r11,r11,32+ stvx v24,r10,$sp+ addi r10,r10,32+ stvx v25,r11,$sp+ addi r11,r11,32+ stvx v26,r10,$sp+ addi r10,r10,32+ stvx v27,r11,$sp+ addi r11,r11,32+ stvx v28,r10,$sp+ addi r10,r10,32+ stvx v29,r11,$sp+ addi r11,r11,32+ stvx v30,r10,$sp+ stvx v31,r11,$sp+ li r0,-1+ stw $vrsave,`$FRAME+21*16-4`($sp) # save vrsave+ li $x10,0x10+ $PUSH r26,`$FRAME+21*16+0*$SIZE_T`($sp)+ li $x20,0x20+ $PUSH r27,`$FRAME+21*16+1*$SIZE_T`($sp)+ li $x30,0x30+ $PUSH r28,`$FRAME+21*16+2*$SIZE_T`($sp)+ li $x40,0x40+ $PUSH r29,`$FRAME+21*16+3*$SIZE_T`($sp)+ li $x50,0x50+ $PUSH r30,`$FRAME+21*16+4*$SIZE_T`($sp)+ li $x60,0x60+ $PUSH r31,`$FRAME+21*16+5*$SIZE_T`($sp)+ li $x70,0x70+ mtspr 256,r0++ subi $rounds,$rounds,3 # -4 in total++ lvx $rndkey0,$x00,$key # load key schedule+ lvx v30,$x10,$key+ addi $key,$key,0x20+ lvx v31,$x00,$key+ ?vperm $rndkey0,$rndkey0,v30,$keyperm+ addi $key_,$sp,$FRAME+15+ mtctr $rounds++Load_ctr32_enc_key:+ ?vperm v24,v30,v31,$keyperm+ lvx v30,$x10,$key+ addi $key,$key,0x20+ stvx v24,$x00,$key_ # off-load round[1]+ ?vperm v25,v31,v30,$keyperm+ lvx v31,$x00,$key+ stvx v25,$x10,$key_ # off-load round[2]+ addi $key_,$key_,0x20+ bdnz Load_ctr32_enc_key++ lvx v26,$x10,$key+ ?vperm v24,v30,v31,$keyperm+ lvx v27,$x20,$key+ stvx v24,$x00,$key_ # off-load round[3]+ ?vperm v25,v31,v26,$keyperm+ lvx v28,$x30,$key+ stvx v25,$x10,$key_ # off-load round[4]+ addi $key_,$sp,$FRAME+15 # rewind $key_+ ?vperm v26,v26,v27,$keyperm+ lvx v29,$x40,$key+ ?vperm v27,v27,v28,$keyperm+ lvx v30,$x50,$key+ ?vperm v28,v28,v29,$keyperm+ lvx v31,$x60,$key+ ?vperm v29,v29,v30,$keyperm+ lvx $out0,$x70,$key # borrow $out0+ ?vperm v30,v30,v31,$keyperm+ lvx v24,$x00,$key_ # pre-load round[1]+ ?vperm v31,v31,$out0,$keyperm+ lvx v25,$x10,$key_ # pre-load round[2]++ vadduwm $two,$one,$one+ subi $inp,$inp,15 # undo "caller"+ $SHL $len,$len,4++ vadduwm $out1,$ivec,$one # counter values ...+ vadduwm $out2,$ivec,$two+ vxor $out0,$ivec,$rndkey0 # ... xored with rndkey[0]+ le?li $idx,8+ vadduwm $out3,$out1,$two+ vxor $out1,$out1,$rndkey0+ le?lvsl $inpperm,0,$idx+ vadduwm $out4,$out2,$two+ vxor $out2,$out2,$rndkey0+ le?vspltisb $tmp,0x0f+ vadduwm $out5,$out3,$two+ vxor $out3,$out3,$rndkey0+ le?vxor $inpperm,$inpperm,$tmp # transform for lvx_u/stvx_u+ vadduwm $out6,$out4,$two+ vxor $out4,$out4,$rndkey0+ vadduwm $out7,$out5,$two+ vxor $out5,$out5,$rndkey0+ vadduwm $ivec,$out6,$two # next counter value+ vxor $out6,$out6,$rndkey0+ vxor $out7,$out7,$rndkey0++ mtctr $rounds+ b Loop_ctr32_enc8x+.align 5+Loop_ctr32_enc8x:+ vcipher $out0,$out0,v24+ vcipher $out1,$out1,v24+ vcipher $out2,$out2,v24+ vcipher $out3,$out3,v24+ vcipher $out4,$out4,v24+ vcipher $out5,$out5,v24+ vcipher $out6,$out6,v24+ vcipher $out7,$out7,v24+Loop_ctr32_enc8x_middle:+ lvx v24,$x20,$key_ # round[3]+ addi $key_,$key_,0x20++ vcipher $out0,$out0,v25+ vcipher $out1,$out1,v25+ vcipher $out2,$out2,v25+ vcipher $out3,$out3,v25+ vcipher $out4,$out4,v25+ vcipher $out5,$out5,v25+ vcipher $out6,$out6,v25+ vcipher $out7,$out7,v25+ lvx v25,$x10,$key_ # round[4]+ bdnz Loop_ctr32_enc8x++ subic r11,$len,256 # $len-256, borrow $key_+ vcipher $out0,$out0,v24+ vcipher $out1,$out1,v24+ vcipher $out2,$out2,v24+ vcipher $out3,$out3,v24+ vcipher $out4,$out4,v24+ vcipher $out5,$out5,v24+ vcipher $out6,$out6,v24+ vcipher $out7,$out7,v24++ subfe r0,r0,r0 # borrow?-1:0+ vcipher $out0,$out0,v25+ vcipher $out1,$out1,v25+ vcipher $out2,$out2,v25+ vcipher $out3,$out3,v25+ vcipher $out4,$out4,v25+ vcipher $out5,$out5,v25+ vcipher $out6,$out6,v25+ vcipher $out7,$out7,v25++ and r0,r0,r11+ addi $key_,$sp,$FRAME+15 # rewind $key_+ vcipher $out0,$out0,v26+ vcipher $out1,$out1,v26+ vcipher $out2,$out2,v26+ vcipher $out3,$out3,v26+ vcipher $out4,$out4,v26+ vcipher $out5,$out5,v26+ vcipher $out6,$out6,v26+ vcipher $out7,$out7,v26+ lvx v24,$x00,$key_ # re-pre-load round[1]++ subic $len,$len,129 # $len-=129+ vcipher $out0,$out0,v27+ addi $len,$len,1 # $len-=128 really+ vcipher $out1,$out1,v27+ vcipher $out2,$out2,v27+ vcipher $out3,$out3,v27+ vcipher $out4,$out4,v27+ vcipher $out5,$out5,v27+ vcipher $out6,$out6,v27+ vcipher $out7,$out7,v27+ lvx v25,$x10,$key_ # re-pre-load round[2]++ vcipher $out0,$out0,v28+ lvx_u $in0,$x00,$inp # load input+ vcipher $out1,$out1,v28+ lvx_u $in1,$x10,$inp+ vcipher $out2,$out2,v28+ lvx_u $in2,$x20,$inp+ vcipher $out3,$out3,v28+ lvx_u $in3,$x30,$inp+ vcipher $out4,$out4,v28+ lvx_u $in4,$x40,$inp+ vcipher $out5,$out5,v28+ lvx_u $in5,$x50,$inp+ vcipher $out6,$out6,v28+ lvx_u $in6,$x60,$inp+ vcipher $out7,$out7,v28+ lvx_u $in7,$x70,$inp+ addi $inp,$inp,0x80++ vcipher $out0,$out0,v29+ le?vperm $in0,$in0,$in0,$inpperm+ vcipher $out1,$out1,v29+ le?vperm $in1,$in1,$in1,$inpperm+ vcipher $out2,$out2,v29+ le?vperm $in2,$in2,$in2,$inpperm+ vcipher $out3,$out3,v29+ le?vperm $in3,$in3,$in3,$inpperm+ vcipher $out4,$out4,v29+ le?vperm $in4,$in4,$in4,$inpperm+ vcipher $out5,$out5,v29+ le?vperm $in5,$in5,$in5,$inpperm+ vcipher $out6,$out6,v29+ le?vperm $in6,$in6,$in6,$inpperm+ vcipher $out7,$out7,v29+ le?vperm $in7,$in7,$in7,$inpperm++ add $inp,$inp,r0 # $inp is adjusted in such+ # way that at exit from the+ # loop inX-in7 are loaded+ # with last "words"+ subfe. r0,r0,r0 # borrow?-1:0+ vcipher $out0,$out0,v30+ vxor $in0,$in0,v31 # xor with last round key+ vcipher $out1,$out1,v30+ vxor $in1,$in1,v31+ vcipher $out2,$out2,v30+ vxor $in2,$in2,v31+ vcipher $out3,$out3,v30+ vxor $in3,$in3,v31+ vcipher $out4,$out4,v30+ vxor $in4,$in4,v31+ vcipher $out5,$out5,v30+ vxor $in5,$in5,v31+ vcipher $out6,$out6,v30+ vxor $in6,$in6,v31+ vcipher $out7,$out7,v30+ vxor $in7,$in7,v31++ bne Lctr32_enc8x_break # did $len-129 borrow?++ vcipherlast $in0,$out0,$in0+ vcipherlast $in1,$out1,$in1+ vadduwm $out1,$ivec,$one # counter values ...+ vcipherlast $in2,$out2,$in2+ vadduwm $out2,$ivec,$two+ vxor $out0,$ivec,$rndkey0 # ... xored with rndkey[0]+ vcipherlast $in3,$out3,$in3+ vadduwm $out3,$out1,$two+ vxor $out1,$out1,$rndkey0+ vcipherlast $in4,$out4,$in4+ vadduwm $out4,$out2,$two+ vxor $out2,$out2,$rndkey0+ vcipherlast $in5,$out5,$in5+ vadduwm $out5,$out3,$two+ vxor $out3,$out3,$rndkey0+ vcipherlast $in6,$out6,$in6+ vadduwm $out6,$out4,$two+ vxor $out4,$out4,$rndkey0+ vcipherlast $in7,$out7,$in7+ vadduwm $out7,$out5,$two+ vxor $out5,$out5,$rndkey0+ le?vperm $in0,$in0,$in0,$inpperm+ vadduwm $ivec,$out6,$two # next counter value+ vxor $out6,$out6,$rndkey0+ le?vperm $in1,$in1,$in1,$inpperm+ vxor $out7,$out7,$rndkey0+ mtctr $rounds++ vcipher $out0,$out0,v24+ stvx_u $in0,$x00,$out+ le?vperm $in2,$in2,$in2,$inpperm+ vcipher $out1,$out1,v24+ stvx_u $in1,$x10,$out+ le?vperm $in3,$in3,$in3,$inpperm+ vcipher $out2,$out2,v24+ stvx_u $in2,$x20,$out+ le?vperm $in4,$in4,$in4,$inpperm+ vcipher $out3,$out3,v24+ stvx_u $in3,$x30,$out+ le?vperm $in5,$in5,$in5,$inpperm+ vcipher $out4,$out4,v24+ stvx_u $in4,$x40,$out+ le?vperm $in6,$in6,$in6,$inpperm+ vcipher $out5,$out5,v24+ stvx_u $in5,$x50,$out+ le?vperm $in7,$in7,$in7,$inpperm+ vcipher $out6,$out6,v24+ stvx_u $in6,$x60,$out+ vcipher $out7,$out7,v24+ stvx_u $in7,$x70,$out+ addi $out,$out,0x80++ b Loop_ctr32_enc8x_middle++.align 5+Lctr32_enc8x_break:+ cmpwi $len,-0x60+ blt Lctr32_enc8x_one+ nop+ beq Lctr32_enc8x_two+ cmpwi $len,-0x40+ blt Lctr32_enc8x_three+ nop+ beq Lctr32_enc8x_four+ cmpwi $len,-0x20+ blt Lctr32_enc8x_five+ nop+ beq Lctr32_enc8x_six+ cmpwi $len,0x00+ blt Lctr32_enc8x_seven++Lctr32_enc8x_eight:+ vcipherlast $out0,$out0,$in0+ vcipherlast $out1,$out1,$in1+ vcipherlast $out2,$out2,$in2+ vcipherlast $out3,$out3,$in3+ vcipherlast $out4,$out4,$in4+ vcipherlast $out5,$out5,$in5+ vcipherlast $out6,$out6,$in6+ vcipherlast $out7,$out7,$in7++ le?vperm $out0,$out0,$out0,$inpperm+ le?vperm $out1,$out1,$out1,$inpperm+ stvx_u $out0,$x00,$out+ le?vperm $out2,$out2,$out2,$inpperm+ stvx_u $out1,$x10,$out+ le?vperm $out3,$out3,$out3,$inpperm+ stvx_u $out2,$x20,$out+ le?vperm $out4,$out4,$out4,$inpperm+ stvx_u $out3,$x30,$out+ le?vperm $out5,$out5,$out5,$inpperm+ stvx_u $out4,$x40,$out+ le?vperm $out6,$out6,$out6,$inpperm+ stvx_u $out5,$x50,$out+ le?vperm $out7,$out7,$out7,$inpperm+ stvx_u $out6,$x60,$out+ stvx_u $out7,$x70,$out+ addi $out,$out,0x80+ b Lctr32_enc8x_done++.align 5+Lctr32_enc8x_seven:+ vcipherlast $out0,$out0,$in1+ vcipherlast $out1,$out1,$in2+ vcipherlast $out2,$out2,$in3+ vcipherlast $out3,$out3,$in4+ vcipherlast $out4,$out4,$in5+ vcipherlast $out5,$out5,$in6+ vcipherlast $out6,$out6,$in7++ le?vperm $out0,$out0,$out0,$inpperm+ le?vperm $out1,$out1,$out1,$inpperm+ stvx_u $out0,$x00,$out+ le?vperm $out2,$out2,$out2,$inpperm+ stvx_u $out1,$x10,$out+ le?vperm $out3,$out3,$out3,$inpperm+ stvx_u $out2,$x20,$out+ le?vperm $out4,$out4,$out4,$inpperm+ stvx_u $out3,$x30,$out+ le?vperm $out5,$out5,$out5,$inpperm+ stvx_u $out4,$x40,$out+ le?vperm $out6,$out6,$out6,$inpperm+ stvx_u $out5,$x50,$out+ stvx_u $out6,$x60,$out+ addi $out,$out,0x70+ b Lctr32_enc8x_done++.align 5+Lctr32_enc8x_six:+ vcipherlast $out0,$out0,$in2+ vcipherlast $out1,$out1,$in3+ vcipherlast $out2,$out2,$in4+ vcipherlast $out3,$out3,$in5+ vcipherlast $out4,$out4,$in6+ vcipherlast $out5,$out5,$in7++ le?vperm $out0,$out0,$out0,$inpperm+ le?vperm $out1,$out1,$out1,$inpperm+ stvx_u $out0,$x00,$out+ le?vperm $out2,$out2,$out2,$inpperm+ stvx_u $out1,$x10,$out+ le?vperm $out3,$out3,$out3,$inpperm+ stvx_u $out2,$x20,$out+ le?vperm $out4,$out4,$out4,$inpperm+ stvx_u $out3,$x30,$out+ le?vperm $out5,$out5,$out5,$inpperm+ stvx_u $out4,$x40,$out+ stvx_u $out5,$x50,$out+ addi $out,$out,0x60+ b Lctr32_enc8x_done++.align 5+Lctr32_enc8x_five:+ vcipherlast $out0,$out0,$in3+ vcipherlast $out1,$out1,$in4+ vcipherlast $out2,$out2,$in5+ vcipherlast $out3,$out3,$in6+ vcipherlast $out4,$out4,$in7++ le?vperm $out0,$out0,$out0,$inpperm+ le?vperm $out1,$out1,$out1,$inpperm+ stvx_u $out0,$x00,$out+ le?vperm $out2,$out2,$out2,$inpperm+ stvx_u $out1,$x10,$out+ le?vperm $out3,$out3,$out3,$inpperm+ stvx_u $out2,$x20,$out+ le?vperm $out4,$out4,$out4,$inpperm+ stvx_u $out3,$x30,$out+ stvx_u $out4,$x40,$out+ addi $out,$out,0x50+ b Lctr32_enc8x_done++.align 5+Lctr32_enc8x_four:+ vcipherlast $out0,$out0,$in4+ vcipherlast $out1,$out1,$in5+ vcipherlast $out2,$out2,$in6+ vcipherlast $out3,$out3,$in7++ le?vperm $out0,$out0,$out0,$inpperm+ le?vperm $out1,$out1,$out1,$inpperm+ stvx_u $out0,$x00,$out+ le?vperm $out2,$out2,$out2,$inpperm+ stvx_u $out1,$x10,$out+ le?vperm $out3,$out3,$out3,$inpperm+ stvx_u $out2,$x20,$out+ stvx_u $out3,$x30,$out+ addi $out,$out,0x40+ b Lctr32_enc8x_done++.align 5+Lctr32_enc8x_three:+ vcipherlast $out0,$out0,$in5+ vcipherlast $out1,$out1,$in6+ vcipherlast $out2,$out2,$in7++ le?vperm $out0,$out0,$out0,$inpperm+ le?vperm $out1,$out1,$out1,$inpperm+ stvx_u $out0,$x00,$out+ le?vperm $out2,$out2,$out2,$inpperm+ stvx_u $out1,$x10,$out+ stvx_u $out2,$x20,$out+ addi $out,$out,0x30+ b Lctr32_enc8x_done++.align 5+Lctr32_enc8x_two:+ vcipherlast $out0,$out0,$in6+ vcipherlast $out1,$out1,$in7++ le?vperm $out0,$out0,$out0,$inpperm+ le?vperm $out1,$out1,$out1,$inpperm+ stvx_u $out0,$x00,$out+ stvx_u $out1,$x10,$out+ addi $out,$out,0x20+ b Lctr32_enc8x_done++.align 5+Lctr32_enc8x_one:+ vcipherlast $out0,$out0,$in7++ le?vperm $out0,$out0,$out0,$inpperm+ stvx_u $out0,0,$out+ addi $out,$out,0x10++Lctr32_enc8x_done:+ li r10,`$FRAME+15`+ li r11,`$FRAME+31`+ stvx $inpperm,r10,$sp # wipe copies of round keys+ addi r10,r10,32+ stvx $inpperm,r11,$sp+ addi r11,r11,32+ stvx $inpperm,r10,$sp+ addi r10,r10,32+ stvx $inpperm,r11,$sp+ addi r11,r11,32+ stvx $inpperm,r10,$sp+ addi r10,r10,32+ stvx $inpperm,r11,$sp+ addi r11,r11,32+ stvx $inpperm,r10,$sp+ addi r10,r10,32+ stvx $inpperm,r11,$sp+ addi r11,r11,32++ mtspr 256,$vrsave+ lvx v20,r10,$sp # ABI says so+ addi r10,r10,32+ lvx v21,r11,$sp+ addi r11,r11,32+ lvx v22,r10,$sp+ addi r10,r10,32+ lvx v23,r11,$sp+ addi r11,r11,32+ lvx v24,r10,$sp+ addi r10,r10,32+ lvx v25,r11,$sp+ addi r11,r11,32+ lvx v26,r10,$sp+ addi r10,r10,32+ lvx v27,r11,$sp+ addi r11,r11,32+ lvx v28,r10,$sp+ addi r10,r10,32+ lvx v29,r11,$sp+ addi r11,r11,32+ lvx v30,r10,$sp+ lvx v31,r11,$sp+ $POP r26,`$FRAME+21*16+0*$SIZE_T`($sp)+ $POP r27,`$FRAME+21*16+1*$SIZE_T`($sp)+ $POP r28,`$FRAME+21*16+2*$SIZE_T`($sp)+ $POP r29,`$FRAME+21*16+3*$SIZE_T`($sp)+ $POP r30,`$FRAME+21*16+4*$SIZE_T`($sp)+ $POP r31,`$FRAME+21*16+5*$SIZE_T`($sp)+ addi $sp,$sp,`$FRAME+21*16+6*$SIZE_T`+ blr+ .long 0+ .byte 0,12,0x04,0,0x80,6,6,0+ .long 0+.size .${prefix}_ctr32_encrypt_blocks,.-.${prefix}_ctr32_encrypt_blocks+___+}} }}}++#########################################################################+{{{ # XTS procedures #+# int aes_p8_xts_[en|de]crypt(const char *inp, char *out, size_t len, #+# const AES_KEY *key1, const AES_KEY *key2, #+# [const] unsigned char iv[16]); #+# If $key2 is NULL, then a "tweak chaining" mode is engaged, in which #+# input tweak value is assumed to be encrypted already, and last tweak #+# value, one suitable for consecutive call on same chunk of data, is #+# written back to original buffer. In addition, in "tweak chaining" #+# mode only complete input blocks are processed. #++my ($inp,$out,$len,$key1,$key2,$ivp,$rounds,$idx) = map("r$_",(3..10));+my ($rndkey0,$rndkey1,$inout) = map("v$_",(0..2));+my ($output,$inptail,$inpperm,$leperm,$keyperm) = map("v$_",(3..7));+my ($tweak,$seven,$eighty7,$tmp,$tweak1) = map("v$_",(8..12));+my $taillen = $key2;++ ($inp,$idx) = ($idx,$inp); # reassign++$code.=<<___;+.globl .${prefix}_xts_encrypt+.align 5+.${prefix}_xts_encrypt:+ mr $inp,r3 # reassign+ li r3,-1+ ${UCMP}i $len,16+ bltlr-++ lis r0,0xfff0+ mfspr r12,256 # save vrsave+ li r11,0+ mtspr 256,r0++ vspltisb $seven,0x07 # 0x070707..07+ le?lvsl $leperm,r11,r11+ le?vspltisb $tmp,0x0f+ le?vxor $leperm,$leperm,$seven++ li $idx,15+ lvx $tweak,0,$ivp # load [unaligned] iv+ lvsl $inpperm,0,$ivp+ lvx $inptail,$idx,$ivp+ le?vxor $inpperm,$inpperm,$tmp+ vperm $tweak,$tweak,$inptail,$inpperm++ neg r11,$inp+ lvsr $inpperm,0,r11 # prepare for unaligned load+ lvx $inout,0,$inp+ addi $inp,$inp,15 # 15 is not typo+ le?vxor $inpperm,$inpperm,$tmp++ ${UCMP}i $key2,0 # key2==NULL?+ beq Lxts_enc_no_key2++ ?lvsl $keyperm,0,$key2 # prepare for unaligned key+ lwz $rounds,240($key2)+ srwi $rounds,$rounds,1+ subi $rounds,$rounds,1+ li $idx,16++ lvx $rndkey0,0,$key2+ lvx $rndkey1,$idx,$key2+ addi $idx,$idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vxor $tweak,$tweak,$rndkey0+ lvx $rndkey0,$idx,$key2+ addi $idx,$idx,16+ mtctr $rounds++Ltweak_xts_enc:+ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vcipher $tweak,$tweak,$rndkey1+ lvx $rndkey1,$idx,$key2+ addi $idx,$idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vcipher $tweak,$tweak,$rndkey0+ lvx $rndkey0,$idx,$key2+ addi $idx,$idx,16+ bdnz Ltweak_xts_enc++ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vcipher $tweak,$tweak,$rndkey1+ lvx $rndkey1,$idx,$key2+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vcipherlast $tweak,$tweak,$rndkey0++ li $ivp,0 # don't chain the tweak+ b Lxts_enc++Lxts_enc_no_key2:+ li $idx,-16+ and $len,$len,$idx # in "tweak chaining"+ # mode only complete+ # blocks are processed+Lxts_enc:+ lvx $inptail,0,$inp+ addi $inp,$inp,16++ ?lvsl $keyperm,0,$key1 # prepare for unaligned key+ lwz $rounds,240($key1)+ srwi $rounds,$rounds,1+ subi $rounds,$rounds,1+ li $idx,16++ vslb $eighty7,$seven,$seven # 0x808080..80+ vor $eighty7,$eighty7,$seven # 0x878787..87+ vspltisb $tmp,1 # 0x010101..01+ vsldoi $eighty7,$eighty7,$tmp,15 # 0x870101..01++ ${UCMP}i $len,96+ bge _aesp8_xts_encrypt6x++ andi. $taillen,$len,15+ subic r0,$len,32+ subi $taillen,$taillen,16+ subfe r0,r0,r0+ and r0,r0,$taillen+ add $inp,$inp,r0++ lvx $rndkey0,0,$key1+ lvx $rndkey1,$idx,$key1+ addi $idx,$idx,16+ vperm $inout,$inout,$inptail,$inpperm+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vxor $inout,$inout,$tweak+ vxor $inout,$inout,$rndkey0+ lvx $rndkey0,$idx,$key1+ addi $idx,$idx,16+ mtctr $rounds+ b Loop_xts_enc++.align 5+Loop_xts_enc:+ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vcipher $inout,$inout,$rndkey1+ lvx $rndkey1,$idx,$key1+ addi $idx,$idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vcipher $inout,$inout,$rndkey0+ lvx $rndkey0,$idx,$key1+ addi $idx,$idx,16+ bdnz Loop_xts_enc++ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vcipher $inout,$inout,$rndkey1+ lvx $rndkey1,$idx,$key1+ li $idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vxor $rndkey0,$rndkey0,$tweak+ vcipherlast $output,$inout,$rndkey0++ le?vperm $tmp,$output,$output,$leperm+ be?nop+ le?stvx_u $tmp,0,$out+ be?stvx_u $output,0,$out+ addi $out,$out,16++ subic. $len,$len,16+ beq Lxts_enc_done++ vmr $inout,$inptail+ lvx $inptail,0,$inp+ addi $inp,$inp,16+ lvx $rndkey0,0,$key1+ lvx $rndkey1,$idx,$key1+ addi $idx,$idx,16++ subic r0,$len,32+ subfe r0,r0,r0+ and r0,r0,$taillen+ add $inp,$inp,r0++ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ vand $tmp,$tmp,$eighty7+ vxor $tweak,$tweak,$tmp++ vperm $inout,$inout,$inptail,$inpperm+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vxor $inout,$inout,$tweak+ vxor $output,$output,$rndkey0 # just in case $len<16+ vxor $inout,$inout,$rndkey0+ lvx $rndkey0,$idx,$key1+ addi $idx,$idx,16++ mtctr $rounds+ ${UCMP}i $len,16+ bge Loop_xts_enc++ vxor $output,$output,$tweak+ lvsr $inpperm,0,$len # $inpperm is no longer needed+ vxor $inptail,$inptail,$inptail # $inptail is no longer needed+ vspltisb $tmp,-1+ vperm $inptail,$inptail,$tmp,$inpperm+ vsel $inout,$inout,$output,$inptail++ subi r11,$out,17+ subi $out,$out,16+ mtctr $len+ li $len,16+Loop_xts_enc_steal:+ lbzu r0,1(r11)+ stb r0,16(r11)+ bdnz Loop_xts_enc_steal++ mtctr $rounds+ b Loop_xts_enc # one more time...++Lxts_enc_done:+ ${UCMP}i $ivp,0+ beq Lxts_enc_ret++ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ vand $tmp,$tmp,$eighty7+ vxor $tweak,$tweak,$tmp++ le?vperm $tweak,$tweak,$tweak,$leperm+ stvx_u $tweak,0,$ivp++Lxts_enc_ret:+ mtspr 256,r12 # restore vrsave+ li r3,0+ blr+ .long 0+ .byte 0,12,0x04,0,0x80,6,6,0+ .long 0+.size .${prefix}_xts_encrypt,.-.${prefix}_xts_encrypt++.globl .${prefix}_xts_decrypt+.align 5+.${prefix}_xts_decrypt:+ mr $inp,r3 # reassign+ li r3,-1+ ${UCMP}i $len,16+ bltlr-++ lis r0,0xfff8+ mfspr r12,256 # save vrsave+ li r11,0+ mtspr 256,r0++ andi. r0,$len,15+ neg r0,r0+ andi. r0,r0,16+ sub $len,$len,r0++ vspltisb $seven,0x07 # 0x070707..07+ le?lvsl $leperm,r11,r11+ le?vspltisb $tmp,0x0f+ le?vxor $leperm,$leperm,$seven++ li $idx,15+ lvx $tweak,0,$ivp # load [unaligned] iv+ lvsl $inpperm,0,$ivp+ lvx $inptail,$idx,$ivp+ le?vxor $inpperm,$inpperm,$tmp+ vperm $tweak,$tweak,$inptail,$inpperm++ neg r11,$inp+ lvsr $inpperm,0,r11 # prepare for unaligned load+ lvx $inout,0,$inp+ addi $inp,$inp,15 # 15 is not typo+ le?vxor $inpperm,$inpperm,$tmp++ ${UCMP}i $key2,0 # key2==NULL?+ beq Lxts_dec_no_key2++ ?lvsl $keyperm,0,$key2 # prepare for unaligned key+ lwz $rounds,240($key2)+ srwi $rounds,$rounds,1+ subi $rounds,$rounds,1+ li $idx,16++ lvx $rndkey0,0,$key2+ lvx $rndkey1,$idx,$key2+ addi $idx,$idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vxor $tweak,$tweak,$rndkey0+ lvx $rndkey0,$idx,$key2+ addi $idx,$idx,16+ mtctr $rounds++Ltweak_xts_dec:+ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vcipher $tweak,$tweak,$rndkey1+ lvx $rndkey1,$idx,$key2+ addi $idx,$idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vcipher $tweak,$tweak,$rndkey0+ lvx $rndkey0,$idx,$key2+ addi $idx,$idx,16+ bdnz Ltweak_xts_dec++ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vcipher $tweak,$tweak,$rndkey1+ lvx $rndkey1,$idx,$key2+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vcipherlast $tweak,$tweak,$rndkey0++ li $ivp,0 # don't chain the tweak+ b Lxts_dec++Lxts_dec_no_key2:+ neg $idx,$len+ andi. $idx,$idx,15+ add $len,$len,$idx # in "tweak chaining"+ # mode only complete+ # blocks are processed+Lxts_dec:+ lvx $inptail,0,$inp+ addi $inp,$inp,16++ ?lvsl $keyperm,0,$key1 # prepare for unaligned key+ lwz $rounds,240($key1)+ srwi $rounds,$rounds,1+ subi $rounds,$rounds,1+ li $idx,16++ vslb $eighty7,$seven,$seven # 0x808080..80+ vor $eighty7,$eighty7,$seven # 0x878787..87+ vspltisb $tmp,1 # 0x010101..01+ vsldoi $eighty7,$eighty7,$tmp,15 # 0x870101..01++ ${UCMP}i $len,96+ bge _aesp8_xts_decrypt6x++ lvx $rndkey0,0,$key1+ lvx $rndkey1,$idx,$key1+ addi $idx,$idx,16+ vperm $inout,$inout,$inptail,$inpperm+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vxor $inout,$inout,$tweak+ vxor $inout,$inout,$rndkey0+ lvx $rndkey0,$idx,$key1+ addi $idx,$idx,16+ mtctr $rounds++ ${UCMP}i $len,16+ blt Ltail_xts_dec+ be?b Loop_xts_dec++.align 5+Loop_xts_dec:+ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vncipher $inout,$inout,$rndkey1+ lvx $rndkey1,$idx,$key1+ addi $idx,$idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vncipher $inout,$inout,$rndkey0+ lvx $rndkey0,$idx,$key1+ addi $idx,$idx,16+ bdnz Loop_xts_dec++ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vncipher $inout,$inout,$rndkey1+ lvx $rndkey1,$idx,$key1+ li $idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vxor $rndkey0,$rndkey0,$tweak+ vncipherlast $output,$inout,$rndkey0++ le?vperm $tmp,$output,$output,$leperm+ be?nop+ le?stvx_u $tmp,0,$out+ be?stvx_u $output,0,$out+ addi $out,$out,16++ subic. $len,$len,16+ beq Lxts_dec_done++ vmr $inout,$inptail+ lvx $inptail,0,$inp+ addi $inp,$inp,16+ lvx $rndkey0,0,$key1+ lvx $rndkey1,$idx,$key1+ addi $idx,$idx,16++ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ vand $tmp,$tmp,$eighty7+ vxor $tweak,$tweak,$tmp++ vperm $inout,$inout,$inptail,$inpperm+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vxor $inout,$inout,$tweak+ vxor $inout,$inout,$rndkey0+ lvx $rndkey0,$idx,$key1+ addi $idx,$idx,16++ mtctr $rounds+ ${UCMP}i $len,16+ bge Loop_xts_dec++Ltail_xts_dec:+ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak1,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ vand $tmp,$tmp,$eighty7+ vxor $tweak1,$tweak1,$tmp++ subi $inp,$inp,16+ add $inp,$inp,$len++ vxor $inout,$inout,$tweak # :-(+ vxor $inout,$inout,$tweak1 # :-)++Loop_xts_dec_short:+ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vncipher $inout,$inout,$rndkey1+ lvx $rndkey1,$idx,$key1+ addi $idx,$idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vncipher $inout,$inout,$rndkey0+ lvx $rndkey0,$idx,$key1+ addi $idx,$idx,16+ bdnz Loop_xts_dec_short++ ?vperm $rndkey1,$rndkey1,$rndkey0,$keyperm+ vncipher $inout,$inout,$rndkey1+ lvx $rndkey1,$idx,$key1+ li $idx,16+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm+ vxor $rndkey0,$rndkey0,$tweak1+ vncipherlast $output,$inout,$rndkey0++ le?vperm $tmp,$output,$output,$leperm+ be?nop+ le?stvx_u $tmp,0,$out+ be?stvx_u $output,0,$out++ vmr $inout,$inptail+ lvx $inptail,0,$inp+ #addi $inp,$inp,16+ lvx $rndkey0,0,$key1+ lvx $rndkey1,$idx,$key1+ addi $idx,$idx,16+ vperm $inout,$inout,$inptail,$inpperm+ ?vperm $rndkey0,$rndkey0,$rndkey1,$keyperm++ lvsr $inpperm,0,$len # $inpperm is no longer needed+ vxor $inptail,$inptail,$inptail # $inptail is no longer needed+ vspltisb $tmp,-1+ vperm $inptail,$inptail,$tmp,$inpperm+ vsel $inout,$inout,$output,$inptail++ vxor $rndkey0,$rndkey0,$tweak+ vxor $inout,$inout,$rndkey0+ lvx $rndkey0,$idx,$key1+ addi $idx,$idx,16++ subi r11,$out,1+ mtctr $len+ li $len,16+Loop_xts_dec_steal:+ lbzu r0,1(r11)+ stb r0,16(r11)+ bdnz Loop_xts_dec_steal++ mtctr $rounds+ b Loop_xts_dec # one more time...++Lxts_dec_done:+ ${UCMP}i $ivp,0+ beq Lxts_dec_ret++ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ vand $tmp,$tmp,$eighty7+ vxor $tweak,$tweak,$tmp++ le?vperm $tweak,$tweak,$tweak,$leperm+ stvx_u $tweak,0,$ivp++Lxts_dec_ret:+ mtspr 256,r12 # restore vrsave+ li r3,0+ blr+ .long 0+ .byte 0,12,0x04,0,0x80,6,6,0+ .long 0+.size .${prefix}_xts_decrypt,.-.${prefix}_xts_decrypt+___+#########################################################################+{{ # Optimized XTS procedures #+my $key_=$key2;+my ($x00,$x10,$x20,$x30,$x40,$x50,$x60,$x70)=map("r$_",(0,3,26..31));+ $x00=0 if ($flavour =~ /osx/);+my ($in0, $in1, $in2, $in3, $in4, $in5 )=map("v$_",(0..5));+my ($out0, $out1, $out2, $out3, $out4, $out5)=map("v$_",(7,12..16));+my ($twk0, $twk1, $twk2, $twk3, $twk4, $twk5)=map("v$_",(17..22));+my $rndkey0="v23"; # v24-v25 rotating buffer for first found keys+ # v26-v31 last 6 round keys+my ($keyperm)=($out0); # aliases with "caller", redundant assignment+my $taillen=$x70;++$code.=<<___;+.align 5+_aesp8_xts_encrypt6x:+ $STU $sp,-`($FRAME+21*16+6*$SIZE_T)`($sp)+ mflr r11+ li r7,`$FRAME+8*16+15`+ li r3,`$FRAME+8*16+31`+ $PUSH r11,`$FRAME+21*16+6*$SIZE_T+$LRSAVE`($sp)+ stvx v20,r7,$sp # ABI says so+ addi r7,r7,32+ stvx v21,r3,$sp+ addi r3,r3,32+ stvx v22,r7,$sp+ addi r7,r7,32+ stvx v23,r3,$sp+ addi r3,r3,32+ stvx v24,r7,$sp+ addi r7,r7,32+ stvx v25,r3,$sp+ addi r3,r3,32+ stvx v26,r7,$sp+ addi r7,r7,32+ stvx v27,r3,$sp+ addi r3,r3,32+ stvx v28,r7,$sp+ addi r7,r7,32+ stvx v29,r3,$sp+ addi r3,r3,32+ stvx v30,r7,$sp+ stvx v31,r3,$sp+ li r0,-1+ stw $vrsave,`$FRAME+21*16-4`($sp) # save vrsave+ li $x10,0x10+ $PUSH r26,`$FRAME+21*16+0*$SIZE_T`($sp)+ li $x20,0x20+ $PUSH r27,`$FRAME+21*16+1*$SIZE_T`($sp)+ li $x30,0x30+ $PUSH r28,`$FRAME+21*16+2*$SIZE_T`($sp)+ li $x40,0x40+ $PUSH r29,`$FRAME+21*16+3*$SIZE_T`($sp)+ li $x50,0x50+ $PUSH r30,`$FRAME+21*16+4*$SIZE_T`($sp)+ li $x60,0x60+ $PUSH r31,`$FRAME+21*16+5*$SIZE_T`($sp)+ li $x70,0x70+ mtspr 256,r0++ subi $rounds,$rounds,3 # -4 in total++ lvx $rndkey0,$x00,$key1 # load key schedule+ lvx v30,$x10,$key1+ addi $key1,$key1,0x20+ lvx v31,$x00,$key1+ ?vperm $rndkey0,$rndkey0,v30,$keyperm+ addi $key_,$sp,$FRAME+15+ mtctr $rounds++Load_xts_enc_key:+ ?vperm v24,v30,v31,$keyperm+ lvx v30,$x10,$key1+ addi $key1,$key1,0x20+ stvx v24,$x00,$key_ # off-load round[1]+ ?vperm v25,v31,v30,$keyperm+ lvx v31,$x00,$key1+ stvx v25,$x10,$key_ # off-load round[2]+ addi $key_,$key_,0x20+ bdnz Load_xts_enc_key++ lvx v26,$x10,$key1+ ?vperm v24,v30,v31,$keyperm+ lvx v27,$x20,$key1+ stvx v24,$x00,$key_ # off-load round[3]+ ?vperm v25,v31,v26,$keyperm+ lvx v28,$x30,$key1+ stvx v25,$x10,$key_ # off-load round[4]+ addi $key_,$sp,$FRAME+15 # rewind $key_+ ?vperm v26,v26,v27,$keyperm+ lvx v29,$x40,$key1+ ?vperm v27,v27,v28,$keyperm+ lvx v30,$x50,$key1+ ?vperm v28,v28,v29,$keyperm+ lvx v31,$x60,$key1+ ?vperm v29,v29,v30,$keyperm+ lvx $twk5,$x70,$key1 # borrow $twk5+ ?vperm v30,v30,v31,$keyperm+ lvx v24,$x00,$key_ # pre-load round[1]+ ?vperm v31,v31,$twk5,$keyperm+ lvx v25,$x10,$key_ # pre-load round[2]++ vperm $in0,$inout,$inptail,$inpperm+ subi $inp,$inp,31 # undo "caller"+ vxor $twk0,$tweak,$rndkey0+ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ vand $tmp,$tmp,$eighty7+ vxor $out0,$in0,$twk0+ vxor $tweak,$tweak,$tmp++ lvx_u $in1,$x10,$inp+ vxor $twk1,$tweak,$rndkey0+ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ le?vperm $in1,$in1,$in1,$leperm+ vand $tmp,$tmp,$eighty7+ vxor $out1,$in1,$twk1+ vxor $tweak,$tweak,$tmp++ lvx_u $in2,$x20,$inp+ andi. $taillen,$len,15+ vxor $twk2,$tweak,$rndkey0+ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ le?vperm $in2,$in2,$in2,$leperm+ vand $tmp,$tmp,$eighty7+ vxor $out2,$in2,$twk2+ vxor $tweak,$tweak,$tmp++ lvx_u $in3,$x30,$inp+ sub $len,$len,$taillen+ vxor $twk3,$tweak,$rndkey0+ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ le?vperm $in3,$in3,$in3,$leperm+ vand $tmp,$tmp,$eighty7+ vxor $out3,$in3,$twk3+ vxor $tweak,$tweak,$tmp++ lvx_u $in4,$x40,$inp+ subi $len,$len,0x60+ vxor $twk4,$tweak,$rndkey0+ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ le?vperm $in4,$in4,$in4,$leperm+ vand $tmp,$tmp,$eighty7+ vxor $out4,$in4,$twk4+ vxor $tweak,$tweak,$tmp++ lvx_u $in5,$x50,$inp+ addi $inp,$inp,0x60+ vxor $twk5,$tweak,$rndkey0+ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ le?vperm $in5,$in5,$in5,$leperm+ vand $tmp,$tmp,$eighty7+ vxor $out5,$in5,$twk5+ vxor $tweak,$tweak,$tmp++ vxor v31,v31,$rndkey0+ mtctr $rounds+ b Loop_xts_enc6x++.align 5+Loop_xts_enc6x:+ vcipher $out0,$out0,v24+ vcipher $out1,$out1,v24+ vcipher $out2,$out2,v24+ vcipher $out3,$out3,v24+ vcipher $out4,$out4,v24+ vcipher $out5,$out5,v24+ lvx v24,$x20,$key_ # round[3]+ addi $key_,$key_,0x20++ vcipher $out0,$out0,v25+ vcipher $out1,$out1,v25+ vcipher $out2,$out2,v25+ vcipher $out3,$out3,v25+ vcipher $out4,$out4,v25+ vcipher $out5,$out5,v25+ lvx v25,$x10,$key_ # round[4]+ bdnz Loop_xts_enc6x++ subic $len,$len,96 # $len-=96+ vxor $in0,$twk0,v31 # xor with last round key+ vcipher $out0,$out0,v24+ vcipher $out1,$out1,v24+ vsrab $tmp,$tweak,$seven # next tweak value+ vxor $twk0,$tweak,$rndkey0+ vaddubm $tweak,$tweak,$tweak+ vcipher $out2,$out2,v24+ vcipher $out3,$out3,v24+ vsldoi $tmp,$tmp,$tmp,15+ vcipher $out4,$out4,v24+ vcipher $out5,$out5,v24++ subfe. r0,r0,r0 # borrow?-1:0+ vand $tmp,$tmp,$eighty7+ vcipher $out0,$out0,v25+ vcipher $out1,$out1,v25+ vxor $tweak,$tweak,$tmp+ vcipher $out2,$out2,v25+ vcipher $out3,$out3,v25+ vxor $in1,$twk1,v31+ vsrab $tmp,$tweak,$seven # next tweak value+ vxor $twk1,$tweak,$rndkey0+ vcipher $out4,$out4,v25+ vcipher $out5,$out5,v25++ and r0,r0,$len+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ vcipher $out0,$out0,v26+ vcipher $out1,$out1,v26+ vand $tmp,$tmp,$eighty7+ vcipher $out2,$out2,v26+ vcipher $out3,$out3,v26+ vxor $tweak,$tweak,$tmp+ vcipher $out4,$out4,v26+ vcipher $out5,$out5,v26++ add $inp,$inp,r0 # $inp is adjusted in such+ # way that at exit from the+ # loop inX-in5 are loaded+ # with last "words"+ vxor $in2,$twk2,v31+ vsrab $tmp,$tweak,$seven # next tweak value+ vxor $twk2,$tweak,$rndkey0+ vaddubm $tweak,$tweak,$tweak+ vcipher $out0,$out0,v27+ vcipher $out1,$out1,v27+ vsldoi $tmp,$tmp,$tmp,15+ vcipher $out2,$out2,v27+ vcipher $out3,$out3,v27+ vand $tmp,$tmp,$eighty7+ vcipher $out4,$out4,v27+ vcipher $out5,$out5,v27++ addi $key_,$sp,$FRAME+15 # rewind $key_+ vxor $tweak,$tweak,$tmp+ vcipher $out0,$out0,v28+ vcipher $out1,$out1,v28+ vxor $in3,$twk3,v31+ vsrab $tmp,$tweak,$seven # next tweak value+ vxor $twk3,$tweak,$rndkey0+ vcipher $out2,$out2,v28+ vcipher $out3,$out3,v28+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ vcipher $out4,$out4,v28+ vcipher $out5,$out5,v28+ lvx v24,$x00,$key_ # re-pre-load round[1]+ vand $tmp,$tmp,$eighty7++ vcipher $out0,$out0,v29+ vcipher $out1,$out1,v29+ vxor $tweak,$tweak,$tmp+ vcipher $out2,$out2,v29+ vcipher $out3,$out3,v29+ vxor $in4,$twk4,v31+ vsrab $tmp,$tweak,$seven # next tweak value+ vxor $twk4,$tweak,$rndkey0+ vcipher $out4,$out4,v29+ vcipher $out5,$out5,v29+ lvx v25,$x10,$key_ # re-pre-load round[2]+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15++ vcipher $out0,$out0,v30+ vcipher $out1,$out1,v30+ vand $tmp,$tmp,$eighty7+ vcipher $out2,$out2,v30+ vcipher $out3,$out3,v30+ vxor $tweak,$tweak,$tmp+ vcipher $out4,$out4,v30+ vcipher $out5,$out5,v30+ vxor $in5,$twk5,v31+ vsrab $tmp,$tweak,$seven # next tweak value+ vxor $twk5,$tweak,$rndkey0++ vcipherlast $out0,$out0,$in0+ lvx_u $in0,$x00,$inp # load next input block+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ vcipherlast $out1,$out1,$in1+ lvx_u $in1,$x10,$inp+ vcipherlast $out2,$out2,$in2+ le?vperm $in0,$in0,$in0,$leperm+ lvx_u $in2,$x20,$inp+ vand $tmp,$tmp,$eighty7+ vcipherlast $out3,$out3,$in3+ le?vperm $in1,$in1,$in1,$leperm+ lvx_u $in3,$x30,$inp+ vcipherlast $out4,$out4,$in4+ le?vperm $in2,$in2,$in2,$leperm+ lvx_u $in4,$x40,$inp+ vxor $tweak,$tweak,$tmp+ vcipherlast $tmp,$out5,$in5 # last block might be needed+ # in stealing mode+ le?vperm $in3,$in3,$in3,$leperm+ lvx_u $in5,$x50,$inp+ addi $inp,$inp,0x60+ le?vperm $in4,$in4,$in4,$leperm+ le?vperm $in5,$in5,$in5,$leperm++ le?vperm $out0,$out0,$out0,$leperm+ le?vperm $out1,$out1,$out1,$leperm+ stvx_u $out0,$x00,$out # store output+ vxor $out0,$in0,$twk0+ le?vperm $out2,$out2,$out2,$leperm+ stvx_u $out1,$x10,$out+ vxor $out1,$in1,$twk1+ le?vperm $out3,$out3,$out3,$leperm+ stvx_u $out2,$x20,$out+ vxor $out2,$in2,$twk2+ le?vperm $out4,$out4,$out4,$leperm+ stvx_u $out3,$x30,$out+ vxor $out3,$in3,$twk3+ le?vperm $out5,$tmp,$tmp,$leperm+ stvx_u $out4,$x40,$out+ vxor $out4,$in4,$twk4+ le?stvx_u $out5,$x50,$out+ be?stvx_u $tmp, $x50,$out+ vxor $out5,$in5,$twk5+ addi $out,$out,0x60++ mtctr $rounds+ beq Loop_xts_enc6x # did $len-=96 borrow?++ addic. $len,$len,0x60+ beq Lxts_enc6x_zero+ cmpwi $len,0x20+ blt Lxts_enc6x_one+ nop+ beq Lxts_enc6x_two+ cmpwi $len,0x40+ blt Lxts_enc6x_three+ nop+ beq Lxts_enc6x_four++Lxts_enc6x_five:+ vxor $out0,$in1,$twk0+ vxor $out1,$in2,$twk1+ vxor $out2,$in3,$twk2+ vxor $out3,$in4,$twk3+ vxor $out4,$in5,$twk4++ bl _aesp8_xts_enc5x++ le?vperm $out0,$out0,$out0,$leperm+ vmr $twk0,$twk5 # unused tweak+ le?vperm $out1,$out1,$out1,$leperm+ stvx_u $out0,$x00,$out # store output+ le?vperm $out2,$out2,$out2,$leperm+ stvx_u $out1,$x10,$out+ le?vperm $out3,$out3,$out3,$leperm+ stvx_u $out2,$x20,$out+ vxor $tmp,$out4,$twk5 # last block prep for stealing+ le?vperm $out4,$out4,$out4,$leperm+ stvx_u $out3,$x30,$out+ stvx_u $out4,$x40,$out+ addi $out,$out,0x50+ bne Lxts_enc6x_steal+ b Lxts_enc6x_done++.align 4+Lxts_enc6x_four:+ vxor $out0,$in2,$twk0+ vxor $out1,$in3,$twk1+ vxor $out2,$in4,$twk2+ vxor $out3,$in5,$twk3+ vxor $out4,$out4,$out4++ bl _aesp8_xts_enc5x++ le?vperm $out0,$out0,$out0,$leperm+ vmr $twk0,$twk4 # unused tweak+ le?vperm $out1,$out1,$out1,$leperm+ stvx_u $out0,$x00,$out # store output+ le?vperm $out2,$out2,$out2,$leperm+ stvx_u $out1,$x10,$out+ vxor $tmp,$out3,$twk4 # last block prep for stealing+ le?vperm $out3,$out3,$out3,$leperm+ stvx_u $out2,$x20,$out+ stvx_u $out3,$x30,$out+ addi $out,$out,0x40+ bne Lxts_enc6x_steal+ b Lxts_enc6x_done++.align 4+Lxts_enc6x_three:+ vxor $out0,$in3,$twk0+ vxor $out1,$in4,$twk1+ vxor $out2,$in5,$twk2+ vxor $out3,$out3,$out3+ vxor $out4,$out4,$out4++ bl _aesp8_xts_enc5x++ le?vperm $out0,$out0,$out0,$leperm+ vmr $twk0,$twk3 # unused tweak+ le?vperm $out1,$out1,$out1,$leperm+ stvx_u $out0,$x00,$out # store output+ vxor $tmp,$out2,$twk3 # last block prep for stealing+ le?vperm $out2,$out2,$out2,$leperm+ stvx_u $out1,$x10,$out+ stvx_u $out2,$x20,$out+ addi $out,$out,0x30+ bne Lxts_enc6x_steal+ b Lxts_enc6x_done++.align 4+Lxts_enc6x_two:+ vxor $out0,$in4,$twk0+ vxor $out1,$in5,$twk1+ vxor $out2,$out2,$out2+ vxor $out3,$out3,$out3+ vxor $out4,$out4,$out4++ bl _aesp8_xts_enc5x++ le?vperm $out0,$out0,$out0,$leperm+ vmr $twk0,$twk2 # unused tweak+ vxor $tmp,$out1,$twk2 # last block prep for stealing+ le?vperm $out1,$out1,$out1,$leperm+ stvx_u $out0,$x00,$out # store output+ stvx_u $out1,$x10,$out+ addi $out,$out,0x20+ bne Lxts_enc6x_steal+ b Lxts_enc6x_done++.align 4+Lxts_enc6x_one:+ vxor $out0,$in5,$twk0+ nop+Loop_xts_enc1x:+ vcipher $out0,$out0,v24+ lvx v24,$x20,$key_ # round[3]+ addi $key_,$key_,0x20++ vcipher $out0,$out0,v25+ lvx v25,$x10,$key_ # round[4]+ bdnz Loop_xts_enc1x++ add $inp,$inp,$taillen+ cmpwi $taillen,0+ vcipher $out0,$out0,v24++ subi $inp,$inp,16+ vcipher $out0,$out0,v25++ lvsr $inpperm,0,$taillen+ vcipher $out0,$out0,v26++ lvx_u $in0,0,$inp+ vcipher $out0,$out0,v27++ addi $key_,$sp,$FRAME+15 # rewind $key_+ vcipher $out0,$out0,v28+ lvx v24,$x00,$key_ # re-pre-load round[1]++ vcipher $out0,$out0,v29+ lvx v25,$x10,$key_ # re-pre-load round[2]+ vxor $twk0,$twk0,v31++ le?vperm $in0,$in0,$in0,$leperm+ vcipher $out0,$out0,v30++ vperm $in0,$in0,$in0,$inpperm+ vcipherlast $out0,$out0,$twk0++ vmr $twk0,$twk1 # unused tweak+ vxor $tmp,$out0,$twk1 # last block prep for stealing+ le?vperm $out0,$out0,$out0,$leperm+ stvx_u $out0,$x00,$out # store output+ addi $out,$out,0x10+ bne Lxts_enc6x_steal+ b Lxts_enc6x_done++.align 4+Lxts_enc6x_zero:+ cmpwi $taillen,0+ beq Lxts_enc6x_done++ add $inp,$inp,$taillen+ subi $inp,$inp,16+ lvx_u $in0,0,$inp+ lvsr $inpperm,0,$taillen # $in5 is no more+ le?vperm $in0,$in0,$in0,$leperm+ vperm $in0,$in0,$in0,$inpperm+ vxor $tmp,$tmp,$twk0+Lxts_enc6x_steal:+ vxor $in0,$in0,$twk0+ vxor $out0,$out0,$out0+ vspltisb $out1,-1+ vperm $out0,$out0,$out1,$inpperm+ vsel $out0,$in0,$tmp,$out0 # $tmp is last block, remember?++ subi r30,$out,17+ subi $out,$out,16+ mtctr $taillen+Loop_xts_enc6x_steal:+ lbzu r0,1(r30)+ stb r0,16(r30)+ bdnz Loop_xts_enc6x_steal++ li $taillen,0+ mtctr $rounds+ b Loop_xts_enc1x # one more time...++.align 4+Lxts_enc6x_done:+ ${UCMP}i $ivp,0+ beq Lxts_enc6x_ret++ vxor $tweak,$twk0,$rndkey0+ le?vperm $tweak,$tweak,$tweak,$leperm+ stvx_u $tweak,0,$ivp++Lxts_enc6x_ret:+ mtlr r11+ li r10,`$FRAME+15`+ li r11,`$FRAME+31`+ stvx $seven,r10,$sp # wipe copies of round keys+ addi r10,r10,32+ stvx $seven,r11,$sp+ addi r11,r11,32+ stvx $seven,r10,$sp+ addi r10,r10,32+ stvx $seven,r11,$sp+ addi r11,r11,32+ stvx $seven,r10,$sp+ addi r10,r10,32+ stvx $seven,r11,$sp+ addi r11,r11,32+ stvx $seven,r10,$sp+ addi r10,r10,32+ stvx $seven,r11,$sp+ addi r11,r11,32++ mtspr 256,$vrsave+ lvx v20,r10,$sp # ABI says so+ addi r10,r10,32+ lvx v21,r11,$sp+ addi r11,r11,32+ lvx v22,r10,$sp+ addi r10,r10,32+ lvx v23,r11,$sp+ addi r11,r11,32+ lvx v24,r10,$sp+ addi r10,r10,32+ lvx v25,r11,$sp+ addi r11,r11,32+ lvx v26,r10,$sp+ addi r10,r10,32+ lvx v27,r11,$sp+ addi r11,r11,32+ lvx v28,r10,$sp+ addi r10,r10,32+ lvx v29,r11,$sp+ addi r11,r11,32+ lvx v30,r10,$sp+ lvx v31,r11,$sp+ $POP r26,`$FRAME+21*16+0*$SIZE_T`($sp)+ $POP r27,`$FRAME+21*16+1*$SIZE_T`($sp)+ $POP r28,`$FRAME+21*16+2*$SIZE_T`($sp)+ $POP r29,`$FRAME+21*16+3*$SIZE_T`($sp)+ $POP r30,`$FRAME+21*16+4*$SIZE_T`($sp)+ $POP r31,`$FRAME+21*16+5*$SIZE_T`($sp)+ addi $sp,$sp,`$FRAME+21*16+6*$SIZE_T`+ blr+ .long 0+ .byte 0,12,0x04,1,0x80,6,6,0+ .long 0++.align 5+_aesp8_xts_enc5x:+ vcipher $out0,$out0,v24+ vcipher $out1,$out1,v24+ vcipher $out2,$out2,v24+ vcipher $out3,$out3,v24+ vcipher $out4,$out4,v24+ lvx v24,$x20,$key_ # round[3]+ addi $key_,$key_,0x20++ vcipher $out0,$out0,v25+ vcipher $out1,$out1,v25+ vcipher $out2,$out2,v25+ vcipher $out3,$out3,v25+ vcipher $out4,$out4,v25+ lvx v25,$x10,$key_ # round[4]+ bdnz _aesp8_xts_enc5x++ add $inp,$inp,$taillen+ cmpwi $taillen,0+ vcipher $out0,$out0,v24+ vcipher $out1,$out1,v24+ vcipher $out2,$out2,v24+ vcipher $out3,$out3,v24+ vcipher $out4,$out4,v24++ subi $inp,$inp,16+ vcipher $out0,$out0,v25+ vcipher $out1,$out1,v25+ vcipher $out2,$out2,v25+ vcipher $out3,$out3,v25+ vcipher $out4,$out4,v25+ vxor $twk0,$twk0,v31++ vcipher $out0,$out0,v26+ lvsr $inpperm,0,$taillen # $in5 is no more+ vcipher $out1,$out1,v26+ vcipher $out2,$out2,v26+ vcipher $out3,$out3,v26+ vcipher $out4,$out4,v26+ vxor $in1,$twk1,v31++ vcipher $out0,$out0,v27+ lvx_u $in0,0,$inp+ vcipher $out1,$out1,v27+ vcipher $out2,$out2,v27+ vcipher $out3,$out3,v27+ vcipher $out4,$out4,v27+ vxor $in2,$twk2,v31++ addi $key_,$sp,$FRAME+15 # rewind $key_+ vcipher $out0,$out0,v28+ vcipher $out1,$out1,v28+ vcipher $out2,$out2,v28+ vcipher $out3,$out3,v28+ vcipher $out4,$out4,v28+ lvx v24,$x00,$key_ # re-pre-load round[1]+ vxor $in3,$twk3,v31++ vcipher $out0,$out0,v29+ le?vperm $in0,$in0,$in0,$leperm+ vcipher $out1,$out1,v29+ vcipher $out2,$out2,v29+ vcipher $out3,$out3,v29+ vcipher $out4,$out4,v29+ lvx v25,$x10,$key_ # re-pre-load round[2]+ vxor $in4,$twk4,v31++ vcipher $out0,$out0,v30+ vperm $in0,$in0,$in0,$inpperm+ vcipher $out1,$out1,v30+ vcipher $out2,$out2,v30+ vcipher $out3,$out3,v30+ vcipher $out4,$out4,v30++ vcipherlast $out0,$out0,$twk0+ vcipherlast $out1,$out1,$in1+ vcipherlast $out2,$out2,$in2+ vcipherlast $out3,$out3,$in3+ vcipherlast $out4,$out4,$in4+ blr+ .long 0+ .byte 0,12,0x14,0,0,0,0,0++.align 5+_aesp8_xts_decrypt6x:+ $STU $sp,-`($FRAME+21*16+6*$SIZE_T)`($sp)+ mflr r11+ li r7,`$FRAME+8*16+15`+ li r3,`$FRAME+8*16+31`+ $PUSH r11,`$FRAME+21*16+6*$SIZE_T+$LRSAVE`($sp)+ stvx v20,r7,$sp # ABI says so+ addi r7,r7,32+ stvx v21,r3,$sp+ addi r3,r3,32+ stvx v22,r7,$sp+ addi r7,r7,32+ stvx v23,r3,$sp+ addi r3,r3,32+ stvx v24,r7,$sp+ addi r7,r7,32+ stvx v25,r3,$sp+ addi r3,r3,32+ stvx v26,r7,$sp+ addi r7,r7,32+ stvx v27,r3,$sp+ addi r3,r3,32+ stvx v28,r7,$sp+ addi r7,r7,32+ stvx v29,r3,$sp+ addi r3,r3,32+ stvx v30,r7,$sp+ stvx v31,r3,$sp+ li r0,-1+ stw $vrsave,`$FRAME+21*16-4`($sp) # save vrsave+ li $x10,0x10+ $PUSH r26,`$FRAME+21*16+0*$SIZE_T`($sp)+ li $x20,0x20+ $PUSH r27,`$FRAME+21*16+1*$SIZE_T`($sp)+ li $x30,0x30+ $PUSH r28,`$FRAME+21*16+2*$SIZE_T`($sp)+ li $x40,0x40+ $PUSH r29,`$FRAME+21*16+3*$SIZE_T`($sp)+ li $x50,0x50+ $PUSH r30,`$FRAME+21*16+4*$SIZE_T`($sp)+ li $x60,0x60+ $PUSH r31,`$FRAME+21*16+5*$SIZE_T`($sp)+ li $x70,0x70+ mtspr 256,r0++ subi $rounds,$rounds,3 # -4 in total++ lvx $rndkey0,$x00,$key1 # load key schedule+ lvx v30,$x10,$key1+ addi $key1,$key1,0x20+ lvx v31,$x00,$key1+ ?vperm $rndkey0,$rndkey0,v30,$keyperm+ addi $key_,$sp,$FRAME+15+ mtctr $rounds++Load_xts_dec_key:+ ?vperm v24,v30,v31,$keyperm+ lvx v30,$x10,$key1+ addi $key1,$key1,0x20+ stvx v24,$x00,$key_ # off-load round[1]+ ?vperm v25,v31,v30,$keyperm+ lvx v31,$x00,$key1+ stvx v25,$x10,$key_ # off-load round[2]+ addi $key_,$key_,0x20+ bdnz Load_xts_dec_key++ lvx v26,$x10,$key1+ ?vperm v24,v30,v31,$keyperm+ lvx v27,$x20,$key1+ stvx v24,$x00,$key_ # off-load round[3]+ ?vperm v25,v31,v26,$keyperm+ lvx v28,$x30,$key1+ stvx v25,$x10,$key_ # off-load round[4]+ addi $key_,$sp,$FRAME+15 # rewind $key_+ ?vperm v26,v26,v27,$keyperm+ lvx v29,$x40,$key1+ ?vperm v27,v27,v28,$keyperm+ lvx v30,$x50,$key1+ ?vperm v28,v28,v29,$keyperm+ lvx v31,$x60,$key1+ ?vperm v29,v29,v30,$keyperm+ lvx $twk5,$x70,$key1 # borrow $twk5+ ?vperm v30,v30,v31,$keyperm+ lvx v24,$x00,$key_ # pre-load round[1]+ ?vperm v31,v31,$twk5,$keyperm+ lvx v25,$x10,$key_ # pre-load round[2]++ vperm $in0,$inout,$inptail,$inpperm+ subi $inp,$inp,31 # undo "caller"+ vxor $twk0,$tweak,$rndkey0+ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ vand $tmp,$tmp,$eighty7+ vxor $out0,$in0,$twk0+ vxor $tweak,$tweak,$tmp++ lvx_u $in1,$x10,$inp+ vxor $twk1,$tweak,$rndkey0+ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ le?vperm $in1,$in1,$in1,$leperm+ vand $tmp,$tmp,$eighty7+ vxor $out1,$in1,$twk1+ vxor $tweak,$tweak,$tmp++ lvx_u $in2,$x20,$inp+ andi. $taillen,$len,15+ vxor $twk2,$tweak,$rndkey0+ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ le?vperm $in2,$in2,$in2,$leperm+ vand $tmp,$tmp,$eighty7+ vxor $out2,$in2,$twk2+ vxor $tweak,$tweak,$tmp++ lvx_u $in3,$x30,$inp+ sub $len,$len,$taillen+ vxor $twk3,$tweak,$rndkey0+ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ le?vperm $in3,$in3,$in3,$leperm+ vand $tmp,$tmp,$eighty7+ vxor $out3,$in3,$twk3+ vxor $tweak,$tweak,$tmp++ lvx_u $in4,$x40,$inp+ subi $len,$len,0x60+ vxor $twk4,$tweak,$rndkey0+ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ le?vperm $in4,$in4,$in4,$leperm+ vand $tmp,$tmp,$eighty7+ vxor $out4,$in4,$twk4+ vxor $tweak,$tweak,$tmp++ lvx_u $in5,$x50,$inp+ addi $inp,$inp,0x60+ vxor $twk5,$tweak,$rndkey0+ vsrab $tmp,$tweak,$seven # next tweak value+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ le?vperm $in5,$in5,$in5,$leperm+ vand $tmp,$tmp,$eighty7+ vxor $out5,$in5,$twk5+ vxor $tweak,$tweak,$tmp++ vxor v31,v31,$rndkey0+ mtctr $rounds+ b Loop_xts_dec6x++.align 5+Loop_xts_dec6x:+ vncipher $out0,$out0,v24+ vncipher $out1,$out1,v24+ vncipher $out2,$out2,v24+ vncipher $out3,$out3,v24+ vncipher $out4,$out4,v24+ vncipher $out5,$out5,v24+ lvx v24,$x20,$key_ # round[3]+ addi $key_,$key_,0x20++ vncipher $out0,$out0,v25+ vncipher $out1,$out1,v25+ vncipher $out2,$out2,v25+ vncipher $out3,$out3,v25+ vncipher $out4,$out4,v25+ vncipher $out5,$out5,v25+ lvx v25,$x10,$key_ # round[4]+ bdnz Loop_xts_dec6x++ subic $len,$len,96 # $len-=96+ vxor $in0,$twk0,v31 # xor with last round key+ vncipher $out0,$out0,v24+ vncipher $out1,$out1,v24+ vsrab $tmp,$tweak,$seven # next tweak value+ vxor $twk0,$tweak,$rndkey0+ vaddubm $tweak,$tweak,$tweak+ vncipher $out2,$out2,v24+ vncipher $out3,$out3,v24+ vsldoi $tmp,$tmp,$tmp,15+ vncipher $out4,$out4,v24+ vncipher $out5,$out5,v24++ subfe. r0,r0,r0 # borrow?-1:0+ vand $tmp,$tmp,$eighty7+ vncipher $out0,$out0,v25+ vncipher $out1,$out1,v25+ vxor $tweak,$tweak,$tmp+ vncipher $out2,$out2,v25+ vncipher $out3,$out3,v25+ vxor $in1,$twk1,v31+ vsrab $tmp,$tweak,$seven # next tweak value+ vxor $twk1,$tweak,$rndkey0+ vncipher $out4,$out4,v25+ vncipher $out5,$out5,v25++ and r0,r0,$len+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ vncipher $out0,$out0,v26+ vncipher $out1,$out1,v26+ vand $tmp,$tmp,$eighty7+ vncipher $out2,$out2,v26+ vncipher $out3,$out3,v26+ vxor $tweak,$tweak,$tmp+ vncipher $out4,$out4,v26+ vncipher $out5,$out5,v26++ add $inp,$inp,r0 # $inp is adjusted in such+ # way that at exit from the+ # loop inX-in5 are loaded+ # with last "words"+ vxor $in2,$twk2,v31+ vsrab $tmp,$tweak,$seven # next tweak value+ vxor $twk2,$tweak,$rndkey0+ vaddubm $tweak,$tweak,$tweak+ vncipher $out0,$out0,v27+ vncipher $out1,$out1,v27+ vsldoi $tmp,$tmp,$tmp,15+ vncipher $out2,$out2,v27+ vncipher $out3,$out3,v27+ vand $tmp,$tmp,$eighty7+ vncipher $out4,$out4,v27+ vncipher $out5,$out5,v27++ addi $key_,$sp,$FRAME+15 # rewind $key_+ vxor $tweak,$tweak,$tmp+ vncipher $out0,$out0,v28+ vncipher $out1,$out1,v28+ vxor $in3,$twk3,v31+ vsrab $tmp,$tweak,$seven # next tweak value+ vxor $twk3,$tweak,$rndkey0+ vncipher $out2,$out2,v28+ vncipher $out3,$out3,v28+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ vncipher $out4,$out4,v28+ vncipher $out5,$out5,v28+ lvx v24,$x00,$key_ # re-pre-load round[1]+ vand $tmp,$tmp,$eighty7++ vncipher $out0,$out0,v29+ vncipher $out1,$out1,v29+ vxor $tweak,$tweak,$tmp+ vncipher $out2,$out2,v29+ vncipher $out3,$out3,v29+ vxor $in4,$twk4,v31+ vsrab $tmp,$tweak,$seven # next tweak value+ vxor $twk4,$tweak,$rndkey0+ vncipher $out4,$out4,v29+ vncipher $out5,$out5,v29+ lvx v25,$x10,$key_ # re-pre-load round[2]+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15++ vncipher $out0,$out0,v30+ vncipher $out1,$out1,v30+ vand $tmp,$tmp,$eighty7+ vncipher $out2,$out2,v30+ vncipher $out3,$out3,v30+ vxor $tweak,$tweak,$tmp+ vncipher $out4,$out4,v30+ vncipher $out5,$out5,v30+ vxor $in5,$twk5,v31+ vsrab $tmp,$tweak,$seven # next tweak value+ vxor $twk5,$tweak,$rndkey0++ vncipherlast $out0,$out0,$in0+ lvx_u $in0,$x00,$inp # load next input block+ vaddubm $tweak,$tweak,$tweak+ vsldoi $tmp,$tmp,$tmp,15+ vncipherlast $out1,$out1,$in1+ lvx_u $in1,$x10,$inp+ vncipherlast $out2,$out2,$in2+ le?vperm $in0,$in0,$in0,$leperm+ lvx_u $in2,$x20,$inp+ vand $tmp,$tmp,$eighty7+ vncipherlast $out3,$out3,$in3+ le?vperm $in1,$in1,$in1,$leperm+ lvx_u $in3,$x30,$inp+ vncipherlast $out4,$out4,$in4+ le?vperm $in2,$in2,$in2,$leperm+ lvx_u $in4,$x40,$inp+ vxor $tweak,$tweak,$tmp+ vncipherlast $out5,$out5,$in5+ le?vperm $in3,$in3,$in3,$leperm+ lvx_u $in5,$x50,$inp+ addi $inp,$inp,0x60+ le?vperm $in4,$in4,$in4,$leperm+ le?vperm $in5,$in5,$in5,$leperm++ le?vperm $out0,$out0,$out0,$leperm+ le?vperm $out1,$out1,$out1,$leperm+ stvx_u $out0,$x00,$out # store output+ vxor $out0,$in0,$twk0+ le?vperm $out2,$out2,$out2,$leperm+ stvx_u $out1,$x10,$out+ vxor $out1,$in1,$twk1+ le?vperm $out3,$out3,$out3,$leperm+ stvx_u $out2,$x20,$out+ vxor $out2,$in2,$twk2+ le?vperm $out4,$out4,$out4,$leperm+ stvx_u $out3,$x30,$out+ vxor $out3,$in3,$twk3+ le?vperm $out5,$out5,$out5,$leperm+ stvx_u $out4,$x40,$out+ vxor $out4,$in4,$twk4+ stvx_u $out5,$x50,$out+ vxor $out5,$in5,$twk5+ addi $out,$out,0x60++ mtctr $rounds+ beq Loop_xts_dec6x # did $len-=96 borrow?++ addic. $len,$len,0x60+ beq Lxts_dec6x_zero+ cmpwi $len,0x20+ blt Lxts_dec6x_one+ nop+ beq Lxts_dec6x_two+ cmpwi $len,0x40+ blt Lxts_dec6x_three+ nop+ beq Lxts_dec6x_four++Lxts_dec6x_five:+ vxor $out0,$in1,$twk0+ vxor $out1,$in2,$twk1+ vxor $out2,$in3,$twk2+ vxor $out3,$in4,$twk3+ vxor $out4,$in5,$twk4++ bl _aesp8_xts_dec5x++ le?vperm $out0,$out0,$out0,$leperm+ vmr $twk0,$twk5 # unused tweak+ vxor $twk1,$tweak,$rndkey0+ le?vperm $out1,$out1,$out1,$leperm+ stvx_u $out0,$x00,$out # store output+ vxor $out0,$in0,$twk1+ le?vperm $out2,$out2,$out2,$leperm+ stvx_u $out1,$x10,$out+ le?vperm $out3,$out3,$out3,$leperm+ stvx_u $out2,$x20,$out+ le?vperm $out4,$out4,$out4,$leperm+ stvx_u $out3,$x30,$out+ stvx_u $out4,$x40,$out+ addi $out,$out,0x50+ bne Lxts_dec6x_steal+ b Lxts_dec6x_done++.align 4+Lxts_dec6x_four:+ vxor $out0,$in2,$twk0+ vxor $out1,$in3,$twk1+ vxor $out2,$in4,$twk2+ vxor $out3,$in5,$twk3+ vxor $out4,$out4,$out4++ bl _aesp8_xts_dec5x++ le?vperm $out0,$out0,$out0,$leperm+ vmr $twk0,$twk4 # unused tweak+ vmr $twk1,$twk5+ le?vperm $out1,$out1,$out1,$leperm+ stvx_u $out0,$x00,$out # store output+ vxor $out0,$in0,$twk5+ le?vperm $out2,$out2,$out2,$leperm+ stvx_u $out1,$x10,$out+ le?vperm $out3,$out3,$out3,$leperm+ stvx_u $out2,$x20,$out+ stvx_u $out3,$x30,$out+ addi $out,$out,0x40+ bne Lxts_dec6x_steal+ b Lxts_dec6x_done++.align 4+Lxts_dec6x_three:+ vxor $out0,$in3,$twk0+ vxor $out1,$in4,$twk1+ vxor $out2,$in5,$twk2+ vxor $out3,$out3,$out3+ vxor $out4,$out4,$out4++ bl _aesp8_xts_dec5x++ le?vperm $out0,$out0,$out0,$leperm+ vmr $twk0,$twk3 # unused tweak+ vmr $twk1,$twk4+ le?vperm $out1,$out1,$out1,$leperm+ stvx_u $out0,$x00,$out # store output+ vxor $out0,$in0,$twk4+ le?vperm $out2,$out2,$out2,$leperm+ stvx_u $out1,$x10,$out+ stvx_u $out2,$x20,$out+ addi $out,$out,0x30+ bne Lxts_dec6x_steal+ b Lxts_dec6x_done++.align 4+Lxts_dec6x_two:+ vxor $out0,$in4,$twk0+ vxor $out1,$in5,$twk1+ vxor $out2,$out2,$out2+ vxor $out3,$out3,$out3+ vxor $out4,$out4,$out4++ bl _aesp8_xts_dec5x++ le?vperm $out0,$out0,$out0,$leperm+ vmr $twk0,$twk2 # unused tweak+ vmr $twk1,$twk3+ le?vperm $out1,$out1,$out1,$leperm+ stvx_u $out0,$x00,$out # store output+ vxor $out0,$in0,$twk3+ stvx_u $out1,$x10,$out+ addi $out,$out,0x20+ bne Lxts_dec6x_steal+ b Lxts_dec6x_done++.align 4+Lxts_dec6x_one:+ vxor $out0,$in5,$twk0+ nop+Loop_xts_dec1x:+ vncipher $out0,$out0,v24+ lvx v24,$x20,$key_ # round[3]+ addi $key_,$key_,0x20++ vncipher $out0,$out0,v25+ lvx v25,$x10,$key_ # round[4]+ bdnz Loop_xts_dec1x++ subi r0,$taillen,1+ vncipher $out0,$out0,v24++ andi. r0,r0,16+ cmpwi $taillen,0+ vncipher $out0,$out0,v25++ sub $inp,$inp,r0+ vncipher $out0,$out0,v26++ lvx_u $in0,0,$inp+ vncipher $out0,$out0,v27++ addi $key_,$sp,$FRAME+15 # rewind $key_+ vncipher $out0,$out0,v28+ lvx v24,$x00,$key_ # re-pre-load round[1]++ vncipher $out0,$out0,v29+ lvx v25,$x10,$key_ # re-pre-load round[2]+ vxor $twk0,$twk0,v31++ le?vperm $in0,$in0,$in0,$leperm+ vncipher $out0,$out0,v30++ mtctr $rounds+ vncipherlast $out0,$out0,$twk0++ vmr $twk0,$twk1 # unused tweak+ vmr $twk1,$twk2+ le?vperm $out0,$out0,$out0,$leperm+ stvx_u $out0,$x00,$out # store output+ addi $out,$out,0x10+ vxor $out0,$in0,$twk2+ bne Lxts_dec6x_steal+ b Lxts_dec6x_done++.align 4+Lxts_dec6x_zero:+ cmpwi $taillen,0+ beq Lxts_dec6x_done++ lvx_u $in0,0,$inp+ le?vperm $in0,$in0,$in0,$leperm+ vxor $out0,$in0,$twk1+Lxts_dec6x_steal:+ vncipher $out0,$out0,v24+ lvx v24,$x20,$key_ # round[3]+ addi $key_,$key_,0x20++ vncipher $out0,$out0,v25+ lvx v25,$x10,$key_ # round[4]+ bdnz Lxts_dec6x_steal++ add $inp,$inp,$taillen+ vncipher $out0,$out0,v24++ cmpwi $taillen,0+ vncipher $out0,$out0,v25++ lvx_u $in0,0,$inp+ vncipher $out0,$out0,v26++ lvsr $inpperm,0,$taillen # $in5 is no more+ vncipher $out0,$out0,v27++ addi $key_,$sp,$FRAME+15 # rewind $key_+ vncipher $out0,$out0,v28+ lvx v24,$x00,$key_ # re-pre-load round[1]++ vncipher $out0,$out0,v29+ lvx v25,$x10,$key_ # re-pre-load round[2]+ vxor $twk1,$twk1,v31++ le?vperm $in0,$in0,$in0,$leperm+ vncipher $out0,$out0,v30++ vperm $in0,$in0,$in0,$inpperm+ vncipherlast $tmp,$out0,$twk1++ le?vperm $out0,$tmp,$tmp,$leperm+ le?stvx_u $out0,0,$out+ be?stvx_u $tmp,0,$out++ vxor $out0,$out0,$out0+ vspltisb $out1,-1+ vperm $out0,$out0,$out1,$inpperm+ vsel $out0,$in0,$tmp,$out0+ vxor $out0,$out0,$twk0++ subi r30,$out,1+ mtctr $taillen+Loop_xts_dec6x_steal:+ lbzu r0,1(r30)+ stb r0,16(r30)+ bdnz Loop_xts_dec6x_steal++ li $taillen,0+ mtctr $rounds+ b Loop_xts_dec1x # one more time...++.align 4+Lxts_dec6x_done:+ ${UCMP}i $ivp,0+ beq Lxts_dec6x_ret++ vxor $tweak,$twk0,$rndkey0+ le?vperm $tweak,$tweak,$tweak,$leperm+ stvx_u $tweak,0,$ivp++Lxts_dec6x_ret:+ mtlr r11+ li r10,`$FRAME+15`+ li r11,`$FRAME+31`+ stvx $seven,r10,$sp # wipe copies of round keys+ addi r10,r10,32+ stvx $seven,r11,$sp+ addi r11,r11,32+ stvx $seven,r10,$sp+ addi r10,r10,32+ stvx $seven,r11,$sp+ addi r11,r11,32+ stvx $seven,r10,$sp+ addi r10,r10,32+ stvx $seven,r11,$sp+ addi r11,r11,32+ stvx $seven,r10,$sp+ addi r10,r10,32+ stvx $seven,r11,$sp+ addi r11,r11,32++ mtspr 256,$vrsave+ lvx v20,r10,$sp # ABI says so+ addi r10,r10,32+ lvx v21,r11,$sp+ addi r11,r11,32+ lvx v22,r10,$sp+ addi r10,r10,32+ lvx v23,r11,$sp+ addi r11,r11,32+ lvx v24,r10,$sp+ addi r10,r10,32+ lvx v25,r11,$sp+ addi r11,r11,32+ lvx v26,r10,$sp+ addi r10,r10,32+ lvx v27,r11,$sp+ addi r11,r11,32+ lvx v28,r10,$sp+ addi r10,r10,32+ lvx v29,r11,$sp+ addi r11,r11,32+ lvx v30,r10,$sp+ lvx v31,r11,$sp+ $POP r26,`$FRAME+21*16+0*$SIZE_T`($sp)+ $POP r27,`$FRAME+21*16+1*$SIZE_T`($sp)+ $POP r28,`$FRAME+21*16+2*$SIZE_T`($sp)+ $POP r29,`$FRAME+21*16+3*$SIZE_T`($sp)+ $POP r30,`$FRAME+21*16+4*$SIZE_T`($sp)+ $POP r31,`$FRAME+21*16+5*$SIZE_T`($sp)+ addi $sp,$sp,`$FRAME+21*16+6*$SIZE_T`+ blr+ .long 0+ .byte 0,12,0x04,1,0x80,6,6,0+ .long 0++.align 5+_aesp8_xts_dec5x:+ vncipher $out0,$out0,v24+ vncipher $out1,$out1,v24+ vncipher $out2,$out2,v24+ vncipher $out3,$out3,v24+ vncipher $out4,$out4,v24+ lvx v24,$x20,$key_ # round[3]+ addi $key_,$key_,0x20++ vncipher $out0,$out0,v25+ vncipher $out1,$out1,v25+ vncipher $out2,$out2,v25+ vncipher $out3,$out3,v25+ vncipher $out4,$out4,v25+ lvx v25,$x10,$key_ # round[4]+ bdnz _aesp8_xts_dec5x++ subi r0,$taillen,1+ vncipher $out0,$out0,v24+ vncipher $out1,$out1,v24+ vncipher $out2,$out2,v24+ vncipher $out3,$out3,v24+ vncipher $out4,$out4,v24++ andi. r0,r0,16+ cmpwi $taillen,0+ vncipher $out0,$out0,v25+ vncipher $out1,$out1,v25+ vncipher $out2,$out2,v25+ vncipher $out3,$out3,v25+ vncipher $out4,$out4,v25+ vxor $twk0,$twk0,v31++ sub $inp,$inp,r0+ vncipher $out0,$out0,v26+ vncipher $out1,$out1,v26+ vncipher $out2,$out2,v26+ vncipher $out3,$out3,v26+ vncipher $out4,$out4,v26+ vxor $in1,$twk1,v31++ vncipher $out0,$out0,v27+ lvx_u $in0,0,$inp+ vncipher $out1,$out1,v27+ vncipher $out2,$out2,v27+ vncipher $out3,$out3,v27+ vncipher $out4,$out4,v27+ vxor $in2,$twk2,v31++ addi $key_,$sp,$FRAME+15 # rewind $key_+ vncipher $out0,$out0,v28+ vncipher $out1,$out1,v28+ vncipher $out2,$out2,v28+ vncipher $out3,$out3,v28+ vncipher $out4,$out4,v28+ lvx v24,$x00,$key_ # re-pre-load round[1]+ vxor $in3,$twk3,v31++ vncipher $out0,$out0,v29+ le?vperm $in0,$in0,$in0,$leperm+ vncipher $out1,$out1,v29+ vncipher $out2,$out2,v29+ vncipher $out3,$out3,v29+ vncipher $out4,$out4,v29+ lvx v25,$x10,$key_ # re-pre-load round[2]+ vxor $in4,$twk4,v31++ vncipher $out0,$out0,v30+ vncipher $out1,$out1,v30+ vncipher $out2,$out2,v30+ vncipher $out3,$out3,v30+ vncipher $out4,$out4,v30++ vncipherlast $out0,$out0,$twk0+ vncipherlast $out1,$out1,$in1+ vncipherlast $out2,$out2,$in2+ vncipherlast $out3,$out3,$in3+ vncipherlast $out4,$out4,$in4+ mtctr $rounds+ blr+ .long 0+ .byte 0,12,0x14,0,0,0,0,0+___+}} }}}++my $consts=1;+foreach(split("\n",$code)) {+ s/\`([^\`]*)\`/eval($1)/geo;++ # constants table endian-specific conversion+ if ($consts && m/\.(long|byte)\s+(.+)\s+(\?[a-z]*)$/o) {+ my $conv=$3;+ my @bytes=();++ # convert to endian-agnostic format+ if ($1 eq "long") {+ foreach (split(/,\s*/,$2)) {+ my $l = /^0/?oct:int;+ push @bytes,($l>>24)&0xff,($l>>16)&0xff,($l>>8)&0xff,$l&0xff;+ }+ } else {+ @bytes = map(/^0/?oct:int,split(/,\s*/,$2));+ }++ # little-endian conversion+ if ($flavour =~ /le$/o) {+ SWITCH: for($conv) {+ /\?inv/ && do { @bytes=map($_^0xf,@bytes); last; };+ /\?rev/ && do { @bytes=reverse(@bytes); last; };+ }+ }++ #emit+ print ".byte\t",join(',',map (sprintf("0x%02x",$_),@bytes)),"\n";+ next;+ }+ $consts=0 if (m/Lconsts:/o); # end of table++ # instructions prefixed with '?' are endian-specific and need+ # to be adjusted accordingly...+ if ($flavour =~ /le$/o) { # little-endian+ s/le\?//o or+ s/be\?/#be#/o or+ s/\?lvsr/lvsl/o or+ s/\?lvsl/lvsr/o or+ s/\?(vperm\s+v[0-9]+,\s*)(v[0-9]+,\s*)(v[0-9]+,\s*)(v[0-9]+)/$1$3$2$4/o or+ s/\?(vsldoi\s+v[0-9]+,\s*)(v[0-9]+,)\s*(v[0-9]+,\s*)([0-9]+)/$1$3$2 16-$4/o or+ s/\?(vspltw\s+v[0-9]+,\s*)(v[0-9]+,)\s*([0-9])/$1$2 3-$3/o;+ } else { # big-endian+ s/le\?/#le#/o or+ s/be\?//o or+ s/\?([a-z]+)/$1/o;+ }++ print $_,"\n";+}++close STDOUT;
cbits/asm/generate.sh view
@@ -12,6 +12,8 @@ # arm/sha1-armv8.pl arm/sha512-armv8.pl # arm/keccak1600-armv8.pl arm/arm-xlate.pl # arm/arm_arch.h+# ppc/aesp8-ppc.pl ppc/ghashp8-ppc.pl+# ppc/ppc-xlate.pl # # The .pl files are the generator, not the product: each one emits # assembly for a given "flavour", which is the calling convention and the@@ -150,4 +152,25 @@ .section .note.GNU-stack,"",%progbits NOTE+done++# POWER8's AES and GHASH, little-endian only. The generator emits a+# big-endian flavour from the same source, and crypton does not build for+# big-endian POWER: nothing here has been able to run that code, and an AES+# path that has never been executed is not one to check in. Adding+# "linux64" to the list below is what it would take.+#+# These two read no capability word of their own -- unlike the ARM and x86-64+# modules above -- so the entry points are all that is renamed.+for flavour in linux64le; do+ perl aesp8-ppc.pl $flavour tmp-$flavour.S+ sed -e 's/\baes_p8_/crypton_aes_p8_/g' \+ tmp-$flavour.S > aesp8-ppc-$flavour.S++ perl ghashp8-ppc.pl $flavour tmp-$flavour.S+ sed -e 's/\bgcm_init_p8/crypton_gcm_init_p8/g' \+ -e 's/\bgcm_gmult_p8/crypton_gcm_gmult_p8/g' \+ -e 's/\bgcm_ghash_p8/crypton_gcm_ghash_p8/g' \+ tmp-$flavour.S > ghashp8-ppc-$flavour.S+ rm -f tmp-$flavour.S done
+ cbits/asm/ghashp8-ppc-linux64le.S view
@@ -0,0 +1,575 @@+.machine "any"++.abiversion 2+.text++.globl crypton_gcm_init_p8+.type crypton_gcm_init_p8,@function+.align 5+crypton_gcm_init_p8:+.localentry crypton_gcm_init_p8,0++ li 0,-4096+ li 8,0x10+ li 12,-1+ li 9,0x20+ or 0,0,0+ li 10,0x30+ .long 0x7D202699++ vspltisb 8,-16+ vspltisb 5,1+ vaddubm 8,8,8+ vxor 4,4,4+ vor 8,8,5+ vsldoi 8,8,4,15+ vsldoi 6,4,5,1+ vaddubm 8,8,8+ vspltisb 7,7+ vor 8,8,6+ vspltb 6,9,0+ vsl 9,9,5+ vsrab 6,6,7+ vand 6,6,8+ vxor 3,9,6++ vsldoi 9,3,3,8+ vsldoi 8,4,8,8+ vsldoi 11,4,9,8+ vsldoi 10,9,4,8++ .long 0x7D001F99+ .long 0x7D681F99+ li 8,0x40+ .long 0x7D291F99+ li 9,0x50+ .long 0x7D4A1F99+ li 10,0x60++ .long 0x10035CC8+ .long 0x10234CC8+ .long 0x104354C8++ .long 0x10E044C8++ vsldoi 5,1,4,8+ vsldoi 6,4,1,8+ vxor 0,0,5+ vxor 2,2,6++ vsldoi 0,0,0,8+ vxor 0,0,7++ vsldoi 6,0,0,8+ .long 0x100044C8+ vxor 6,6,2+ vxor 16,0,6++ vsldoi 17,16,16,8+ vsldoi 19,4,17,8+ vsldoi 18,17,4,8++ .long 0x7E681F99+ li 8,0x70+ .long 0x7E291F99+ li 9,0x80+ .long 0x7E4A1F99+ li 10,0x90+ .long 0x10039CC8+ .long 0x11B09CC8+ .long 0x10238CC8+ .long 0x11D08CC8+ .long 0x104394C8+ .long 0x11F094C8++ .long 0x10E044C8+ .long 0x114D44C8++ vsldoi 5,1,4,8+ vsldoi 6,4,1,8+ vsldoi 11,14,4,8+ vsldoi 9,4,14,8+ vxor 0,0,5+ vxor 2,2,6+ vxor 13,13,11+ vxor 15,15,9++ vsldoi 0,0,0,8+ vsldoi 13,13,13,8+ vxor 0,0,7+ vxor 13,13,10++ vsldoi 6,0,0,8+ vsldoi 9,13,13,8+ .long 0x100044C8+ .long 0x11AD44C8+ vxor 6,6,2+ vxor 9,9,15+ vxor 0,0,6+ vxor 13,13,9++ vsldoi 9,0,0,8+ vsldoi 17,13,13,8+ vsldoi 11,4,9,8+ vsldoi 10,9,4,8+ vsldoi 19,4,17,8+ vsldoi 18,17,4,8++ .long 0x7D681F99+ li 8,0xa0+ .long 0x7D291F99+ li 9,0xb0+ .long 0x7D4A1F99+ li 10,0xc0+ .long 0x7E681F99+ .long 0x7E291F99+ .long 0x7E4A1F99++ or 12,12,12+ blr +.long 0+.byte 0,12,0x14,0,0,0,2,0+.long 0+.size crypton_gcm_init_p8,.-crypton_gcm_init_p8+.globl crypton_gcm_gmult_p8+.type crypton_gcm_gmult_p8,@function+.align 5+crypton_gcm_gmult_p8:+.localentry crypton_gcm_gmult_p8,0++ lis 0,0xfff8+ li 8,0x10+ li 12,-1+ li 9,0x20+ or 0,0,0+ li 10,0x30+ .long 0x7C601E99++ .long 0x7D682699+ lvsl 12,0,0+ .long 0x7D292699+ vspltisb 5,0x07+ .long 0x7D4A2699+ vxor 12,12,5+ .long 0x7D002699+ vperm 3,3,3,12+ vxor 4,4,4++ .long 0x10035CC8+ .long 0x10234CC8+ .long 0x104354C8++ .long 0x10E044C8++ vsldoi 5,1,4,8+ vsldoi 6,4,1,8+ vxor 0,0,5+ vxor 2,2,6++ vsldoi 0,0,0,8+ vxor 0,0,7++ vsldoi 6,0,0,8+ .long 0x100044C8+ vxor 6,6,2+ vxor 0,0,6++ vperm 0,0,0,12+ .long 0x7C001F99++ or 12,12,12+ blr +.long 0+.byte 0,12,0x14,0,0,0,2,0+.long 0+.size crypton_gcm_gmult_p8,.-crypton_gcm_gmult_p8++.globl crypton_gcm_ghash_p8+.type crypton_gcm_ghash_p8,@function+.align 5+crypton_gcm_ghash_p8:+.localentry crypton_gcm_ghash_p8,0++ li 0,-4096+ li 8,0x10+ li 12,-1+ li 9,0x20+ or 0,0,0+ li 10,0x30+ .long 0x7C001E99++ .long 0x7D682699+ li 8,0x40+ lvsl 12,0,0+ .long 0x7D292699+ li 9,0x50+ vspltisb 5,0x07+ .long 0x7D4A2699+ li 10,0x60+ vxor 12,12,5+ .long 0x7D002699+ vperm 0,0,0,12+ vxor 4,4,4++ cmpldi 6,64+ bge .Lgcm_ghash_p8_4x++ .long 0x7C602E99+ addi 5,5,16+ subic. 6,6,16+ vperm 3,3,3,12+ vxor 3,3,0+ beq .Lshort++ .long 0x7E682699+ li 8,16+ .long 0x7E292699+ add 9,5,6+ .long 0x7E4A2699+++.align 5+.Loop_2x:+ .long 0x7E002E99+ vperm 16,16,16,12++ subic 6,6,32+ .long 0x10039CC8+ .long 0x11B05CC8+ subfe 0,0,0+ .long 0x10238CC8+ .long 0x11D04CC8+ and 0,0,6+ .long 0x104394C8+ .long 0x11F054C8+ add 5,5,0++ vxor 0,0,13+ vxor 1,1,14++ .long 0x10E044C8++ vsldoi 5,1,4,8+ vsldoi 6,4,1,8+ vxor 2,2,15+ vxor 0,0,5+ vxor 2,2,6++ vsldoi 0,0,0,8+ vxor 0,0,7+ .long 0x7C682E99+ addi 5,5,32++ vsldoi 6,0,0,8+ .long 0x100044C8+ vperm 3,3,3,12+ vxor 6,6,2+ vxor 3,3,6+ vxor 3,3,0+ cmpld 9,5+ bgt .Loop_2x++ cmplwi 6,0+ bne .Leven++.Lshort:+ .long 0x10035CC8+ .long 0x10234CC8+ .long 0x104354C8++ .long 0x10E044C8++ vsldoi 5,1,4,8+ vsldoi 6,4,1,8+ vxor 0,0,5+ vxor 2,2,6++ vsldoi 0,0,0,8+ vxor 0,0,7++ vsldoi 6,0,0,8+ .long 0x100044C8+ vxor 6,6,2++.Leven:+ vxor 0,0,6+ vperm 0,0,0,12+ .long 0x7C001F99++ or 12,12,12+ blr +.long 0+.byte 0,12,0x14,0,0,0,4,0+.long 0+.align 5+.crypton_gcm_ghash_p8_4x:+.Lgcm_ghash_p8_4x:+ stdu 1,-256(1)+ li 10,63+ li 11,79+ stvx 20,10,1+ addi 10,10,32+ stvx 21,11,1+ addi 11,11,32+ stvx 22,10,1+ addi 10,10,32+ stvx 23,11,1+ addi 11,11,32+ stvx 24,10,1+ addi 10,10,32+ stvx 25,11,1+ addi 11,11,32+ stvx 26,10,1+ addi 10,10,32+ stvx 27,11,1+ addi 11,11,32+ stvx 28,10,1+ addi 10,10,32+ stvx 29,11,1+ addi 11,11,32+ stvx 30,10,1+ li 10,0x60+ stvx 31,11,1+ li 0,-1+ stw 12,252(1)+ or 0,0,0++ lvsl 5,0,8++ li 8,0x70+ .long 0x7E292699+ li 9,0x80+ vspltisb 6,8++ li 10,0x90+ .long 0x7EE82699+ li 8,0xa0+ .long 0x7F092699+ li 9,0xb0+ .long 0x7F2A2699+ li 10,0xc0+ .long 0x7FA82699+ li 8,0x10+ .long 0x7FC92699+ li 9,0x20+ .long 0x7FEA2699+ li 10,0x30++ vsldoi 7,4,6,8+ vaddubm 18,5,7+ vaddubm 19,6,18++ srdi 6,6,4++ .long 0x7C602E99+ .long 0x7E082E99+ subic. 6,6,8+ .long 0x7EC92E99+ .long 0x7F8A2E99+ addi 5,5,0x40+ vperm 3,3,3,12+ vperm 16,16,16,12+ vperm 22,22,22,12+ vperm 28,28,28,12++ vxor 2,3,0++ .long 0x11B0BCC8+ .long 0x11D0C4C8+ .long 0x11F0CCC8++ vperm 11,17,9,18+ vperm 5,22,28,19+ vperm 10,17,9,19+ vperm 6,22,28,18+ .long 0x12B68CC8+ .long 0x12855CC8+ .long 0x137C4CC8+ .long 0x134654C8++ vxor 21,21,14+ vxor 20,20,13+ vxor 27,27,21+ vxor 26,26,15++ blt .Ltail_4x++.Loop_4x:+ .long 0x7C602E99+ .long 0x7E082E99+ subic. 6,6,4+ .long 0x7EC92E99+ .long 0x7F8A2E99+ addi 5,5,0x40+ vperm 16,16,16,12+ vperm 22,22,22,12+ vperm 28,28,28,12+ vperm 3,3,3,12++ .long 0x1002ECC8+ .long 0x1022F4C8+ .long 0x1042FCC8+ .long 0x11B0BCC8+ .long 0x11D0C4C8+ .long 0x11F0CCC8++ vxor 0,0,20+ vxor 1,1,27+ vxor 2,2,26+ vperm 5,22,28,19+ vperm 6,22,28,18++ .long 0x10E044C8+ .long 0x12855CC8+ .long 0x134654C8++ vsldoi 5,1,4,8+ vsldoi 6,4,1,8+ vxor 0,0,5+ vxor 2,2,6++ vsldoi 0,0,0,8+ vxor 0,0,7++ vsldoi 6,0,0,8+ .long 0x12B68CC8+ .long 0x137C4CC8+ .long 0x100044C8++ vxor 20,20,13+ vxor 26,26,15+ vxor 2,2,3+ vxor 21,21,14+ vxor 2,2,6+ vxor 27,27,21+ vxor 2,2,0+ bge .Loop_4x++.Ltail_4x:+ .long 0x1002ECC8+ .long 0x1022F4C8+ .long 0x1042FCC8++ vxor 0,0,20+ vxor 1,1,27++ .long 0x10E044C8++ vsldoi 5,1,4,8+ vsldoi 6,4,1,8+ vxor 2,2,26+ vxor 0,0,5+ vxor 2,2,6++ vsldoi 0,0,0,8+ vxor 0,0,7++ vsldoi 6,0,0,8+ .long 0x100044C8+ vxor 6,6,2+ vxor 0,0,6++ addic. 6,6,4+ beq .Ldone_4x++ .long 0x7C602E99+ cmpldi 6,2+ li 6,-4+ blt .Lone+ .long 0x7E082E99+ beq .Ltwo++.Lthree:+ .long 0x7EC92E99+ vperm 3,3,3,12+ vperm 16,16,16,12+ vperm 22,22,22,12++ vxor 2,3,0+ vor 29,23,23+ vor 30,24,24+ vor 31,25,25++ vperm 5,16,22,19+ vperm 6,16,22,18+ .long 0x12B08CC8+ .long 0x13764CC8+ .long 0x12855CC8+ .long 0x134654C8++ vxor 27,27,21+ b .Ltail_4x++.align 4+.Ltwo:+ vperm 3,3,3,12+ vperm 16,16,16,12++ vxor 2,3,0+ vperm 5,4,16,19+ vperm 6,4,16,18++ vsldoi 29,4,17,8+ vor 30,17,17+ vsldoi 31,17,4,8++ .long 0x12855CC8+ .long 0x13704CC8+ .long 0x134654C8++ b .Ltail_4x++.align 4+.Lone:+ vperm 3,3,3,12++ vsldoi 29,4,9,8+ vor 30,9,9+ vsldoi 31,9,4,8++ vxor 2,3,0+ vxor 20,20,20+ vxor 27,27,27+ vxor 26,26,26++ b .Ltail_4x++.Ldone_4x:+ vperm 0,0,0,12+ .long 0x7C001F99++ li 10,63+ li 11,79+ or 12,12,12+ lvx 20,10,1+ addi 10,10,32+ lvx 21,11,1+ addi 11,11,32+ lvx 22,10,1+ addi 10,10,32+ lvx 23,11,1+ addi 11,11,32+ lvx 24,10,1+ addi 10,10,32+ lvx 25,11,1+ addi 11,11,32+ lvx 26,10,1+ addi 10,10,32+ lvx 27,11,1+ addi 11,11,32+ lvx 28,10,1+ addi 10,10,32+ lvx 29,11,1+ addi 11,11,32+ lvx 30,10,1+ lvx 31,11,1+ addi 1,1,256+ blr +.long 0+.byte 0,12,0x04,0,0x80,0,4,0+.long 0+.size crypton_gcm_ghash_p8,.-crypton_gcm_ghash_p8++.byte 71,72,65,83,72,32,102,111,114,32,80,111,119,101,114,73,83,65,32,50,46,48,55,44,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0+.align 2+.align 2
+ cbits/asm/ghashp8-ppc.pl view
@@ -0,0 +1,664 @@+#!/usr/bin/env perl+#+# ====================================================================+# Written by Andy Polyakov, @dot-asm, initially for use in the OpenSSL+# project. The module is dual licensed under OpenSSL and CRYPTOGAMS+# licenses depending on where you obtain it. For further details see+# https://github.com/dot-asm/cryptogams/.+# ====================================================================+#+# GHASH for for PowerISA v2.07.+#+# July 2014+#+# Accurate performance measurements are problematic, because it's+# always virtualized setup with possibly throttled processor.+# Relative comparison is therefore more informative. This initial+# version is ~2.1x slower than hardware-assisted AES-128-CTR, ~12x+# faster than "4-bit" integer-only compiler-generated 64-bit code.+# "Initial version" means that there is room for further improvement.++# May 2016+#+# 2x aggregated reduction improves performance by 50% (resulting+# performance on POWER8 is 1 cycle per processed byte), and 4x+# aggregated reduction - by 170% or 2.7x (resulting in 0.55 cpb).+# POWER9 delivers 0.51 cpb.++$flavour=shift;+$output =shift;++if ($flavour =~ /64/) {+ $SIZE_T=8;+ $LRSAVE=2*$SIZE_T;+ $STU="stdu";+ $POP="ld";+ $PUSH="std";+ $UCMP="cmpld";+ $SHRI="srdi";+} elsif ($flavour =~ /32/) {+ $SIZE_T=4;+ $LRSAVE=$SIZE_T;+ $STU="stwu";+ $POP="lwz";+ $PUSH="stw";+ $UCMP="cmplw";+ $SHRI="srwi";+} else { die "nonsense $flavour"; }++$sp="r1";+$FRAME=6*$SIZE_T+13*16; # 13*16 is for v20-v31 offload++$0 =~ m/(.*[\/\\])[^\/\\]+$/; $dir=$1;+( $xlate="${dir}ppc-xlate.pl" and -f $xlate ) or+( $xlate="${dir}../../perlasm/ppc-xlate.pl" and -f $xlate) or+die "can't locate ppc-xlate.pl";++open STDOUT,"| $^X $xlate $flavour $output" || die "can't call $xlate: $!";++my ($Xip,$Htbl,$inp,$len)=map("r$_",(3..6)); # argument block++my ($Xl,$Xm,$Xh,$IN)=map("v$_",(0..3));+my ($zero,$t0,$t1,$t2,$xC2,$H,$Hh,$Hl,$lemask)=map("v$_",(4..12));+my ($Xl1,$Xm1,$Xh1,$IN1,$H2,$H2h,$H2l)=map("v$_",(13..19));+my $vrsave="r12";++$code=<<___;+.machine "any"++.text++.globl .gcm_init_p8+.align 5+.gcm_init_p8:+ li r0,-4096+ li r8,0x10+ mfspr $vrsave,256+ li r9,0x20+ mtspr 256,r0+ li r10,0x30+ lvx_u $H,0,r4 # load H++ vspltisb $xC2,-16 # 0xf0+ vspltisb $t0,1 # one+ vaddubm $xC2,$xC2,$xC2 # 0xe0+ vxor $zero,$zero,$zero+ vor $xC2,$xC2,$t0 # 0xe1+ vsldoi $xC2,$xC2,$zero,15 # 0xe1...+ vsldoi $t1,$zero,$t0,1 # ...1+ vaddubm $xC2,$xC2,$xC2 # 0xc2...+ vspltisb $t2,7+ vor $xC2,$xC2,$t1 # 0xc2....01+ vspltb $t1,$H,0 # most significant byte+ vsl $H,$H,$t0 # H<<=1+ vsrab $t1,$t1,$t2 # broadcast carry bit+ vand $t1,$t1,$xC2+ vxor $IN,$H,$t1 # twisted H++ vsldoi $H,$IN,$IN,8 # twist even more ...+ vsldoi $xC2,$zero,$xC2,8 # 0xc2.0+ vsldoi $Hl,$zero,$H,8 # ... and split+ vsldoi $Hh,$H,$zero,8++ stvx_u $xC2,0,r3 # save pre-computed table+ stvx_u $Hl,r8,r3+ li r8,0x40+ stvx_u $H, r9,r3+ li r9,0x50+ stvx_u $Hh,r10,r3+ li r10,0x60++ vpmsumd $Xl,$IN,$Hl # H.lo·H.lo+ vpmsumd $Xm,$IN,$H # H.hi·H.lo+H.lo·H.hi+ vpmsumd $Xh,$IN,$Hh # H.hi·H.hi++ vpmsumd $t2,$Xl,$xC2 # 1st reduction phase++ vsldoi $t0,$Xm,$zero,8+ vsldoi $t1,$zero,$Xm,8+ vxor $Xl,$Xl,$t0+ vxor $Xh,$Xh,$t1++ vsldoi $Xl,$Xl,$Xl,8+ vxor $Xl,$Xl,$t2++ vsldoi $t1,$Xl,$Xl,8 # 2nd reduction phase+ vpmsumd $Xl,$Xl,$xC2+ vxor $t1,$t1,$Xh+ vxor $IN1,$Xl,$t1++ vsldoi $H2,$IN1,$IN1,8+ vsldoi $H2l,$zero,$H2,8+ vsldoi $H2h,$H2,$zero,8++ stvx_u $H2l,r8,r3 # save H^2+ li r8,0x70+ stvx_u $H2,r9,r3+ li r9,0x80+ stvx_u $H2h,r10,r3+ li r10,0x90+___+{+my ($t4,$t5,$t6) = ($Hl,$H,$Hh);+$code.=<<___;+ vpmsumd $Xl,$IN,$H2l # H.lo·H^2.lo+ vpmsumd $Xl1,$IN1,$H2l # H^2.lo·H^2.lo+ vpmsumd $Xm,$IN,$H2 # H.hi·H^2.lo+H.lo·H^2.hi+ vpmsumd $Xm1,$IN1,$H2 # H^2.hi·H^2.lo+H^2.lo·H^2.hi+ vpmsumd $Xh,$IN,$H2h # H.hi·H^2.hi+ vpmsumd $Xh1,$IN1,$H2h # H^2.hi·H^2.hi++ vpmsumd $t2,$Xl,$xC2 # 1st reduction phase+ vpmsumd $t6,$Xl1,$xC2 # 1st reduction phase++ vsldoi $t0,$Xm,$zero,8+ vsldoi $t1,$zero,$Xm,8+ vsldoi $t4,$Xm1,$zero,8+ vsldoi $t5,$zero,$Xm1,8+ vxor $Xl,$Xl,$t0+ vxor $Xh,$Xh,$t1+ vxor $Xl1,$Xl1,$t4+ vxor $Xh1,$Xh1,$t5++ vsldoi $Xl,$Xl,$Xl,8+ vsldoi $Xl1,$Xl1,$Xl1,8+ vxor $Xl,$Xl,$t2+ vxor $Xl1,$Xl1,$t6++ vsldoi $t1,$Xl,$Xl,8 # 2nd reduction phase+ vsldoi $t5,$Xl1,$Xl1,8 # 2nd reduction phase+ vpmsumd $Xl,$Xl,$xC2+ vpmsumd $Xl1,$Xl1,$xC2+ vxor $t1,$t1,$Xh+ vxor $t5,$t5,$Xh1+ vxor $Xl,$Xl,$t1+ vxor $Xl1,$Xl1,$t5++ vsldoi $H,$Xl,$Xl,8+ vsldoi $H2,$Xl1,$Xl1,8+ vsldoi $Hl,$zero,$H,8+ vsldoi $Hh,$H,$zero,8+ vsldoi $H2l,$zero,$H2,8+ vsldoi $H2h,$H2,$zero,8++ stvx_u $Hl,r8,r3 # save H^3+ li r8,0xa0+ stvx_u $H,r9,r3+ li r9,0xb0+ stvx_u $Hh,r10,r3+ li r10,0xc0+ stvx_u $H2l,r8,r3 # save H^4+ stvx_u $H2,r9,r3+ stvx_u $H2h,r10,r3++ mtspr 256,$vrsave+ blr+ .long 0+ .byte 0,12,0x14,0,0,0,2,0+ .long 0+.size .gcm_init_p8,.-.gcm_init_p8+___+}+$code.=<<___;+.globl .gcm_gmult_p8+.align 5+.gcm_gmult_p8:+ lis r0,0xfff8+ li r8,0x10+ mfspr $vrsave,256+ li r9,0x20+ mtspr 256,r0+ li r10,0x30+ lvx_u $IN,0,$Xip # load Xi++ lvx_u $Hl,r8,$Htbl # load pre-computed table+ le?lvsl $lemask,r0,r0+ lvx_u $H, r9,$Htbl+ le?vspltisb $t0,0x07+ lvx_u $Hh,r10,$Htbl+ le?vxor $lemask,$lemask,$t0+ lvx_u $xC2,0,$Htbl+ le?vperm $IN,$IN,$IN,$lemask+ vxor $zero,$zero,$zero++ vpmsumd $Xl,$IN,$Hl # H.lo·Xi.lo+ vpmsumd $Xm,$IN,$H # H.hi·Xi.lo+H.lo·Xi.hi+ vpmsumd $Xh,$IN,$Hh # H.hi·Xi.hi++ vpmsumd $t2,$Xl,$xC2 # 1st reduction phase++ vsldoi $t0,$Xm,$zero,8+ vsldoi $t1,$zero,$Xm,8+ vxor $Xl,$Xl,$t0+ vxor $Xh,$Xh,$t1++ vsldoi $Xl,$Xl,$Xl,8+ vxor $Xl,$Xl,$t2++ vsldoi $t1,$Xl,$Xl,8 # 2nd reduction phase+ vpmsumd $Xl,$Xl,$xC2+ vxor $t1,$t1,$Xh+ vxor $Xl,$Xl,$t1++ le?vperm $Xl,$Xl,$Xl,$lemask+ stvx_u $Xl,0,$Xip # write out Xi++ mtspr 256,$vrsave+ blr+ .long 0+ .byte 0,12,0x14,0,0,0,2,0+ .long 0+.size .gcm_gmult_p8,.-.gcm_gmult_p8++.globl .gcm_ghash_p8+.align 5+.gcm_ghash_p8:+ li r0,-4096+ li r8,0x10+ mfspr $vrsave,256+ li r9,0x20+ mtspr 256,r0+ li r10,0x30+ lvx_u $Xl,0,$Xip # load Xi++ lvx_u $Hl,r8,$Htbl # load pre-computed table+ li r8,0x40+ le?lvsl $lemask,r0,r0+ lvx_u $H, r9,$Htbl+ li r9,0x50+ le?vspltisb $t0,0x07+ lvx_u $Hh,r10,$Htbl+ li r10,0x60+ le?vxor $lemask,$lemask,$t0+ lvx_u $xC2,0,$Htbl+ le?vperm $Xl,$Xl,$Xl,$lemask+ vxor $zero,$zero,$zero++ ${UCMP}i $len,64+ bge Lgcm_ghash_p8_4x++ lvx_u $IN,0,$inp+ addi $inp,$inp,16+ subic. $len,$len,16+ le?vperm $IN,$IN,$IN,$lemask+ vxor $IN,$IN,$Xl+ beq Lshort++ lvx_u $H2l,r8,$Htbl # load H^2+ li r8,16+ lvx_u $H2, r9,$Htbl+ add r9,$inp,$len # end of input+ lvx_u $H2h,r10,$Htbl+ be?b Loop_2x++.align 5+Loop_2x:+ lvx_u $IN1,0,$inp+ le?vperm $IN1,$IN1,$IN1,$lemask++ subic $len,$len,32+ vpmsumd $Xl,$IN,$H2l # H^2.lo·Xi.lo+ vpmsumd $Xl1,$IN1,$Hl # H.lo·Xi+1.lo+ subfe r0,r0,r0 # borrow?-1:0+ vpmsumd $Xm,$IN,$H2 # H^2.hi·Xi.lo+H^2.lo·Xi.hi+ vpmsumd $Xm1,$IN1,$H # H.hi·Xi+1.lo+H.lo·Xi+1.hi+ and r0,r0,$len+ vpmsumd $Xh,$IN,$H2h # H^2.hi·Xi.hi+ vpmsumd $Xh1,$IN1,$Hh # H.hi·Xi+1.hi+ add $inp,$inp,r0++ vxor $Xl,$Xl,$Xl1+ vxor $Xm,$Xm,$Xm1++ vpmsumd $t2,$Xl,$xC2 # 1st reduction phase++ vsldoi $t0,$Xm,$zero,8+ vsldoi $t1,$zero,$Xm,8+ vxor $Xh,$Xh,$Xh1+ vxor $Xl,$Xl,$t0+ vxor $Xh,$Xh,$t1++ vsldoi $Xl,$Xl,$Xl,8+ vxor $Xl,$Xl,$t2+ lvx_u $IN,r8,$inp+ addi $inp,$inp,32++ vsldoi $t1,$Xl,$Xl,8 # 2nd reduction phase+ vpmsumd $Xl,$Xl,$xC2+ le?vperm $IN,$IN,$IN,$lemask+ vxor $t1,$t1,$Xh+ vxor $IN,$IN,$t1+ vxor $IN,$IN,$Xl+ $UCMP r9,$inp+ bgt Loop_2x # done yet?++ cmplwi $len,0+ bne Leven++Lshort:+ vpmsumd $Xl,$IN,$Hl # H.lo·Xi.lo+ vpmsumd $Xm,$IN,$H # H.hi·Xi.lo+H.lo·Xi.hi+ vpmsumd $Xh,$IN,$Hh # H.hi·Xi.hi++ vpmsumd $t2,$Xl,$xC2 # 1st reduction phase++ vsldoi $t0,$Xm,$zero,8+ vsldoi $t1,$zero,$Xm,8+ vxor $Xl,$Xl,$t0+ vxor $Xh,$Xh,$t1++ vsldoi $Xl,$Xl,$Xl,8+ vxor $Xl,$Xl,$t2++ vsldoi $t1,$Xl,$Xl,8 # 2nd reduction phase+ vpmsumd $Xl,$Xl,$xC2+ vxor $t1,$t1,$Xh++Leven:+ vxor $Xl,$Xl,$t1+ le?vperm $Xl,$Xl,$Xl,$lemask+ stvx_u $Xl,0,$Xip # write out Xi++ mtspr 256,$vrsave+ blr+ .long 0+ .byte 0,12,0x14,0,0,0,4,0+ .long 0+___+{+my ($Xl3,$Xm2,$IN2,$H3l,$H3,$H3h,+ $Xh3,$Xm3,$IN3,$H4l,$H4,$H4h) = map("v$_",(20..31));+my $IN0=$IN;+my ($H21l,$H21h,$loperm,$hiperm) = ($Hl,$Hh,$H2l,$H2h);++$code.=<<___;+.align 5+.gcm_ghash_p8_4x:+Lgcm_ghash_p8_4x:+ $STU $sp,-$FRAME($sp)+ li r10,`15+6*$SIZE_T`+ li r11,`31+6*$SIZE_T`+ stvx v20,r10,$sp+ addi r10,r10,32+ stvx v21,r11,$sp+ addi r11,r11,32+ stvx v22,r10,$sp+ addi r10,r10,32+ stvx v23,r11,$sp+ addi r11,r11,32+ stvx v24,r10,$sp+ addi r10,r10,32+ stvx v25,r11,$sp+ addi r11,r11,32+ stvx v26,r10,$sp+ addi r10,r10,32+ stvx v27,r11,$sp+ addi r11,r11,32+ stvx v28,r10,$sp+ addi r10,r10,32+ stvx v29,r11,$sp+ addi r11,r11,32+ stvx v30,r10,$sp+ li r10,0x60+ stvx v31,r11,$sp+ li r0,-1+ stw $vrsave,`$FRAME-4`($sp) # save vrsave+ mtspr 256,r0 # preserve all AltiVec registers++ lvsl $t0,0,r8 # 0x0001..0e0f+ #lvx_u $H2l,r8,$Htbl # load H^2+ li r8,0x70+ lvx_u $H2, r9,$Htbl+ li r9,0x80+ vspltisb $t1,8 # 0x0808..0808+ #lvx_u $H2h,r10,$Htbl+ li r10,0x90+ lvx_u $H3l,r8,$Htbl # load H^3+ li r8,0xa0+ lvx_u $H3, r9,$Htbl+ li r9,0xb0+ lvx_u $H3h,r10,$Htbl+ li r10,0xc0+ lvx_u $H4l,r8,$Htbl # load H^4+ li r8,0x10+ lvx_u $H4, r9,$Htbl+ li r9,0x20+ lvx_u $H4h,r10,$Htbl+ li r10,0x30++ vsldoi $t2,$zero,$t1,8 # 0x0000..0808+ vaddubm $hiperm,$t0,$t2 # 0x0001..1617+ vaddubm $loperm,$t1,$hiperm # 0x0809..1e1f++ $SHRI $len,$len,4 # this allows to use sign bit+ # as carry+ lvx_u $IN0,0,$inp # load input+ lvx_u $IN1,r8,$inp+ subic. $len,$len,8+ lvx_u $IN2,r9,$inp+ lvx_u $IN3,r10,$inp+ addi $inp,$inp,0x40+ le?vperm $IN0,$IN0,$IN0,$lemask+ le?vperm $IN1,$IN1,$IN1,$lemask+ le?vperm $IN2,$IN2,$IN2,$lemask+ le?vperm $IN3,$IN3,$IN3,$lemask++ vxor $Xh,$IN0,$Xl++ vpmsumd $Xl1,$IN1,$H3l+ vpmsumd $Xm1,$IN1,$H3+ vpmsumd $Xh1,$IN1,$H3h++ vperm $H21l,$H2,$H,$hiperm+ vperm $t0,$IN2,$IN3,$loperm+ vperm $H21h,$H2,$H,$loperm+ vperm $t1,$IN2,$IN3,$hiperm+ vpmsumd $Xm2,$IN2,$H2 # H^2.lo·Xi+2.hi+H^2.hi·Xi+2.lo+ vpmsumd $Xl3,$t0,$H21l # H^2.lo·Xi+2.lo+H.lo·Xi+3.lo+ vpmsumd $Xm3,$IN3,$H # H.hi·Xi+3.lo +H.lo·Xi+3.hi+ vpmsumd $Xh3,$t1,$H21h # H^2.hi·Xi+2.hi+H.hi·Xi+3.hi++ vxor $Xm2,$Xm2,$Xm1+ vxor $Xl3,$Xl3,$Xl1+ vxor $Xm3,$Xm3,$Xm2+ vxor $Xh3,$Xh3,$Xh1++ blt Ltail_4x++Loop_4x:+ lvx_u $IN0,0,$inp+ lvx_u $IN1,r8,$inp+ subic. $len,$len,4+ lvx_u $IN2,r9,$inp+ lvx_u $IN3,r10,$inp+ addi $inp,$inp,0x40+ le?vperm $IN1,$IN1,$IN1,$lemask+ le?vperm $IN2,$IN2,$IN2,$lemask+ le?vperm $IN3,$IN3,$IN3,$lemask+ le?vperm $IN0,$IN0,$IN0,$lemask++ vpmsumd $Xl,$Xh,$H4l # H^4.lo·Xi.lo+ vpmsumd $Xm,$Xh,$H4 # H^4.hi·Xi.lo+H^4.lo·Xi.hi+ vpmsumd $Xh,$Xh,$H4h # H^4.hi·Xi.hi+ vpmsumd $Xl1,$IN1,$H3l+ vpmsumd $Xm1,$IN1,$H3+ vpmsumd $Xh1,$IN1,$H3h++ vxor $Xl,$Xl,$Xl3+ vxor $Xm,$Xm,$Xm3+ vxor $Xh,$Xh,$Xh3+ vperm $t0,$IN2,$IN3,$loperm+ vperm $t1,$IN2,$IN3,$hiperm++ vpmsumd $t2,$Xl,$xC2 # 1st reduction phase+ vpmsumd $Xl3,$t0,$H21l # H.lo·Xi+3.lo +H^2.lo·Xi+2.lo+ vpmsumd $Xh3,$t1,$H21h # H.hi·Xi+3.hi +H^2.hi·Xi+2.hi++ vsldoi $t0,$Xm,$zero,8+ vsldoi $t1,$zero,$Xm,8+ vxor $Xl,$Xl,$t0+ vxor $Xh,$Xh,$t1++ vsldoi $Xl,$Xl,$Xl,8+ vxor $Xl,$Xl,$t2++ vsldoi $t1,$Xl,$Xl,8 # 2nd reduction phase+ vpmsumd $Xm2,$IN2,$H2 # H^2.hi·Xi+2.lo+H^2.lo·Xi+2.hi+ vpmsumd $Xm3,$IN3,$H # H.hi·Xi+3.lo +H.lo·Xi+3.hi+ vpmsumd $Xl,$Xl,$xC2++ vxor $Xl3,$Xl3,$Xl1+ vxor $Xh3,$Xh3,$Xh1+ vxor $Xh,$Xh,$IN0+ vxor $Xm2,$Xm2,$Xm1+ vxor $Xh,$Xh,$t1+ vxor $Xm3,$Xm3,$Xm2+ vxor $Xh,$Xh,$Xl+ bge Loop_4x++Ltail_4x:+ vpmsumd $Xl,$Xh,$H4l # H^4.lo·Xi.lo+ vpmsumd $Xm,$Xh,$H4 # H^4.hi·Xi.lo+H^4.lo·Xi.hi+ vpmsumd $Xh,$Xh,$H4h # H^4.hi·Xi.hi++ vxor $Xl,$Xl,$Xl3+ vxor $Xm,$Xm,$Xm3++ vpmsumd $t2,$Xl,$xC2 # 1st reduction phase++ vsldoi $t0,$Xm,$zero,8+ vsldoi $t1,$zero,$Xm,8+ vxor $Xh,$Xh,$Xh3+ vxor $Xl,$Xl,$t0+ vxor $Xh,$Xh,$t1++ vsldoi $Xl,$Xl,$Xl,8+ vxor $Xl,$Xl,$t2++ vsldoi $t1,$Xl,$Xl,8 # 2nd reduction phase+ vpmsumd $Xl,$Xl,$xC2+ vxor $t1,$t1,$Xh+ vxor $Xl,$Xl,$t1++ addic. $len,$len,4+ beq Ldone_4x++ lvx_u $IN0,0,$inp+ ${UCMP}i $len,2+ li $len,-4+ blt Lone+ lvx_u $IN1,r8,$inp+ beq Ltwo++Lthree:+ lvx_u $IN2,r9,$inp+ le?vperm $IN0,$IN0,$IN0,$lemask+ le?vperm $IN1,$IN1,$IN1,$lemask+ le?vperm $IN2,$IN2,$IN2,$lemask++ vxor $Xh,$IN0,$Xl+ vmr $H4l,$H3l+ vmr $H4, $H3+ vmr $H4h,$H3h++ vperm $t0,$IN1,$IN2,$loperm+ vperm $t1,$IN1,$IN2,$hiperm+ vpmsumd $Xm2,$IN1,$H2 # H^2.lo·Xi+1.hi+H^2.hi·Xi+1.lo+ vpmsumd $Xm3,$IN2,$H # H.hi·Xi+2.lo +H.lo·Xi+2.hi+ vpmsumd $Xl3,$t0,$H21l # H^2.lo·Xi+1.lo+H.lo·Xi+2.lo+ vpmsumd $Xh3,$t1,$H21h # H^2.hi·Xi+1.hi+H.hi·Xi+2.hi++ vxor $Xm3,$Xm3,$Xm2+ b Ltail_4x++.align 4+Ltwo:+ le?vperm $IN0,$IN0,$IN0,$lemask+ le?vperm $IN1,$IN1,$IN1,$lemask++ vxor $Xh,$IN0,$Xl+ vperm $t0,$zero,$IN1,$loperm+ vperm $t1,$zero,$IN1,$hiperm++ vsldoi $H4l,$zero,$H2,8+ vmr $H4, $H2+ vsldoi $H4h,$H2,$zero,8++ vpmsumd $Xl3,$t0, $H21l # H.lo·Xi+1.lo+ vpmsumd $Xm3,$IN1,$H # H.hi·Xi+1.lo+H.lo·Xi+2.hi+ vpmsumd $Xh3,$t1, $H21h # H.hi·Xi+1.hi++ b Ltail_4x++.align 4+Lone:+ le?vperm $IN0,$IN0,$IN0,$lemask++ vsldoi $H4l,$zero,$H,8+ vmr $H4, $H+ vsldoi $H4h,$H,$zero,8++ vxor $Xh,$IN0,$Xl+ vxor $Xl3,$Xl3,$Xl3+ vxor $Xm3,$Xm3,$Xm3+ vxor $Xh3,$Xh3,$Xh3++ b Ltail_4x++Ldone_4x:+ le?vperm $Xl,$Xl,$Xl,$lemask+ stvx_u $Xl,0,$Xip # write out Xi++ li r10,`15+6*$SIZE_T`+ li r11,`31+6*$SIZE_T`+ mtspr 256,$vrsave+ lvx v20,r10,$sp+ addi r10,r10,32+ lvx v21,r11,$sp+ addi r11,r11,32+ lvx v22,r10,$sp+ addi r10,r10,32+ lvx v23,r11,$sp+ addi r11,r11,32+ lvx v24,r10,$sp+ addi r10,r10,32+ lvx v25,r11,$sp+ addi r11,r11,32+ lvx v26,r10,$sp+ addi r10,r10,32+ lvx v27,r11,$sp+ addi r11,r11,32+ lvx v28,r10,$sp+ addi r10,r10,32+ lvx v29,r11,$sp+ addi r11,r11,32+ lvx v30,r10,$sp+ lvx v31,r11,$sp+ addi $sp,$sp,$FRAME+ blr+ .long 0+ .byte 0,12,0x04,0,0x80,0,4,0+ .long 0+___+}+$code.=<<___;+.size .gcm_ghash_p8,.-.gcm_ghash_p8++.asciz "GHASH for PowerISA 2.07, CRYPTOGAMS by <appro\@openssl.org>"+.align 2+___++foreach (split("\n",$code)) {+ s/\`([^\`]*)\`/eval $1/geo;++ if ($flavour =~ /le$/o) { # little-endian+ s/le\?//o or+ s/be\?/#be#/o;+ } else {+ s/le\?/#le#/o or+ s/be\?//o;+ }+ print $_,"\n";+}++close STDOUT; # enforce flush
+ cbits/asm/ppc-xlate.pl view
@@ -0,0 +1,352 @@+#!/usr/bin/env perl++# PowerPC assembler distiller by \@dot-asm.++################################################################+# Recognized "flavour"-s are:+#+# linux{32|64}[le] GNU assembler and ELF symbol decorations,+# with little-endian option+# linux64v2 GNU asssembler and big-endian instantiation+# of latest ELF specification+# aix{32|64} AIX assembler and symbol decorations+# osx{32|64} Mac OS X assembler and symbol decoratons++my $flavour = shift;+my $output = shift;+open STDOUT,">$output" || die "can't open $output: $!";++my %GLOBALS;+my %TYPES;+my $dotinlocallabels=($flavour=~/linux/)?1:0;++################################################################+# directives which need special treatment on different platforms+################################################################+my $type = sub {+ my ($dir,$name,$type) = @_;++ $TYPES{$name} = $type;+ if ($flavour =~ /linux/) {+ $name =~ s|^\.||;+ ".type $name,$type";+ } else {+ "";+ }+};+my $globl = sub {+ my $junk = shift;+ my $name = shift;+ my $global = \$GLOBALS{$name};+ my $type = \$TYPES{$name};+ my $ret;++ $name =~ s|^\.||;++ SWITCH: for ($flavour) {+ /aix/ && do { if (!$$type) {+ $$type = "\@function";+ }+ if ($$type =~ /function/) {+ $name = ".$name";+ }+ last;+ };+ /osx/ && do { $name = "_$name";+ last;+ };+ /linux.*(32|64(le|v2))/+ && do { $ret .= ".globl $name";+ if (!$$type) {+ $ret .= "\n.type $name,\@function";+ $$type = "\@function";+ }+ last;+ };+ /linux.*64/ && do { $ret .= ".globl $name";+ if (!$$type) {+ $ret .= "\n.type $name,\@function";+ $$type = "\@function";+ }+ if ($$type =~ /function/) {+ $ret .= "\n.section \".opd\",\"aw\"";+ $ret .= "\n.align 3";+ $ret .= "\n$name:";+ $ret .= "\n.quad .$name,.TOC.\@tocbase,0";+ $ret .= "\n.previous";+ $name = ".$name";+ }+ last;+ };+ }++ $ret = ".globl $name" if (!$ret);+ $$global = $name;+ $ret;+};+my $text = sub {+ my $ret = ($flavour =~ /aix/) ? ".csect\t.text[PR],7" : ".text";+ $ret = ".abiversion 2\n".$ret if ($flavour =~ /linux.*64(le|v2)/);+ $ret;+};+my $machine = sub {+ my $junk = shift;+ my $arch = shift;+ if ($flavour =~ /osx/)+ { $arch =~ s/\"//g;+ $arch = ($flavour=~/64/) ? "ppc970-64" : "ppc970" if ($arch eq "any");+ }+ ".machine $arch";+};+my $size = sub {+ if ($flavour =~ /linux/)+ { shift;+ my $name = shift;+ my $real = $GLOBALS{$name} ? \$GLOBALS{$name} : \$name;+ my $ret = ".size $$real,.-$$real";+ $name =~ s|^\.||;+ if ($$real ne $name) {+ $ret .= "\n.size $name,.-$$real";+ }+ $ret;+ }+ else+ { ""; }+};+my $asciz = sub {+ shift;+ my $line = join(",",@_);+ if ($line =~ /^"(.*)"$/)+ { ".byte " . join(",",unpack("C*",$1),0) . "\n.align 2"; }+ else+ { ""; }+};+my $quad = sub {+ shift;+ my @ret;+ my ($hi,$lo);+ for (@_) {+ if (/^0x([0-9a-f]*?)([0-9a-f]{1,8})$/io)+ { $hi=$1?"0x$1":"0"; $lo="0x$2"; }+ elsif (/^([0-9]+)$/o)+ { $hi=$1>>32; $lo=$1&0xffffffff; } # error-prone with 32-bit perl+ else+ { $hi=undef; $lo=$_; }++ if (defined($hi))+ { push(@ret,$flavour=~/le$/o?".long\t$lo,$hi":".long\t$hi,$lo"); }+ else+ { push(@ret,".quad $lo"); }+ }+ join("\n",@ret);+};++################################################################+# simplified mnemonics not handled by at least one assembler+################################################################+my $cmplw = sub {+ my $f = shift;+ my $cr = 0; $cr = shift if ($#_>1);+ # Some out-of-date 32-bit GNU assembler just can't handle cmplw...+ ($flavour =~ /linux.*32/) ?+ " .long ".sprintf "0x%x",31<<26|$cr<<23|$_[0]<<16|$_[1]<<11|64 :+ " cmplw ".join(',',$cr,@_);+};+my $bdnz = sub {+ my $f = shift;+ my $bo = $f=~/[\+\-]/ ? 16+9 : 16; # optional "to be taken" hint+ " bc $bo,0,".shift;+} if ($flavour!~/linux/);+my $bltlr = sub {+ my $f = shift;+ my $bo = $f=~/\-/ ? 12+2 : 12; # optional "not to be taken" hint+ ($flavour =~ /linux/) ? # GNU as doesn't allow most recent hints+ " .long ".sprintf "0x%x",19<<26|$bo<<21|16<<1 :+ " bclr $bo,0";+};+my $bnelr = sub {+ my $f = shift;+ my $bo = $f=~/\-/ ? 4+2 : 4; # optional "not to be taken" hint+ ($flavour =~ /linux/) ? # GNU as doesn't allow most recent hints+ " .long ".sprintf "0x%x",19<<26|$bo<<21|2<<16|16<<1 :+ " bclr $bo,2";+};+my $beqlr = sub {+ my $f = shift;+ my $bo = $f=~/-/ ? 12+2 : 12; # optional "not to be taken" hint+ ($flavour =~ /linux/) ? # GNU as doesn't allow most recent hints+ " .long ".sprintf "0x%X",19<<26|$bo<<21|2<<16|16<<1 :+ " bclr $bo,2";+};+# GNU assembler can't handle extrdi rA,rS,16,48, or when sum of last two+# arguments is 64, with "operand out of range" error.+my $extrdi = sub {+ my ($f,$ra,$rs,$n,$b) = @_;+ $b = ($b+$n)&63; $n = 64-$n;+ " rldicl $ra,$rs,$b,$n";+};+my $vmr = sub {+ my ($f,$vx,$vy) = @_;+ " vor $vx,$vy,$vy";+};++# Some ABIs specify vrsave, special-purpose register #256, as reserved+# for system use.+my $no_vrsave = ($flavour =~ /aix|linux64(le|v2)/);+my $mtspr = sub {+ my ($f,$idx,$ra) = @_;+ if ($idx == 256 && $no_vrsave) {+ " or $ra,$ra,$ra";+ } else {+ " mtspr $idx,$ra";+ }+};+my $mfspr = sub {+ my ($f,$rd,$idx) = @_;+ if ($idx == 256 && $no_vrsave) {+ " li $rd,-1";+ } else {+ " mfspr $rd,$idx";+ }+};++# PowerISA 2.06 stuff+sub vsxmem_op {+ my ($f, $vrt, $ra, $rb, $op) = @_;+ " .long ".sprintf "0x%X",(31<<26)|($vrt<<21)|($ra<<16)|($rb<<11)|($op*2+1);+}+# made-up unaligned memory reference AltiVec/VMX instructions+my $lvx_u = sub { vsxmem_op(@_, 844); }; # lxvd2x+my $stvx_u = sub { vsxmem_op(@_, 972); }; # stxvd2x+my $lvdx_u = sub { vsxmem_op(@_, 588); }; # lxsdx+my $stvdx_u = sub { vsxmem_op(@_, 716); }; # stxsdx+my $lvx_4w = sub { vsxmem_op(@_, 780); }; # lxvw4x+my $stvx_4w = sub { vsxmem_op(@_, 908); }; # stxvw4x+my $lvx_splt = sub { vsxmem_op(@_, 332); }; # lxvdsx+# VSX instruction[s] masqueraded as made-up AltiVec/VMX+my $vpermdi = sub { # xxpermdi+ my ($f, $vrt, $vra, $vrb, $dm) = @_;+ $dm = oct($dm) if ($dm =~ /^0/);+ " .long ".sprintf "0x%X",(60<<26)|($vrt<<21)|($vra<<16)|($vrb<<11)|($dm<<8)|(10<<3)|7;+};++# PowerISA 2.07 stuff+sub vcrypto_op {+ my ($f, $vrt, $vra, $vrb, $op) = @_;+ " .long ".sprintf "0x%X",(4<<26)|($vrt<<21)|($vra<<16)|($vrb<<11)|$op;+}+sub vfour {+ my ($f, $vrt, $vra, $vrb, $vrc, $op) = @_;+ " .long ".sprintf "0x%X",(4<<26)|($vrt<<21)|($vra<<16)|($vrb<<11)|($vrc<<6)|$op;+};+my $vcipher = sub { vcrypto_op(@_, 1288); };+my $vcipherlast = sub { vcrypto_op(@_, 1289); };+my $vncipher = sub { vcrypto_op(@_, 1352); };+my $vncipherlast= sub { vcrypto_op(@_, 1353); };+my $vsbox = sub { vcrypto_op(@_, 0, 1480); };+my $vshasigmad = sub { my ($st,$six)=splice(@_,-2); vcrypto_op(@_, $st<<4|$six, 1730); };+my $vshasigmaw = sub { my ($st,$six)=splice(@_,-2); vcrypto_op(@_, $st<<4|$six, 1666); };+my $vpmsumb = sub { vcrypto_op(@_, 1032); };+my $vpmsumd = sub { vcrypto_op(@_, 1224); };+my $vpmsubh = sub { vcrypto_op(@_, 1096); };+my $vpmsumw = sub { vcrypto_op(@_, 1160); };+# These are not really crypto, but vcrypto_op template works+my $vaddudm = sub { vcrypto_op(@_, 192); };+my $vadduqm = sub { vcrypto_op(@_, 256); };+my $vmuleuw = sub { vcrypto_op(@_, 648); };+my $vmulouw = sub { vcrypto_op(@_, 136); };+my $vrld = sub { vcrypto_op(@_, 196); };+my $vsld = sub { vcrypto_op(@_, 1476); };+my $vsrd = sub { vcrypto_op(@_, 1732); };+my $vsubudm = sub { vcrypto_op(@_, 1216); };+my $vaddcuq = sub { vcrypto_op(@_, 320); };+my $vaddeuqm = sub { vfour(@_,60); };+my $vaddecuq = sub { vfour(@_,61); };+my $vmrgew = sub { vfour(@_,0,1932); };+my $vmrgow = sub { vfour(@_,0,1676); };++my $mtsle = sub {+ my ($f, $arg) = @_;+ " .long ".sprintf "0x%X",(31<<26)|($arg<<21)|(147*2);+};++# VSX instructions masqueraded as AltiVec/VMX+my $mtvrd = sub {+ my ($f, $vrt, $ra) = @_;+ " .long ".sprintf "0x%X",(31<<26)|($vrt<<21)|($ra<<16)|(179<<1)|1;+};+my $mtvrwz = sub {+ my ($f, $vrt, $ra) = @_;+ " .long ".sprintf "0x%X",(31<<26)|($vrt<<21)|($ra<<16)|(243<<1)|1;+};+my $lvwzx_u = sub { vsxmem_op(@_, 12); }; # lxsiwzx+my $stvwx_u = sub { vsxmem_op(@_, 140); }; # stxsiwx++# PowerISA 3.0 stuff+my $maddhdu = sub { vfour(@_,49); };+my $maddld = sub { vfour(@_,51); };+my $darn = sub {+ my ($f, $rt, $l) = @_;+ " .long ".sprintf "0x%X",(31<<26)|($rt<<21)|($l<<16)|(755<<1);+};+my $iseleq = sub {+ my ($f, $rt, $ra, $rb) = @_;+ " .long ".sprintf "0x%X",(31<<26)|($rt<<21)|($ra<<16)|($rb<<11)|(2<<6)|30;+};+# VSX instruction[s] masqueraded as made-up AltiVec/VMX+my $vspltib = sub { # xxspltib+ my ($f, $vrt, $imm8) = @_;+ $imm8 = oct($imm8) if ($imm8 =~ /^0/);+ $imm8 &= 0xff;+ " .long ".sprintf "0x%X",(60<<26)|($vrt<<21)|($imm8<<11)|(360<<1)|1;+};++# PowerISA 3.0B stuff+my $addex = sub {+ my ($f, $rt, $ra, $rb, $cy) = @_; # only cy==0 is specified in 3.0B+ " .long ".sprintf "0x%X",(31<<26)|($rt<<21)|($ra<<16)|($rb<<11)|($cy<<9)|(170<<1);+};+my $vmsumudm = sub { vfour(@_,35); };++while($line=<>) {++ $line =~ s|[#!;].*$||; # get rid of asm-style comments...+ $line =~ s|/\*.*\*/||; # ... and C-style comments...+ $line =~ s|^\s+||; # ... and skip white spaces in beginning...+ $line =~ s|\s+$||; # ... and at the end++ {+ $line =~ s|\.L(\w+)|L$1|g; # common denominator for Locallabel+ $line =~ s|\bL(\w+)|\.L$1|g if ($dotinlocallabels);+ }++ {+ $line =~ s|(^[\.\w]+)\:\s*||;+ my $label = $1;+ if ($label) {+ my $xlated = ($GLOBALS{$label} or $label);+ print "$xlated:";+ if ($flavour =~ /linux.*64(le|v2)/) {+ if ($TYPES{$label} =~ /function/) {+ printf "\n.localentry %s,0\n",$xlated;+ }+ }+ }+ }++ {+ $line =~ s|^\s*(\.?)(\w+)([\.\+\-]?)\s*||;+ my $c = $1; $c = "\t" if ($c eq "");+ my $mnemonic = $2;+ my $f = $3;+ my $opcode = eval("\$$mnemonic");+ $line =~ s/\b(c?[rf]|v|vs)([0-9]+)\b/$2/g if ($c ne "." and $flavour !~ /osx/);+ if (ref($opcode) eq 'CODE') { $line = &$opcode($f,split(/,\s*/,$line)); }+ elsif ($mnemonic) { $line = $c.$mnemonic.$f."\t".$line; }+ }++ print $line if ($line);+ print "\n";+}++close STDOUT;
+ cbits/bearssl/LICENSE view
@@ -0,0 +1,21 @@+Copyright (c) 2016 Thomas Pornin <pornin@bolet.org>++Permission is hereby granted, free of charge, to any person obtaining +a copy of this software and associated documentation files (the+"Software"), to deal in the Software without restriction, including+without limitation the rights to use, copy, modify, merge, publish,+distribute, sublicense, and/or sell copies of the Software, and to+permit persons to whom the Software is furnished to do so, subject to+the following conditions:++The above copyright notice and this permission notice shall be +included in all copies or substantial portions of the Software.++THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, +EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF+MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND +NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS+BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN+ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN+CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE+SOFTWARE.
+ cbits/bearssl/README.md view
@@ -0,0 +1,57 @@+# BearSSL++Constant-time AES and GHASH from [BearSSL](https://bearssl.org/), vendored+here. `VERSION` holds the upstream release these files came from, and+`import.sh` fetches them again.++## Why++crypton's portable AES and GHASH are table-driven, and both index their+tables with a secret: `cbits/aes/generic.c` with a byte of the state, in+every round and in the key expansion, and `cbits/aes/gf.c` with a nibble of+the GHASH accumulator. Both are variable-time by construction, and the+cache-timing attacks on that shape of code are old and well documented.++The processors that have AES instructions do not run any of it -- crypton+asks them first. What runs it is everything else: ppc64le and s390x, where+the instructions exist but crypton has no path to them; 32-bit ARM, where+the same is true; riscv64 and loongarch64; and the boards that genuinely+have no AES instructions at all, of which the Raspberry Pi 3 and 4 are by+some distance the largest population.++BearSSL's answer is bitslicing. `aes_ct64` holds four blocks interleaved+across eight 64-bit words and computes the S-box as boolean algebra, so+there is no table and no address derived from a secret; `ghash_ctmul64`+builds the GF(2^128) multiply out of shifts, masks and integer multiplies+rather than a table of H. Both are plain C99 and assume nothing beyond+`uint64_t`.++## What is here++| file | from |+| --- | --- |+| `aes_ct64.c` | `src/symcipher/aes_ct64.c` |+| `aes_ct64_enc.c` | `src/symcipher/aes_ct64_enc.c` |+| `aes_ct64_dec.c` | `src/symcipher/aes_ct64_dec.c` |+| `ghash_ctmul64.c` | `src/hash/ghash_ctmul64.c` |+| `dec32le.c` | `src/codec/dec32le.c`, for `br_range_dec32le` |+| `LICENSE` | `LICENSE.txt` -- MIT, (c) 2016 Thomas Pornin |++`inner.h` is **crypton's, not upstream's**. Each of those five opens with+`#include "inner.h"`, and upstream's is some two thousand lines declaring+the whole library; these five want six things from it. So this one gives+those six and nothing else, which is what lets the five stay byte for byte+what upstream ships. `import.sh` does not overwrite it.++The byte-order helpers in it are written rather than copied, in their plain+portable form without upstream's unaligned-access fast paths, so that no+platform configuration comes with them.++## Keeping it honest++`cbits/tests/bearssl_diff.c` checks this against the implementation it+replaces: the FIPS-197 vectors first, so that agreement means AES and not+merely that both sides compute the same wrong thing, then a run of random+keys and blocks, then GHASH against the 4-bit table. Given any argument it+corrupts three results on purpose and the comparison has to notice -- a+differential test that cannot fail has said nothing.
+ cbits/bearssl/VERSION view
@@ -0,0 +1,1 @@+0.6
+ cbits/bearssl/aes_ct64.c view
@@ -0,0 +1,398 @@+/*+ * Copyright (c) 2016 Thomas Pornin <pornin@bolet.org>+ *+ * Permission is hereby granted, free of charge, to any person obtaining + * a copy of this software and associated documentation files (the+ * "Software"), to deal in the Software without restriction, including+ * without limitation the rights to use, copy, modify, merge, publish,+ * distribute, sublicense, and/or sell copies of the Software, and to+ * permit persons to whom the Software is furnished to do so, subject to+ * the following conditions:+ *+ * The above copyright notice and this permission notice shall be + * included in all copies or substantial portions of the Software.+ *+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, + * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF+ * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND + * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS+ * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN+ * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN+ * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE+ * SOFTWARE.+ */++#include "inner.h"++/* see inner.h */+void+br_aes_ct64_bitslice_Sbox(uint64_t *q)+{+ /*+ * This S-box implementation is a straightforward translation of+ * the circuit described by Boyar and Peralta in "A new+ * combinational logic minimization technique with applications+ * to cryptology" (https://eprint.iacr.org/2009/191.pdf).+ *+ * Note that variables x* (input) and s* (output) are numbered+ * in "reverse" order (x0 is the high bit, x7 is the low bit).+ */++ uint64_t x0, x1, x2, x3, x4, x5, x6, x7;+ uint64_t y1, y2, y3, y4, y5, y6, y7, y8, y9;+ uint64_t y10, y11, y12, y13, y14, y15, y16, y17, y18, y19;+ uint64_t y20, y21;+ uint64_t z0, z1, z2, z3, z4, z5, z6, z7, z8, z9;+ uint64_t z10, z11, z12, z13, z14, z15, z16, z17;+ uint64_t t0, t1, t2, t3, t4, t5, t6, t7, t8, t9;+ uint64_t t10, t11, t12, t13, t14, t15, t16, t17, t18, t19;+ uint64_t t20, t21, t22, t23, t24, t25, t26, t27, t28, t29;+ uint64_t t30, t31, t32, t33, t34, t35, t36, t37, t38, t39;+ uint64_t t40, t41, t42, t43, t44, t45, t46, t47, t48, t49;+ uint64_t t50, t51, t52, t53, t54, t55, t56, t57, t58, t59;+ uint64_t t60, t61, t62, t63, t64, t65, t66, t67;+ uint64_t s0, s1, s2, s3, s4, s5, s6, s7;++ x0 = q[7];+ x1 = q[6];+ x2 = q[5];+ x3 = q[4];+ x4 = q[3];+ x5 = q[2];+ x6 = q[1];+ x7 = q[0];++ /*+ * Top linear transformation.+ */+ y14 = x3 ^ x5;+ y13 = x0 ^ x6;+ y9 = x0 ^ x3;+ y8 = x0 ^ x5;+ t0 = x1 ^ x2;+ y1 = t0 ^ x7;+ y4 = y1 ^ x3;+ y12 = y13 ^ y14;+ y2 = y1 ^ x0;+ y5 = y1 ^ x6;+ y3 = y5 ^ y8;+ t1 = x4 ^ y12;+ y15 = t1 ^ x5;+ y20 = t1 ^ x1;+ y6 = y15 ^ x7;+ y10 = y15 ^ t0;+ y11 = y20 ^ y9;+ y7 = x7 ^ y11;+ y17 = y10 ^ y11;+ y19 = y10 ^ y8;+ y16 = t0 ^ y11;+ y21 = y13 ^ y16;+ y18 = x0 ^ y16;++ /*+ * Non-linear section.+ */+ t2 = y12 & y15;+ t3 = y3 & y6;+ t4 = t3 ^ t2;+ t5 = y4 & x7;+ t6 = t5 ^ t2;+ t7 = y13 & y16;+ t8 = y5 & y1;+ t9 = t8 ^ t7;+ t10 = y2 & y7;+ t11 = t10 ^ t7;+ t12 = y9 & y11;+ t13 = y14 & y17;+ t14 = t13 ^ t12;+ t15 = y8 & y10;+ t16 = t15 ^ t12;+ t17 = t4 ^ t14;+ t18 = t6 ^ t16;+ t19 = t9 ^ t14;+ t20 = t11 ^ t16;+ t21 = t17 ^ y20;+ t22 = t18 ^ y19;+ t23 = t19 ^ y21;+ t24 = t20 ^ y18;++ t25 = t21 ^ t22;+ t26 = t21 & t23;+ t27 = t24 ^ t26;+ t28 = t25 & t27;+ t29 = t28 ^ t22;+ t30 = t23 ^ t24;+ t31 = t22 ^ t26;+ t32 = t31 & t30;+ t33 = t32 ^ t24;+ t34 = t23 ^ t33;+ t35 = t27 ^ t33;+ t36 = t24 & t35;+ t37 = t36 ^ t34;+ t38 = t27 ^ t36;+ t39 = t29 & t38;+ t40 = t25 ^ t39;++ t41 = t40 ^ t37;+ t42 = t29 ^ t33;+ t43 = t29 ^ t40;+ t44 = t33 ^ t37;+ t45 = t42 ^ t41;+ z0 = t44 & y15;+ z1 = t37 & y6;+ z2 = t33 & x7;+ z3 = t43 & y16;+ z4 = t40 & y1;+ z5 = t29 & y7;+ z6 = t42 & y11;+ z7 = t45 & y17;+ z8 = t41 & y10;+ z9 = t44 & y12;+ z10 = t37 & y3;+ z11 = t33 & y4;+ z12 = t43 & y13;+ z13 = t40 & y5;+ z14 = t29 & y2;+ z15 = t42 & y9;+ z16 = t45 & y14;+ z17 = t41 & y8;++ /*+ * Bottom linear transformation.+ */+ t46 = z15 ^ z16;+ t47 = z10 ^ z11;+ t48 = z5 ^ z13;+ t49 = z9 ^ z10;+ t50 = z2 ^ z12;+ t51 = z2 ^ z5;+ t52 = z7 ^ z8;+ t53 = z0 ^ z3;+ t54 = z6 ^ z7;+ t55 = z16 ^ z17;+ t56 = z12 ^ t48;+ t57 = t50 ^ t53;+ t58 = z4 ^ t46;+ t59 = z3 ^ t54;+ t60 = t46 ^ t57;+ t61 = z14 ^ t57;+ t62 = t52 ^ t58;+ t63 = t49 ^ t58;+ t64 = z4 ^ t59;+ t65 = t61 ^ t62;+ t66 = z1 ^ t63;+ s0 = t59 ^ t63;+ s6 = t56 ^ ~t62;+ s7 = t48 ^ ~t60;+ t67 = t64 ^ t65;+ s3 = t53 ^ t66;+ s4 = t51 ^ t66;+ s5 = t47 ^ t65;+ s1 = t64 ^ ~s3;+ s2 = t55 ^ ~t67;++ q[7] = s0;+ q[6] = s1;+ q[5] = s2;+ q[4] = s3;+ q[3] = s4;+ q[2] = s5;+ q[1] = s6;+ q[0] = s7;+}++/* see inner.h */+void+br_aes_ct64_ortho(uint64_t *q)+{+#define SWAPN(cl, ch, s, x, y) do { \+ uint64_t a, b; \+ a = (x); \+ b = (y); \+ (x) = (a & (uint64_t)cl) | ((b & (uint64_t)cl) << (s)); \+ (y) = ((a & (uint64_t)ch) >> (s)) | (b & (uint64_t)ch); \+ } while (0)++#define SWAP2(x, y) SWAPN(0x5555555555555555, 0xAAAAAAAAAAAAAAAA, 1, x, y)+#define SWAP4(x, y) SWAPN(0x3333333333333333, 0xCCCCCCCCCCCCCCCC, 2, x, y)+#define SWAP8(x, y) SWAPN(0x0F0F0F0F0F0F0F0F, 0xF0F0F0F0F0F0F0F0, 4, x, y)++ SWAP2(q[0], q[1]);+ SWAP2(q[2], q[3]);+ SWAP2(q[4], q[5]);+ SWAP2(q[6], q[7]);++ SWAP4(q[0], q[2]);+ SWAP4(q[1], q[3]);+ SWAP4(q[4], q[6]);+ SWAP4(q[5], q[7]);++ SWAP8(q[0], q[4]);+ SWAP8(q[1], q[5]);+ SWAP8(q[2], q[6]);+ SWAP8(q[3], q[7]);+}++/* see inner.h */+void+br_aes_ct64_interleave_in(uint64_t *q0, uint64_t *q1, const uint32_t *w)+{+ uint64_t x0, x1, x2, x3;++ x0 = w[0];+ x1 = w[1];+ x2 = w[2];+ x3 = w[3];+ x0 |= (x0 << 16);+ x1 |= (x1 << 16);+ x2 |= (x2 << 16);+ x3 |= (x3 << 16);+ x0 &= (uint64_t)0x0000FFFF0000FFFF;+ x1 &= (uint64_t)0x0000FFFF0000FFFF;+ x2 &= (uint64_t)0x0000FFFF0000FFFF;+ x3 &= (uint64_t)0x0000FFFF0000FFFF;+ x0 |= (x0 << 8);+ x1 |= (x1 << 8);+ x2 |= (x2 << 8);+ x3 |= (x3 << 8);+ x0 &= (uint64_t)0x00FF00FF00FF00FF;+ x1 &= (uint64_t)0x00FF00FF00FF00FF;+ x2 &= (uint64_t)0x00FF00FF00FF00FF;+ x3 &= (uint64_t)0x00FF00FF00FF00FF;+ *q0 = x0 | (x2 << 8);+ *q1 = x1 | (x3 << 8);+}++/* see inner.h */+void+br_aes_ct64_interleave_out(uint32_t *w, uint64_t q0, uint64_t q1)+{+ uint64_t x0, x1, x2, x3;++ x0 = q0 & (uint64_t)0x00FF00FF00FF00FF;+ x1 = q1 & (uint64_t)0x00FF00FF00FF00FF;+ x2 = (q0 >> 8) & (uint64_t)0x00FF00FF00FF00FF;+ x3 = (q1 >> 8) & (uint64_t)0x00FF00FF00FF00FF;+ x0 |= (x0 >> 8);+ x1 |= (x1 >> 8);+ x2 |= (x2 >> 8);+ x3 |= (x3 >> 8);+ x0 &= (uint64_t)0x0000FFFF0000FFFF;+ x1 &= (uint64_t)0x0000FFFF0000FFFF;+ x2 &= (uint64_t)0x0000FFFF0000FFFF;+ x3 &= (uint64_t)0x0000FFFF0000FFFF;+ w[0] = (uint32_t)x0 | (uint32_t)(x0 >> 16);+ w[1] = (uint32_t)x1 | (uint32_t)(x1 >> 16);+ w[2] = (uint32_t)x2 | (uint32_t)(x2 >> 16);+ w[3] = (uint32_t)x3 | (uint32_t)(x3 >> 16);+}++static const unsigned char Rcon[] = {+ 0x01, 0x02, 0x04, 0x08, 0x10, 0x20, 0x40, 0x80, 0x1B, 0x36+};++static uint32_t+sub_word(uint32_t x)+{+ uint64_t q[8];++ memset(q, 0, sizeof q);+ q[0] = x;+ br_aes_ct64_ortho(q);+ br_aes_ct64_bitslice_Sbox(q);+ br_aes_ct64_ortho(q);+ return (uint32_t)q[0];+}++/* see inner.h */+unsigned+br_aes_ct64_keysched(uint64_t *comp_skey, const void *key, size_t key_len)+{+ unsigned num_rounds;+ int i, j, k, nk, nkf;+ uint32_t tmp;+ uint32_t skey[60];++ switch (key_len) {+ case 16:+ num_rounds = 10;+ break;+ case 24:+ num_rounds = 12;+ break;+ case 32:+ num_rounds = 14;+ break;+ default:+ /* abort(); */+ return 0;+ }+ nk = (int)(key_len >> 2);+ nkf = (int)((num_rounds + 1) << 2);+ br_range_dec32le(skey, (key_len >> 2), key);+ tmp = skey[(key_len >> 2) - 1];+ for (i = nk, j = 0, k = 0; i < nkf; i ++) {+ if (j == 0) {+ tmp = (tmp << 24) | (tmp >> 8);+ tmp = sub_word(tmp) ^ Rcon[k];+ } else if (nk > 6 && j == 4) {+ tmp = sub_word(tmp);+ }+ tmp ^= skey[i - nk];+ skey[i] = tmp;+ if (++ j == nk) {+ j = 0;+ k ++;+ }+ }++ for (i = 0, j = 0; i < nkf; i += 4, j += 2) {+ uint64_t q[8];++ br_aes_ct64_interleave_in(&q[0], &q[4], skey + i);+ q[1] = q[0];+ q[2] = q[0];+ q[3] = q[0];+ q[5] = q[4];+ q[6] = q[4];+ q[7] = q[4];+ br_aes_ct64_ortho(q);+ comp_skey[j + 0] =+ (q[0] & (uint64_t)0x1111111111111111)+ | (q[1] & (uint64_t)0x2222222222222222)+ | (q[2] & (uint64_t)0x4444444444444444)+ | (q[3] & (uint64_t)0x8888888888888888);+ comp_skey[j + 1] =+ (q[4] & (uint64_t)0x1111111111111111)+ | (q[5] & (uint64_t)0x2222222222222222)+ | (q[6] & (uint64_t)0x4444444444444444)+ | (q[7] & (uint64_t)0x8888888888888888);+ }+ return num_rounds;+}++/* see inner.h */+void+br_aes_ct64_skey_expand(uint64_t *skey,+ unsigned num_rounds, const uint64_t *comp_skey)+{+ unsigned u, v, n;++ n = (num_rounds + 1) << 1;+ for (u = 0, v = 0; u < n; u ++, v += 4) {+ uint64_t x0, x1, x2, x3;++ x0 = x1 = x2 = x3 = comp_skey[u];+ x0 &= (uint64_t)0x1111111111111111;+ x1 &= (uint64_t)0x2222222222222222;+ x2 &= (uint64_t)0x4444444444444444;+ x3 &= (uint64_t)0x8888888888888888;+ x1 >>= 1;+ x2 >>= 2;+ x3 >>= 3;+ skey[v + 0] = (x0 << 4) - x0;+ skey[v + 1] = (x1 << 4) - x1;+ skey[v + 2] = (x2 << 4) - x2;+ skey[v + 3] = (x3 << 4) - x3;+ }+}
+ cbits/bearssl/aes_ct64_dec.c view
@@ -0,0 +1,159 @@+/*+ * Copyright (c) 2016 Thomas Pornin <pornin@bolet.org>+ *+ * Permission is hereby granted, free of charge, to any person obtaining + * a copy of this software and associated documentation files (the+ * "Software"), to deal in the Software without restriction, including+ * without limitation the rights to use, copy, modify, merge, publish,+ * distribute, sublicense, and/or sell copies of the Software, and to+ * permit persons to whom the Software is furnished to do so, subject to+ * the following conditions:+ *+ * The above copyright notice and this permission notice shall be + * included in all copies or substantial portions of the Software.+ *+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, + * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF+ * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND + * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS+ * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN+ * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN+ * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE+ * SOFTWARE.+ */++#include "inner.h"++/* see inner.h */+void+br_aes_ct64_bitslice_invSbox(uint64_t *q)+{+ /*+ * See br_aes_ct_bitslice_invSbox(). This is the natural extension+ * to 64-bit registers.+ */+ uint64_t q0, q1, q2, q3, q4, q5, q6, q7;++ q0 = ~q[0];+ q1 = ~q[1];+ q2 = q[2];+ q3 = q[3];+ q4 = q[4];+ q5 = ~q[5];+ q6 = ~q[6];+ q7 = q[7];+ q[7] = q1 ^ q4 ^ q6;+ q[6] = q0 ^ q3 ^ q5;+ q[5] = q7 ^ q2 ^ q4;+ q[4] = q6 ^ q1 ^ q3;+ q[3] = q5 ^ q0 ^ q2;+ q[2] = q4 ^ q7 ^ q1;+ q[1] = q3 ^ q6 ^ q0;+ q[0] = q2 ^ q5 ^ q7;++ br_aes_ct64_bitslice_Sbox(q);++ q0 = ~q[0];+ q1 = ~q[1];+ q2 = q[2];+ q3 = q[3];+ q4 = q[4];+ q5 = ~q[5];+ q6 = ~q[6];+ q7 = q[7];+ q[7] = q1 ^ q4 ^ q6;+ q[6] = q0 ^ q3 ^ q5;+ q[5] = q7 ^ q2 ^ q4;+ q[4] = q6 ^ q1 ^ q3;+ q[3] = q5 ^ q0 ^ q2;+ q[2] = q4 ^ q7 ^ q1;+ q[1] = q3 ^ q6 ^ q0;+ q[0] = q2 ^ q5 ^ q7;+}++static void+add_round_key(uint64_t *q, const uint64_t *sk)+{+ int i;++ for (i = 0; i < 8; i ++) {+ q[i] ^= sk[i];+ }+}++static void+inv_shift_rows(uint64_t *q)+{+ int i;++ for (i = 0; i < 8; i ++) {+ uint64_t x;++ x = q[i];+ q[i] = (x & (uint64_t)0x000000000000FFFF)+ | ((x & (uint64_t)0x000000000FFF0000) << 4)+ | ((x & (uint64_t)0x00000000F0000000) >> 12)+ | ((x & (uint64_t)0x000000FF00000000) << 8)+ | ((x & (uint64_t)0x0000FF0000000000) >> 8)+ | ((x & (uint64_t)0x000F000000000000) << 12)+ | ((x & (uint64_t)0xFFF0000000000000) >> 4);+ }+}++static inline uint64_t+rotr32(uint64_t x)+{+ return (x << 32) | (x >> 32);+}++static void+inv_mix_columns(uint64_t *q)+{+ uint64_t q0, q1, q2, q3, q4, q5, q6, q7;+ uint64_t r0, r1, r2, r3, r4, r5, r6, r7;++ q0 = q[0];+ q1 = q[1];+ q2 = q[2];+ q3 = q[3];+ q4 = q[4];+ q5 = q[5];+ q6 = q[6];+ q7 = q[7];+ r0 = (q0 >> 16) | (q0 << 48);+ r1 = (q1 >> 16) | (q1 << 48);+ r2 = (q2 >> 16) | (q2 << 48);+ r3 = (q3 >> 16) | (q3 << 48);+ r4 = (q4 >> 16) | (q4 << 48);+ r5 = (q5 >> 16) | (q5 << 48);+ r6 = (q6 >> 16) | (q6 << 48);+ r7 = (q7 >> 16) | (q7 << 48);++ q[0] = q5 ^ q6 ^ q7 ^ r0 ^ r5 ^ r7 ^ rotr32(q0 ^ q5 ^ q6 ^ r0 ^ r5);+ q[1] = q0 ^ q5 ^ r0 ^ r1 ^ r5 ^ r6 ^ r7 ^ rotr32(q1 ^ q5 ^ q7 ^ r1 ^ r5 ^ r6);+ q[2] = q0 ^ q1 ^ q6 ^ r1 ^ r2 ^ r6 ^ r7 ^ rotr32(q0 ^ q2 ^ q6 ^ r2 ^ r6 ^ r7);+ q[3] = q0 ^ q1 ^ q2 ^ q5 ^ q6 ^ r0 ^ r2 ^ r3 ^ r5 ^ rotr32(q0 ^ q1 ^ q3 ^ q5 ^ q6 ^ q7 ^ r0 ^ r3 ^ r5 ^ r7);+ q[4] = q1 ^ q2 ^ q3 ^ q5 ^ r1 ^ r3 ^ r4 ^ r5 ^ r6 ^ r7 ^ rotr32(q1 ^ q2 ^ q4 ^ q5 ^ q7 ^ r1 ^ r4 ^ r5 ^ r6);+ q[5] = q2 ^ q3 ^ q4 ^ q6 ^ r2 ^ r4 ^ r5 ^ r6 ^ r7 ^ rotr32(q2 ^ q3 ^ q5 ^ q6 ^ r2 ^ r5 ^ r6 ^ r7);+ q[6] = q3 ^ q4 ^ q5 ^ q7 ^ r3 ^ r5 ^ r6 ^ r7 ^ rotr32(q3 ^ q4 ^ q6 ^ q7 ^ r3 ^ r6 ^ r7);+ q[7] = q4 ^ q5 ^ q6 ^ r4 ^ r6 ^ r7 ^ rotr32(q4 ^ q5 ^ q7 ^ r4 ^ r7);+}++/* see inner.h */+void+br_aes_ct64_bitslice_decrypt(unsigned num_rounds,+ const uint64_t *skey, uint64_t *q)+{+ unsigned u;++ add_round_key(q, skey + (num_rounds << 3));+ for (u = num_rounds - 1; u > 0; u --) {+ inv_shift_rows(q);+ br_aes_ct64_bitslice_invSbox(q);+ add_round_key(q, skey + (u << 3));+ inv_mix_columns(q);+ }+ inv_shift_rows(q);+ br_aes_ct64_bitslice_invSbox(q);+ add_round_key(q, skey);+}
+ cbits/bearssl/aes_ct64_enc.c view
@@ -0,0 +1,115 @@+/*+ * Copyright (c) 2016 Thomas Pornin <pornin@bolet.org>+ *+ * Permission is hereby granted, free of charge, to any person obtaining + * a copy of this software and associated documentation files (the+ * "Software"), to deal in the Software without restriction, including+ * without limitation the rights to use, copy, modify, merge, publish,+ * distribute, sublicense, and/or sell copies of the Software, and to+ * permit persons to whom the Software is furnished to do so, subject to+ * the following conditions:+ *+ * The above copyright notice and this permission notice shall be + * included in all copies or substantial portions of the Software.+ *+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, + * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF+ * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND + * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS+ * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN+ * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN+ * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE+ * SOFTWARE.+ */++#include "inner.h"++static inline void+add_round_key(uint64_t *q, const uint64_t *sk)+{+ q[0] ^= sk[0];+ q[1] ^= sk[1];+ q[2] ^= sk[2];+ q[3] ^= sk[3];+ q[4] ^= sk[4];+ q[5] ^= sk[5];+ q[6] ^= sk[6];+ q[7] ^= sk[7];+}++static inline void+shift_rows(uint64_t *q)+{+ int i;++ for (i = 0; i < 8; i ++) {+ uint64_t x;++ x = q[i];+ q[i] = (x & (uint64_t)0x000000000000FFFF)+ | ((x & (uint64_t)0x00000000FFF00000) >> 4)+ | ((x & (uint64_t)0x00000000000F0000) << 12)+ | ((x & (uint64_t)0x0000FF0000000000) >> 8)+ | ((x & (uint64_t)0x000000FF00000000) << 8)+ | ((x & (uint64_t)0xF000000000000000) >> 12)+ | ((x & (uint64_t)0x0FFF000000000000) << 4);+ }+}++static inline uint64_t+rotr32(uint64_t x)+{+ return (x << 32) | (x >> 32);+}++static inline void+mix_columns(uint64_t *q)+{+ uint64_t q0, q1, q2, q3, q4, q5, q6, q7;+ uint64_t r0, r1, r2, r3, r4, r5, r6, r7;++ q0 = q[0];+ q1 = q[1];+ q2 = q[2];+ q3 = q[3];+ q4 = q[4];+ q5 = q[5];+ q6 = q[6];+ q7 = q[7];+ r0 = (q0 >> 16) | (q0 << 48);+ r1 = (q1 >> 16) | (q1 << 48);+ r2 = (q2 >> 16) | (q2 << 48);+ r3 = (q3 >> 16) | (q3 << 48);+ r4 = (q4 >> 16) | (q4 << 48);+ r5 = (q5 >> 16) | (q5 << 48);+ r6 = (q6 >> 16) | (q6 << 48);+ r7 = (q7 >> 16) | (q7 << 48);++ q[0] = q7 ^ r7 ^ r0 ^ rotr32(q0 ^ r0);+ q[1] = q0 ^ r0 ^ q7 ^ r7 ^ r1 ^ rotr32(q1 ^ r1);+ q[2] = q1 ^ r1 ^ r2 ^ rotr32(q2 ^ r2);+ q[3] = q2 ^ r2 ^ q7 ^ r7 ^ r3 ^ rotr32(q3 ^ r3);+ q[4] = q3 ^ r3 ^ q7 ^ r7 ^ r4 ^ rotr32(q4 ^ r4);+ q[5] = q4 ^ r4 ^ r5 ^ rotr32(q5 ^ r5);+ q[6] = q5 ^ r5 ^ r6 ^ rotr32(q6 ^ r6);+ q[7] = q6 ^ r6 ^ r7 ^ rotr32(q7 ^ r7);+}++/* see inner.h */+void+br_aes_ct64_bitslice_encrypt(unsigned num_rounds,+ const uint64_t *skey, uint64_t *q)+{+ unsigned u;++ add_round_key(q, skey);+ for (u = 1; u < num_rounds; u ++) {+ br_aes_ct64_bitslice_Sbox(q);+ shift_rows(q);+ mix_columns(q);+ add_round_key(q, skey + (u << 3));+ }+ br_aes_ct64_bitslice_Sbox(q);+ shift_rows(q);+ add_round_key(q, skey + (num_rounds << 3));+}
+ cbits/bearssl/dec32le.c view
@@ -0,0 +1,38 @@+/*+ * Copyright (c) 2016 Thomas Pornin <pornin@bolet.org>+ *+ * Permission is hereby granted, free of charge, to any person obtaining + * a copy of this software and associated documentation files (the+ * "Software"), to deal in the Software without restriction, including+ * without limitation the rights to use, copy, modify, merge, publish,+ * distribute, sublicense, and/or sell copies of the Software, and to+ * permit persons to whom the Software is furnished to do so, subject to+ * the following conditions:+ *+ * The above copyright notice and this permission notice shall be + * included in all copies or substantial portions of the Software.+ *+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, + * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF+ * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND + * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS+ * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN+ * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN+ * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE+ * SOFTWARE.+ */++#include "inner.h"++/* see inner.h */+void+br_range_dec32le(uint32_t *v, size_t num, const void *src)+{+ const unsigned char *buf;++ buf = src;+ while (num -- > 0) {+ *v ++ = br_dec32le(buf);+ buf += 4;+ }+}
+ cbits/bearssl/ghash_ctmul64.c view
@@ -0,0 +1,154 @@+/*+ * Copyright (c) 2016 Thomas Pornin <pornin@bolet.org>+ *+ * Permission is hereby granted, free of charge, to any person obtaining + * a copy of this software and associated documentation files (the+ * "Software"), to deal in the Software without restriction, including+ * without limitation the rights to use, copy, modify, merge, publish,+ * distribute, sublicense, and/or sell copies of the Software, and to+ * permit persons to whom the Software is furnished to do so, subject to+ * the following conditions:+ *+ * The above copyright notice and this permission notice shall be + * included in all copies or substantial portions of the Software.+ *+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, + * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF+ * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND + * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS+ * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN+ * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN+ * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE+ * SOFTWARE.+ */++#include "inner.h"++/*+ * This is the 64-bit variant of br_ghash_ctmul32(), with 64-bit operands+ * and bit reversal of 64-bit words.+ */++static inline uint64_t+bmul64(uint64_t x, uint64_t y)+{+ uint64_t x0, x1, x2, x3;+ uint64_t y0, y1, y2, y3;+ uint64_t z0, z1, z2, z3;++ x0 = x & (uint64_t)0x1111111111111111;+ x1 = x & (uint64_t)0x2222222222222222;+ x2 = x & (uint64_t)0x4444444444444444;+ x3 = x & (uint64_t)0x8888888888888888;+ y0 = y & (uint64_t)0x1111111111111111;+ y1 = y & (uint64_t)0x2222222222222222;+ y2 = y & (uint64_t)0x4444444444444444;+ y3 = y & (uint64_t)0x8888888888888888;+ z0 = (x0 * y0) ^ (x1 * y3) ^ (x2 * y2) ^ (x3 * y1);+ z1 = (x0 * y1) ^ (x1 * y0) ^ (x2 * y3) ^ (x3 * y2);+ z2 = (x0 * y2) ^ (x1 * y1) ^ (x2 * y0) ^ (x3 * y3);+ z3 = (x0 * y3) ^ (x1 * y2) ^ (x2 * y1) ^ (x3 * y0);+ z0 &= (uint64_t)0x1111111111111111;+ z1 &= (uint64_t)0x2222222222222222;+ z2 &= (uint64_t)0x4444444444444444;+ z3 &= (uint64_t)0x8888888888888888;+ return z0 | z1 | z2 | z3;+}++static uint64_t+rev64(uint64_t x)+{+#define RMS(m, s) do { \+ x = ((x & (uint64_t)(m)) << (s)) \+ | ((x >> (s)) & (uint64_t)(m)); \+ } while (0)++ RMS(0x5555555555555555, 1);+ RMS(0x3333333333333333, 2);+ RMS(0x0F0F0F0F0F0F0F0F, 4);+ RMS(0x00FF00FF00FF00FF, 8);+ RMS(0x0000FFFF0000FFFF, 16);+ return (x << 32) | (x >> 32);++#undef RMS+}++/* see bearssl_ghash.h */+void+br_ghash_ctmul64(void *y, const void *h, const void *data, size_t len)+{+ const unsigned char *buf, *hb;+ unsigned char *yb;+ uint64_t y0, y1;+ uint64_t h0, h1, h2, h0r, h1r, h2r;++ buf = data;+ yb = y;+ hb = h;+ y1 = br_dec64be(yb);+ y0 = br_dec64be(yb + 8);+ h1 = br_dec64be(hb);+ h0 = br_dec64be(hb + 8);+ h0r = rev64(h0);+ h1r = rev64(h1);+ h2 = h0 ^ h1;+ h2r = h0r ^ h1r;+ while (len > 0) {+ const unsigned char *src;+ unsigned char tmp[16];+ uint64_t y0r, y1r, y2, y2r;+ uint64_t z0, z1, z2, z0h, z1h, z2h;+ uint64_t v0, v1, v2, v3;++ if (len >= 16) {+ src = buf;+ buf += 16;+ len -= 16;+ } else {+ memcpy(tmp, buf, len);+ memset(tmp + len, 0, (sizeof tmp) - len);+ src = tmp;+ len = 0;+ }+ y1 ^= br_dec64be(src);+ y0 ^= br_dec64be(src + 8);++ y0r = rev64(y0);+ y1r = rev64(y1);+ y2 = y0 ^ y1;+ y2r = y0r ^ y1r;++ z0 = bmul64(y0, h0);+ z1 = bmul64(y1, h1);+ z2 = bmul64(y2, h2);+ z0h = bmul64(y0r, h0r);+ z1h = bmul64(y1r, h1r);+ z2h = bmul64(y2r, h2r);+ z2 ^= z0 ^ z1;+ z2h ^= z0h ^ z1h;+ z0h = rev64(z0h) >> 1;+ z1h = rev64(z1h) >> 1;+ z2h = rev64(z2h) >> 1;++ v0 = z0;+ v1 = z0h ^ z2;+ v2 = z1 ^ z2h;+ v3 = z1h;++ v3 = (v3 << 1) | (v2 >> 63);+ v2 = (v2 << 1) | (v1 >> 63);+ v1 = (v1 << 1) | (v0 >> 63);+ v0 = (v0 << 1);++ v2 ^= v0 ^ (v0 >> 1) ^ (v0 >> 2) ^ (v0 >> 7);+ v1 ^= (v0 << 63) ^ (v0 << 62) ^ (v0 << 57);+ v3 ^= v1 ^ (v1 >> 1) ^ (v1 >> 2) ^ (v1 >> 7);+ v2 ^= (v1 << 63) ^ (v1 << 62) ^ (v1 << 57);++ y0 = v2;+ y1 = v3;+ }++ br_enc64be(yb, y1);+ br_enc64be(yb + 8, y0);+}
+ cbits/bearssl/import.sh view
@@ -0,0 +1,34 @@+#!/bin/sh+# Re-import the vendored parts of BearSSL.+#+# Only the five files crypton calls are kept, and they are kept unmodified,+# so that `diff` against a new release is readable. What they want from+# upstream's two-thousand-line src/inner.h is supplied by the inner.h here,+# which is crypton's and is NOT overwritten by this script. Run this from+# cbits/bearssl:+#+# ./import.sh [version]+#+# and commit the result together with the VERSION line it writes, so that+# the tree always says which upstream release it holds.+set -eu++VER=${1:-0.6}+HERE=$(cd "$(dirname "$0")" && pwd)+TMP=$(mktemp -d)+trap 'rm -rf "$TMP"' EXIT++curl -sSL "https://bearssl.org/bearssl-$VER.tar.gz" -o "$TMP/b.tgz"+tar xzf "$TMP/b.tgz" -C "$TMP"+SRC="$TMP/bearssl-$VER"++cp "$SRC/LICENSE.txt" "$HERE/LICENSE"+cp "$SRC/src/symcipher/aes_ct64.c" "$HERE/"+cp "$SRC/src/symcipher/aes_ct64_enc.c" "$HERE/"+cp "$SRC/src/symcipher/aes_ct64_dec.c" "$HERE/"+cp "$SRC/src/hash/ghash_ctmul64.c" "$HERE/"+cp "$SRC/src/codec/dec32le.c" "$HERE/"+echo "$VER" > "$HERE/VERSION"++echo "imported BearSSL $VER"+echo "now run the differential test in cbits/tests/bearssl_diff.c"
+ cbits/bearssl/inner.h view
@@ -0,0 +1,92 @@+/*+ * Stands in for BearSSL's own src/inner.h.+ *+ * The five .c files beside this one are upstream's, byte for byte, and each+ * of them opens with #include "inner.h". Upstream's is some two thousand+ * lines and declares the whole library; these five want six things from it.+ * So this header gives those six and nothing else, and the upstream files+ * stay unmodified -- which is what makes `diff` against a new BearSSL+ * release readable. See README.md.+ *+ * The byte-order helpers are written here rather than copied, in their plain+ * portable form without upstream's unaligned-access fast paths, so that they+ * carry no platform configuration with them. cbits/tests/bearssl_diff.c+ * checks the whole thing against crypton's existing implementation.+ */+#ifndef CRYPTON_BEARSSL_INNER_H+#define CRYPTON_BEARSSL_INNER_H++#include <stddef.h>+#include <stdint.h>+#include <string.h>++static inline uint32_t+br_dec32le(const void *src)+{+ const unsigned char *b = src;++ return (uint32_t)b[0]+ | ((uint32_t)b[1] << 8)+ | ((uint32_t)b[2] << 16)+ | ((uint32_t)b[3] << 24);+}++static inline void+br_enc32le(void *dst, uint32_t x)+{+ unsigned char *b = dst;++ b[0] = (unsigned char)x;+ b[1] = (unsigned char)(x >> 8);+ b[2] = (unsigned char)(x >> 16);+ b[3] = (unsigned char)(x >> 24);+}++static inline uint64_t+br_dec64be(const void *src)+{+ const unsigned char *b = src;+ uint64_t x = 0;+ int i;++ for (i = 0; i < 8; i++)+ x = (x << 8) | (uint64_t)b[i];+ return x;+}++static inline void+br_enc64be(void *dst, uint64_t x)+{+ unsigned char *b = dst;+ int i;++ for (i = 7; i >= 0; i--) {+ b[i] = (unsigned char)(x & 0xff);+ x >>= 8;+ }+}++/* dec32le.c */+void br_range_dec32le(uint32_t *v, size_t num, const void *src);++/* aes_ct64.c */+void br_aes_ct64_bitslice_Sbox(uint64_t *q);+void br_aes_ct64_ortho(uint64_t *q);+void br_aes_ct64_interleave_in(uint64_t *q0, uint64_t *q1, const uint32_t *w);+void br_aes_ct64_interleave_out(uint32_t *w, uint64_t q0, uint64_t q1);+unsigned br_aes_ct64_keysched(uint64_t *comp_skey, const void *key,+ size_t key_len);+void br_aes_ct64_skey_expand(uint64_t *skey, unsigned num_rounds,+ const uint64_t *comp_skey);++/* aes_ct64_enc.c, aes_ct64_dec.c */+void br_aes_ct64_bitslice_encrypt(unsigned num_rounds, const uint64_t *skey,+ uint64_t *q);+void br_aes_ct64_bitslice_invSbox(uint64_t *q);+void br_aes_ct64_bitslice_decrypt(unsigned num_rounds, const uint64_t *skey,+ uint64_t *q);++/* ghash_ctmul64.c */+void br_ghash_ctmul64(void *y, const void *h, const void *data, size_t len);++#endif
cbits/crypton_aes.c view
@@ -38,6 +38,9 @@ #include <aes/generic.h> #include <aes/gf.h> #include <aes/x86ni.h>+#ifdef WITH_PPC8_CRYPTO+#include <aes/ppc8.h>+#endif #ifdef WITH_GCM_FUSED #include <aes/gcm_fused_x86.h> #endif@@ -54,6 +57,8 @@ uint32_t spoint, aes_block *input, uint32_t nb_blocks); void crypton_aes_generic_gcm_encrypt(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length); void crypton_aes_generic_gcm_decrypt(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length);+void crypton_aes_bitsliced_gcm_encrypt(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length);+void crypton_aes_bitsliced_gcm_decrypt(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length); void crypton_aes_generic_ocb_encrypt(uint8_t *output, aes_ocb *ocb, aes_key *key, uint8_t *input, uint32_t length); void crypton_aes_generic_ocb_decrypt(uint8_t *output, aes_ocb *ocb, aes_key *key, uint8_t *input, uint32_t length); void crypton_aes_generic_ccm_encrypt(uint8_t *output, aes_ccm *ccm, aes_key *key, uint8_t *input, uint32_t length);@@ -208,7 +213,7 @@ typedef void (*gf_mul_f)(block128 *a, const table_4bit htable); typedef void (*gf_mul4_f)(block128 *a, const block128 *blocks, const table_4bit htable); -#if defined(WITH_AESNI) || defined(WITH_ARMV8_CRYPTO)+#if defined(WITH_AESNI) || defined(WITH_ARMV8_CRYPTO) || defined(WITH_PPC8_CRYPTO) #define GET_INIT(strength) \ ((init_f) (crypton_aes_branch_table[INIT_128 + strength])) #define GET_ECB_ENCRYPT(strength) \@@ -255,12 +260,16 @@ #define GET_ECB_DECRYPT(strength) crypton_aes_generic_decrypt_ecb #define GET_CBC_ENCRYPT(strength) crypton_aes_generic_encrypt_cbc #define GET_CBC_DECRYPT(strength) crypton_aes_generic_decrypt_cbc-#define GET_CTR_ENCRYPT(strength) crypton_aes_generic_encrypt_ctr-#define GET_C32_ENCRYPT(strength) crypton_aes_generic_encrypt_c32+/* No accelerator is compiled in, so the portable schedule is the only one+ * a key can hold and the four-block CTR is named here rather than installed+ * at run time. This is the build ppc64le, s390x, riscv64, 32-bit ARM and+ * -f-support_aesni all get, and nothing in it reads the branch table. */+#define GET_CTR_ENCRYPT(strength) crypton_aes_bitsliced_encrypt_ctr+#define GET_C32_ENCRYPT(strength) crypton_aes_bitsliced_encrypt_c32 #define GET_XTS_ENCRYPT(strength) crypton_aes_generic_encrypt_xts #define GET_XTS_DECRYPT(strength) crypton_aes_generic_decrypt_xts-#define GET_GCM_ENCRYPT(strength) crypton_aes_generic_gcm_encrypt-#define GET_GCM_DECRYPT(strength) crypton_aes_generic_gcm_decrypt+#define GET_GCM_ENCRYPT(strength) crypton_aes_bitsliced_gcm_encrypt+#define GET_GCM_DECRYPT(strength) crypton_aes_bitsliced_gcm_decrypt #define GET_OCB_ENCRYPT(strength) crypton_aes_generic_ocb_encrypt #define GET_OCB_DECRYPT(strength) crypton_aes_generic_ocb_decrypt #define GET_CCM_ENCRYPT(strength) crypton_aes_generic_ccm_encrypt@@ -272,10 +281,6 @@ #define crypton_gf_mul4(a,b,t) crypton_aes_generic_gf_mul4(a,b,t) #endif -#define CPU_AESNI 0-#define CPU_PCLMUL 1-#define CPU_OPTION_COUNT 2- static uint8_t crypton_aes_cpu_options[CPU_OPTION_COUNT] = {}; #if defined(ARCH_X86) && defined(WITH_AESNI)@@ -440,6 +445,74 @@ * A constructor runs while there is one thread, which is the cheapest way to * have no race at all: no flag to test, no lock to take, and one less thing * for crypton_aes_initkey to do per key. */+/*+ * The two CTR entries, where the portable implementation is the one that+ * will run. This is for the build that has an accelerator compiled in and+ * did not find it on the processor; a build with no accelerator at all+ * names them directly in the GET_ macros above and never reads this table.+ *+ * They cannot be the defaults. crypton_aes_generic_encrypt_c32 is not+ * overridden on AArch64 -- the ARMv8 table has no C32 of its own -- and+ * neither CCM nor OCB is overridden anywhere, so those reach the block+ * function through the branch table and work whatever is installed. A CTR+ * that read the portable schedule directly would be wrong on exactly those+ * machines. So these go in only once nothing has claimed AES.+ */+static void initialize_table_bitsliced(void)+{+ crypton_aes_branch_table[ENCRYPT_CTR_128] = crypton_aes_bitsliced_encrypt_ctr;+ crypton_aes_branch_table[ENCRYPT_CTR_192] = crypton_aes_bitsliced_encrypt_ctr;+ crypton_aes_branch_table[ENCRYPT_CTR_256] = crypton_aes_bitsliced_encrypt_ctr;+ crypton_aes_branch_table[ENCRYPT_C32_128] = crypton_aes_bitsliced_encrypt_c32;+ crypton_aes_branch_table[ENCRYPT_C32_192] = crypton_aes_bitsliced_encrypt_c32;+ crypton_aes_branch_table[ENCRYPT_C32_256] = crypton_aes_bitsliced_encrypt_c32;+ crypton_aes_branch_table[ENCRYPT_GCM_128] = crypton_aes_bitsliced_gcm_encrypt;+ crypton_aes_branch_table[ENCRYPT_GCM_192] = crypton_aes_bitsliced_gcm_encrypt;+ crypton_aes_branch_table[ENCRYPT_GCM_256] = crypton_aes_bitsliced_gcm_encrypt;+ crypton_aes_branch_table[DECRYPT_GCM_128] = crypton_aes_bitsliced_gcm_decrypt;+ crypton_aes_branch_table[DECRYPT_GCM_192] = crypton_aes_bitsliced_gcm_decrypt;+ crypton_aes_branch_table[DECRYPT_GCM_256] = crypton_aes_bitsliced_gcm_decrypt;+}++#ifdef WITH_PPC8_CRYPTO+/*+ * POWER8. One function per operation rather than three: the assembly reads+ * the round count out of the key, so the same entry serves every key size.+ *+ * GCM and the 32-bit counter are left where they are. Neither has an entry+ * in the assembly, and the generic loops reach the block function and the+ * GHASH through this table, so both get the instructions anyway.+ */+static void initialize_table_ppc8(void)+{+ int sz;++ if (!crypton_aes_ppc8_available())+ return;+ /* what stops initialize_table_bitsliced below from putting its own CTR+ * in, which would read this key as a bitsliced schedule */+ crypton_aes_cpu_options[CPU_AESNI] = 1;+ crypton_aes_cpu_options[CPU_PCLMUL] = 1;++ for (sz = 0; sz < 3; sz++) {+ crypton_aes_branch_table[INIT_128 + sz] = crypton_aes_ppc8_init;+ crypton_aes_branch_table[ENCRYPT_BLOCK_128 + sz] = crypton_aes_ppc8_encrypt_block;+ crypton_aes_branch_table[DECRYPT_BLOCK_128 + sz] = crypton_aes_ppc8_decrypt_block;+ crypton_aes_branch_table[ENCRYPT_ECB_128 + sz] = crypton_aes_ppc8_encrypt_ecb;+ crypton_aes_branch_table[DECRYPT_ECB_128 + sz] = crypton_aes_ppc8_decrypt_ecb;+ crypton_aes_branch_table[ENCRYPT_CBC_128 + sz] = crypton_aes_ppc8_encrypt_cbc;+ crypton_aes_branch_table[DECRYPT_CBC_128 + sz] = crypton_aes_ppc8_decrypt_cbc;+ crypton_aes_branch_table[ENCRYPT_CTR_128 + sz] = crypton_aes_ppc8_encrypt_ctr;+ crypton_aes_branch_table[ENCRYPT_XTS_128 + sz] = crypton_aes_ppc8_encrypt_xts;+ crypton_aes_branch_table[DECRYPT_XTS_128 + sz] = crypton_aes_ppc8_decrypt_xts;+ }++ crypton_aes_branch_table[GHASH_HINIT] = crypton_aes_ppc8_hinit;+ crypton_aes_branch_table[GHASH_GF_MUL] = crypton_aes_ppc8_gf_mul;+ crypton_aes_branch_table[GHASH_GF_MUL4] = crypton_aes_ppc8_gf_mul4;+}+#endif+ static void crypton_aes_cpu_setup(void) { #if defined(ARCH_X86) && defined(WITH_AESNI)@@ -448,6 +521,11 @@ #ifdef WITH_ARMV8_CRYPTO initialize_table_armv8(); #endif+#ifdef WITH_PPC8_CRYPTO+ initialize_table_ppc8();+#endif+ if (crypton_aes_cpu_options[CPU_AESNI] == 0)+ initialize_table_bitsliced(); } __attribute__((constructor))@@ -1136,18 +1214,30 @@ block128_xor((block128 *) tag, &ocb->sum_aad); } +/*+ * These three reach cbits/aes/generic.c directly rather than through the+ * branch table, so they run only where the portable implementation was the+ * one installed -- which is what lets them read its schedule and hand its+ * core four blocks at a time. The CTR entries below cannot do the same:+ * they go through the branch table for the block, and so run on accelerated+ * machines too.+ */ void crypton_aes_generic_encrypt_ecb(aes_block *output, aes_key *key, aes_block *input, uint32_t nb_blocks) {- for ( ; nb_blocks-- > 0; input++, output++) {- crypton_aes_generic_encrypt_block(output, key, input);- }+ aes_sched sched;++ crypton_aes_generic_schedule(&sched, key);+ crypton_aes_generic_blocks((uint8_t *) output, (const uint8_t *) input,+ nb_blocks, &sched, 0); } void crypton_aes_generic_decrypt_ecb(aes_block *output, aes_key *key, aes_block *input, uint32_t nb_blocks) {- for ( ; nb_blocks-- > 0; input++, output++) {- crypton_aes_generic_decrypt_block(output, key, input);- }+ aes_sched sched;++ crypton_aes_generic_schedule(&sched, key);+ crypton_aes_generic_blocks((uint8_t *) output, (const uint8_t *) input,+ nb_blocks, &sched, 1); } void crypton_aes_generic_encrypt_cbc(aes_block *output, aes_key *key, aes_block *iv, aes_block *input, uint32_t nb_blocks)@@ -1163,18 +1253,35 @@ } } +/*+ * Decryption, unlike encryption, does not wait for the block before it:+ * every ciphertext block is ready at once and the chaining is a XOR+ * afterwards. So four at a pass, with the ciphertext copied aside first,+ * since the caller is allowed to decrypt in place and the next group's IV+ * is the last ciphertext block of this one.+ */ void crypton_aes_generic_decrypt_cbc(aes_block *output, aes_key *key, aes_block *ivini, aes_block *input, uint32_t nb_blocks) {- aes_block block, blocko;+ aes_sched sched; aes_block iv;+ uint8_t ct[64], plain[64]; - /* preload IV in block */+ crypton_aes_generic_schedule(&sched, key); block128_copy(&iv, ivini);- for ( ; nb_blocks-- > 0; input++, output++) {- block128_copy(&block, (block128 *) input);- crypton_aes_generic_decrypt_block(&blocko, key, &block);- block128_vxor((block128 *) output, &blocko, &iv);- block128_copy(&iv, &block);+ while (nb_blocks > 0) {+ uint32_t n = nb_blocks < 4 ? nb_blocks : 4;+ uint32_t i;++ memcpy(ct, input, n * 16);+ crypton_aes_generic_blocks(plain, ct, n, &sched, 1);+ for (i = 0; i < n; i++) {+ block128_vxor((block128 *) (output + i),+ (block128 *) (plain + 16 * i), &iv);+ block128_copy(&iv, (block128 *) (ct + 16 * i));+ }+ input += n;+ output += n;+ nb_blocks -= n; } } @@ -1229,7 +1336,9 @@ void crypton_aes_generic_encrypt_xts(aes_block *output, aes_key *k1, aes_key *k2, aes_block *dataunit, uint32_t spoint, aes_block *input, uint32_t nb_blocks) {- aes_block block, tweak;+ aes_sched sched;+ aes_block tweak;+ uint8_t buf[64], tw[64]; /* load IV and encrypt it using k2 as the tweak */ block128_copy(&tweak, dataunit);@@ -1239,17 +1348,34 @@ while (spoint-- > 0) crypton_aes_generic_gf_mulx(&tweak); - for ( ; nb_blocks-- > 0; input++, output++, crypton_aes_generic_gf_mulx(&tweak)) {- block128_vxor(&block, input, &tweak);- crypton_aes_encrypt_block(&block, k1, &block);- block128_vxor(output, &block, &tweak);+ crypton_aes_generic_schedule(&sched, k1);+ while (nb_blocks > 0) {+ uint32_t n = nb_blocks < 4 ? nb_blocks : 4;+ uint32_t i;++ /* the tweaks for a group depend on nothing but each other, so+ * all four are known before any block is enciphered */+ for (i = 0; i < n; i++) {+ block128_copy((block128 *) (tw + 16 * i), &tweak);+ block128_vxor((block128 *) (buf + 16 * i), input + i, &tweak);+ crypton_aes_generic_gf_mulx(&tweak);+ }+ crypton_aes_generic_blocks(buf, buf, n, &sched, 0);+ for (i = 0; i < n; i++)+ block128_vxor(output + i, (block128 *) (buf + 16 * i),+ (block128 *) (tw + 16 * i));+ input += n;+ output += n;+ nb_blocks -= n; } } void crypton_aes_generic_decrypt_xts(aes_block *output, aes_key *k1, aes_key *k2, aes_block *dataunit, uint32_t spoint, aes_block *input, uint32_t nb_blocks) {- aes_block block, tweak;+ aes_sched sched;+ aes_block tweak;+ uint8_t buf[64], tw[64]; /* load IV and encrypt it using k2 as the tweak */ block128_copy(&tweak, dataunit);@@ -1259,10 +1385,25 @@ while (spoint-- > 0) crypton_aes_generic_gf_mulx(&tweak); - for ( ; nb_blocks-- > 0; input++, output++, crypton_aes_generic_gf_mulx(&tweak)) {- block128_vxor(&block, input, &tweak);- crypton_aes_decrypt_block(&block, k1, &block);- block128_vxor(output, &block, &tweak);+ crypton_aes_generic_schedule(&sched, k1);+ while (nb_blocks > 0) {+ uint32_t n = nb_blocks < 4 ? nb_blocks : 4;+ uint32_t i;++ /* the tweaks for a group depend on nothing but each other, so+ * all four are known before any block is enciphered */+ for (i = 0; i < n; i++) {+ block128_copy((block128 *) (tw + 16 * i), &tweak);+ block128_vxor((block128 *) (buf + 16 * i), input + i, &tweak);+ crypton_aes_generic_gf_mulx(&tweak);+ }+ crypton_aes_generic_blocks(buf, buf, n, &sched, 1);+ for (i = 0; i < n; i++)+ block128_vxor(output + i, (block128 *) (buf + 16 * i),+ (block128 *) (tw + 16 * i));+ input += n;+ output += n;+ nb_blocks -= n; } } @@ -1491,6 +1632,116 @@ block128_zero(&tmp); block128_copy_bytes(&tmp, input, length); ccm_cbcmac_add(ccm, key, &tmp);+ }+}++/*+ * GCM where the portable implementation is the one that will run.+ *+ * The same shape as the two above, with two differences: the four counter+ * blocks of a group are encrypted together, which is what the bitsliced+ * core is for, and the schedule is expanded once for the message rather+ * than once per block.+ *+ * These cannot replace the generic pair. That pair also runs on an x86+ * machine that has AES-NI and no carry-less multiply, where crypton_aes.c+ * installs the AES-NI block function and leaves GCM alone -- and there the+ * key holds the AES-NI schedule, not this one. So the choice is made where+ * the rest of the portable entries are chosen.+ */+void crypton_aes_bitsliced_gcm_encrypt(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length)+{+ aes_sched sched;+ aes_block out;++ crypton_aes_generic_schedule(&sched, key);+ gcm->length_input += length;+ for (; length >= 64; input += 64, output += 64, length -= 64) {+ aes_block buf[4];+ int i;++ for (i = 0; i < 4; i++) {+ block128_inc32_be(&gcm->civ);+ block128_copy(&buf[i], &gcm->civ);+ }+ crypton_aes_generic_blocks((uint8_t *) buf, (const uint8_t *) buf,+ 4, &sched, 0);+ for (i = 0; i < 4; i++)+ block128_xor(&buf[i], (block128 *) (input + 16 * i));+ gcm_ghash_add4(gcm, buf);+ for (i = 0; i < 4; i++)+ block128_copy((block128 *) (output + 16 * i), &buf[i]);+ }+ for (; length >= 16; input += 16, output += 16, length -= 16) {+ block128_inc32_be(&gcm->civ);+ crypton_aes_generic_blocks((uint8_t *) &out,+ (const uint8_t *) &gcm->civ, 1, &sched, 0);+ block128_xor(&out, (block128 *) input);+ gcm_ghash_add(gcm, &out);+ block128_copy((block128 *) output, &out);+ }+ if (length > 0) {+ aes_block tmp;+ uint32_t i;++ block128_inc32_be(&gcm->civ);+ crypton_aes_generic_blocks((uint8_t *) &out,+ (const uint8_t *) &gcm->civ, 1, &sched, 0);+ block128_zero(&tmp);+ block128_copy_bytes(&tmp, input, length);+ block128_xor_bytes(&tmp, out.b, length);+ gcm_ghash_add(gcm, &tmp);+ for (i = 0; i < length; i++)+ output[i] = tmp.b[i];+ }+}++void crypton_aes_bitsliced_gcm_decrypt(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length)+{+ aes_sched sched;+ aes_block out;++ crypton_aes_generic_schedule(&sched, key);+ gcm->length_input += length;+ /* GHASH all four ciphertext blocks before writing any plaintext, since+ * output may be input */+ for (; length >= 64; input += 64, output += 64, length -= 64) {+ aes_block buf[4];+ int i;++ gcm_ghash_add4(gcm, (const block128 *) input);+ for (i = 0; i < 4; i++) {+ block128_inc32_be(&gcm->civ);+ block128_copy(&buf[i], &gcm->civ);+ }+ crypton_aes_generic_blocks((uint8_t *) buf, (const uint8_t *) buf,+ 4, &sched, 0);+ for (i = 0; i < 4; i++) {+ block128_xor(&buf[i], (block128 *) (input + 16 * i));+ block128_copy((block128 *) (output + 16 * i), &buf[i]);+ }+ }+ for (; length >= 16; input += 16, output += 16, length -= 16) {+ block128_inc32_be(&gcm->civ);+ crypton_aes_generic_blocks((uint8_t *) &out,+ (const uint8_t *) &gcm->civ, 1, &sched, 0);+ gcm_ghash_add(gcm, (block128 *) input);+ block128_xor(&out, (block128 *) input);+ block128_copy((block128 *) output, &out);+ }+ if (length > 0) {+ aes_block tmp;+ uint32_t i;++ block128_inc32_be(&gcm->civ);+ crypton_aes_generic_blocks((uint8_t *) &out,+ (const uint8_t *) &gcm->civ, 1, &sched, 0);+ block128_zero(&tmp);+ block128_copy_bytes(&tmp, input, length);+ gcm_ghash_add(gcm, &tmp);+ block128_xor_bytes(&tmp, out.b, length);+ for (i = 0; i < length; i++)+ output[i] = tmp.b[i]; } }
cbits/crypton_armv8_target.h view
@@ -23,7 +23,19 @@ #define CRYPTON_ARMV8_TARGET_H #ifdef WITH_TARGET_ATTRIBUTES-#if defined(__clang__)+#if defined(__arm__)+/*+ * AArch32 asks for an FPU rather than an architecture extension, and the+ * spelling differs between the compilers in ways that cannot be checked from+ * here. Rather than guess, the cabal file passes -mfpu=crypto-neon-fp-armv8+ * for the whole component on this architecture whatever this flag says, and+ * these expand to nothing. A global target option is safe because a+ * compiler emits these instructions where an intrinsic asks for them and+ * nowhere else.+ */+#define CRYPTON_TARGET_ARMV8_CRYPTO+#define CRYPTON_TARGET_ARMV8_SHA3+#elif defined(__clang__) #define CRYPTON_TARGET_ARMV8_CRYPTO __attribute__((target("+crypto"))) #define CRYPTON_TARGET_ARMV8_SHA3 __attribute__((target("+sha3"))) #else
cbits/crypton_cpu.c view
@@ -60,7 +60,7 @@ CRYPTON_ARMCAP_NEON; #endif -#if defined(__aarch64__)+#if defined(__aarch64__) || defined(__arm__) #if defined(__APPLE__) #include <sys/sysctl.h> #elif defined(__linux__)@@ -71,14 +71,19 @@ #endif /*- * The HWCAP bit positions are the ARM ELF ABI's, so every system that- * reports through the auxiliary vector agrees on them. What differs is- * AT_HWCAP itself -- 16 on Linux, 25 on FreeBSD -- and that comes from each- * system's own header, which is why the tag is never written out here.- * Defined only where the system's headers did not define them, the way- * compiler-rt does it, so that a system which reports through the auxiliary- * vector without shipping the ARM names still compiles.+ * The bit positions are the ARM ELF ABI's, so every system that reports+ * through the auxiliary vector agrees on them. What differs is the tag --+ * AT_HWCAP is 16 on Linux and 25 on FreeBSD -- and that comes from each+ * system's own header, which is why no tag is written out here. Defined+ * only where the system's headers did not define them, the way compiler-rt+ * does it, so that a system which reports through the auxiliary vector+ * without shipping the ARM names still compiles.+ *+ * The two execution states do not share a word or an order. AArch64 puts+ * these in AT_HWCAP; AArch32 has filled that word with older features and+ * puts the cryptographic ones in AT_HWCAP2, starting again from bit zero. */+#if defined(__aarch64__) #ifndef HWCAP_AES #define HWCAP_AES (1 << 3) #endif@@ -97,6 +102,20 @@ #ifndef HWCAP_SHA512 #define HWCAP_SHA512 (1 << 21) #endif+#else+#ifndef HWCAP2_AES+#define HWCAP2_AES (1 << 0)+#endif+#ifndef HWCAP2_PMULL+#define HWCAP2_PMULL (1 << 1)+#endif+#ifndef HWCAP2_SHA1+#define HWCAP2_SHA1 (1 << 2)+#endif+#ifndef HWCAP2_SHA2+#define HWCAP2_SHA2 (1 << 3)+#endif+#endif #if defined(__APPLE__) static int apple_has(const char *name)@@ -145,7 +164,7 @@ f |= CRYPTON_ARM_SHA512; if (apple_has("hw.optional.arm.FEAT_SHA3")) f |= CRYPTON_ARM_SHA3;-#else+#elif defined(__aarch64__) unsigned long cap = 0; #if defined(__linux__)@@ -160,14 +179,67 @@ if (cap & HWCAP_SHA2) f |= CRYPTON_ARM_SHA2; if (cap & HWCAP_SHA512) f |= CRYPTON_ARM_SHA512; if (cap & HWCAP_SHA3) f |= CRYPTON_ARM_SHA3;+#else+ /* AArch32: the second word, and nothing later than SHA-256 to+ * ask about */+ unsigned long cap = 0;++#if defined(__linux__)+ cap = getauxval(AT_HWCAP2);+#elif defined(__FreeBSD__)+ if (elf_aux_info(AT_HWCAP2, &cap, sizeof(cap)) != 0)+ cap = 0; #endif+ if (cap & HWCAP2_AES) f |= CRYPTON_ARM_AES;+ if (cap & HWCAP2_PMULL) f |= CRYPTON_ARM_PMULL;+ if (cap & HWCAP2_SHA1) f |= CRYPTON_ARM_SHA1;+ if (cap & HWCAP2_SHA2) f |= CRYPTON_ARM_SHA2;+#endif features = f; resolved = 1; } return features; }-#endif /* __aarch64__ */+#endif /* __aarch64__ || __arm__ */ +#if defined(__powerpc64__) || defined(__PPC64__)+#if defined(__linux__)+#include <sys/auxv.h>+#elif defined(__FreeBSD__)+#include <sys/auxv.h>+#endif++/*+ * PPC_FEATURE2_VEC_CRYPTO, from the Power ELF ABI. Defined here only where+ * the system's headers did not, so that a system reporting through the+ * auxiliary vector without shipping the name still compiles.+ */+#ifndef PPC_FEATURE2_VEC_CRYPTO+#define PPC_FEATURE2_VEC_CRYPTO 0x02000000+#endif++unsigned int crypton_ppc_features(void)+{+ static unsigned int features;+ static int resolved;++ if (!resolved) {+ unsigned long cap = 0;++#if defined(__linux__)+ cap = getauxval(AT_HWCAP2);+#elif defined(__FreeBSD__)+ if (elf_aux_info(AT_HWCAP2, &cap, sizeof(cap)) != 0)+ cap = 0;+#endif+ features = (cap & PPC_FEATURE2_VEC_CRYPTO)+ ? CRYPTON_PPC_VCRYPTO : 0;+ resolved = 1;+ }+ return features;+}+#endif /* __powerpc64__ */+ #ifdef ARCH_X86 static void cpuid(uint32_t info, uint32_t *eax, uint32_t *ebx, uint32_t *ecx, uint32_t *edx) {@@ -421,3 +493,108 @@ #endif #endif++/*+ * What Crypto.System.CPU reports.+ *+ * The numbers are that module's, and this answers for them. The old shape+ * was the other way round -- Haskell read crypton_aes.c's two-entry array by+ * the Enum index of its own constructors -- which tied the list of names to+ * the indices of an array that had no reason to grow, and so it never did.+ *+ * Nothing here is cached: every answer is either a static array filled by a+ * constructor, a value resolved once and remembered by the function that+ * owns it, or a runtime check that does its own remembering. Haskell calls+ * this once, for a CAF.+ */+uint8_t *crypton_aes_cpu_init(void);++#ifdef WITH_ARMV8_CRYPTO+int crypton_aes_armv8_available(void);+int crypton_aes_armv8_pmull_available(void);+#endif+#ifdef WITH_ARMV8_SHA1+extern int crypton_sha1_armv8_available(void);+#endif+#ifdef WITH_ARMV8_SHA2+extern int crypton_sha256_armv8_available(void);+#endif+#ifdef WITH_ARMV8_SHA512+extern int crypton_sha512_armv8_available(void);+#endif++/* Keep in step with Crypto/System/CPU.hs, which is where these are named.+ * 2 is missing: it was RDRAND, which crypton no longer dispatches on. The+ * number is left out rather than reused, so that the rest keep the values+ * they had. */+#define OPT_AESNI 0+#define OPT_PCLMUL 1+#define OPT_SSSE3 3+#define OPT_AVX 4+#define OPT_AVX2 5+#define OPT_SHANI 6+#define OPT_MOVBE 7+#define OPT_ADX 8+#define OPT_VAES 9+#define OPT_VAES512 10+#define OPT_NEON 11+#define OPT_ARMAES 12+#define OPT_ARMPMULL 13+#define OPT_ARMSHA1 14+#define OPT_ARMSHA2 15+#define OPT_ARMSHA512 16+#define OPT_PPCAES 17+#define OPT_PPCVPMSUM 18++int crypton_cpu_option(unsigned int option)+{+#ifdef ARCH_X86+ uint32_t f = crypton_x86_simd_features();++ switch (option) {+ case OPT_AESNI: return crypton_aes_cpu_init()[CPU_AESNI] != 0;+ case OPT_PCLMUL: return crypton_aes_cpu_init()[CPU_PCLMUL] != 0;+ case OPT_SSSE3: return (f & CRYPTON_X86_SSSE3) != 0;+ case OPT_AVX: return (f & CRYPTON_X86_AVX) != 0;+ case OPT_AVX2: return (f & CRYPTON_X86_AVX2) != 0;+ case OPT_SHANI: return (f & CRYPTON_X86_SHA_NI) != 0;+ case OPT_MOVBE: return (f & CRYPTON_X86_MOVBE) != 0;+ case OPT_ADX: return (f & CRYPTON_X86_ADX) != 0;+ case OPT_VAES: return (f & CRYPTON_X86_VAES) != 0;+ case OPT_VAES512: return (f & CRYPTON_X86_VAES512) != 0;+ default: return 0;+ }+#else+ switch (option) {+ /* NEON is not optional on AArch64, and crypton_armcap_P is set from+ * the start for it. Where there is no ARM assembly at all there is+ * nothing to say. */+#ifdef CRYPTON_ARM_ASM+ case OPT_NEON: return (crypton_armcap_P & CRYPTON_ARMCAP_NEON) != 0;+#endif+#ifdef WITH_ARMV8_CRYPTO+ /* The array rather than the check: it is what the branch table was+ * actually built from, so it says what will run, not merely what the+ * processor has. */+ case OPT_ARMAES: return crypton_aes_cpu_init()[CPU_AESNI] != 0;+ case OPT_ARMPMULL: return crypton_aes_cpu_init()[CPU_PCLMUL] != 0;+#endif+#ifdef WITH_ARMV8_SHA1+ case OPT_ARMSHA1: return crypton_sha1_armv8_available() != 0;+#endif+#ifdef WITH_ARMV8_SHA2+ case OPT_ARMSHA2: return crypton_sha256_armv8_available() != 0;+#endif+#ifdef WITH_ARMV8_SHA512+ case OPT_ARMSHA512: return crypton_sha512_armv8_available() != 0;+#endif+#ifdef WITH_PPC8_CRYPTO+ /* The array rather than the check, for the same reason as the ARM pair+ * above: it says what the branch table was built from. */+ case OPT_PPCAES: return crypton_aes_cpu_init()[CPU_AESNI] != 0;+ case OPT_PPCVPMSUM: return crypton_aes_cpu_init()[CPU_PCLMUL] != 0;+#endif+ default: return 0;+ }+#endif+}
cbits/crypton_cpu.h view
@@ -107,8 +107,13 @@ * which systems someone had thought of. It is what left FreeBSD running * the table-driven AES and the C SHA on hardware that has the * instructions. In one place the next system is added once.+ *+ * AArch32 answers here too. It has the same AES, PMULL, SHA-1 and SHA-256+ * instructions, reports them in a different word of the auxiliary vector,+ * and has no SHA-512 or SHA-3 to report at all -- those two are AArch64's+ * alone, so the bits exist and are never set. */-#if defined(__aarch64__)+#if defined(__aarch64__) || defined(__arm__) #define CRYPTON_ARM_AES (1u << 0) #define CRYPTON_ARM_PMULL (1u << 1) #define CRYPTON_ARM_SHA1 (1u << 2)@@ -117,6 +122,40 @@ #define CRYPTON_ARM_SHA3 (1u << 5) unsigned int crypton_arm_features(void); #endif++/*+ * Which of the optional PowerISA instruction sets this processor has.+ *+ * One bit so far: the vector AES and the vector carry-less multiply that+ * PowerISA 2.07 brought and POWER8 was the first to implement. The system+ * reports it in the second capability word, as AArch32 does, and with its+ * own numbering again.+ */+#if defined(__powerpc64__) || defined(__PPC64__)+#define CRYPTON_PPC_VCRYPTO (1u << 0)+unsigned int crypton_ppc_features(void);+#endif++/*+ * The two slots of the array crypton_aes.c fills and crypton_aes_cpu_init+ * hands out. Here rather than in that file because crypton_cpu.c reads+ * them too, to answer for Crypto.System.CPU.+ */+#define CPU_AESNI 0+#define CPU_PCLMUL 1+#define CPU_OPTION_COUNT 2++/*+ * One question at a time, numbered the way Crypto.System.CPU numbers its+ * ProcessorOption rather than the way any array here is indexed. The+ * numbering is the module's to choose and this answers for it; reading an+ * array by the Haskell constructor's Enum index is what kept that type at+ * three names. Returns 1 if the processor has it and crypton was built to+ * use it, 0 otherwise -- including for anything this does not know, so an+ * older library answers a newer caller rather than failing to link.+ */+int crypton_cpu_option(unsigned int option);+ #ifdef USE_AESNI void crypton_aesni_initialize_hw(void (*init_table)(int, int)); #else
− cbits/crypton_rdrand.c
@@ -1,126 +0,0 @@-/*- * Copyright (C) Thomas DuBuisson- * Copyright (C) 2013 Vincent Hanquez <tab@snarc.org>- *- * All rights reserved.- * - * Redistribution and use in source and binary forms, with or without- * modification, are permitted provided that the following conditions- * are met:- * 1. Redistributions of source code must retain the above copyright- * notice, this list of conditions and the following disclaimer.- * 2. Redistributions in binary form must reproduce the above copyright- * notice, this list of conditions and the following disclaimer in the- * documentation and/or other materials provided with the distribution.- * 3. Neither the name of the author nor the names of his contributors- * may be used to endorse or promote products derived from this software- * without specific prior written permission.- * - * THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND- * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE- * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE- * ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHORS OR CONTRIBUTORS BE LIABLE- * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL- * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS- * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)- * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT- * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY- * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF- * SUCH DAMAGE.- */--#include <stdint.h>-#include <stdlib.h>-#include <stdio.h>-#include <string.h>--int crypton_cpu_has_rdrand()-{- uint32_t ax,bx,cx,dx,func=1;-#if defined(__PIC__) && defined(__i386__)- __asm__ volatile ("mov %%ebx, %%edi;" "cpuid;" "xchgl %%ebx, %%edi;"- : "=a" (ax), "=D" (bx), "=c" (cx), "=d" (dx) : "a" (func));-#else- __asm__ volatile ("cpuid": "=a" (ax), "=b" (bx), "=c" (cx), "=d" (dx) : "a" (func));-#endif- return (cx & 0x40000000);-}--/* inline encoding of 'rdrand %rax' to cover old binutils- * - no inputs- * - 'cc' to the clobber list as we modify condition code.- * - output of rdrand in rax and have a 8 bit error condition- */-#define inline_rdrand_rax(val, err) \- asm(".byte 0x48,0x0f,0xc7,0xf0; setc %1" \- : "=a" (val), "=q" (err) \- : \- : "cc")--/* inline encoding of 'rdrand %eax' to cover old binutils- * - no inputs- * - 'cc' to the clobber list as we modify condition code.- * - output of rdrand in eax and have a 8 bit error condition- */-#define inline_rdrand_eax(val, err) \- asm(".byte 0x0f,0xc7,0xf0; setc %1" \- : "=a" (val), "=q" (err) \- : \- : "cc")--#ifdef __x86_64__-# define RDRAND_SZ 8-# define RDRAND_T uint64_t-#define inline_rdrand(val, err) err = crypton_rdrand_step(&val)-#else-# define RDRAND_SZ 4-# define RDRAND_T uint32_t-#define inline_rdrand(val, err) err = crypton_rdrand_step(&val)-#endif--/* sadly many people are still using an old binutils,- * leading to report that instruction is not recognized.- */-#if 1-/* Returns 1 on success */-static inline int crypton_rdrand_step(RDRAND_T *buffer)-{- unsigned char err;- asm volatile ("rdrand %0; setc %1" : "=r" (*buffer), "=qm" (err));- return (int) err;-}-#endif--/* Returns the number of bytes successfully generated */-int crypton_get_rand_bytes(uint8_t *buffer, size_t len)-{- RDRAND_T tmp;- int aligned = (intptr_t) buffer % RDRAND_SZ;- int orig_len = len;- int to_alignment = RDRAND_SZ - aligned;- uint8_t ok;-- if (aligned != 0) {- inline_rdrand(tmp, ok);- if (!ok)- return 0;- memcpy(buffer, (uint8_t *) &tmp, to_alignment);- buffer += to_alignment;- len -= to_alignment;- }-- for (; len >= RDRAND_SZ; buffer += RDRAND_SZ, len -= RDRAND_SZ) {- inline_rdrand(tmp, ok);- if (!ok)- return (orig_len - len);- *((RDRAND_T *) buffer) = tmp;- }-- if (len > 0) {- inline_rdrand(tmp, ok);- if (!ok)- return (orig_len - len);- memcpy(buffer, (uint8_t *) &tmp, len);- }- return orig_len;-}
+ cbits/crypton_sysdrg.c view
@@ -0,0 +1,388 @@+/*+ * The generator behind MonadRandom IO.+ *+ * A ChaCha20 DRBG per operating system thread, seeded from a process-wide+ * DRBG, which is itself seeded from the system entropy pool. This is the+ * shape RFC 9180's neighbours and the other libraries have settled on, and+ * it was asked for in #298.+ *+ * Per operating system thread and not per Haskell thread: a forkIO thread+ * moves between capabilities, so state kept against it would be shared by+ * threads running at the same time. That is why the state is here and+ * reached through pthread_getspecific rather than held in Haskell.+ *+ * Three things force a reseed:+ *+ * - a thread has produced CRYPTON_THREAD_RESEED bytes,+ * - the process DRBG has issued CRYPTON_GLOBAL_RESEED bytes of seed,+ * - the process has forked.+ *+ * The last one is the one that bites. A child inherits its parent's state+ * and would otherwise produce the same stream; the generation counter below+ * is bumped in the child by a pthread_atfork handler, and every generator+ * compares against it before it answers.+ */++#include <stdint.h>+#include <string.h>+#include <stdlib.h>++#include "crypton_chacha.h"+#include "crypton_sha512.h"++#ifdef _WIN32+#include <windows.h>+#else+#include <pthread.h>+#include <unistd.h>+#endif++/* from crypton_sysrandom.c */+int crypton_sysrandom_available(void);+int crypton_sysrandom_bytes(uint8_t *buf, int len);++#define CHACHA_ROUNDS 20+#define SEED_KEY 32+#define SEED_IV 8+#define SEED_LEN (SEED_KEY + SEED_IV)++#define CRYPTON_THREAD_RESEED (1u << 20)+#define CRYPTON_GLOBAL_RESEED (1u << 20)++typedef struct {+ crypton_chacha_context ctx;+ uint64_t used;+ uint32_t generation;+ int seeded;+} drg_t;++static drg_t global_drg;+static volatile uint32_t fork_generation = 0;++#ifdef _WIN32+static CRITICAL_SECTION global_lock;+/* Fls and not Tls: TlsAlloc has no destructor, so the state of every thread+ * that ever drew a byte would be left allocated and unscrubbed when the+ * thread ended. FlsAlloc takes the callback that pthread_key_create does. */+static DWORD thread_slot;+static INIT_ONCE init_once = INIT_ONCE_STATIC_INIT;+#else+static pthread_mutex_t global_lock = PTHREAD_MUTEX_INITIALIZER;+static pthread_key_t thread_slot;+static pthread_once_t init_once = PTHREAD_ONCE_INIT;+#endif++static void scrub(void *p, size_t n)+{+ volatile uint8_t *q = (volatile uint8_t *) p;+ while (n--) *q++ = 0;+}++/*+ * Seed material for the process DRBG.+ *+ * What the system gives goes through SHA-512 rather than into the key+ * directly, so that the key is a fixed size whatever the call returns, and+ * so that a second source could be added without any of this changing shape.+ * There is one source today and the note below says why.+ */+static int seed_from_system(uint8_t out[SEED_LEN])+{+ struct sha512_ctx h;+ uint8_t buf[64];+ uint8_t digest[64];+ int got;++ crypton_sha512_init(&h);++ /*+ * The system call only. Where there is none -- an old kernel, a BSD+ * this does not know -- seeding fails and the caller keeps the path+ * it has, rather than this file growing a second copy of the device+ * reading that Crypto.Random.Entropy.Unix already does.+ *+ * And RDRAND is not mixed in beside it, which it was until+ * @vdukhovni pointed out that it cannot add anything: every system+ * this code runs on already feeds RDRAND into the pool the call below+ * draws from. Linux does that whatever random.trust_cpu says -- that+ * setting decides whether the contribution is *credited*, not whether+ * it is mixed -- so the instruction's output is in this buffer+ * already.+ *+ * The argument for keeping it was defence in depth against that pool+ * having gone wrong. It does not survive the above: a pool that has+ * gone wrong has gone wrong with RDRAND already in it, so asking the+ * instruction a second time covers only the case where the kernel's+ * mixing is broken and the instruction is not. That is not worth a+ * second code path, a flag, and the standing invitation to read this+ * as a second source when it is the same one twice.+ */+ got = crypton_sysrandom_available() ? crypton_sysrandom_bytes(buf, 64) : 0;+ if (got <= 0) {+ /* Nothing else here is a seed on its own, so there is nothing to+ * go on with: return before drawing anything that would only be+ * thrown away. */+ scrub(&h, sizeof h);+ scrub(buf, sizeof buf);+ return 0;+ }+ crypton_sha512_update(&h, buf, (uint32_t) got);+++ crypton_sha512_finalize(&h, digest);+ memcpy(out, digest, SEED_LEN);++ scrub(&h, sizeof h);+ scrub(buf, sizeof buf);+ scrub(digest, sizeof digest);+ return 1;+}++static void drg_seed(drg_t *d, const uint8_t seed[SEED_LEN])+{+ crypton_chacha_init(&d->ctx, CHACHA_ROUNDS, SEED_KEY, seed,+ SEED_IV, seed + SEED_KEY);+ d->used = 0;+ d->generation = fork_generation;+ d->seeded = 1;+}++/*+ * Forget the key that produced the bytes just handed out.+ *+ * Without this the key stands until the next reseed, and anyone who reads a+ * generator's state can wind the counter back and reproduce everything it+ * has issued since -- up to CRYPTON_THREAD_RESEED bytes that were meant to+ * be secret. Taking the next forty bytes of keystream as the new key and+ * nonce, and dropping the old ones, puts that out of reach one step after+ * it is issued: ChaCha20 does not run backwards, and the key that would+ * have been needed is gone. It is what arc4random does.+ *+ * crypton_chacha_init memsets the whole context, so the counter returns to+ * zero and the tail of a part-used block goes with the old key rather than+ * being handed out under the new one.+ *+ * d->used is not touched. It counts what callers were given, which is what+ * the reseed interval is written in terms of; these forty bytes are the+ * cost of the rekey and not an answer to anybody.+ */+static void drg_rekey(drg_t *d)+{+ uint8_t next[SEED_LEN];++ crypton_chacha_generate(next, &d->ctx, SEED_LEN);+ crypton_chacha_init(&d->ctx, CHACHA_ROUNDS, SEED_KEY, next,+ SEED_IV, next + SEED_KEY);+ scrub(next, sizeof next);+}++static int drg_stale(const drg_t *d, uint64_t limit)+{+ return !d->seeded || d->used >= limit || d->generation != fork_generation;+}++/* Bytes from the process DRBG, which is only ever asked for seed material. */+static int global_bytes(uint8_t *out, uint32_t len)+{+ int ok = 1;++#ifdef _WIN32+ EnterCriticalSection(&global_lock);+#else+ pthread_mutex_lock(&global_lock);+#endif+ if (drg_stale(&global_drg, CRYPTON_GLOBAL_RESEED)) {+ uint8_t seed[SEED_LEN];+ ok = seed_from_system(seed);+ if (ok)+ drg_seed(&global_drg, seed);+ scrub(seed, sizeof seed);+ }+ if (ok) {+ crypton_chacha_generate(out, &global_drg.ctx, len);+ global_drg.used += len;+ drg_rekey(&global_drg);+ }+#ifdef _WIN32+ LeaveCriticalSection(&global_lock);+#else+ pthread_mutex_unlock(&global_lock);+#endif+ return ok;+}++#ifdef _WIN32+static void WINAPI thread_free(void *p)+#else+static void thread_free(void *p)+#endif+{+ if (p) {+ scrub(p, sizeof(drg_t));+ free(p);+ }+}++#ifndef _WIN32+/* All three handlers, not just the child's. The child's first draw has to+ * reseed -- the generation has changed -- and reseeding takes global_lock.+ * A fork made while another thread held it would hand the child a mutex+ * locked by a thread that did not come across, and the child would wait on+ * it for ever. So the lock is taken before the fork and released on both+ * sides of it, which is the state the child needs it in. */+static void before_fork(void)+{+ pthread_mutex_lock(&global_lock);+}++static void after_fork_in_parent(void)+{+ pthread_mutex_unlock(&global_lock);+}++static void after_fork_in_child(void)+{+ pthread_mutex_unlock(&global_lock);+ fork_generation++;+}+#endif++#ifdef CRYPTON_SYSDRG_TESTING+/* For cbits/tests/sysdrg and nothing else: hold and release the process+ * generator's lock, so that a fork can be made to happen while another+ * thread holds it. Nothing outside that test declares these, and the+ * library is never built with this defined. */+void crypton_sysdrg_test_lock(void);+void crypton_sysdrg_test_unlock(void);++static drg_t *this_thread(void);++/* The calling thread's ChaCha key, which is d[4..11] of the state. A test+ * uses it to ask whether the key that produced a draw is still there+ * afterwards; see "a draw replaces the key that made it" in+ * cbits/tests/sysdrg. */+void crypton_sysdrg_test_key(uint8_t out[32]);++void crypton_sysdrg_test_key(uint8_t out[32])+{+ drg_t *d = this_thread();+ int i;++ if (!d)+ return;+ for (i = 0; i < 8; i++) {+ uint32_t w = d->ctx.st.d[4 + i];+ out[i * 4 + 0] = (uint8_t) (w);+ out[i * 4 + 1] = (uint8_t) (w >> 8);+ out[i * 4 + 2] = (uint8_t) (w >> 16);+ out[i * 4 + 3] = (uint8_t) (w >> 24);+ }+}++void crypton_sysdrg_test_lock(void)+{+#ifndef _WIN32+ pthread_mutex_lock(&global_lock);+#endif+}++void crypton_sysdrg_test_unlock(void)+{+#ifndef _WIN32+ pthread_mutex_unlock(&global_lock);+#endif+}+#endif++#ifdef _WIN32+static BOOL CALLBACK init_slot(PINIT_ONCE o, PVOID p, PVOID *c)+{+ (void) o; (void) p; (void) c;+ InitializeCriticalSection(&global_lock);+ thread_slot = FlsAlloc(thread_free);+ return TRUE;+}+#else+static void init_slot(void)+{+ pthread_key_create(&thread_slot, thread_free);+ pthread_atfork(before_fork, after_fork_in_parent, after_fork_in_child);+}+#endif++static drg_t *this_thread(void)+{+ drg_t *d;++#ifdef _WIN32+ InitOnceExecuteOnce(&init_once, init_slot, NULL, NULL);+ if (thread_slot == FLS_OUT_OF_INDEXES)+ return NULL;+ d = (drg_t *) FlsGetValue(thread_slot);+#else+ pthread_once(&init_once, init_slot);+ d = (drg_t *) pthread_getspecific(thread_slot);+#endif+ if (!d) {+ d = (drg_t *) calloc(1, sizeof(drg_t));+ if (!d)+ return NULL;+#ifdef _WIN32+ if (!FlsSetValue(thread_slot, d)) {+ free(d);+ return NULL;+ }+#else+ if (pthread_setspecific(thread_slot, d) != 0) {+ free(d);+ return NULL;+ }+#endif+ }+ return d;+}++/* Returns the number of bytes written, which is len unless there was no+ * seed to be had -- and then it is 0, so that the caller can say so rather+ * than hand back a buffer it cannot vouch for. */+int crypton_sysdrg_bytes(uint8_t *out, int len)+{+ drg_t *d;++ if (len < 0)+ return 0;+ d = this_thread();+ if (!d)+ return 0;++ if (drg_stale(d, CRYPTON_THREAD_RESEED)) {+ uint8_t seed[SEED_LEN];+ int ok = global_bytes(seed, SEED_LEN);+ if (ok)+ drg_seed(d, seed);+ scrub(seed, sizeof seed);+ if (!ok)+ return 0;+ }++ crypton_chacha_generate(out, &d->ctx, (uint32_t) len);+ d->used += (uint64_t) len;+ drg_rekey(d);+ return len;+}++/* For the tests: how many times the process has been seen to fork. */+uint32_t crypton_sysdrg_generation(void)+{+ return fork_generation;+}++/* For the tests: bytes this thread's generator has produced since it was+ * last seeded. A generator shared between threads would carry the first+ * thread's count into the second; a per-thread one starts again at zero,+ * and that is the only way from outside to tell the two apart. */+uint64_t crypton_sysdrg_thread_used(void)+{+ drg_t *d = this_thread();+ return d ? d->used : 0;+}
+ cbits/crypton_sysrandom.c view
@@ -0,0 +1,119 @@+/*+ * The kernel's own random number generator, reached without a file+ * descriptor: getrandom(2) on Linux and FreeBSD, getentropy(3) on the+ * systems that have that instead.+ *+ * This is preferred over reading /dev/urandom because it needs no path and+ * no descriptor, so it still works where /dev is not mounted or not+ * populated -- a minimal container, a chroot, a sandbox -- and it cannot be+ * defeated by exhausting the descriptor table.+ *+ * Availability is decided at run time as well as at compile time: a binary+ * built against headers that declare getrandom can still run on a kernel+ * that does not implement it, and says so with ENOSYS.+ */++#include <stddef.h>+#include <stdint.h>+#include <errno.h>++#if defined(__linux__)+#include <unistd.h>+#include <sys/syscall.h>+#ifdef SYS_getrandom+#define CRYPTON_SYSRANDOM_GETRANDOM 1+#endif+#elif defined(__FreeBSD__)+#include <sys/param.h>+#if __FreeBSD_version >= 1200000+#include <sys/random.h>+#define CRYPTON_SYSRANDOM_GETRANDOM 1+#endif+#elif defined(__APPLE__)+/* getentropy(3) is declared in sys/random.h on Darwin, since 10.12 -- but+ * macOS only. Crypto.Random says iOS does not allow it, which is why that+ * platform builds with INSECURE_ENTROPY; until someone can try it there,+ * iOS keeps the path it has rather than gaining an untested one. */+#include <TargetConditionals.h>+#if defined(TARGET_OS_OSX) && TARGET_OS_OSX+#include <sys/random.h>+#define CRYPTON_SYSRANDOM_GETENTROPY 1+#endif+#elif defined(__OpenBSD__)+#include <unistd.h>+#define CRYPTON_SYSRANDOM_GETENTROPY 1+#elif defined(__NetBSD__)+#include <sys/param.h>+#if __NetBSD_Version__ >= 1000000000+#include <sys/random.h>+#define CRYPTON_SYSRANDOM_GETENTROPY 1+#endif+#endif++/* getentropy(3) refuses more than 256 bytes in one call. */+#define CRYPTON_GETENTROPY_MAX 256++#if defined(CRYPTON_SYSRANDOM_GETRANDOM)++static int sysrandom_once(uint8_t *buf, size_t n)+{+#if defined(__linux__)+ return (int) syscall(SYS_getrandom, buf, n, 0);+#else+ return (int) getrandom(buf, n, 0);+#endif+}++#elif defined(CRYPTON_SYSRANDOM_GETENTROPY)++static int sysrandom_once(uint8_t *buf, size_t n)+{+ if (n > CRYPTON_GETENTROPY_MAX)+ n = CRYPTON_GETENTROPY_MAX;+ if (getentropy(buf, n) != 0)+ return -1;+ return (int) n;+}++#else++static int sysrandom_once(uint8_t *buf, size_t n)+{+ (void) buf; (void) n;+ errno = ENOSYS;+ return -1;+}++#endif++/* Is the call there, on this kernel, right now? A zero-length request+ * answers that without consuming anything. */+int crypton_sysrandom_available(void)+{+ uint8_t b;+ int r = sysrandom_once(&b, 0);+ return r < 0 ? 0 : 1;+}++/* Fill the buffer. Returns the number of bytes written, which is n unless+ * the call failed for a reason other than being interrupted. */+int crypton_sysrandom_bytes(uint8_t *buf, int len)+{+ size_t n = (size_t) len;+ size_t done = 0;++ if (len < 0)+ return 0;+ while (done < n) {+ int r = sysrandom_once(buf + done, n - done);+ if (r < 0) {+ if (errno == EINTR)+ continue;+ break;+ }+ if (r == 0)+ break;+ done += (size_t) r;+ }+ return (int) done;+}
+ cbits/tests/bearssl_diff.c view
@@ -0,0 +1,432 @@+/*+ * crypton's portable AES and GHASH against the BearSSL they are written on.+ *+ * This began as the migration check: for one commit, cbits/aes/generic.c and+ * cbits/aes/gf.c were still the table-driven implementations, and this said+ * the vendored code computed what they computed before they were replaced.+ *+ * Since the replacement the two sides are no longer independent, and what is+ * left is still worth checking: everything in generic.c and gf.c is now+ * crypton's own glue -- a schedule kept compressed and carried through+ * memcpy, the interleave-and-ortho idiom around a single block, three of four+ * lanes left idle, and a GHASH entry that reaches the same multiply by handing+ * it a block of zeros. Each of those is somewhere a mistake would live, and+ * each is compared here against calling BearSSL directly.+ *+ * FIPS-197's own vectors come first either way, so that agreement means AES+ * rather than two callers agreeing on something that is not.+ *+ * Run with an argument to corrupt results on purpose and see the comparison+ * notice: a differential test that cannot fail has said nothing.+ */+#include <stdio.h>+#include <stdlib.h>+#include <string.h>+#include <stdint.h>++#include "bearssl/inner.h"+#include "crypton_aes.h"+#include "aes/generic.h"+#include "aes/gf.h"+#include "aes/block128.h"++/*+ * crypton_aes.c's table, which is a global, and the three key sizes of the+ * two CTR entries this looks for in it. Searching rather than indexing+ * because the index is an enum private to that file.+ */+#define BRANCH_TABLE_SEARCH 64+extern void *crypton_aes_branch_table[];++/* the two GCM implementations, which crypton_aes.c declares but no header+ * does: in this build the generic one reaches the portable block function+ * too, so the two have to agree exactly, tag and all */+void crypton_aes_generic_gcm_encrypt(uint8_t *, aes_gcm *, aes_key *, uint8_t *, uint32_t);+void crypton_aes_generic_gcm_decrypt(uint8_t *, aes_gcm *, aes_key *, uint8_t *, uint32_t);+void crypton_aes_bitsliced_gcm_encrypt(uint8_t *, aes_gcm *, aes_key *, uint8_t *, uint32_t);+void crypton_aes_bitsliced_gcm_decrypt(uint8_t *, aes_gcm *, aes_key *, uint8_t *, uint32_t);++/* ---- the vendored code, one block at a time, as aes_ct64_cbcenc.c does ---- */++static void bear_key(uint64_t *comp_skey, unsigned *nr,+ const uint8_t *key, size_t len)+{+ *nr = br_aes_ct64_keysched(comp_skey, key, len);+}++static void bear_block(uint8_t *out, const uint64_t *comp_skey, unsigned nr,+ const uint8_t *in, int decrypt)+{+ uint64_t sk_exp[120];+ uint32_t w[4];+ uint64_t q[8];++ br_aes_ct64_skey_expand(sk_exp, nr, comp_skey);+ w[0] = br_dec32le(in);+ w[1] = br_dec32le(in + 4);+ w[2] = br_dec32le(in + 8);+ w[3] = br_dec32le(in + 12);+ memset(q, 0, sizeof q);+ br_aes_ct64_interleave_in(&q[0], &q[4], w);+ br_aes_ct64_ortho(q);+ if (decrypt)+ br_aes_ct64_bitslice_decrypt(nr, sk_exp, q);+ else+ br_aes_ct64_bitslice_encrypt(nr, sk_exp, q);+ br_aes_ct64_ortho(q);+ br_aes_ct64_interleave_out(w, q[0], q[4]);+ br_enc32le(out, w[0]);+ br_enc32le(out + 4, w[1]);+ br_enc32le(out + 8, w[2]);+ br_enc32le(out + 12, w[3]);+}++/* ---- crypton's GHASH, driven the way crypton_aes.c drives it ---- */++static void crypton_ghash(uint8_t *y, const uint8_t *h,+ const uint8_t *data, size_t len)+{+ table_4bit ht;+ block128 acc;+ size_t i;++ crypton_aes_generic_hinit(ht, (const block128 *) h);+ block128_zero(&acc);+ for (i = 0; i < len; i += 16) {+ block128_xor_bytes(&acc, data + i, 16);+ crypton_aes_generic_gf_mul(&acc, ht);+ }+ memcpy(y, &acc, 16);+}++/* ---- the comparison ---- */++static int failures;+static int sabotage;++static void same(const char *what, const uint8_t *a, const uint8_t *b, size_t n)+{+ if (memcmp(a, b, n) != 0) {+ size_t i;++ printf(" MISMATCH %s\n bearssl ", what);+ for (i = 0; i < n; i++) printf("%02x", a[i]);+ printf("\n crypton ");+ for (i = 0; i < n; i++) printf("%02x", b[i]);+ printf("\n");+ failures++;+ }+}++static uint32_t rnd_state = 1;+static uint8_t rnd(void)+{+ rnd_state = rnd_state * 1103515245u + 12345u;+ return (uint8_t)(rnd_state >> 16);+}+static void rnd_fill(uint8_t *p, size_t n)+{+ while (n--) *p++ = rnd();+}++int main(int argc, char **argv)+{+ /* FIPS-197 C.1, C.2, C.3 */+ static const uint8_t pt[16] = {+ 0x00,0x11,0x22,0x33,0x44,0x55,0x66,0x77,+ 0x88,0x99,0xaa,0xbb,0xcc,0xdd,0xee,0xff };+ static const uint8_t k128[16] = {+ 0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15 };+ static const uint8_t k192[24] = {+ 0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23 };+ static const uint8_t k256[32] = {+ 0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,+ 16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31 };+ static const uint8_t c128[16] = {+ 0x69,0xc4,0xe0,0xd8,0x6a,0x7b,0x04,0x30,+ 0xd8,0xcd,0xb7,0x80,0x70,0xb4,0xc5,0x5a };+ static const uint8_t c192[16] = {+ 0xdd,0xa9,0x7c,0xa4,0x86,0x4c,0xdf,0xe0,+ 0x6e,0xaf,0x70,0xa0,0xec,0x0d,0x71,0x91 };+ static const uint8_t c256[16] = {+ 0x8e,0xa2,0xb7,0xca,0x51,0x67,0x45,0xbf,+ 0xea,0xfc,0x49,0x90,0x4b,0x49,0x60,0x89 };+ const uint8_t *keys[3] = { k128, k192, k256 };+ const uint8_t *cts[3] = { c128, c192, c256 };+ size_t klens[3] = { 16, 24, 32 };+ uint64_t comp[30];+ unsigned nr;+ uint8_t a[16], b[16];+ int i, round;++ sabotage = argc > 1;+ printf("== FIPS-197, so that agreement means AES ==\n");+ for (i = 0; i < 3; i++) {+ bear_key(comp, &nr, keys[i], klens[i]);+ bear_block(a, comp, nr, pt, 0);+ if (sabotage && i == 1) a[0] ^= 1;+ same("bearssl vs FIPS-197", a, cts[i], 16);+ bear_block(b, comp, nr, cts[i], 1);+ same("bearssl decrypt vs plaintext", b, pt, 16);+ }++ printf("== crypton's glue vs BearSSL direct, 2000 random keys and blocks ==\n");+ for (round = 0; round < 2000; round++) {+ uint8_t key[32], in[16], e1[16], e2[16], d1[16], d2[16];+ size_t kl = klens[round % 3];+ aes_key ck;++ rnd_fill(key, kl);+ rnd_fill(in, 16);++ bear_key(comp, &nr, key, kl);+ bear_block(e1, comp, nr, in, 0);+ bear_block(d1, comp, nr, in, 1);++ crypton_aes_generic_init(&ck, key, (uint8_t) kl);+ crypton_aes_generic_encrypt_block((aes_block *) e2, &ck,+ (aes_block *) in);+ crypton_aes_generic_decrypt_block((aes_block *) d2, &ck,+ (aes_block *) in);+ if (sabotage && round == 7) e1[3] ^= 0x10;+ same("encrypt", e1, e2, 16);+ same("decrypt", d1, d2, 16);+ }++ printf("== GHASH: crypton's entries vs br_ghash_ctmul64, 500 messages ==\n");+ for (round = 0; round < 500; round++) {+ uint8_t h[16], data[256], y1[16], y2[16];+ size_t len = 16u * (size_t)(1 + (round % 16));++ rnd_fill(h, 16);+ rnd_fill(data, len);++ memset(y1, 0, 16);+ br_ghash_ctmul64(y1, h, data, len);+ crypton_ghash(y2, h, data, len);+ if (sabotage && round == 3) y1[15] ^= 0x80;+ same("ghash", y1, y2, 16);+ }++ printf("== gf_mul4: the four-block entry against the same four blocks ==\n");+ for (round = 0; round < 500; round++) {+ uint8_t h[16], data[64], y1[16];+ table_4bit ht;+ block128 acc;++ rnd_fill(h, 16);+ rnd_fill(data, sizeof data);++ memset(y1, 0, 16);+ br_ghash_ctmul64(y1, h, data, sizeof data);++ crypton_aes_generic_hinit(ht, (const block128 *) h);+ block128_zero(&acc);+ crypton_aes_generic_gf_mul4(&acc, (const block128 *) data, ht);+ if (sabotage && round == 11) y1[0] ^= 0x40;+ same("gf_mul4", y1, (const uint8_t *) &acc, 16);+ }++ /*+ * And that the wide entries are the ones installed. Nothing else+ * here would notice if they were not: the entries they replace are+ * correct too, just a block at a time, so every answer above would+ * be the same and only the speed would be gone.+ */+ /*+ * The Haskell suite reaches the four-block pass, but thinly: its+ * vectors are mostly a block or three long, and breaking that pass+ * alone fails sixteen of its examples where breaking every pass+ * fails seven hundred. So the lane logic and the counter are+ * covered here instead, where the lengths can be chosen.+ */+ printf("== many blocks at a pass against one at a time ==\n");+ for (round = 0; round < 200; round++) {+ uint8_t key[32], in[16 * 9], wide[16 * 9], single[16 * 9];+ size_t kl = klens[round % 3];+ uint32_t nb = 1 + (round % 9);+ aes_sched sched;+ aes_key ck;+ uint32_t i;+ int dec = round & 1;++ rnd_fill(key, kl);+ rnd_fill(in, nb * 16);+ crypton_aes_generic_init(&ck, key, (uint8_t) kl);++ crypton_aes_generic_schedule(&sched, &ck);+ crypton_aes_generic_blocks(wide, in, nb, &sched, dec);++ for (i = 0; i < nb; i++) {+ if (dec)+ crypton_aes_generic_decrypt_block(+ (aes_block *) (single + 16 * i), &ck,+ (aes_block *) (in + 16 * i));+ else+ crypton_aes_generic_encrypt_block(+ (aes_block *) (single + 16 * i), &ck,+ (aes_block *) (in + 16 * i));+ }+ if (sabotage && round == 5) wide[16] ^= 2;+ same("wide pass", wide, single, nb * 16);+ }++ printf("== the four-block CTR against one block at a time ==\n");+ for (round = 0; round < 200; round++) {+ uint8_t key[32], iv[16], in[200], got[200], want[200];+ size_t kl = klens[round % 3];+ /* lengths that land on, before and after a group of four */+ uint32_t len = 1 + (round % 200);+ aes_key ck;+ aes_block counter, ks;+ uint32_t done, n, i;+ int c32 = round & 1;++ rnd_fill(key, kl);+ rnd_fill(iv, 16);+ rnd_fill(in, len);+ crypton_aes_generic_init(&ck, key, (uint8_t) kl);++ if (c32)+ crypton_aes_bitsliced_encrypt_c32(got, &ck,+ (aes_block *) iv, in, len);+ else+ crypton_aes_bitsliced_encrypt_ctr(got, &ck,+ (aes_block *) iv, in, len);++ block128_copy(&counter, (block128 *) iv);+ for (done = 0; done < len; done += 16) {+ crypton_aes_generic_encrypt_block(&ks, &ck, &counter);+ n = len - done < 16 ? len - done : 16;+ for (i = 0; i < n; i++)+ want[done + i] = ((uint8_t *) &ks)[i]+ ^ in[done + i];+ if (c32)+ block128_inc32_le(&counter);+ else+ block128_inc_be(&counter);+ }+ if (sabotage && round == 9) got[len - 1] ^= 4;+ same("ctr", got, want, len);+ }++ printf("== XTS four at a pass against one at a time ==\n");+ for (round = 0; round < 200; round++) {+ uint8_t key[32], key2[32], du[16], in[16 * 11];+ uint8_t got[16 * 11], want[16 * 11];+ size_t kl = klens[round % 3];+ uint32_t nb = 1 + (round % 11);+ uint32_t spoint = round % 3;+ aes_key k1, k2;+ aes_block tweak;+ uint32_t i, j;+ int dec = round & 1;++ rnd_fill(key, kl);+ rnd_fill(key2, kl);+ rnd_fill(du, 16);+ rnd_fill(in, nb * 16);+ crypton_aes_initkey(&k1, key, (uint8_t) kl);+ crypton_aes_initkey(&k2, key2, (uint8_t) kl);++ {+ aes_block d;+ memcpy(&d, du, 16);+ if (dec)+ crypton_aes_decrypt_xts((aes_block *) got, &k1, &k2,+ &d, spoint, (aes_block *) in, nb);+ else+ crypton_aes_encrypt_xts((aes_block *) got, &k1, &k2,+ &d, spoint, (aes_block *) in, nb);+ }++ /* the same thing a block at a time */+ memcpy(&tweak, du, 16);+ crypton_aes_generic_encrypt_block(&tweak, &k2, &tweak);+ for (j = 0; j < spoint; j++)+ crypton_aes_generic_gf_mulx((block128 *) &tweak);+ for (i = 0; i < nb; i++) {+ aes_block t;++ block128_vxor(&t, (block128 *) (in + 16 * i), &tweak);+ if (dec)+ crypton_aes_generic_decrypt_block(&t, &k1, &t);+ else+ crypton_aes_generic_encrypt_block(&t, &k1, &t);+ block128_vxor((block128 *) (want + 16 * i), &t, &tweak);+ crypton_aes_generic_gf_mulx((block128 *) &tweak);+ }+ if (sabotage && round == 17) got[16] ^= 8;+ same("xts", got, want, nb * 16);+ }++ printf("== the four-block GCM against the one-block GCM ==\n");+ for (round = 0; round < 200; round++) {+ uint8_t key[32], iv[12], in[300], ga[300], gb[300];+ size_t kl = klens[round % 3];+ uint32_t len = 1 + (round % 300);+ aes_key ck;+ aes_gcm g1, g2;+ int dec = round & 1;++ rnd_fill(key, kl);+ rnd_fill(iv, sizeof iv);+ rnd_fill(in, len);+ crypton_aes_initkey(&ck, key, (uint8_t) kl);+ crypton_aes_gcm_init(&g1, &ck, iv, sizeof iv);+ memcpy(&g2, &g1, sizeof g1);++ if (dec) {+ crypton_aes_generic_gcm_decrypt(ga, &g1, &ck, in, len);+ crypton_aes_bitsliced_gcm_decrypt(gb, &g2, &ck, in, len);+ } else {+ crypton_aes_generic_gcm_encrypt(ga, &g1, &ck, in, len);+ crypton_aes_bitsliced_gcm_encrypt(gb, &g2, &ck, in, len);+ }+ if (sabotage && round == 13) gb[0] ^= 1;+ same("gcm text", ga, gb, len);+ /* the running GHASH and counter, which the tag is made from */+ same("gcm state", (const uint8_t *) &g1, (const uint8_t *) &g2,+ sizeof g1);+ }++ printf("== which implementation the build chose ==\n");+#if !defined(WITH_AESNI) && !defined(WITH_ARMV8_CRYPTO)+ /*+ * With no accelerator compiled in, crypton_aes.c does not read the+ * branch table at all -- its GET_ macros name the portable entries+ * directly -- so there is nothing here to look at, and which+ * implementation runs is settled by the preprocessor. Scanning the+ * table in this build was a check that passed while telling nothing,+ * which is how the four-block CTR came to be written, installed, and+ * never called.+ */+ printf(" named at compile time; the table is not read in this build\n");+ (void) crypton_aes_branch_table;+ if (sabotage) { /* nothing to corrupt here */ }+#else+ {+ int i, ctr = 0, c32 = 0;++ for (i = 0; i < BRANCH_TABLE_SEARCH; i++) {+ if (crypton_aes_branch_table[i] ==+ (void *) crypton_aes_bitsliced_encrypt_ctr)+ ctr++;+ if (crypton_aes_branch_table[i] ==+ (void *) crypton_aes_bitsliced_encrypt_c32)+ c32++;+ }+ printf(" CTR entries %d, C32 entries %d\n", ctr, c32);+ if (sabotage) { ctr = 0; }+ if (ctr != 3 || c32 != 3) {+ printf(" MISMATCH the portable CTR entries were not installed\n");+ failures++;+ }+ }+#endif++ printf("%s: %d mismatch(es)%s\n",+ failures ? "FAIL" : "ok", failures,+ sabotage ? " (sabotage was asked for)" : "");+ return failures != 0;+}
cbits/tests/ct/README view
@@ -26,29 +26,33 @@ chapoly the ChaCha20 key, the plaintext, and the Poly1305 key aes the AES key and the plaintext, through ECB and GCM aes_armv8 the same driver again, on AArch64, against the instructions+ aes_x86ni the same driver again, on x86-64, against AES-NI -The aes driver is built against cbits/aes/generic.c and cbits/aes/gf.c on-purpose, rather than whatever the machine offers. AES-NI and the ARMv8-instructions do not look anything up and would report nothing, which would say-nothing about the table-driven code every other machine runs. That code is-variable-time by construction -- a table index is a byte of the state -- and-so is the table-driven GHASH beside it. A report from the aes driver is-therefore expected and is a property of those implementations, not a defect-found in them; it is here so that the size of it is written down rather than-assumed. Everything else is expected to be silent.+One driver, built three times, against the portable C and against each of+the two instruction sets. All three must report nothing at all -- not+"nothing outside known.txt", nothing. The portable AES and GHASH are+bitsliced and table-free; AESE, AESMC, PMULL, AESENC and PCLMULQDQ look+nothing up. None of the three has anywhere for a secret to decide a branch+or an address. -On AArch64 the same driver is then built a second time, as aes_armv8, against-cbits/aes/armv8.c with the crypto extension turned on. AESE, AESMC and PMULL-look nothing up and branch on nothing, so that run must report nothing at all--- not "nothing outside known.txt", nothing; a table site appearing there-would mean the dispatch had not picked the instructions. The two runs keep-each other honest: the table-driven one has to report and the instruction one-has to be silent, and either going the wrong way says the run is not-measuring what it claims to.+This entry used to read the other way round. The portable implementation+was table-driven -- an S-box indexed by a byte of the state, and Shoup's+4-bit table for the GHASH -- so the aes driver was *required* to report, and+known.txt carried four entries saying so. That requirement was also what+kept the pair honest: if the portable build had quietly taken an accelerated+path it would have fallen silent, and the silence would have failed it. -Until that was added the harness ran only on x86-64, so crypton's AArch64 AES-and GHASH had never been put to it -- which was noticed when the GHASH was-rewritten.+With the tables gone that check goes too, so the question is now asked+outright rather than read off a leak. Each build of the driver prints which+implementation its dispatch took, and run.sh requires the answer it expects+before it looks at anything else. A build that takes the wrong path fails+on that line rather than passing by saying nothing.++The x86-64 build is new with the rewrite. Before it, AES-NI and PCLMULQDQ+had never been put to this harness at all: their being constant time was an+inference from the instruction specifications rather than a measurement of+crypton's code. The AArch64 build came earlier, when the GHASH was+rewritten and it was noticed that the harness had only ever run on x86-64. What round eight found ----------------------
cbits/tests/ct/ct_aes.c view
@@ -5,13 +5,24 @@ * is table-driven and is variable-time by construction, which is a property * of that code rather than a defect in it. See cbits/tests/ct/README. */ #include "tests/ct/ct.h"+#include <stdio.h> #include <string.h> #include "crypton_aes.h"+#include "crypton_cpu.h" +/* crypton_aes.c fills this; run.sh reads the line below to tell the three+ * builds of this driver apart. They are all silent now, so which one took+ * which implementation cannot be read off a leak any more. */+uint8_t *crypton_aes_cpu_init(void);+ int main(void) { aes_key k; aes_gcm_key gk; uint8_t key[32], pt[256], ct[256 + 16], iv[12];++ printf("dispatch aes=%d pclmul=%d\n",+ crypton_aes_cpu_init()[CPU_AESNI] != 0,+ crypton_aes_cpu_init()[CPU_PCLMUL] != 0); ct_fill(key, sizeof key); ct_fill(pt, sizeof pt);
cbits/tests/ct/known.txt view
@@ -22,15 +22,6 @@ # arithmetic, and it too goes the same way on every valid input. decaf.c:136 assert(ret) in crypton_gf_invert -# The table-driven AES and the table-driven GHASH index with a byte of the-# state, which is what makes them fast and what makes them variable-time.-# That is a property of those implementations rather than a defect in them;-# a machine with AES-NI or the ARMv8 instructions runs neither.-generic.c the AES tables, in key expansion and in the rounds-gf.c the GHASH table-crypton_aes.c the same tables, attributed to the code that inlines them-block128.h likewise- # AArch64 only. gcc keeps a carry in the flags and takes it out with `cset` # or `cinc`, where on x86-64 it uses `adc` and the carry never leaves the # data path. memcheck calls `cset` a conditional move and reports it, and
cbits/tests/ct/run.sh view
@@ -27,17 +27,25 @@ decaf_inc="-DCRYPTON_DECAF_WORD_BITS=64 -I$D/include -I$D/p448 -I$D/include/arch_ref64 -I$D/p448/arch_ref64" -# The generic C, not whatever the machine happens to offer. A build that-# takes AES-NI reports nothing from the AES driver and says nothing about the-# table-driven code every other machine runs.-aes_src="cbits/crypton_aes.c cbits/aes/generic.c cbits/aes/gf.c"+# The portable C, not whatever the machine happens to offer, and the BearSSL+# it is written on. All three builds below have to be silent, so what tells+# them apart is which implementation each one's dispatch chose -- which the+# driver prints and run_one checks, since otherwise a build that quietly took+# the wrong path would pass by saying nothing.+aes_src="cbits/crypton_aes.c cbits/aes/generic.c cbits/aes/gf.c+ cbits/bearssl/aes_ct64.c cbits/bearssl/aes_ct64_enc.c+ cbits/bearssl/aes_ct64_dec.c cbits/bearssl/ghash_ctmul64.c+ cbits/bearssl/dec32le.c" -# And, on AArch64, the same driver again against the instructions. That one-# has to be silent; this one has to report. Either going the wrong way says-# the run is not measuring what it claims to.+# The same driver again against the instructions, on each architecture that+# has them. armv8_src="$aes_src cbits/aes/armv8.c cbits/crypton_cpu.c" armv8_inc="-DWITH_ARMV8_CRYPTO -march=armv8-a+crypto -Icbits/aes" +x86ni_src="$aes_src cbits/aes/x86ni.c cbits/crypton_cpu.c+ cbits/aes/gcm_vaes_x86.c cbits/aes/gcm_vaes512_x86.c"+x86ni_inc="-DWITH_AESNI -DWITH_PCLMUL -maes -mpclmul -mssse3 -msse4.1 -Icbits/aes"+ status=0 have_valgrind=no ct_define=@@ -67,6 +75,32 @@ mlkem_native_inc="$mlkem_inc -DCRYPTON_MLKEM_NATIVE_BACKEND" +# Does this processor have AES instructions at all?+#+# Asked of the system rather than of crypton, because the question being+# settled is whether crypton found what is there -- and a witness that comes+# from the code under test cannot answer that. "Cannot tell" counts as+# having them, so an unfamiliar system turns into a loud failure rather than+# a silent skip.+#+# CRYPTON_CT_CPUINFO is for testing this function itself; nothing else sets+# it.+machine_has_aes() {+ cpuinfo=${CRYPTON_CT_CPUINFO:-/proc/cpuinfo}+ if [ -r "$cpuinfo" ]; then+ # "aes" in Features on ARM, in flags on x86; -w so that a model+ # name containing the letters does not answer for the flags+ grep -qw aes "$cpuinfo"+ return+ fi+ if [ "$(uname -s)" = Darwin ]; then+ [ "$(sysctl -n hw.optional.arm.FEAT_AES 2>/dev/null)" = 1 ] && return 0+ [ "$(sysctl -n hw.optional.aes 2>/dev/null)" = 1 ] && return 0+ return 1+ fi+ return 0+}+ # run_one <name> <sources> <includes> [driver] # # The driver defaults to ct_<name>.c. It is given separately where one@@ -79,6 +113,36 @@ -o "$out/$name" "cbits/tests/ct/ct_$drv.c" $srcs 2> "$out/$name.cc" || { echo "FAIL $name did not build"; sed -n '1,12p' "$out/$name.cc"; status=1; return }+ # Which implementation did this build's dispatch actually take? The+ # driver says, and the three AES builds want different answers. This+ # used to be inferred: the portable build was required to report,+ # because the tables leaked, and silence meant it had taken an+ # accelerated path instead. The tables are gone and all three are+ # silent now, so the question is asked outright rather than read off a+ # leak that no longer happens.+ case $name in+ aes | aes_armv8 | aes_x86ni)+ want=0+ [ "$name" = aes ] || want=1+ got=$("$out/$name" 2>/dev/null | sed -n 's/^dispatch aes=\([01]\).*/\1/p')+ if [ "$got" != "$want" ]; then+ # A processor with no AES instructions is not a failure of+ # this driver; there is simply nothing here to measure, and+ # the portable one above has already measured what does run.+ # Boards in this position are common -- the Raspberry Pi 2,+ # 3 and 4 among them, on either word size.+ if [ "$want" = 1 ] && [ "$got" = 0 ] && ! machine_has_aes; then+ echo "skip $name: this processor has no AES instructions,"+ echo " so there is nothing here to measure"+ return+ fi+ echo "FAIL $name: dispatch took aes=$got, wanted aes=$want --"+ echo " this build is not running the implementation it is named for"+ status=1+ return+ fi+ ;;+ esac if [ "$have_valgrind" = no ]; then "$out/$name" > /dev/null 2>&1 && echo "built $name (no valgrind here; nothing checked)" \ || { echo "FAIL $name did not run"; status=1; }@@ -123,27 +187,17 @@ echo "ok canary: reported $n, so the marking works" fi ;;- aes)- # Silence would mean the build took an accelerated path and so- # measured nothing; the tables reporting is the point.- if [ "$n" -eq 0 ]; then- echo "FAIL aes: reported nothing, so this build did not take the"- echo " table-driven code the driver exists to measure"- status=1- else- echo "note aes: $n report(s), from $(echo "$sites" | tr '\n' ' ')"- fi- ;;- aes_armv8)- # The opposite demand, and known.txt does not apply: the entries in- # it are for the tables, and this build is not supposed to reach- # them. Anything at all here is a finding, including a table site,- # which would mean the dispatch did not pick the instructions.+ aes | aes_armv8 | aes_x86ni)+ # Nothing here looks anything up: the portable AES and GHASH are+ # bitsliced, and the two instruction sets branch on nothing. So+ # anything at all is a finding and known.txt does not apply --+ # until the tables went this entry read the other way round, with+ # the portable build required to report. if [ "$n" -eq 0 ]; then- echo "ok aes_armv8: the instructions decided nothing"+ echo "ok $name: the secret decided nothing" else- echo "FAIL aes_armv8: $n report(s) from the AArch64 AES or GHASH,"- echo " which look nothing up and should branch on nothing:"+ echo "FAIL $name: $n report(s) from AES or GHASH,"+ echo " which should branch on nothing:" for site in $sites; do echo " $site"; done sed -n '/Conditional jump\|Use of uninitialised/,/^==[0-9]*== $/p' \ "$out/$name.log" | head -30 | sed 's/^/ /'@@ -183,7 +237,10 @@ # and the build would not even compile. case $(uname -m) in aarch64 | arm64)- run_one aes_armv8 "$armv8_src" "$armv8_inc"+ run_one aes_armv8 "$armv8_src" "$armv8_inc" aes+ ;;+x86_64 | amd64)+ run_one aes_x86ni "$x86ni_src" "$x86ni_inc" aes ;; esac
cbits/tests/endian/endian.c view
@@ -25,6 +25,9 @@ #include "crypton_chacha.h" #include "crypton_salsa.h" #include "crypton_poly1305.h"+#include "crypton_aes.h"+#include "aes/gf.h"+#include "aes/block128.h" /* The skein headers spell the prefix "cryponite", which nothing defines. */ void crypton_skein256_init(struct skein256_ctx *ctx, uint32_t hashlen);@@ -92,6 +95,131 @@ } \ } while (0) +/*+ * AES, which reaches further into the byte order than the hashes above do.+ *+ * The portable implementation keeps its schedule as 64-bit words and reads+ * its input through br_dec32le; the GHASH beside it reads H and the+ * accumulator as big-endian words; the counter modes carry a counter that is+ * incremented big-endian and stored little-endian in one case and the other+ * way round in another; and XTS doubles its tweak in GF(2^128) through+ * cpu_to_le64. Every one of those is a place where a big-endian machine can+ * differ, and none of them was asked about here until now.+ *+ * The entries called are the public ones, so this is whichever+ * implementation the build installed -- which, for the build this harness+ * makes, is the portable one. That is the one a big-endian machine runs:+ * crypton has no AES instructions for s390x, and the POWER8 ones are+ * little-endian only.+ */+static void aes_answers(void) {+ static const uint8_t keylens[] = {16, 24, 32};+ /* multiples of the block, for the modes that take whole blocks */+ static const uint32_t blocks[] = {1, 2, 4, 7};+ /* and byte counts, including a partial block, for the ones that do not */+ static const uint32_t bytes[] = {0, 1, 15, 16, 17, 64, 100};+ uint8_t key[32], key2[32], iv[16], out[128], tmp[128];+ char nm[128];+ size_t ki, li;+ uint32_t i;++ for (i = 0; i < 32; i++) { key[i] = (uint8_t)(i * 3 + 1);+ key2[i] = (uint8_t)(i * 5 + 2); }+ for (i = 0; i < 16; i++) iv[i] = (uint8_t)(i * 11 + 7);++ for (ki = 0; ki < sizeof keylens / sizeof *keylens; ki++) {+ uint8_t kl = keylens[ki];+ aes_key k, k2;+ aes_gcm_key gk;++ crypton_aes_initkey(&k, key, kl);+ crypton_aes_initkey(&k2, key2, kl);+ crypton_aes_gcm_key_init(&gk, &k);++ for (li = 0; li < sizeof blocks / sizeof *blocks; li++) {+ uint32_t nb = blocks[li];+ aes_block ivb;++ crypton_aes_encrypt_ecb((aes_block *)out, &k, (aes_block *)buf, nb);+ snprintf(nm, sizeof nm, "aes%u-ecb/%u", kl * 8, nb);+ answer(nm, out, nb * 16);++ crypton_aes_decrypt_ecb((aes_block *)out, &k, (aes_block *)buf, nb);+ snprintf(nm, sizeof nm, "aes%u-ecbd/%u", kl * 8, nb);+ answer(nm, out, nb * 16);++ memcpy(&ivb, iv, 16);+ crypton_aes_encrypt_cbc((aes_block *)out, &k, &ivb, (aes_block *)buf, nb);+ snprintf(nm, sizeof nm, "aes%u-cbc/%u", kl * 8, nb);+ answer(nm, out, nb * 16);++ memcpy(&ivb, iv, 16);+ crypton_aes_decrypt_cbc((aes_block *)out, &k, &ivb, (aes_block *)buf, nb);+ snprintf(nm, sizeof nm, "aes%u-cbcd/%u", kl * 8, nb);+ answer(nm, out, nb * 16);++ memcpy(&ivb, iv, 16);+ crypton_aes_encrypt_xts((aes_block *)out, &k, &k2, &ivb, 0,+ (aes_block *)buf, nb);+ snprintf(nm, sizeof nm, "aes%u-xts/%u", kl * 8, nb);+ answer(nm, out, nb * 16);++ memcpy(&ivb, iv, 16);+ crypton_aes_decrypt_xts((aes_block *)out, &k, &k2, &ivb, 0,+ (aes_block *)buf, nb);+ snprintf(nm, sizeof nm, "aes%u-xtsd/%u", kl * 8, nb);+ answer(nm, out, nb * 16);++ /* a starting point too, since that is extra tweak doubling */+ memcpy(&ivb, iv, 16);+ crypton_aes_encrypt_xts((aes_block *)out, &k, &k2, &ivb, 3,+ (aes_block *)buf, nb);+ snprintf(nm, sizeof nm, "aes%u-xts-sp3/%u", kl * 8, nb);+ answer(nm, out, nb * 16);+ }++ for (li = 0; li < sizeof bytes / sizeof *bytes; li++) {+ uint32_t n = bytes[li];+ aes_block ivb;++ memcpy(&ivb, iv, 16);+ crypton_aes_encrypt_ctr(out, &k, &ivb, buf, n);+ snprintf(nm, sizeof nm, "aes%u-ctr/%u", kl * 8, n);+ answer(nm, out, n);++ /* the tag goes after the ciphertext, so this answers for both */+ crypton_aes_gcm_full_encrypt(out, &gk, &k, iv, 12, buf, 13,+ buf, n, 16);+ snprintf(nm, sizeof nm, "aes%u-gcm/%u", kl * 8, n);+ answer(nm, out, n + 16);+ }+ }++ /* GHASH and POLYVAL on their own, which the modes above reach only+ * through whatever length they were given */+ {+ table_4bit ht;+ block128 acc;+ aes_polyval pv;++ crypton_aes_generic_hinit(ht, (const block128 *)buf);+ memcpy(&acc, buf + 16, 16);+ crypton_aes_generic_gf_mul(&acc, ht);+ answer("ghash-mul", (const uint8_t *)&acc, 16);++ crypton_aes_generic_hinit(ht, (const block128 *)buf);+ memcpy(&acc, buf + 16, 16);+ crypton_aes_generic_gf_mul4(&acc, (const block128 *)(buf + 32), ht);+ answer("ghash-mul4", (const uint8_t *)&acc, 16);++ memcpy(tmp, buf, 16);+ crypton_aes_polyval_init(&pv, (const aes_block *)tmp);+ crypton_aes_polyval_update(&pv, buf + 16, 64);+ crypton_aes_polyval_finalize(&pv, (aes_block *)out);+ answer("polyval", out, 16);+ }+}+ int main(int argc, char **argv) { generating = (argc > 1 && strcmp(argv[1], "generate") == 0); vf = fopen(argc > 2 ? argv[2] : "cbits/tests/endian/vectors.txt",@@ -182,6 +310,8 @@ answer(nmbuf, mac, sizeof mac); } }++ aes_answers(); if (!generating && failures == 0) printf("ok %d answers match the little-endian ones\n", checked);
cbits/tests/endian/run.sh view
@@ -6,6 +6,14 @@ # loaders in crypton_align.h -- rewritten from word-typed casts to memcpy in # #256 -- and eight more decide something from the byte order themselves. #+# AES is here as well, and it reaches further into the byte order than the+# hashes do: a schedule kept as 64-bit words, a GHASH that reads H and its+# accumulator as big-endian words, counters incremented one way and stored+# the other, and an XTS tweak doubled through cpu_to_le64. It is the+# portable implementation that answers, which is what a big-endian machine+# runs: crypton has no AES instructions for s390x, and the POWER8 ones are+# little-endian only.+# # The two sides cannot be compared in one run the way the 32-bit harness # compares two builds, because the machine doing the comparing has only one # byte order. So the answers are frozen: vectors.txt is what this code gives@@ -26,7 +34,11 @@ cbits/crypton_sha256.c cbits/crypton_sha512.c cbits/crypton_sha3.c cbits/crypton_ripemd.c cbits/crypton_skein256.c cbits/crypton_skein512.c cbits/crypton_tiger.c cbits/crypton_whirlpool.c- cbits/crypton_chacha.c cbits/crypton_salsa.c cbits/crypton_poly1305.c"+ cbits/crypton_chacha.c cbits/crypton_salsa.c cbits/crypton_poly1305.c+ cbits/crypton_aes.c cbits/aes/generic.c cbits/aes/gf.c+ cbits/bearssl/aes_ct64.c cbits/bearssl/aes_ct64_enc.c+ cbits/bearssl/aes_ct64_dec.c cbits/bearssl/ghash_ctmul64.c+ cbits/bearssl/dec32le.c" # Generating is done under the sanitizers, since a driver that writes out of # bounds would otherwise freeze whatever it happened to leave behind. That@@ -38,7 +50,7 @@ fi # shellcheck disable=SC2086-$cc -O2 -g $san -Icbits -Icbits/include64 -o "$out/endian" \+$cc -O2 -g $san -Icbits -Icbits/aes -Icbits/include64 -o "$out/endian" \ cbits/tests/endian/endian.c $srcs if [ "$mode" = generate ]; then
cbits/tests/endian/vectors.txt view
@@ -250,3 +250,132 @@ chacha20/1000 064429ccbd6ef85054af07f8dee3a283e75e844df9a59b1b76addd7d41fb554946f5dd12b49b97f4a72f05b0b41c557e8c69da034ae9dddc8a54dbf245d2bcff salsa20/1000 9ee7f7b5774db507d8e78fe7398594de9b40444a5301c14dfbc675e6310f9aa659e9cca9892563df9386a60607a2e18781d35d4897ea2b029db7b8819cafae17 poly1305/1000 c403a63071ea2f2e732472db8e7b79c0+aes128-ecb/1 51d0767be0142b54f38422f098a979c1+aes128-ecbd/1 0772f11dd1e53677251c6d2b42c12a84+aes128-cbc/1 53423fee1b0b9dda9006158bb0e3c6e6+aes128-cbcd/1 0060ec35e2db7f237a7618abc9578b28+aes128-xts/1 2634e65251d6fed878a1b3546061cd8f+aes128-xtsd/1 9d22db2b665e917344f60d1561481854+aes128-xts-sp3/1 1d40e186532c4338b08ebb40039c49b6+aes128-ecb/2 51d0767be0142b54f38422f098a979c1f5e0ee3fe17beef5b33b72d8f60ec17c+aes128-ecbd/2 0772f11dd1e53677251c6d2b42c12a843d8abf116e0bd219a05d2c1d89e8faee+aes128-cbc/2 53423fee1b0b9dda9006158bb0e3c6e6b1acc41a9ff16d75c2170e3242346760+aes128-cbcd/2 0060ec35e2db7f237a7618abc9578b283d8db1047228f82898626a50ddb39887+aes128-xts/2 2634e65251d6fed878a1b3546061cd8ff2e82679d8d7ade78233b89a76958855+aes128-xtsd/2 9d22db2b665e917344f60d1561481854c8d8c3fe1890b52db7ca720cd253192d+aes128-xts-sp3/2 1d40e186532c4338b08ebb40039c49b6cc2b2b42b37e9412005737634804ad97+aes128-ecb/4 51d0767be0142b54f38422f098a979c1f5e0ee3fe17beef5b33b72d8f60ec17c527ee4edb0bd6e9346d1148c3763613b5112f9a67b30c06fd511e725279957b9+aes128-ecbd/4 0772f11dd1e53677251c6d2b42c12a843d8abf116e0bd219a05d2c1d89e8faeed21af82debb90e26d30fd338766a0ff47234606001901e7888e61a64468495be+aes128-cbc/4 53423fee1b0b9dda9006158bb0e3c6e6b1acc41a9ff16d75c2170e324234676075d60f392793fd0623859fffb40b23585955ff09a2c20c02b9c2baf084d6943b+aes128-cbcd/4 0060ec35e2db7f237a7618abc9578b283d8db1047228f82898626a50ddb39887a26d86a8672a94877ba06585b2a1dd2d8d326d741ab23748bfd85f2815def4d6+aes128-xts/4 2634e65251d6fed878a1b3546061cd8ff2e82679d8d7ade78233b89a76958855da809daab8adf228232779804848093c5aa478a2d7e0c11b58dfbfd19b94ea36+aes128-xtsd/4 9d22db2b665e917344f60d1561481854c8d8c3fe1890b52db7ca720cd253192dedd68b0854e3340afb2d8a34942bcd5479bed9f04ee52769479c2192544eaa2f+aes128-xts-sp3/4 1d40e186532c4338b08ebb40039c49b6cc2b2b42b37e9412005737634804ad970343c410aa37b63b6595f3e6e8bbcb8bc8581ec02491cb1e3df920165a8d05c7+aes128-ecb/7 51d0767be0142b54f38422f098a979c1f5e0ee3fe17beef5b33b72d8f60ec17c527ee4edb0bd6e9346d1148c3763613b5112f9a67b30c06fd511e725279957b977a15ad7bbc6a54c141a2d07c4b233000adedca5ef085da03709e978ed1aca3500e2bd1f0ad7a29ade429446d2ddfb93+aes128-ecbd/7 0772f11dd1e53677251c6d2b42c12a843d8abf116e0bd219a05d2c1d89e8faeed21af82debb90e26d30fd338766a0ff47234606001901e7888e61a64468495be2de8cb16ce03066f65758235de9386de764ddfa85c70fae9b4c80eed9b53a216d5c34dcea96fa27e88fe002306a3548d+aes128-cbc/7 53423fee1b0b9dda9006158bb0e3c6e6b1acc41a9ff16d75c2170e324234676075d60f392793fd0623859fffb40b23585955ff09a2c20c02b9c2baf084d6943b84e8cef5ab838dc0864175cab14a1026f6230f8df62be8d6794dfb9aba0622e62e6a358aab88832f2b18d4819eca4f91+aes128-cbcd/7 0060ec35e2db7f237a7618abc9578b283d8db1047228f82898626a50ddb39887a26d86a8672a94877ba06585b2a1dd2d8d326d741ab23748bfd85f2815def4d6429eb69245919fcfc2db37891d5957068848d3bb4651d2c682f54aa6c90ac271bbb6314d23fe3ae12e53b498c46a845a+aes128-xts/7 2634e65251d6fed878a1b3546061cd8ff2e82679d8d7ade78233b89a76958855da809daab8adf228232779804848093c5aa478a2d7e0c11b58dfbfd19b94ea3651d66c7adda6172d3e8672cc65cce7dbd8d9d6f32bf8f087867527f0a8128fdd80ed3a97398f66577f2041342c585e72+aes128-xtsd/7 9d22db2b665e917344f60d1561481854c8d8c3fe1890b52db7ca720cd253192dedd68b0854e3340afb2d8a34942bcd5479bed9f04ee52769479c2192544eaa2fa426c177c9b397f0bb380e3991c69479759294d63723eeb04aae18e48f84fa213dcf254eb1d532252c6328a8fdd8976f+aes128-xts-sp3/7 1d40e186532c4338b08ebb40039c49b6cc2b2b42b37e9412005737634804ad970343c410aa37b63b6595f3e6e8bbcb8bc8581ec02491cb1e3df920165a8d05c7fa1c26910201c2ac91dab7ffcb1995c57cfc19a1a8ffbb3e11debe1e9de63147bd99f1f0f9af66d827b18a427ed02d32+aes128-ctr/0 -+aes128-gcm/0 91527d0ffa250047dc10c74aaf7dc630+aes128-ctr/1 f3+aes128-gcm/1 110be12ced502543dc2b098ce214c704a9+aes128-ctr/15 f3e218ff08f7128f6c9e088df658bf+aes128-gcm/15 11e696e7a098883e54b19200262e669d12ac829a6a3eed6933b267f13690e7+aes128-ctr/16 f3e218ff08f7128f6c9e088df658bfcc+aes128-gcm/16 11e696e7a098883e54b19200262e6622968a6e12ea8c1d162967acbccc339980+aes128-ctr/17 f3e218ff08f7128f6c9e088df658bfccea+aes128-gcm/17 11e696e7a098883e54b19200262e6622dd540e6e9683dd7026b3b24579826c0593+aes128-ctr/64 f3e218ff08f7128f6c9e088df658bfccea245fbfae1cb12ed69d5f01a5a6d7f161f8ae682ba4555a088d89fe91fdfb2890935fe9d0ae19f0ac4f73e19d26d1cc+aes128-gcm/64 11e696e7a098883e54b19200262e6622dd7f1ac8145981a493fdaea4641bf49da46e0d311c3daf1665fd5fdf025965d9312f324d4c646d3f3f85ab13d236716fdb10935bd1c334f13b56a98aa09ef476+aes128-ctr/100 f3e218ff08f7128f6c9e088df658bfccea245fbfae1cb12ed69d5f01a5a6d7f161f8ae682ba4555a088d89fe91fdfb2890935fe9d0ae19f0ac4f73e19d26d1cc8c30a3d0749027d7cff23bb8dbccb1a8540feb89f5ca9447a5cdef585d59125c58b73b22+aes128-gcm/100 11e696e7a098883e54b19200262e6622dd7f1ac8145981a493fdaea4641bf49da46e0d311c3daf1665fd5fdf025965d9312f324d4c646d3f3f85ab13d236716fc40f23dfc154e04b18dfc89ae50a386551eab4dd02434c526f9fc45a278b21a485347381adbe43ff0386975d584a98adacbde34a+aes192-ecb/1 b7bd8a4e647f413adc38c4464e4434b6+aes192-ecbd/1 4308adb49827c2741aabb7c40c5b63a5+aes192-cbc/1 9fdd2c700fd7e916fd89805f18439a27+aes192-cbcd/1 441ab09cab198b2045c1c24487cdc209+aes192-xts/1 c2eea1b9f5f0fd170168a3e2af1d059a+aes192-xtsd/1 5a3a19588f9700d5482eae9dd316051d+aes192-xts-sp3/1 f1dba9fc3e3d43d3a46608690e98b0e8+aes192-ecb/2 b7bd8a4e647f413adc38c4464e4434b67d307330822566b908545f9e8de18c1a+aes192-ecbd/2 4308adb49827c2741aabb7c40c5b63a5e6015605e3e6c8df51f2de769db588e2+aes192-cbc/2 9fdd2c700fd7e916fd89805f18439a2754e55a17706f184fe6b2185ef87eb075+aes192-cbcd/2 441ab09cab198b2045c1c24487cdc209e6065810ffc5e2ee69cd983bc9eeea8b+aes192-xts/2 c2eea1b9f5f0fd170168a3e2af1d059af4c8308582684da17d5b6d631459d6d5+aes192-xtsd/2 5a3a19588f9700d5482eae9dd316051dab8c3b6155002903c90664413bd96653+aes192-xts-sp3/2 f1dba9fc3e3d43d3a46608690e98b0e823173a551464b1a3ef6095560902a48c+aes192-ecb/4 b7bd8a4e647f413adc38c4464e4434b67d307330822566b908545f9e8de18c1ac8d6dd6f7bee6e99e4682aef343f7cd0ea985b24e2368b8d1acd3b0bf164a30b+aes192-ecbd/4 4308adb49827c2741aabb7c40c5b63a5e6015605e3e6c8df51f2de769db588e2f4a1ce5088ed9eee6f721f73986dcf23ffc534ac5d16cb7010015dc1b50330ce+aes192-cbc/4 9fdd2c700fd7e916fd89805f18439a2754e55a17706f184fe6b2185ef87eb075061361ddd895a18991e3068ef563dcea087fc1517810a4ac222a391215026095+aes192-cbcd/4 441ab09cab198b2045c1c24487cdc209e6065810ffc5e2ee69cd983bc9eeea8b84d6b0d5047e044fc7dda9ce5ca61dfa00c339b84634e240273f188de65951a6+aes192-xts/4 c2eea1b9f5f0fd170168a3e2af1d059af4c8308582684da17d5b6d631459d6d52e43efd39bfc3ebbbbf189a4bfb1071bbc3851a67841dc4206c4f0f86fca0fed+aes192-xtsd/4 5a3a19588f9700d5482eae9dd316051dab8c3b6155002903c90664413bd9665351b41f0a6b8b2afd38a19274800e86739bb50da065cc97f9818895e062c6be6c+aes192-xts-sp3/4 f1dba9fc3e3d43d3a46608690e98b0e823173a551464b1a3ef6095560902a48cdc267dd87440e9a910bedc98758b512799ad24644966748404717b701ad70804+aes192-ecb/7 b7bd8a4e647f413adc38c4464e4434b67d307330822566b908545f9e8de18c1ac8d6dd6f7bee6e99e4682aef343f7cd0ea985b24e2368b8d1acd3b0bf164a30b225dbf0a85eb782bbaf30c02286d91d271bbfb08e60fc12977e7eea142abc72bed4ade399f048bfa5d5a48f02ac47e22+aes192-ecbd/7 4308adb49827c2741aabb7c40c5b63a5e6015605e3e6c8df51f2de769db588e2f4a1ce5088ed9eee6f721f73986dcf23ffc534ac5d16cb7010015dc1b50330ce1b0b2582b28d8f6174ee21d4cefdc31632f8ca413704961b7da50fdc2de61e730ab37b8b3148660f36335f6d4a01d85a+aes192-cbc/7 9fdd2c700fd7e916fd89805f18439a2754e55a17706f184fe6b2185ef87eb075061361ddd895a18991e3068ef563dcea087fc1517810a4ac222a391215026095582b2b9c34d2b9b40ef88f9ec4459fc671f5ecb0d23defc6ea99f6075174381b861dec529ac5a49c9c55efae85aca707+aes192-cbcd/7 441ab09cab198b2045c1c24487cdc209e6065810ffc5e2ee69cd983bc9eeea8b84d6b0d5047e044fc7dda9ce5ca61dfa00c339b84634e240273f188de65951a6747d5806391f16c1d34094680d3712ceccfdc6522d25be344b984b977fbf7e1464c60708bbd9fe90909eebd688c8088d+aes192-xts/7 c2eea1b9f5f0fd170168a3e2af1d059af4c8308582684da17d5b6d631459d6d52e43efd39bfc3ebbbbf189a4bfb1071bbc3851a67841dc4206c4f0f86fca0fedf90c2617695920f2f5611aab587198b0263b555f0a93d1580c22822b6fe3451e31e5f67c9526f5128d87bc586e61a255+aes192-xtsd/7 5a3a19588f9700d5482eae9dd316051dab8c3b6155002903c90664413bd9665351b41f0a6b8b2afd38a19274800e86739bb50da065cc97f9818895e062c6be6cfa00838607d2c4a55642a4973aac6fbdd5e9c514e69e8569d1559acde57fc207a5709b8f0eff9a93ad7dc6c170fda2c0+aes192-xts-sp3/7 f1dba9fc3e3d43d3a46608690e98b0e823173a551464b1a3ef6095560902a48cdc267dd87440e9a910bedc98758b512799ad24644966748404717b701ad70804506ba3f939ed54c43dc721d3532f83cf267aa4c32d9543edb71de9548dacd5171e76bef732fdad2c49ebb781d639a445+aes192-ctr/0 -+aes192-gcm/0 b03daf3f9346095b3083d09391d47cee+aes192-ctr/1 72+aes192-gcm/1 bdc01f6ed1a9e3dc9be2c4d9e8d8d3b584+aes192-ctr/15 72138e46596ca4b6de00c2aeadc2c1+aes192-gcm/15 bd19761039873f6946e29ba643ee41c6ca3791dc7497012fce9f48631f599c+aes192-ctr/16 72138e46596ca4b6de00c2aeadc2c17c+aes192-gcm/16 bd19761039873f6946e29ba643ee4175c85f71850653f05f2410354e2e93cdc9+aes192-ctr/17 72138e46596ca4b6de00c2aeadc2c17c02+aes192-gcm/17 bd19761039873f6946e29ba643ee4175b3e87ccbca9af385c12d31382f6e721151+aes192-ctr/64 72138e46596ca4b6de00c2aeadc2c17c02df852e3c7abed9e3cd5a94d24c71074945c7835771862ec94dbc9c8ba47ef305131b7f3933a150025d8a796c5889b2+aes192-gcm/64 bd19761039873f6946e29ba643ee4175b3c8f8e942762e364f67333b85f6c220999554b2d6e78a47fc789581a39e89a4b6a5539491894c85fbba17f03e183e3b27736f65aaba8f86ba884618e2c75b83+aes192-ctr/100 72138e46596ca4b6de00c2aeadc2c17c02df852e3c7abed9e3cd5a94d24c71074945c7835771862ec94dbc9c8ba47ef305131b7f3933a150025d8a796c5889b2aea7d364add8934a993f8532bae1b59ec0c4d75bf59dca8e265de412ed73052188706192+aes192-gcm/100 bd19761039873f6946e29ba643ee4175b3c8f8e942762e364f67333b85f6c220999554b2d6e78a47fc789581a39e89a4b6a5539491894c85fbba17f03e183e3bb42b4d0d362fc35b7719b76c299e857e82b8b127e9cbc732559c26b6338ef92f153271feb0462a5c14552476a3a41aae5c275596+aes256-ecb/1 791925c38818304f047d756d78ef8f13+aes256-ecbd/1 831e330b93ed108f70c15c463224b5c6+aes256-cbc/1 6b1dfef4ab6d3fd0a32a3c3a7c9603d5+aes256-cbcd/1 840c2e23a0d359db2fab29c6b9b2146a+aes256-xts/1 b90a69094799c4e679fc829084d41055+aes256-xtsd/1 18db4b08618f8573974e5c85d0954302+aes256-xts-sp3/1 d5f9ef046dd7e688938aab18d3d68ce5+aes256-ecb/2 791925c38818304f047d756d78ef8f139da36087c3b26165970cf0987eee7d70+aes256-ecbd/2 831e330b93ed108f70c15c463224b5c6f2f7d848bb39a82dd65b4db54a6f54ff+aes256-cbc/2 6b1dfef4ab6d3fd0a32a3c3a7c9603d5090fa2b707cd96fea1132c143d5683fa+aes256-cbcd/2 840c2e23a0d359db2fab29c6b9b2146af2f0d65da71a821cee640bf81e343696+aes256-xts/2 b90a69094799c4e679fc829084d4105597001399d1f01a458a64913345271c27+aes256-xtsd/2 18db4b08618f8573974e5c85d0954302511a3cdde7128329882a30bd28169980+aes256-xts-sp3/2 d5f9ef046dd7e688938aab18d3d68ce5724313716ca706a5b7a914024b26cfc1+aes256-ecb/4 791925c38818304f047d756d78ef8f139da36087c3b26165970cf0987eee7d70aff88ab2d21b91927e27057d2f91f923cd1e35c75f2b3111adc0fd520ae368b4+aes256-ecbd/4 831e330b93ed108f70c15c463224b5c6f2f7d848bb39a82dd65b4db54a6f54ff6c4c59c5bd048efc1c7cf8353e29eae59fd397d041edd80e0c4a41824a3b7bc1+aes256-cbc/4 6b1dfef4ab6d3fd0a32a3c3a7c9603d5090fa2b707cd96fea1132c143d5683fae5e8441b18737ef3581adef0e182aea3b94a6e5e4925464a95ddb6b409f19e88+aes256-cbcd/4 840c2e23a0d359db2fab29c6b9b2146af2f0d65da71a821cee640bf81e3436961c3b27403197145db4d34e88fae2383c60d59ac45acff13e3b7404ce19611aa9+aes256-xts/4 b90a69094799c4e679fc829084d4105597001399d1f01a458a64913345271c27261700fa455603c6d5370d3867bf3213717c7241850cc16b414cf4598472f9e7+aes256-xtsd/4 18db4b08618f8573974e5c85d0954302511a3cdde7128329882a30bd28169980f52cc7d5d434263c1c22188e735a49d6130e0683a55e636739ba7abb743e7b09+aes256-xts-sp3/4 d5f9ef046dd7e688938aab18d3d68ce5724313716ca706a5b7a914024b26cfc1b47c048f8548cc7a80a22a9dcaf25065efd935f4730020a085cfd489b8a98cab+aes256-ecb/7 791925c38818304f047d756d78ef8f139da36087c3b26165970cf0987eee7d70aff88ab2d21b91927e27057d2f91f923cd1e35c75f2b3111adc0fd520ae368b4b6d87226c642479229919a9c47bfca56ca5607a049cd7f423fb0556511f5db4bb8f7165904e83e4da028f316252e9f5d+aes256-ecbd/7 831e330b93ed108f70c15c463224b5c6f2f7d848bb39a82dd65b4db54a6f54ff6c4c59c5bd048efc1c7cf8353e29eae59fd397d041edd80e0c4a41824a3b7bc1eaa8daa1a6f40e115cfd93684f1b7ad8b66b09a303b2488f93eaac36762e571d4f00c320962c9dc1607e8d884579f306+aes256-cbc/7 6b1dfef4ab6d3fd0a32a3c3a7c9603d5090fa2b707cd96fea1132c143d5683fae5e8441b18737ef3581adef0e182aea3b94a6e5e4925464a95ddb6b409f19e88f104a783b152e599e90b85fc60f57aef7ee95c1608290490adf2688b4683b2d70a6b16291cb5a49008dd74dbf8c16191+aes256-cbcd/7 840c2e23a0d359db2fab29c6b9b2146af2f0d65da71a821cee640bf81e3436961c3b27403197145db4d34e88fae2383c60d59ac45acff13e3b7404ce19611aa985dea7252d6697b1fb5326d48cd1ab00486e05b0199360a0a5d7e87d2477377a2175bfa31cbd055ec6d3393387b023d1+aes256-xts/7 b90a69094799c4e679fc829084d4105597001399d1f01a458a64913345271c27261700fa455603c6d5370d3867bf3213717c7241850cc16b414cf4598472f9e77778636d8d94ec90c8bc220a7075ab56b4758635686c3985c243224397844894defd980efc5ce05ed451cec1af9583e9+aes256-xtsd/7 18db4b08618f8573974e5c85d0954302511a3cdde7128329882a30bd28169980f52cc7d5d434263c1c22188e735a49d6130e0683a55e636739ba7abb743e7b09cdc794b975bdb545b6a5a69fde322ceb0f3da5d8a1dd3103dee190f9fb65d8a0e86ddffed9358c0872862ae921e7ba75+aes256-xts-sp3/7 d5f9ef046dd7e688938aab18d3d68ce5724313716ca706a5b7a914024b26cfc1b47c048f8548cc7a80a22a9dcaf25065efd935f4730020a085cfd489b8a98cab5e7eb266eb52f17b6f78bcb0996b96fc7ca3a671856b32fe2e8c9341c50a70c2d40906b8d450fed98e6a5f6a4d5f84ab+aes256-ctr/0 -+aes256-gcm/0 8316aabedf3935666f9d8b2dd6b5b356+aes256-ctr/1 21+aes256-gcm/1 7b8fcd0cf59d4bccd66dd123de4c1c7aa4+aes256-ctr/15 219430a80fe1a08ed512ff1c9cee61+aes256-gcm/15 7bfc609c1f245b464a8e0763c5e8b7a1dba82dac1ce33ecb0350b7d9e582df+aes256-ctr/16 219430a80fe1a08ed512ff1c9cee61dc+aes256-gcm/16 7bfc609c1f245b464a8e0763c5e8b7f9ae001617c759469bd602cd92dd4bf7e9+aes256-ctr/17 219430a80fe1a08ed512ff1c9cee61dc52+aes256-gcm/17 7bfc609c1f245b464a8e0763c5e8b7f96b70cf1b59a4a56949e6437f0b9b7b257b+aes256-ctr/64 219430a80fe1a08ed512ff1c9cee61dc52867258efbdf3943f48d30ef7da12c7bfb56c293beeddc92b1c270b7110c0e2db805ddad464f0eac9d46e7109001374+aes256-gcm/64 7bfc609c1f245b464a8e0763c5e8b7f96bc3d807f39fd255cbf14006baf6bbccb5d2509a2b3cba5a4c7ee680ee09979e810f94882ff5c0b6e572d397aef389ec72e6a3f5eb4be76004c10d8c5881492a+aes256-ctr/100 219430a80fe1a08ed512ff1c9cee61dc52867258efbdf3943f48d30ef7da12c7bfb56c293beeddc92b1c270b7110c0e2db805ddad464f0eac9d46e71090013745c52cea313ba7efea2eac9da963dcf1bc484a1c1924fd8f91fd8bf774dc77b472356c522+aes256-gcm/100 7bfc609c1f245b464a8e0763c5e8b7f96bc3d807f39fd255cbf14006baf6bbccb5d2509a2b3cba5a4c7ee680ee09979e810f94882ff5c0b6e572d397aef389ecb7bd0a0a1f4d638d3e2557ca92beb2bf4228499ac6cc38d20e0539cd4658ed2a7d24f4ab32d4b6ea69814b4c2718fd1dcdb7e16a+ghash-mul c65b128f165974f311df03df3486ce71+ghash-mul4 6bfd46cd3d314cac2d75ead0e96092c7+polyval bbc31d0761373f0b2737904b9a76fb27
cbits/tests/fuzz/run.sh view
@@ -31,7 +31,10 @@ canary) echo "" ;; ed25519) echo "cbits/ed25519/ed25519.c cbits/crypton_sha512.c" ;; p256) echo "cbits/p256/p256.c cbits/p256/p256_ec.c" ;;- aead) echo "cbits/crypton_aes.c cbits/aes/generic.c cbits/aes/gf.c" ;;+ aead) echo "cbits/crypton_aes.c cbits/aes/generic.c cbits/aes/gf.c+ cbits/bearssl/aes_ct64.c cbits/bearssl/aes_ct64_enc.c+ cbits/bearssl/aes_ct64_dec.c cbits/bearssl/ghash_ctmul64.c+ cbits/bearssl/dec32le.c" ;; decaf) echo "$D/ed448goldilocks/decaf_all.c $D/ed448goldilocks/eddsa.c $D/ed448goldilocks/scalar.c $D/p448/f_arithmetic.c $D/p448/f_generic.c $D/utils.c
+ cbits/tests/ppc8_diff.c view
@@ -0,0 +1,254 @@+/*+ * The POWER8 path against the portable one.+ *+ * crypton's own test suite exercises whichever implementation the machine+ * installed, so on a POWER8 it never runs the portable code and nothing+ * compares the two. CI has no POWER8 either. This does the comparison in+ * one process, with both compiled in, and runs under qemu-ppc64le.+ *+ * powerpc64le-linux-gnu-gcc -O2 -static -Icbits -Icbits/aes \+ * -DWITH_PPC8_CRYPTO -o ppc8_diff \+ * cbits/tests/ppc8_diff.c cbits/crypton_aes.c cbits/aes/generic.c \+ * cbits/aes/gf.c cbits/aes/ppc8.c cbits/crypton_cpu.c \+ * cbits/asm/aesp8-ppc-linux64le.S cbits/asm/ghashp8-ppc-linux64le.S \+ * cbits/bearssl/*.c+ * qemu-ppc64le-static ./ppc8_diff+ *+ * Given any argument it corrupts a result in each section, so that the+ * comparison can be seen to notice: one that cannot fail has said nothing.+ *+ * One trap in writing this, worth naming because it looks like a crash in+ * the code under test. Several of the generic mode loops reach the block+ * function through the branch table, which in this process holds the POWER8+ * entries -- so calling them as the reference hands a portable key to the+ * POWER8 block function. The references below are therefore the entries+ * that reach cbits/aes/generic.c directly, and XTS, whose generic loop takes+ * its tweak through the table, is written out here instead.+ */+#include <stdio.h>+#include <stdlib.h>+#include <string.h>+#include <stdint.h>++#include "crypton_aes.h"+#include "aes/generic.h"+#include "aes/gf.h"+#include "aes/block128.h"+#include "aes/ppc8.h"++/* the portable CTR, which reaches generic.c rather than the branch table */+void crypton_aes_bitsliced_encrypt_ctr(uint8_t *output, aes_key *key,+ aes_block *iv, uint8_t *input,+ uint32_t len);++static int failures;+static int sabotage;++static void same(const char *what, const uint8_t *a, const uint8_t *b, size_t n)+{+ size_t i;++ if (memcmp(a, b, n) == 0)+ return;+ printf(" MISMATCH %s\n ppc8 ", what);+ for (i = 0; i < n && i < 32; i++) printf("%02x", a[i]);+ printf("\n portable ");+ for (i = 0; i < n && i < 32; i++) printf("%02x", b[i]);+ printf("\n");+ failures++;+}++static uint32_t rnd_state = 1;+static uint8_t rnd(void)+{+ rnd_state = rnd_state * 1103515245u + 12345u;+ return (uint8_t) (rnd_state >> 16);+}+static void rnd_fill(uint8_t *p, size_t n)+{+ while (n--) *p++ = rnd();+}++/*+ * XTS with the portable block function, written out because the generic+ * loop takes its tweak through the branch table.+ */+static void xts_reference(uint8_t *out, aes_key *k1, aes_key *k2,+ const uint8_t *dataunit, uint32_t spoint,+ const uint8_t *in, uint32_t nb, int decrypt)+{+ block128 tweak, t;+ uint32_t i;++ memcpy(&tweak, dataunit, 16);+ crypton_aes_generic_encrypt_block(&tweak, k2, &tweak);+ while (spoint-- > 0)+ crypton_aes_generic_gf_mulx(&tweak);++ for (i = 0; i < nb; i++) {+ block128_vxor(&t, (const block128 *) (in + 16 * i), &tweak);+ if (decrypt)+ crypton_aes_generic_decrypt_block(&t, k1, &t);+ else+ crypton_aes_generic_encrypt_block(&t, k1, &t);+ block128_vxor((block128 *) (out + 16 * i), &t, &tweak);+ crypton_aes_generic_gf_mulx(&tweak);+ }+}++/* the two inits write different things into the same aes_key, so each side+ * gets its own */+static void keys(aes_key *p8, aes_key *gen, const uint8_t *k, uint8_t len)+{+ memset(p8, 0, sizeof *p8);+ memset(gen, 0, sizeof *gen);+ p8->nbr = gen->nbr = len == 16 ? 10 : len == 24 ? 12 : 14;+ crypton_aes_ppc8_init(p8, (uint8_t *) k, len);+ crypton_aes_generic_init(gen, (uint8_t *) k, len);+}++int main(int argc, char **argv)+{+ static const uint8_t klens[3] = { 16, 24, 32 };+ int round;++ setvbuf(stdout, NULL, _IONBF, 0); /* so a crash does not eat the section it was in */+ sabotage = argc > 1;++ printf("== the dispatch would take: aes=%d ==\n",+ crypton_aes_ppc8_available());++ printf("== ECB and CBC, both directions ==\n");+ for (round = 0; round < 300; round++) {+ uint8_t key[32], in[16 * 9], a[16 * 9], b[16 * 9], iv[16];+ uint8_t len = klens[round % 3];+ uint32_t nb = 1 + (round % 9);+ aes_key kp, kg;++ rnd_fill(key, len);+ rnd_fill(in, nb * 16);+ rnd_fill(iv, 16);+ keys(&kp, &kg, key, len);++ crypton_aes_ppc8_encrypt_ecb((aes_block *) a, &kp, (aes_block *) in, nb);+ crypton_aes_generic_encrypt_ecb((aes_block *) b, &kg, (aes_block *) in, nb);+ if (sabotage && round == 3) a[0] ^= 1;+ same("ecb encrypt", a, b, nb * 16);++ crypton_aes_ppc8_decrypt_ecb((aes_block *) a, &kp, (aes_block *) in, nb);+ crypton_aes_generic_decrypt_ecb((aes_block *) b, &kg, (aes_block *) in, nb);+ same("ecb decrypt", a, b, nb * 16);++ {+ aes_block iv1, iv2;+ memcpy(&iv1, iv, 16); memcpy(&iv2, iv, 16);+ crypton_aes_ppc8_encrypt_cbc((aes_block *) a, &kp, &iv1, (aes_block *) in, nb);+ crypton_aes_generic_encrypt_cbc((aes_block *) b, &kg, &iv2, (aes_block *) in, nb);+ same("cbc encrypt", a, b, nb * 16);++ memcpy(&iv1, iv, 16); memcpy(&iv2, iv, 16);+ crypton_aes_ppc8_decrypt_cbc((aes_block *) a, &kp, &iv1, (aes_block *) in, nb);+ crypton_aes_generic_decrypt_cbc((aes_block *) b, &kg, &iv2, (aes_block *) in, nb);+ same("cbc decrypt", a, b, nb * 16);+ }+ }++ /*+ * CTR, and the one place the two counters have to be reconciled by+ * hand. crypton counts over all 128 bits and the assembly over the+ * low 32, so the work is handed over in runs that stop where that word+ * wraps. A buffer that never reaches the wrap exercises none of that,+ * so the IV here is placed a few blocks before it on purpose.+ */+ printf("== CTR, including the 32-bit counter wrapping ==\n");+ for (round = 0; round < 200; round++) {+ uint8_t key[32], in[600], a[600], b[600], iv[16];+ uint8_t len = klens[round % 3];+ uint32_t n = 1 + (round % 600);+ aes_key kp, kg;+ aes_block iv1, iv2;+ uint32_t before = round % 5; /* blocks left before the wrap */++ rnd_fill(key, len);+ rnd_fill(in, n);+ rnd_fill(iv, 16);+ /* put the low word within a few blocks of wrapping, and for a+ * third of the rounds exactly on it */+ iv[12] = 0xff; iv[13] = 0xff; iv[14] = 0xff;+ iv[15] = (uint8_t) (0x100 - before - 1);+ keys(&kp, &kg, key, len);++ memcpy(&iv1, iv, 16); memcpy(&iv2, iv, 16);+ crypton_aes_ppc8_encrypt_ctr(a, &kp, &iv1, in, n);+ crypton_aes_bitsliced_encrypt_ctr(b, &kg, &iv2, in, n);+ if (sabotage && round == 7) a[n - 1] ^= 2;+ same("ctr across the wrap", a, b, n);+ }++ printf("== XTS, both directions and a starting point ==\n");+ for (round = 0; round < 200; round++) {+ uint8_t k1[32], k2[32], in[16 * 11], a[16 * 11], b[16 * 11], du[16];+ uint8_t len = klens[round % 3];+ uint32_t nb = 1 + (round % 11);+ uint32_t spoint = round % 4;+ aes_key p1, g1, p2, g2;+ aes_block d1;++ rnd_fill(k1, len);+ rnd_fill(k2, len);+ rnd_fill(in, nb * 16);+ rnd_fill(du, 16);+ keys(&p1, &g1, k1, len);+ keys(&p2, &g2, k2, len);++ memcpy(&d1, du, 16);+ crypton_aes_ppc8_encrypt_xts((aes_block *) a, &p1, &p2, &d1, spoint,+ (aes_block *) in, nb);+ xts_reference(b, &g1, &g2, du, spoint, in, nb, 0);+ /* a[0], not a further block: at this round nb is 1 and anything+ * past the first block is outside what same() is given */+ if (sabotage && round == 11) a[0] ^= 4;+ same("xts encrypt", a, b, nb * 16);++ memcpy(&d1, du, 16);+ crypton_aes_ppc8_decrypt_xts((aes_block *) a, &p1, &p2, &d1, spoint,+ (aes_block *) in, nb);+ xts_reference(b, &g1, &g2, du, spoint, in, nb, 1);+ same("xts decrypt", a, b, nb * 16);+ }++ /*+ * GHASH, where the two arguments do not take the same convention and+ * ppc8.c has to swap one of them. Two different H and a non-zero+ * accumulator, because a symmetric H or a zero start can hide a wrong+ * convention.+ */+ printf("== GHASH ==\n");+ for (round = 0; round < 300; round++) {+ uint8_t h[16], data[64], start[16];+ table_4bit tp, tg;+ block128 ap, ag;++ rnd_fill(h, 16);+ rnd_fill(data, sizeof data);+ rnd_fill(start, 16);++ crypton_aes_ppc8_hinit(tp, (const block128 *) h);+ crypton_aes_generic_hinit(tg, (const block128 *) h);++ memcpy(&ap, start, 16); memcpy(&ag, start, 16);+ crypton_aes_ppc8_gf_mul4(&ap, (const block128 *) data, tp);+ crypton_aes_generic_gf_mul4(&ag, (const block128 *) data, tg);+ if (sabotage && round == 5) ((uint8_t *) &ap)[0] ^= 8;+ same("gf_mul4", (const uint8_t *) &ap, (const uint8_t *) &ag, 16);++ memcpy(&ap, start, 16); memcpy(&ag, start, 16);+ crypton_aes_ppc8_gf_mul(&ap, tp);+ crypton_aes_generic_gf_mul(&ag, tg);+ same("gf_mul", (const uint8_t *) &ap, (const uint8_t *) &ag, 16);+ }++ printf("%s: %d mismatch(es)%s\n", failures ? "FAIL" : "ok", failures,+ sabotage ? " (sabotage was asked for)" : "");+ return failures != 0;+}
+ cbits/tests/sysdrg/run.sh view
@@ -0,0 +1,22 @@+#!/bin/sh+# Does the generator behind MonadRandom IO survive the things that break a+# generator: a second thread, a reseed, and a fork?+#+# None of this can be asked from Haskell alone. A forkIO thread is not an+# operating system thread, and the Haskell test suite cannot fork the way a+# child process must for the question to mean anything.+#+# Usage: cbits/tests/sysdrg/run.sh [build-dir]+set -eu++cbits=$(cd "$(dirname "$0")/../.." && pwd)+out=${1:-$(mktemp -d)}++${CC:-cc} -O2 -Wall -Wextra -DCRYPTON_SYSDRG_TESTING -I"$cbits" -o "$out/sysdrg" \+ "$cbits/tests/sysdrg/sysdrg.c" \+ "$cbits/crypton_sysdrg.c" \+ "$cbits/crypton_sysrandom.c" \+ "$cbits/crypton_chacha.c" \+ "$cbits/crypton_sha512.c"++"$out/sysdrg"
+ cbits/tests/sysdrg/sysdrg.c view
@@ -0,0 +1,140 @@+#include <stdio.h>+#include <stdint.h>+#include <string.h>+#include <stdlib.h>+#include <unistd.h>+#include <pthread.h>+#include <sys/wait.h>++void crypton_sysdrg_test_key(uint8_t out[32]);+void crypton_sysdrg_test_lock(void);+void crypton_sysdrg_test_unlock(void);+int crypton_sysdrg_bytes(uint8_t *out, int len);+uint32_t crypton_sysdrg_generation(void);+uint64_t crypton_sysdrg_thread_used(void);+static uint64_t used_in_thread;++static int fail = 0;+static void check(const char *what, int ok) {+ printf("%-52s %s\n", what, ok ? "ok" : "FAIL");+ if (!ok) fail = 1;+}++/* Holds the process generator's lock for a moment, so that a fork can be+ * made to happen while it is held. Releasing after a delay rather than on+ * demand is deliberate: a prepare handler has to be able to take the lock,+ * and a holder that waited to be told would stop the fork instead of the+ * child. */+static pthread_mutex_t gate = PTHREAD_MUTEX_INITIALIZER;+static pthread_cond_t gate_cv = PTHREAD_COND_INITIALIZER;+static int holding = 0;++static void *lock_holder(void *arg) {+ (void) arg;+ crypton_sysdrg_test_lock();+ pthread_mutex_lock(&gate);+ holding = 1;+ pthread_cond_broadcast(&gate_cv);+ pthread_mutex_unlock(&gate);+ usleep(200000);+ crypton_sysdrg_test_unlock();+ return NULL;+}++static void *thread_body(void *arg) {+ uint8_t *out = (uint8_t *) arg;+ check("a second OS thread gets bytes", crypton_sysdrg_bytes(out, 32) == 32);+ used_in_thread = crypton_sysdrg_thread_used();+ return NULL;+}++int main(void) {+ uint8_t a[32], b[32], big[4096];+ memset(a, 0, 32); memset(b, 0, 32);++ check("asking for 32 bytes gives 32", crypton_sysdrg_bytes(a, 32) == 32);+ check("asking again gives something else",+ crypton_sysdrg_bytes(b, 32) == 32 && memcmp(a, b, 32) != 0);+ check("a 4096-byte request is filled", crypton_sysdrg_bytes(big, 4096) == 4096);+ check("zero length is fine", crypton_sysdrg_bytes(a, 0) == 0);++ /* two OS threads must not share a stream. The main thread has already+ * produced over 4 KiB by here, so a shared generator would show it. */+ uint8_t t1[32], t2[32];+ pthread_t p1, p2;+ pthread_create(&p1, NULL, thread_body, t1);+ pthread_join(p1, NULL);+ pthread_create(&p2, NULL, thread_body, t2);+ pthread_join(p2, NULL);+ check("two OS threads do not produce the same bytes", memcmp(t1, t2, 32) != 0);+ check("a new OS thread starts its own stream, not the main one's",+ used_in_thread == 32);++ /* crossing the per-thread reseed limit (1 MiB) must not repeat */+ uint8_t before[32], after[32];+ crypton_sysdrg_bytes(before, 32);+ for (int i = 0; i < 1100; i++) { uint8_t junk[1024]; crypton_sysdrg_bytes(junk, 1024); }+ crypton_sysdrg_bytes(after, 32);+ check("output after a reseed differs from before", memcmp(before, after, 32) != 0);++ /* fork: parent and child must diverge */+ int fds[2];+ if (pipe(fds) != 0) { perror("pipe"); return 2; }+ uint8_t parent[32], child[32];+ pid_t pid = fork();+ if (pid == 0) {+ close(fds[0]);+ crypton_sysdrg_bytes(child, 32);+ ssize_t w = write(fds[1], child, 32);+ _exit(w == 32 ? 0 : 1);+ }+ close(fds[1]);+ crypton_sysdrg_bytes(parent, 32);+ ssize_t r = read(fds[0], child, 32);+ int status = 0; waitpid(pid, &status, 0);+ check("the child was able to produce bytes", r == 32 && status == 0);+ check("parent and child do not produce the same bytes", memcmp(parent, child, 32) != 0);+ check("the fork was noticed", crypton_sysdrg_generation() == 0);++ /* Backtracking resistance. The key that produced a draw is replaced+ * once the bytes are out, so a state read afterwards is not the state+ * that made them and cannot be wound back to remake them. Comparing+ * the key before and against the key after is the whole of it: without+ * the rekey it is the same key and the counter alone says where to+ * start. */+ uint8_t key_before[32], key_after[32], drawn[32];+ memset(key_before, 0, 32); memset(key_after, 0, 32);+ crypton_sysdrg_test_key(key_before);+ crypton_sysdrg_bytes(drawn, 32);+ crypton_sysdrg_test_key(key_after);+ check("a draw replaces the key that made it",+ memcmp(key_before, key_after, 32) != 0);++ /* A fork while another thread holds the process generator's lock. The+ * child's first draw has to reseed, the generation having changed, and+ * reseeding takes that lock. With no prepare and parent handlers the+ * child inherits it locked and waits on it for ever -- the thread that+ * would release it did not come across the fork. The alarm is what+ * turns that into a failure rather than a hung test run. */+ pthread_t holder;+ if (pthread_create(&holder, NULL, lock_holder, NULL) != 0) {+ perror("pthread_create"); return 2;+ }+ pthread_mutex_lock(&gate);+ while (!holding) pthread_cond_wait(&gate_cv, &gate);+ pthread_mutex_unlock(&gate);++ pid_t locked_pid = fork();+ if (locked_pid == 0) {+ uint8_t c[32];+ alarm(5);+ _exit(crypton_sysdrg_bytes(c, 32) == 32 ? 0 : 1);+ }+ pthread_join(holder, NULL);+ int locked_status = 0;+ waitpid(locked_pid, &locked_status, 0);+ check("a child forked while the lock was held can draw",+ WIFEXITED(locked_status) && WEXITSTATUS(locked_status) == 0);++ return fail;+}
crypton.cabal view
@@ -1,10 +1,11 @@ cabal-version: 3.0 name: crypton-version: 2.1.10+version: 2.2.0 -- crypton's own code is BSD-3-Clause. The parts of -- cbits/aes/gcm_fused_x86.c that follow picotls's fusion are MIT, and the--- vendored s2n-bignum assembly in cbits/s2n is taken under ISC; each has--- its licence beside it, and they are listed below. The CRYPTOGAMS+-- vendored s2n-bignum assembly in cbits/s2n is taken under ISC, and the+-- constant-time AES and GHASH in cbits/bearssl are BearSSL's under MIT;+-- each has its licence beside it, and they are listed below. The CRYPTOGAMS -- assembly in cbits/asm is taken under its BSD-3-Clause option, and the -- AArch64 multiply-accumulate loop in cbits/crypton_bignum.h follows Go's, -- which is BSD-3-Clause too; the first term already covers both.@@ -15,6 +16,7 @@ cbits/aes/LICENSE.fusion cbits/asm/LICENSE.cryptogams cbits/s2n/LICENSE+ cbits/bearssl/LICENSE cbits/mlkem/LICENSE cbits/mldsa/LICENSE copyright:@@ -46,6 +48,9 @@ cbits/asm/README.md cbits/asm/aesni-gcm-x86_64.pl cbits/asm/arm-xlate.pl+ cbits/asm/aesp8-ppc.pl+ cbits/asm/ghashp8-ppc.pl+ cbits/asm/ppc-xlate.pl cbits/asm/arm_arch.h cbits/asm/chacha-armv8.pl cbits/asm/chacha-x86_64.pl@@ -146,6 +151,11 @@ cbits/s2n/import.sh cbits/s2n/include/*.h cbits/s2n/x86_att/*.S+ cbits/bearssl/README.md+ cbits/bearssl/VERSION+ cbits/bearssl/import.sh+ cbits/bearssl/inner.h+ cbits/tests/*.c cbits/tests/ct/*.c cbits/tests/ct/*.h cbits/tests/ct/known.txt@@ -167,9 +177,13 @@ cbits/tests/scrub/README cbits/tests/scrub/known.txt cbits/tests/scrub/run.sh+ cbits/tests/sysdrg/*.c+ cbits/tests/sysdrg/run.sh cbits/tests/width/*.c cbits/tests/width/run.sh tests/*.hs+ tests/tutorial/extract.awk+ tests/tutorial/run.sh extra-doc-files: CHANGELOG.md@@ -185,12 +199,6 @@ manual: True -flag support_rdrand- description:- allow compilation with RDRAND on system and architecture that supports it-- manual: True- flag support_pclmuldq description: Allow compilation with pclmuldq on architecture that supports it@@ -355,6 +363,17 @@ cbits/crypton_chacha.c cbits/crypton_chachapoly.c cbits/crypton_cpu.c+ cbits/crypton_sysdrg.c+ cbits/crypton_sysrandom.c+ -- BearSSL's constant-time AES and GHASH, which cbits/aes/generic.c+ -- and cbits/aes/gf.c are written on top of. Unconditional because+ -- those two are: every one of the three AES branches below names+ -- them, accelerated or not.+ cbits/bearssl/aes_ct64.c+ cbits/bearssl/aes_ct64_dec.c+ cbits/bearssl/aes_ct64_enc.c+ cbits/bearssl/dec32le.c+ cbits/bearssl/ghash_ctmul64.c cbits/crypton_des.c cbits/crypton_ecc.c cbits/crypton_f2m.c@@ -441,6 +460,7 @@ Crypto.Random.Entropy.Source Crypto.Random.HmacDRG Crypto.Random.Probabilistic+ Crypto.Random.SysDRG Crypto.Random.SystemDRG default-language: Haskell2010@@ -723,11 +743,6 @@ cc-options: -mavx2 -mbmi2 asm-options: -mavx2 -mbmi2 - if ((flag(support_rdrand) && (arch(i386) || arch(x86_64))) && !os(windows))- cpp-options: -DSUPPORT_RDRAND- c-sources: cbits/crypton_rdrand.c- other-modules: Crypto.Random.Entropy.RDRand- if (flag(support_aesni) && arch(aarch64)) cc-options: -DWITH_ARMV8_CRYPTO c-sources:@@ -739,6 +754,52 @@ if !flag(use_target_attributes) cc-options: -march=armv8-a+crypto + -- The same instructions, reached from AArch32. cbits/aes/armv8.c needs+ -- nothing changed to compile here: every intrinsic it uses exists in both+ -- execution states, including the PMULL2 it takes through+ -- vmull_high_p64, which the compiler lowers to a pair of VMULL.P64. What+ -- differs is how the operating system reports the instructions --+ -- AArch32 has filled AT_HWCAP with older features and puts these in+ -- AT_HWCAP2 -- which cbits/crypton_cpu.c now asks for.+ --+ -- -mfpu is passed whatever use_target_attributes says, unlike the branch+ -- above. AArch32 names an FPU rather than an architecture extension and+ -- the two compilers spell the function attribute for it differently; a+ -- global option is safe here, since a compiler emits these instructions+ -- where an intrinsic asks for them and nowhere else.+ --+ -- A soft-float ARM that cannot take -mfpu fails loudly at the first C+ -- file rather than quietly, and -f-support_aesni is the way out.+ if (flag(support_aesni) && arch(arm))+ cc-options: -DWITH_ARMV8_CRYPTO -mfpu=crypto-neon-fp-armv8+ c-sources:+ cbits/aes/generic.c+ cbits/aes/gf.c+ cbits/aes/armv8.c+ cbits/crypton_aes.c++ -- POWER8's vector AES and vector carry-less multiply, through the+ -- CRYPTOGAMS assembly in cbits/asm. No cc-options beyond the define:+ -- cbits/aes/ppc8.c calls that assembly rather than using intrinsics of+ -- its own, and the assembly says .machine "any", so the assembler takes+ -- the instructions without being told about the processor.+ --+ -- Little-endian only. The generator emits a big-endian flavour from the+ -- same source, and it is not checked in: nothing here has been able to+ -- run it, and an AES path that has never been executed is not one to+ -- ship. cbits/asm/generate.sh says what adding it would take.+ if (flag(support_aesni) && arch(ppc64le))+ cc-options: -DWITH_PPC8_CRYPTO+ c-sources:+ cbits/aes/generic.c+ cbits/aes/gf.c+ cbits/aes/ppc8.c+ cbits/crypton_aes.c++ asm-sources:+ cbits/asm/aesp8-ppc-linux64le.S+ cbits/asm/ghashp8-ppc-linux64le.S+ if arch(aarch64) cc-options: -DWITH_ARMV8_SHA1 -DWITH_ARMV8_SHA2 -DWITH_ARMV8_SHA3@@ -864,7 +925,7 @@ -- branch had already named every file it names. Cabal drops the repeats, -- so nothing was built twice, but the line read as the fallback for a -- platform with no AES instructions and was not one.- if !((flag(support_aesni) && arch(aarch64)) || (flag(support_aesni) && (arch(i386) || arch(x86_64))))+ if !((flag(support_aesni) && (arch(aarch64) || arch(arm))) || (flag(support_aesni) && (arch(i386) || arch(x86_64))) || (flag(support_aesni) && arch(ppc64le))) c-sources: cbits/aes/generic.c cbits/aes/gf.c@@ -901,7 +962,9 @@ build-depends: Win32 <2.15 else- other-modules: Crypto.Random.Entropy.Unix+ other-modules:+ Crypto.Random.Entropy.SysRandom+ Crypto.Random.Entropy.Unix if (impl(ghc >=0) && flag(integer-gmp)) build-depends: integer-gmp <1.2@@ -986,7 +1049,9 @@ PubKey.RabinSpec PubKey.SecrecySpec PubKey.RSASpec+ EntropySpec RuntimeSpec+ SysDRGSpec StreamCipher.ChaChaPoly1305Spec StreamCipher.ChaChaSpec StreamCipher.RC4Spec@@ -1007,6 +1072,45 @@ default-language: Haskell2010 ghc-options: -Wall -fno-warn-orphans -fno-warn-missing-signatures -rtsopts+ -threaded "-with-rtsopts=-N2"++-- The generator behind MonadRandom and a fork made from Haskell. Its own+-- executables because the suite above runs with -N2, where GHC does not+-- support forkProcess, and because the answer may differ between the+-- threaded runtime and the one without it -- so it is built for both.+-- Not on Windows, which has no fork and no unix package.+test-suite test-forkprocess-threaded+ type: exitcode-stdio-1.0+ main-is: ForkProcess.hs+ hs-source-dirs: tests/forkprocess+ default-language: Haskell2010+ ghc-options:+ -Wall -rtsopts -threaded "-with-rtsopts=-N1"++ if os(windows)+ buildable: False+ else+ build-depends:+ base >=4.13 && <5,+ bytestring,+ crypton,+ unix++test-suite test-forkprocess-unthreaded+ type: exitcode-stdio-1.0+ main-is: ForkProcess.hs+ hs-source-dirs: tests/forkprocess+ default-language: Haskell2010+ ghc-options: -Wall -rtsopts++ if os(windows)+ buildable: False+ else+ build-depends:+ base >=4.13 && <5,+ bytestring,+ crypton,+ unix benchmark bench-crypton type: exitcode-stdio-1.0
+ tests/EntropySpec.hs view
@@ -0,0 +1,58 @@+-- | The system source of entropy.+--+-- There is nothing here to check the bytes against -- they are supposed to+-- be unpredictable, and a test that said otherwise would be a test of the+-- kernel. What can be checked is that the right number of them comes back,+-- which is where an implementation of the call with a limit per request+-- goes wrong: @getentropy(3)@ refuses more than 256 bytes at a time, so a+-- backend that forgets to loop answers a short buffer for anything larger.+module EntropySpec (spec) where++import Control.Exception (try)+import qualified Data.ByteString as BS+import Data.Maybe (catMaybes)+import Foreign.Marshal.Alloc (allocaBytes)+import Test.Hspec++import Crypto.Random.Entropy (EntropyError (..), getEntropy)+import Crypto.Random.Entropy.Unsafe (gatherBackend, replenish, supportedBackends)++spec :: Spec+spec = describe "the system entropy source" $ do+ mapM_ lengthCase [0, 1, 31, 32, 255, 256, 257, 512, 1000]++ it "does not answer the same thing twice" $ do+ a <- getEntropy 64 :: IO BS.ByteString+ b <- getEntropy 64 :: IO BS.ByteString+ a `shouldNotBe` b++ it "is not answering a constant" $ do+ b <- getEntropy 1024 :: IO BS.ByteString+ BS.length (BS.filter (== BS.head b) b) `shouldSatisfy` (< 64)++ -- getEntropy would not notice: replenish tops a short answer up from+ -- the next backend, so the buffer comes back full either way and the+ -- system call is quietly replaced by the device file. The backend has+ -- to be asked on its own.+ it "the best backend fills a buffer larger than one request on its own" $ do+ bs <- catMaybes `fmap` sequence supportedBackends+ case bs of+ [] -> expectationFailure "no source of entropy on this system"+ (b : _) -> do+ n <- allocaBytes 1000 $ \ptr -> gatherBackend b ptr 1000+ n `shouldBe` 1000++ -- A system with no source of entropy is the one failure this library+ -- cannot answer, and it used to be an `error`, which a caller could not+ -- tell from a bug. There is no way to take the sources away from a+ -- running machine, but replenish can be handed the empty list, which is+ -- the same question.+ it "says so with an exception when there is no source at all" $ do+ r <- allocaBytes 16 $ \ptr -> try (replenish 16 [] ptr)+ r `shouldBe` Left NoEntropySource++lengthCase :: Int -> Spec+lengthCase n =+ it ("gives back the " ++ show n ++ " bytes asked for") $ do+ b <- getEntropy n :: IO BS.ByteString+ BS.length b `shouldBe` n
tests/PubKey/RSASpec.hs view
@@ -1,3 +1,7 @@+-- The properties below hold the new entry points against sign, signSafer+-- and verify, which are deprecated as of this release. Comparing against+-- them is the point, so the warning is off here and nowhere else.+{-# OPTIONS_GHC -Wno-deprecations #-} {-# LANGUAGE ExistentialQuantification #-} {-# LANGUAGE OverloadedStrings #-}
tests/RuntimeSpec.hs view
@@ -1,7 +1,72 @@+-- | What 'processorOptions' says about the machine it is running on.+--+-- A test cannot know what processor it is on, so it cannot check that the+-- answer is right. What it can check is that the answer is well formed:+-- that every name is distinct, that nothing is reported under a name from+-- another architecture, and that the two questions which do not name one+-- agree with the list. The list is printed as well, because a CI log that+-- says what each runner reported is the only way the detection itself gets+-- looked at. module RuntimeSpec (spec) where -import Crypto.System.CPU+import Data.List (nub, sort) import Test.Hspec +import Crypto.System.CPU (+ ProcessorOption (..),+ hasAESAcceleration,+ hasGHASHAcceleration,+ processorOptions,+ )++-- | Every name this module gives, which is also the set 'Show' has to+-- cover.+x86Options :: [ProcessorOption]+x86Options =+ [AESNI, PCLMUL, SSSE3, AVX, AVX2, SHANI, MOVBE, ADX, VAES, VAES512]++armOptions :: [ProcessorOption]+armOptions = [NEON, ARMAES, ARMPMULL, ARMSHA1, ARMSHA2, ARMSHA512]++ppcOptions :: [ProcessorOption]+ppcOptions = [PPCAES, PPCVPMSUM]+ spec :: Spec-spec = it "CPU" $ putStrLn (show processorOptions)+spec = describe "processorOptions" $ do+ it "CPU" $ putStrLn (show processorOptions)++ it "gives every name a number of its own" $+ -- two patterns sharing a number would make one of them unreachable+ -- and the other print under the wrong name+ length (nub (x86Options ++ armOptions ++ ppcOptions))+ `shouldBe` length (x86Options ++ armOptions ++ ppcOptions)++ it "has a name for every option it names" $+ -- Show falls back to "ProcessorOption n" for what it does not know,+ -- which is for values from a newer release, not for these+ filter (startsWith "ProcessorOption " . show) (x86Options ++ armOptions ++ ppcOptions)+ `shouldBe` []++ it "reports each option at most once, in order" $ do+ processorOptions `shouldBe` sort processorOptions+ nub processorOptions `shouldBe` processorOptions++ it "does not mix one architecture's names with another's" $ do+ let reported g = any (`elem` g) processorOptions+ -- a machine reporting two of these groups is the+ -- AArch64-says-AESNI fault this replaced+ length (filter reported [x86Options, armOptions, ppcOptions])+ `shouldSatisfy` (<= 1)++ -- Half a check, and which half depends on the machine: where the+ -- processor has AES this fails if the answer is broken to False and+ -- passes if it is broken to True, and the other way round on a machine+ -- without it. Both were tried.+ it "answers the architecture-free questions from the same list" $ do+ hasAESAcceleration+ `shouldBe` any (`elem` processorOptions) [AESNI, ARMAES, PPCAES]+ hasGHASHAcceleration+ `shouldBe` any (`elem` processorOptions) [PCLMUL, ARMPMULL, PPCVPMSUM]++startsWith :: String -> String -> Bool+startsWith p s = take (length p) s == p
+ tests/SysDRGSpec.hs view
@@ -0,0 +1,100 @@+-- | The generator behind 'MonadRandom' for 'IO'.+--+-- What can be asked from Haskell is narrow. A @forkIO@ thread is not an+-- operating system thread, and this process cannot fork, so the questions+-- that matter most -- does a new operating system thread start its own+-- stream, does a child after @fork@ diverge from its parent -- are in+-- @cbits\/tests\/sysdrg@ instead.+--+-- What is left is still worth asking: the generator keeps a lock and a+-- thread-local slot, and it is reached through a @safe@ foreign call, so+-- many Haskell threads drawing at once is exactly the shape that deadlocks+-- or hands two of them the same bytes. That needs @-threaded@ and more+-- than one capability, which is why the suite has them.+module SysDRGSpec (spec) where++import Control.Concurrent (+ forkIO,+ getNumCapabilities,+ setNumCapabilities,+ yield,+ )+import Control.Concurrent.MVar (newEmptyMVar, putMVar, takeMVar)+import Control.Exception (finally)+import Control.Monad (forM, forM_)+import qualified Data.ByteString as BS+import Data.List (group, nub, sort)+import Test.Hspec++import Crypto.Random (getRandomBytes)++spec :: Spec+spec = describe "the generator behind MonadRandom IO" $ do+ it "runs on more than one capability, or the rest proves little" $ do+ n <- getNumCapabilities+ n `shouldSatisfy` (> 1)++ mapM_ lengthCase [0, 1, 32, 1000, 4096]++ it "gives every thread something different" $ do+ -- plainly, rather than through async, which is not a dependency here+ boxes <- forM [1 .. 256 :: Int] $ \_ -> do+ box <- newEmptyMVar+ _ <- forkIO $ do+ b <- getRandomBytes 32 :: IO BS.ByteString+ putMVar box b+ return box+ bss <- mapM takeMVar boxes+ length (nub bss) `shouldBe` 256++ it "keeps the streams apart while setNumCapabilities changes underneath" $ do+ -- The state is held against the operating system thread rather than+ -- against the capability, which is what makes this safe: a forkIO+ -- thread moves between capabilities, and a capability is served by+ -- different worker threads over its life, so state held against one+ -- would be shared by threads running at the same time. Raising the+ -- count makes the runtime create worker threads, each of which has+ -- to start its own stream; lowering it disables capabilities under+ -- threads that are drawing.+ n0 <- getNumCapabilities+ let flips = concat (replicate 3 [1, 2, min 8 (n0 * 2), n0])+ threads = 64 :: Int+ draws = 16 :: Int+ bss <-+ ( do+ flipped <- newEmptyMVar+ _ <- forkIO $ do+ forM_ flips $ \n -> setNumCapabilities n >> yield+ putMVar flipped ()+ boxes <- forM [1 .. threads] $ \_ -> do+ box <- newEmptyMVar+ _ <- forkIO $ do+ bs <- forM [1 .. draws] $ \_ -> do+ b <- getRandomBytes 32 :: IO BS.ByteString+ yield+ return b+ putMVar box bs+ return box+ bss <- concat <$> mapM takeMVar boxes+ takeMVar flipped+ return bss+ )+ `finally` setNumCapabilities n0+ -- sorted and grouped rather than nub, which is quadratic and this+ -- list is long enough for that to show+ length bss `shouldBe` threads * draws+ length (group (sort bss)) `shouldBe` threads * draws++ it "does not repeat across a reseed" $ do+ -- the per-thread generator reseeds after a mebibyte+ first <- getRandomBytes 32 :: IO BS.ByteString+ forM_ [1 .. 300 :: Int] $ \_ ->+ (getRandomBytes 4096 :: IO BS.ByteString) >>= \b -> b `seq` return ()+ second <- getRandomBytes 32 :: IO BS.ByteString+ second `shouldNotBe` first++lengthCase :: Int -> Spec+lengthCase n =+ it ("gives back the " ++ show n ++ " bytes asked for") $ do+ b <- getRandomBytes n :: IO BS.ByteString+ BS.length b `shouldBe` n
+ tests/forkprocess/ForkProcess.hs view
@@ -0,0 +1,81 @@+-- | Does the generator behind 'MonadRandom' notice a fork made from+-- Haskell?+--+-- @cbits\/tests\/sysdrg@ asks the same question of @fork(2)@ called from C,+-- in a process with no runtime system in it at all. That leaves the case+-- anyone actually meets untested: 'forkProcess', with the runtime's own+-- threads about. The handler is registered with @pthread_atfork@, which+-- libc runs for every @fork(2)@ whoever calls it, so it should reach this+-- too -- should, which is why this is a test and not a comment.+--+-- It cannot live in the main suite. That one runs with @-N2@, where GHC+-- says 'forkProcess' is not supported; this needs its own runtime options,+-- and is built twice, once threaded with one capability and once not+-- threaded at all. Both are configurations GHC supports 'forkProcess' in.+--+-- The check is that parent and child disagree. Without fork detection the+-- child carries on the parent's stream, so the next block each of them+-- draws is the same block -- they would agree exactly, which is the fault.+module Main (main) where++import Control.Concurrent (getNumCapabilities, rtsSupportsBoundThreads)+import Control.Monad (when)+import qualified Data.ByteString as B+import System.Exit (exitFailure)+import System.IO (hClose, hFlush, hPutStrLn, stderr, stdout)+import System.Posix.IO (closeFd, createPipe, fdToHandle)+import System.Posix.Process (ProcessStatus (..), forkProcess, getProcessStatus)++import Crypto.Random (getRandomBytes)++draw :: IO B.ByteString+draw = getRandomBytes 32++main :: IO ()+main = do+ caps <- getNumCapabilities+ putStrLn $+ "threaded: "+ ++ show rtsSupportsBoundThreads+ ++ ", capabilities: "+ ++ show caps+ -- GHC supports forkProcess with -threaded only while one capability is+ -- in use. Saying so here means a change to the runtime options shows+ -- up as a failure rather than as a test that quietly means nothing.+ when (rtsSupportsBoundThreads && caps /= 1) $+ die "this test needs one capability when threaded"++ -- Before the fork, or the child inherits whatever is still in the+ -- buffer and writes it out again when it exits.+ hFlush stdout++ -- Draw once first, so that both sides inherit a generator that has been+ -- seeded. A child of an unseeded one would seed itself for the first+ -- time and differ for that reason instead of this one.+ _ <- draw++ (readEnd, writeEnd) <- createPipe+ pid <- forkProcess $ do+ closeFd readEnd+ b <- draw+ h <- fdToHandle writeEnd+ B.hPut h b+ hClose h+ closeFd writeEnd+ hr <- fdToHandle readEnd+ fromChild <- B.hGet hr 32+ hClose hr+ status <- getProcessStatus True False pid++ fromParent <- draw++ case status of+ Just (Exited _) -> return ()+ other -> die ("the child did not exit cleanly: " ++ show other)+ when (B.length fromChild /= 32) $+ die ("the child sent " ++ show (B.length fromChild) ++ " bytes, not 32")+ when (fromChild == fromParent) $+ die "parent and child drew the same bytes: the fork went unnoticed"+ putStrLn "parent and child drew different bytes"+ where+ die msg = hPutStrLn stderr ("FAIL: " ++ msg) >> exitFailure
+ tests/tutorial/extract.awk view
@@ -0,0 +1,53 @@+# Pull every code block out of a Haddock module and write each one as a+# compilable Haskell module.+#+# A block is a maximal run of lines beginning "-- >", which is what Haddock+# renders as a code sample. The run is ended by any line that is not one,+# so two samples separated by a line of prose are two modules and cannot+# refer to each other -- which is the point: each sample has to stand on its+# own, because that is how a reader will copy it.+#+# The name comes from the "-- $section" marker the block sits under, so a+# compiler message names the section of the tutorial that is wrong rather+# than a number. The module header goes after any LANGUAGE pragmas, since+# those have to come first.+#+# Writes one file per block into the directory given as -v out=, and prints+# each module name on stdout.++/^-- \$[A-Za-z_][A-Za-z0-9_]*$/ {+ section = substr($0, 5)+ nth = 0+ next+}++/^-- >/ {+ if (!inblock) {+ inblock = 1+ nlines = 0+ nth+++ mod = "Example_" section "_" nth+ file = out "/" mod ".hs"+ }+ # "-- > foo" carries a space that is not part of the code; "-- >" alone+ # is a blank line inside the block.+ line = (length($0) > 5) ? substr($0, 6) : ""+ lines[++nlines] = line+ next+}++{ if (inblock) flush() }++END { if (inblock) flush() }++function flush( i, k) {+ # Pragmas, then the header, then the rest.+ k = 1+ while (k <= nlines && (lines[k] ~ /^\{-#/ || lines[k] == "")) k+++ for (i = 1; i < k; i++) print lines[i] > file+ print "module " mod " where" > file+ for (i = k; i <= nlines; i++) print lines[i] > file+ close(file)+ print mod+ inblock = 0+}
+ tests/tutorial/run.sh view
@@ -0,0 +1,126 @@+#!/bin/sh+# Does the tutorial still compile?+#+# Crypto/Tutorial.hs is the one place in the package whose code the compiler+# never sees: every example in it is a Haddock block, so an API change+# silently leaves it wrong and the next reader copies something that does+# not build. That has happened -- taking a checked key in Poly1305 made the+# tutorial's crypto_box stop type-checking, and it was noticed by reading+# the module rather than by anything here.+#+# So the blocks are pulled out and type-checked. Each one becomes a module+# of its own, because each one is a thing a reader copies whole; a block+# that needs a definition from the block above it will fail here, which is+# the right answer.+#+# Checked against this tree rather than against whatever crypton is+# installed: cabal repl has the project's library as its home package, so a+# tutorial written for an API this working tree has not got cannot pass by+# finding the API in a released version on the machine. -fno-code because+# nothing is run -- the question is only whether it compiles.+#+# -Wall as well, and a warning in a tutorial counts as a failure: an unused+# import is three words a reader will paste into their own file. Only the+# extracted modules are held to it, not the library loaded beside them.+#+# $GHC names a compiler other than the one on the path, and $BUILDDIR a+# build directory to keep that compiler's products out of the usual one.+# CI has one compiler per job and sets neither; locally they are how an+# example is asked whether it also builds on the oldest GHC the package+# supports, which is the reader this is most likely to have failed.+#+# The first module is wrong on purpose and has to be reported, or the run+# proves nothing: a repl that failed to start, a ghci whose message format+# changed, or an extractor that produced no modules would otherwise all look+# like a tutorial that compiles.+set -eu++cd "$(dirname "$0")/../.."+tutorial=Crypto/Tutorial.hs++work=$(mktemp -d)+trap 'rm -rf "$work"' EXIT INT TERM++modules=$(awk -v out="$work" -f tests/tutorial/extract.awk "$tutorial")+count=$(printf '%s\n' "$modules" | grep -c . || true)+if [ "$count" -eq 0 ]; then+ echo "FAIL no code blocks found in $tutorial"+ exit 1+fi+echo "$count code blocks in $tutorial"++# The deliberate one. hashWith returns a Digest, and has since the module+# was written, so this is the shape of every break this harness is for: the+# tutorial says something the library no longer agrees with.+cat > "$work/Calibration.hs" <<'HS'+module Calibration where+import Crypto.Hash (SHA1 (..), hashWith)+thisIsNotADigest :: Int+thisIsNotADigest = hashWith SHA1 "the harness has to report this"+HS++{+ echo ':!echo @@ Calibration'+ echo ":load $work/Calibration.hs"+ for m in $modules; do+ echo ":!echo @@ $m"+ echo ":load $work/$m.hs"+ done+ echo ':quit'+} > "$work/script"++# Not cabal's -v0: that reaches ghci too and takes the "Ok, N modules+# loaded." line with it, which is what is read below.+cabal repl crypton ${GHC:+-w "$GHC"} ${BUILDDIR:+--builddir="$BUILDDIR"} \+ --repl-options=-i"$work" --repl-options=-fno-code \+ --repl-options=-Wall --repl-options=-Wno-missing-home-modules \+ < "$work/script" > "$work/log" 2>&1 || true++awk -v expected="$count" '+ # ghci writes its prompt before the echo, so the marker is at the end+ # of the line rather than the start of it.+ /@@ [A-Za-z_]/ { mod = $NF; verdict[mod] = "no answer"; order[++n] = mod; next }+ mod == "" { next }+ /^Ok, [0-9]+ modules? loaded\.$/ { verdict[mod] = "ok"; next }+ /^Failed, [0-9]+ modules? loaded\.$/ { verdict[mod] = "failed"; next }+ # Only the extracted modules are held to -Wall; the library is loaded+ # beside them and is not what this is asking about.+ /Example_[A-Za-z0-9_]*\.hs:[0-9]+:[0-9]+: warning:/ { warned[mod] = 1 }+ # Everything ghci said about a module before it gave its verdict, minus+ # the progress lines, which are the bulk of it and say nothing.+ /^(ghci> )?\[ *[0-9]+ of [0-9]+\] Compiling/ { next }+ { if (verdict[mod] == "no answer" && $0 != "") detail[mod] = detail[mod] $0 "\n" }+ END {+ bad = 0+ for (i = 1; i <= n; i++) {+ m = order[i]+ want = (m == "Calibration") ? "failed" : "ok"+ got = verdict[m]+ if (got == want && want == "ok" && warned[m]) got = "warned"+ if (got == want) {+ printf "ok %s%s\n", m, (m == "Calibration" ? " (reported, as it must be)" : "")+ } else {+ bad+++ printf "FAIL %s: expected %s, got %s\n", m, want, got+ printf "%s", detail[m]+ }+ }+ # Nothing answered, or not everything did: the run itself did not+ # happen the way this script assumes, so the log is worth seeing.+ if (n != expected + 1) {+ printf "FAIL %d modules answered, %d expected\n", n - 1, expected+ exit 2+ }+ exit (bad > 0)+ }+' "$work/log" || {+ status=$?+ if [ "$status" -eq 2 ]; then+ echo+ echo "--- the last of the log ---"+ tail -60 "$work/log"+ fi+ exit 1+}++echo "every example in $tutorial compiles"