diff --git a/CHANGELOG.md b/CHANGELOG.md
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -1,5 +1,47 @@
 # CHANGELOG for crypton
 
+## 2.2.0
+
+`ProcessorOption` is no longer a sum of constructors.  It is a number with
+pattern synonyms for the names it knows, so it has `Ord` now and has lost
+`Enum` and `Data`, and a `case` over the names needs a catch-all.  `RDRAND`
+is gone from it, and with it the `support_rdrand` flag.  An AArch64 machine
+reports `ARMAES` and `ARMPMULL` where it used to report `AESNI` and `PCLMUL`.
+
+Two more things change under a caller without the type checker saying so:
+`MonadRandom IO` draws from a per-thread generator rather than reading the
+system on every call, and a missing or failing entropy source raises
+`EntropyError` rather than an `ErrorCall` or an `IOException`.
+
+* feat: take entropy from the kernel without a file descriptor
+  [#300](https://github.com/kazu-yamamoto/crypton/pull/300)
+* feat: a ChaCha20 generator per operating system thread
+  [#300](https://github.com/kazu-yamamoto/crypton/pull/300)
+* feat: MonadRandom IO goes through the per-thread generator
+  [#300](https://github.com/kazu-yamamoto/crypton/pull/300)
+* fix: say that the system gave no entropy, rather than error
+  [#300](https://github.com/kazu-yamamoto/crypton/pull/300)
+* fix: stop using RDRAND at all, since it cannot add anything
+  [#300](https://github.com/kazu-yamamoto/crypton/pull/300)
+* feat: name every processor feature crypton dispatches on, not three of them
+  [#302](https://github.com/kazu-yamamoto/crypton/pull/302)
+* fix: an AArch64 machine no longer reports AESNI and PCLMUL
+  [#302](https://github.com/kazu-yamamoto/crypton/pull/302)
+* chore(rsa): deprecate `sign`, `signSafer` and `verify` in favour of them
+  [#305](https://github.com/kazu-yamamoto/crypton/pull/305)
+* fix: compute the portable AES and GHASH without a table
+  [#308](https://github.com/kazu-yamamoto/crypton/pull/308)
+* feat: use the AES instructions on 32-bit ARM
+  [#309](https://github.com/kazu-yamamoto/crypton/pull/309)
+* test: skip the accelerated constant-time driver where the processor has no instructions
+  [#310](https://github.com/kazu-yamamoto/crypton/pull/310)
+* feat: use the POWER8 AES and GHASH instructions on ppc64le
+  [#311](https://github.com/kazu-yamamoto/crypton/pull/311)
+* test: the big-endian harness now asks about AES
+  [#312](https://github.com/kazu-yamamoto/crypton/pull/312)
+* doc(tutorial): compile the examples, and teach the things people come for
+  [#313](https://github.com/kazu-yamamoto/crypton/pull/313)
+
 ## 2.1.10
 
 * build: ask the processor for AES-NI on every x86 system, not four of them
diff --git a/Crypto/PubKey/RSA/PKCS15.hs b/Crypto/PubKey/RSA/PKCS15.hs
--- a/Crypto/PubKey/RSA/PKCS15.hs
+++ b/Crypto/PubKey/RSA/PKCS15.hs
@@ -512,6 +512,12 @@
 
 -- | sign message using private key, a hash and its ASN1 description
 --
+-- __Deprecated.__  Use 'signDigest', or 'signDigestInfo' where this would
+-- have been given 'Nothing'.  The @Maybe hashAlg@ argument says both \"hash
+-- it with this\" and \"do not hash it\", and down the first path the value
+-- inside the 'Just' is never read: it is there to fix the type, which
+-- leaves a caller that has only a type with nothing to pass.
+--
 -- The blinder is optional and 'Nothing' is accepted, but see t'Blinder' for
 -- what it covers and when leaving it out is a decision rather than a default.
 -- 'signSafer' generates one for you.
@@ -527,8 +533,14 @@
     -- ^ message to sign
     -> Either Error ByteString
 sign blinder hashDescr pk m = dp blinder pk `fmap` makeSignature hashDescr (private_size pk) m
+{-# DEPRECATED
+    sign
+    "Use signDigest, or signDigestInfo when the message is already a DigestInfo"
+    #-}
 
 -- | sign message using the private key and by automatically generating a blinder.
+--
+-- __Deprecated.__  Use 'signSaferDigest', or 'signSaferDigestInfo'.
 signSafer
     :: (HashAlgorithmASN1 hashAlg, MonadRandom m)
     => Maybe hashAlg
@@ -541,9 +553,16 @@
 signSafer hashAlg pk m = do
     blinder <- generateBlinder (private_n pk)
     return (sign (Just blinder) hashAlg pk m)
+{-# DEPRECATED
+    signSafer
+    "Use signSaferDigest, or signSaferDigestInfo when the message is already a DigestInfo"
+    #-}
 
 -- | verify message with the signed message
 --
+-- __Deprecated.__  Use 'verifyDigest', or 'verifyDigestInfo' where this
+-- would have been given 'Nothing'.
+--
 -- Following RFC 8017, the signature is rejected unless it is exactly as long
 -- as the modulus (section 8.2.2, step 1) and its integer representative is
 -- below the modulus (section 5.2.2, step 1).  Verification works by
@@ -562,6 +581,10 @@
     -> Bool
 verify hashAlg pk m sm =
     verifyEncoded pk (makeSignature hashAlg (public_size pk) m) sm
+{-# DEPRECATED
+    verify
+    "Use verifyDigest, or verifyDigestInfo when the message is already a DigestInfo"
+    #-}
 
 -- | The two checks of RFC 8017 and the comparison, shared by the three
 -- verification entry points.  The expected encoding is a thunk and is
diff --git a/Crypto/Random.hs b/Crypto/Random.hs
--- a/Crypto/Random.hs
+++ b/Crypto/Random.hs
@@ -7,35 +7,159 @@
 -- Maintainer  : Vincent Hanquez <vincent@snarc.org>
 -- Stability   : stable
 -- Portability : good
+--
+-- Random bytes, drawn either from the system or from a generator whose
+-- seed you hold.
+--
+-- == Drawing from the system
+--
+-- 'getRandomBytes' in 'IO' is the ordinary way to get bytes nobody can
+-- predict.  The length is in bytes and the type is any 'ByteArray', so the
+-- result is usually pinned down by where it goes:
+--
+-- > import Crypto.Random (getRandomBytes)
+-- > import Data.ByteString (ByteString)
+-- >
+-- > nonce <- getRandomBytes 12 :: IO ByteString
+--
+-- Everything in this library that needs randomness takes it the same way,
+-- through a @MonadRandom m =>@ constraint, so running it in 'IO' is the
+-- whole of choosing this source:
+--
+-- > import qualified Crypto.PubKey.RSA as RSA
+-- >
+-- > (pub, priv) <- RSA.generate 256 0x10001
+--
+-- Where the operating system offers @getrandom(2)@ or @getentropy(3)@ the
+-- bytes come from a ChaCha20 generator belonging to the operating system
+-- thread the call runs on, seeded from the system and reseeded as it goes;
+-- see "Crypto.Random.SysDRG" for what that is and what it is not.  Where
+-- it does not -- Windows, for now -- they come from the system on every
+-- call, as they always did.  Either way the caller writes the same line.
+--
+-- Any Haskell thread may draw, and none has to say which generator it
+-- wants:
+--
+-- > import Control.Concurrent (forkIO)
+-- >
+-- > mapM_ (\_ -> forkIO (getRandomBytes 32 >>= use)) [1 .. 64 :: Int]
+--
+-- The generator belongs to the operating system thread that happens to be
+-- carrying the Haskell thread when the call is made, and is found through
+-- that thread's own storage rather than by anything the caller passes.  A
+-- @forkIO@ thread moves between capabilities, so two draws from one Haskell
+-- thread may be answered by two generators; they are seeded independently
+-- of each other, so it makes no difference which one answers.
+--
+-- == Drawing from a generator you hold
+--
+-- A generator of your own is the way to draw many times from one seed.
+-- The system is asked once, when the generator is made, and not again.
+-- That is how @tls@ does a connection: 'seedNew' when it opens, and
+-- everything the handshake needs afterwards from the generator, whose
+-- state the connection carries from one draw to the next.
+--
+-- > import Crypto.Random (ChaChaDRG, drgNewSeed, seedNew, withDRG)
+-- >
+-- > data Connection = Connection { connRNG :: ChaChaDRG }
+-- >
+-- > newConnection :: IO Connection
+-- > newConnection = do
+-- >     seed <- seedNew                   -- the only draw from the system
+-- >     return (Connection (drgNewSeed seed))
+-- >
+-- > -- every later draw advances the connection's own generator
+-- > connRandom :: Int -> Connection -> (ByteString, Connection)
+-- > connRandom n conn =
+-- >     let (bytes, rng') = withDRG (connRNG conn) (getRandomBytes n)
+-- >      in (bytes, conn{connRNG = rng'})
+--
+-- Inside 'withDRG' the same 'getRandomBytes' resolves to the instance for
+-- 'MonadPseudoRandom' rather than the one for 'IO', so it touches neither
+-- the system nor the per-thread generator: it advances the generator it was
+-- given and hands it back.  Keep that generator and the bytes are
+-- reproducible from the seed, which is what makes a test repeatable -- and
+-- what makes a generator unfit for keys unless its seed came from the
+-- system.
+--
+-- 'drgNew' is 'seedNew' and 'drgNewSeed' in one step; 'drgNewTest' takes
+-- four numbers instead of a seed, for a test that must give the same answer
+-- twice.
+--
+-- == Drawing from a monad stacked on IO
+--
+-- There are two instances, one for 'IO' and one for 'MonadPseudoRandom',
+-- and none for the transformers, so a @ReaderT env IO@ or a @StateT s IO@
+-- reaches the system through 'liftIO':
+--
+-- > import Control.Monad.IO.Class (liftIO)
+-- > import Control.Monad.Trans.Reader (ReaderT)
+-- >
+-- > newKey :: ReaderT env IO ByteString
+-- > newKey = liftIO (getRandomBytes 32)
+--
+-- What that draws is what the first section describes, no more and no less:
+-- the 'IO' instance, the generator of the operating system thread the call
+-- lands on, seeded from the system and reseeded as it goes.  'liftIO' says
+-- which monad the draw happens in and nothing about where the bytes come
+-- from.
+--
+-- Writing an instance for the stack instead would make the choice invisible
+-- at the call site, and the class cannot tell a strong source from a weak
+-- one -- see 'MonadRandom'.
+--
+-- == When the system will not give any
+--
+-- Drawing randomness is the one thing here with nothing to fall back on,
+-- so the failure is an exception rather than a value: there is no sensible
+-- 'Maybe' to return and no partial answer worth having.  Since 2.2 it is an
+-- 'EntropyError' and not a @Control.Exception.ErrorCall@ or an
+-- @Control.Exception.IOException@, which is worth knowing if you were
+-- catching one of those:
+--
+-- > import Control.Exception (catch)
+-- > import Crypto.Random (EntropyError (..), getRandomBytes)
+-- >
+-- > key <- getRandomBytes 32 `catch` \e -> case e of
+-- >     NoEntropySource   -> fail "this system offers no randomness at all"
+-- >     EntropyShort w g  -> fail (show w ++ " bytes wanted, " ++ show g ++ " arrived")
+-- >     EntropySourceLost s -> fail ("the source " ++ s ++ " went away")
+--
+-- Catching it at all is a decision rather than a default.  A program that
+-- cannot get randomness cannot make a key, and stopping is usually the
+-- honest thing; the reason to catch is to say so in the program's own
+-- terms rather than in crypton's.
 module Crypto.Random (
-    -- * Deterministic instances
-    ChaChaDRG,
-    SystemDRG,
-    Seed,
+    -- * Drawing from the system
+    MonadRandom (..),
 
-    -- * Seed
+    -- * Drawing from a generator you hold
+    Seed,
     seedNew,
     seedFromInteger,
     seedToInteger,
     seedFromBinary,
-
-    -- * Deterministic Random class
-    getSystemDRG,
-    drgNew,
     drgNewSeed,
+    drgNew,
     drgNewTest,
     withDRG,
     withRandomBytes,
+    MonadPseudoRandom,
     DRG (..),
 
-    -- * Random abstraction
-    MonadRandom (..),
-    MonadPseudoRandom,
+    -- * The generators
+    ChaChaDRG,
+    SystemDRG,
+    getSystemDRG,
+
+    -- * When the system will not give any
+    EntropyError (..),
 ) where
 
 import Crypto.Error
 import Crypto.Internal.Imports
 import Crypto.Random.ChaChaDRG
+import Crypto.Random.Entropy (EntropyError (..))
 import Crypto.Random.SystemDRG
 import Crypto.Random.Types
 import Data.ByteArray (ByteArray, ByteArrayAccess, ScrubbedBytes)
@@ -50,6 +174,9 @@
 import Foreign.Ptr (Ptr, castPtr)
 #endif
 
+-- | The material a deterministic generator is built from.  Two generators
+-- made from one seed produce the same bytes, which is what 'drgNewSeed' is
+-- for and why a seed kept anywhere is as good as the keys drawn from it.
 newtype Seed = Seed ScrubbedBytes
     deriving (ByteArrayAccess)
 
diff --git a/Crypto/Random/Entropy.hs b/Crypto/Random/Entropy.hs
--- a/Crypto/Random/Entropy.hs
+++ b/Crypto/Random/Entropy.hs
@@ -6,16 +6,37 @@
 -- Portability : Good
 module Crypto.Random.Entropy (
     getEntropy,
+    EntropyError (..),
 ) where
 
 import Crypto.Internal.ByteArray (ByteArray)
 import qualified Crypto.Internal.ByteArray as B
-import Data.Maybe (catMaybes)
+import System.IO.Unsafe (unsafeInterleaveIO, unsafePerformIO)
 
 import Crypto.Random.Entropy.Unsafe
 
+-- | The backends this system has, worked out once and no further than
+-- needed.
+--
+-- Opening one is not free: for a device file it means opening and closing
+-- @\/dev\/random@ or @\/dev\/urandom@ just to learn that it is there.  This
+-- used to be done on every call, and for the whole list before any backend
+-- was asked for a byte, so a program paid for both devices even when the
+-- first backend answered everything.
+--
+-- Once, now, and lazily: 'replenish' stops as soon as the buffer is full,
+-- which leaves the tail of this list unforced, so a system where the first
+-- backend answers never opens a device at all.
+{-# NOINLINE openedBackends #-}
+openedBackends :: [EntropyBackend]
+openedBackends = unsafePerformIO (openAsNeeded supportedBackends)
+  where
+    openAsNeeded [] = return []
+    openAsNeeded (o : os) = do
+        m <- o
+        rest <- unsafeInterleaveIO (openAsNeeded os)
+        return $ maybe rest (: rest) m
+
 -- | Get some entropy from the system source of entropy
 getEntropy :: ByteArray byteArray => Int -> IO byteArray
-getEntropy n = do
-    backends <- catMaybes `fmap` sequence supportedBackends
-    B.alloc n (replenish n backends)
+getEntropy n = B.alloc n (replenish n openedBackends)
diff --git a/Crypto/Random/Entropy/Backend.hs b/Crypto/Random/Entropy/Backend.hs
--- a/Crypto/Random/Entropy/Backend.hs
+++ b/Crypto/Random/Entropy/Backend.hs
@@ -11,27 +11,39 @@
     ( EntropyBackend
     , supportedBackends
     , gatherBackend
+    , EntropyError(..)
     ) where
 
 import Foreign.Ptr
 import Data.Proxy
 import Data.Word (Word8)
 import Crypto.Random.Entropy.Source
-#ifdef SUPPORT_RDRAND
-import Crypto.Random.Entropy.RDRand
-#endif
 #ifdef WINDOWS
 import Crypto.Random.Entropy.Windows
 #else
+import Crypto.Random.Entropy.SysRandom
 import Crypto.Random.Entropy.Unix
 #endif
 
--- | All supported backends 
+-- | All supported backends, best first.
+--
+-- The system call comes before everything else: it is the kernel's own
+-- generator, it needs no descriptor, and it is what the rest of the world
+-- reaches for now.  The device files are what is left when the system has
+-- no such call.
+--
+-- RDRAND is deliberately not here, though it used to be first on x86.  A
+-- list like this one is a list of alternatives, and whichever answers
+-- first decides the bytes on its own -- which is the one thing #298 says
+-- RDRAND should not do.  Nor does it contribute anywhere else: every
+-- system this builds for already feeds the instruction into the pool the
+-- call above draws from, so asking it again adds nothing.  See the note in
+-- @cbits\/crypton_sysdrg.c@.
 supportedBackends :: [IO (Maybe EntropyBackend)]
 supportedBackends =
     [
-#ifdef SUPPORT_RDRAND
-    openBackend (Proxy :: Proxy RDRand),
+#ifndef WINDOWS
+    openBackend (Proxy :: Proxy SysRandom),
 #endif
 #ifdef WINDOWS
     openBackend (Proxy :: Proxy WinCryptoAPI)
diff --git a/Crypto/Random/Entropy/RDRand.hs b/Crypto/Random/Entropy/RDRand.hs
deleted file mode 100644
--- a/Crypto/Random/Entropy/RDRand.hs
+++ /dev/null
@@ -1,39 +0,0 @@
-{-# LANGUAGE ForeignFunctionInterface #-}
-
--- |
--- Module      : Crypto.Random.Entropy.RDRand
--- License     : BSD-style
--- Maintainer  : Vincent Hanquez <vincent@snarc.org>
--- Stability   : experimental
--- Portability : Good
-module Crypto.Random.Entropy.RDRand (
-    RDRand,
-) where
-
-import Crypto.Random.Entropy.Source
-import Data.Word (Word8)
-import Foreign.C.Types
-import Foreign.Ptr
-
-foreign import ccall unsafe "crypton_cpu_has_rdrand"
-    c_cpu_has_rdrand :: IO CInt
-
-foreign import ccall unsafe "crypton_get_rand_bytes"
-    c_get_rand_bytes :: Ptr Word8 -> CInt -> IO CInt
-
--- | Fake handle to Intel RDRand entropy CPU instruction
-data RDRand = RDRand
-
-instance EntropySource RDRand where
-    entropyOpen = rdrandGrab
-    entropyGather _ = rdrandGetBytes
-    entropyClose _ = return ()
-
-rdrandGrab :: IO (Maybe RDRand)
-rdrandGrab = supported `fmap` c_cpu_has_rdrand
-  where
-    supported 0 = Nothing
-    supported _ = Just RDRand
-
-rdrandGetBytes :: Ptr Word8 -> Int -> IO Int
-rdrandGetBytes ptr sz = fromIntegral `fmap` c_get_rand_bytes ptr (fromIntegral sz)
diff --git a/Crypto/Random/Entropy/Source.hs b/Crypto/Random/Entropy/Source.hs
--- a/Crypto/Random/Entropy/Source.hs
+++ b/Crypto/Random/Entropy/Source.hs
@@ -6,8 +6,32 @@
 -- Portability : Good
 module Crypto.Random.Entropy.Source where
 
+import Control.Exception (Exception)
 import Data.Word (Word8)
 import Foreign.Ptr
+
+-- | The system would not give the entropy it was asked for.
+--
+-- This is the one failure in the library with nothing to fall back on and
+-- nothing sensible to return: a key drawn from bytes that are not random
+-- is worse than no key.  It used to be reported with 'error' and 'fail',
+-- which left a caller no way to tell it from a bug in the library, and no
+-- way to say anything useful about it.
+data EntropyError
+    = -- | The system offers no source of entropy at all.  On Unix that
+      -- means no @getrandom(2)@, no @getentropy(3)@ and no @\/dev@ to read
+      -- from; a container built from nothing would look like this.
+      NoEntropySource
+    | -- | The sources between them gave fewer bytes than were asked for,
+      -- three times over.  The two numbers are how many were wanted and
+      -- how many arrived.
+      EntropyShort Int Int
+    | -- | A source that could be opened once could not be opened again.
+      -- The string names it.
+      EntropySourceLost String
+    deriving (Show, Eq)
+
+instance Exception EntropyError
 
 -- | A handle to an entropy maker, either a system capability
 -- or a hardware generator.
diff --git a/Crypto/Random/Entropy/SysRandom.hs b/Crypto/Random/Entropy/SysRandom.hs
new file mode 100644
--- /dev/null
+++ b/Crypto/Random/Entropy/SysRandom.hs
@@ -0,0 +1,50 @@
+{-# LANGUAGE ForeignFunctionInterface #-}
+
+-- |
+-- Module      : Crypto.Random.Entropy.SysRandom
+-- License     : BSD-style
+-- Maintainer  : Kazu Yamamoto <kazu@iij.ad.jp>
+-- Stability   : experimental
+-- Portability : Unix
+--
+-- The kernel's own generator, reached without a file descriptor:
+-- @getrandom(2)@ on Linux and FreeBSD, @getentropy(3)@ where that is what
+-- the system has.
+--
+-- This is the source to prefer over reading @\/dev\/urandom@.  It needs no
+-- path and no descriptor, so it still answers where @\/dev@ is not mounted
+-- or not populated, and it cannot be defeated by a full descriptor table.
+module Crypto.Random.Entropy.SysRandom (
+    SysRandom,
+) where
+
+import Crypto.Random.Entropy.Source
+import Data.Word (Word8)
+import Foreign.C.Types
+import Foreign.Ptr
+
+-- Both are @safe@ rather than @unsafe@: at early boot, before the kernel
+-- pool is initialised, these calls block, and an unsafe call that blocks
+-- holds the capability it runs on.
+foreign import ccall safe "crypton_sysrandom_available"
+    c_sysrandom_available :: IO CInt
+
+foreign import ccall safe "crypton_sysrandom_bytes"
+    c_sysrandom_bytes :: Ptr Word8 -> CInt -> IO CInt
+
+-- | The system call, where there is one.
+data SysRandom = SysRandom
+
+instance EntropySource SysRandom where
+    -- Asked at run time and not only at compile time: a binary built where
+    -- the header declares the call can still run on a kernel that answers
+    -- ENOSYS.
+    entropyOpen = available `fmap` c_sysrandom_available
+      where
+        available 0 = Nothing
+        available _ = Just SysRandom
+
+    entropyGather _ ptr n =
+        fromIntegral `fmap` c_sysrandom_bytes ptr (fromIntegral n)
+
+    entropyClose _ = return ()
diff --git a/Crypto/Random/Entropy/Unix.hs b/Crypto/Random/Entropy/Unix.hs
--- a/Crypto/Random/Entropy/Unix.hs
+++ b/Crypto/Random/Entropy/Unix.hs
@@ -61,7 +61,7 @@
 withDev filepath f =
     openDev filepath >>= \h ->
         case h of
-            Nothing -> error ("device " ++ filepath ++ " cannot be grabbed")
+            Nothing -> E.throwIO (EntropySourceLost filepath)
             Just fd -> f fd `E.finally` closeDev fd
 
 closeDev :: H -> IO ()
diff --git a/Crypto/Random/Entropy/Unsafe.hs b/Crypto/Random/Entropy/Unsafe.hs
--- a/Crypto/Random/Entropy/Unsafe.hs
+++ b/Crypto/Random/Entropy/Unsafe.hs
@@ -9,25 +9,27 @@
     module Crypto.Random.Entropy.Backend,
 ) where
 
+import Control.Exception (throwIO)
+
 import Crypto.Random.Entropy.Backend
 import Data.Word (Word8)
 import Foreign.Ptr (Ptr, plusPtr)
 
 -- | Refill the entropy in a buffer
 --
--- Call each entropy backend in turn until the buffer has
--- been replenished.
+-- Call each entropy backend in turn until the buffer has been replenished.
 --
--- If the buffer cannot be refill after 3 loopings, this will raise
--- an User Error exception
+-- Throws 'EntropyError': 'NoEntropySource' when there is no backend at all,
+-- and 'EntropyShort' when three passes over the backends still leave the
+-- buffer unfilled.
 replenish :: Int -> [EntropyBackend] -> Ptr Word8 -> IO ()
-replenish _ [] _ = fail "crypton: random: cannot get any source of entropy on this system"
+replenish _ [] _ = throwIO NoEntropySource
 replenish poolSize backends ptr = loop 0 backends ptr poolSize
   where
     loop :: Int -> [EntropyBackend] -> Ptr Word8 -> Int -> IO ()
     loop _ _ _ 0 = return ()
     loop retry [] p n
-        | retry == 3 = error "crypton: random: cannot fully replenish"
+        | retry == 3 = throwIO $ EntropyShort poolSize (poolSize - n)
         | otherwise = loop (retry + 1) backends p n
     loop retry (b : bs) p n = do
         r <- gatherBackend b p n
diff --git a/Crypto/Random/Entropy/Windows.hs b/Crypto/Random/Entropy/Windows.hs
--- a/Crypto/Random/Entropy/Windows.hs
+++ b/Crypto/Random/Entropy/Windows.hs
@@ -23,6 +23,7 @@
 import Foreign.Storable (peek)
 import System.Win32.Types (getLastError)
 
+import Control.Exception (throwIO)
 import Crypto.Random.Entropy.Source
 
 
@@ -38,7 +39,8 @@
         case mctx of
             Nothing  -> do
                 lastError <- getLastError
-                fail $ "cannot re-grab win crypto api: error " ++ show lastError
+                throwIO $ EntropySourceLost
+                    ("the Windows crypto API: error " ++ show lastError)
             Just ctx -> do
                 r <- cryptGenRandom ctx ptr n
                 cryptReleaseCtx ctx
@@ -100,4 +102,5 @@
         then return ()
         else do
             lastError <- getLastError
-            fail $ "cryptReleaseCtx: error " ++ show lastError
+            throwIO $ EntropySourceLost
+                ("cryptReleaseCtx: error " ++ show lastError)
diff --git a/Crypto/Random/SysDRG.hs b/Crypto/Random/SysDRG.hs
new file mode 100644
--- /dev/null
+++ b/Crypto/Random/SysDRG.hs
@@ -0,0 +1,82 @@
+{-# LANGUAGE ForeignFunctionInterface #-}
+
+-- |
+-- Module      : Crypto.Random.SysDRG
+-- License     : BSD-style
+-- Maintainer  : Kazu Yamamoto <kazu@iij.ad.jp>
+-- Stability   : experimental
+-- Portability : Unix, Windows
+--
+-- The generator behind the 'Crypto.Random.MonadRandom' instance for 'IO'.
+--
+-- A ChaCha20 generator per operating system thread, seeded from a
+-- process-wide generator, seeded in turn from the system entropy pool.
+-- The state and the reseeding live in
+-- @cbits\/crypton_sysdrg.c@, because a @forkIO@ thread is not an operating
+-- system thread -- it moves between capabilities -- so state held against
+-- one would be shared by threads running at the same time.
+--
+-- == What it is, and what it is not
+--
+-- It is not a DRBG of NIST SP 800-90A.  That standard names three --
+-- @Hash_DRBG@, @HMAC_DRBG@ and @CTR_DRBG@ -- and none of them is built on a
+-- stream cipher.  The name here is the older and looser sense of the word.
+--
+-- What it is is the shape @arc4random@ on OpenBSD and @get_random_bytes@ in
+-- the Linux kernel both have, and the one asked for in
+-- <https://github.com/kazu-yamamoto/crypton/issues/298>.  That shape is
+-- common; it is not specified anywhere, so the parts a standard would have
+-- fixed were chosen here instead:
+--
+-- * ChaCha20, rather than AES in counter mode.
+-- * SHA-512 over the system's bytes, rather than a derivation function a
+--   standard would have named.
+-- * A mebibyte per thread, and a mebibyte of issued seed for the process
+--   generator, as the points at which to reseed.  Those numbers are a
+--   choice, not a result.
+--
+-- The pieces underneath are specified: ChaCha20 is RFC 8439, SHA-512 is
+-- FIPS 180-4.  The way they are put together is not.
+--
+-- == What a compromised state gives away
+--
+-- Each draw ends by taking the next forty bytes of keystream as the key and
+-- nonce and dropping the ones that made the output, so a state read after a
+-- draw is not the state that produced it.  ChaCha20 does not run backwards
+-- and the key that would be needed is gone, which is backtracking
+-- resistance in the terms of SP 800-90A.  Without it, a key stands until
+-- the next reseed and anyone holding the state can wind the counter back
+-- over everything issued since -- a mebibyte of output that was meant to be
+-- secret.  It is what @arc4random@ does, and for the same reason.
+--
+-- Prediction resistance is what the reseeding gives.  A reseed draws from
+-- the system again, so a state that has been read does not determine what
+-- comes after one.
+module Crypto.Random.SysDRG (
+    sysDRGBytes,
+) where
+
+import Data.Word (Word8)
+import Foreign.C.Types (CInt (..))
+import Foreign.Ptr (Ptr)
+
+import Crypto.Internal.ByteArray (ByteArray)
+import qualified Crypto.Internal.ByteArray as B
+
+-- Safe, not unsafe: seeding can reach a getrandom(2) that blocks until the
+-- kernel pool is ready, and an unsafe call that blocks holds the capability
+-- it runs on.
+foreign import ccall safe "crypton_sysdrg_bytes"
+    c_sysdrg_bytes :: Ptr Word8 -> CInt -> IO CInt
+
+-- | Draw bytes, or 'Nothing' if the generator cannot be seeded.
+--
+-- It cannot be seeded where the system has no @getrandom(2)@ or
+-- @getentropy(3)@; the caller falls back to
+-- 'Crypto.Random.Entropy.getEntropy' there, which is the path this system
+-- had before the generator existed.
+sysDRGBytes :: ByteArray byteArray => Int -> IO (Maybe byteArray)
+sysDRGBytes n = do
+    (got, out) <- B.allocRet n $ \ptr ->
+        fromIntegral `fmap` c_sysdrg_bytes ptr (fromIntegral n)
+    return $ if got == n then Just out else Nothing
diff --git a/Crypto/Random/Types.hs b/Crypto/Random/Types.hs
--- a/Crypto/Random/Types.hs
+++ b/Crypto/Random/Types.hs
@@ -13,6 +13,7 @@
 
 import Crypto.Internal.ByteArray
 import Crypto.Random.Entropy
+import Crypto.Random.SysDRG (sysDRGBytes)
 
 -- | A monad constraint that allows to generate random bytes
 --
@@ -35,8 +36,13 @@
     -- | Generate N bytes of randomness from a DRG
     randomBytesGenerate :: ByteArray byteArray => Int -> gen -> (byteArray, gen)
 
+-- | Through the generator of 'Crypto.Random.SysDRG': a ChaCha20 generator
+-- per operating system thread, reseeded from the system.  Where that
+-- cannot be seeded -- a system with no @getrandom(2)@ or @getentropy(3)@,
+-- which includes Windows for now -- this is the system entropy source
+-- directly, as it was before the generator existed.
 instance MonadRandom IO where
-    getRandomBytes = getEntropy
+    getRandomBytes n = sysDRGBytes n >>= maybe (getEntropy n) return
 
 -- | A simple Monad class very similar to a State Monad
 -- with the state being a DRG.
diff --git a/Crypto/System/CPU.hs b/Crypto/System/CPU.hs
--- a/Crypto/System/CPU.hs
+++ b/Crypto/System/CPU.hs
@@ -1,6 +1,6 @@
 {-# LANGUAGE CPP #-}
-{-# LANGUAGE DeriveDataTypeable #-}
 {-# LANGUAGE ForeignFunctionInterface #-}
+{-# LANGUAGE PatternSynonyms #-}
 
 -- |
 -- Module      : Crypto.System.CPU
@@ -11,57 +11,201 @@
 --
 -- Gives information about crypton runtime environment.
 module Crypto.System.CPU (
-    ProcessorOption (..),
+    -- The names are bundled with the type rather than listed as
+    -- `pattern' exports, so an importer writes ProcessorOption (..), or
+    -- names the ones it wants, as it would for a type with constructors.
+    ProcessorOption (
+        -- x86
+        AESNI,
+        PCLMUL,
+        SSSE3,
+        AVX,
+        AVX2,
+        SHANI,
+        MOVBE,
+        ADX,
+        VAES,
+        VAES512,
+        -- AArch64
+        NEON,
+        ARMAES,
+        ARMPMULL,
+        ARMSHA1,
+        ARMSHA2,
+        ARMSHA512,
+        -- PowerISA
+        PPCAES,
+        PPCVPMSUM
+    ),
     processorOptions,
-) where
 
-import Data.Data
-import Data.List (findIndices)
-#ifdef SUPPORT_RDRAND
-import Data.Maybe (isJust)
-#endif
-import Data.Word (Word8)
-import Foreign.Ptr
-import Foreign.Storable
+    -- * Questions that do not name an architecture
+    hasAESAcceleration,
+    hasGHASHAcceleration,
+) where
 
+import Control.Monad (filterM)
+import Data.List (sort)
+import Data.Word (Word16)
+import Foreign.C.Types (CInt (..), CUInt (..))
 import Crypto.Internal.Compat
 
-#ifdef SUPPORT_RDRAND
-import Crypto.Random.Entropy.RDRand
-import Crypto.Random.Entropy.Source
-#endif
+-- | A processor feature crypton looked for, and dispatches on where it
+-- finds it.
+--
+-- This is a number with names rather than a sum of constructors, and the
+-- names are pattern synonyms with no @COMPLETE@ pragma, so a @case@ over
+-- them needs a catch-all and a feature named in a later release breaks
+-- nothing that compiled against this one.  The same reason 'Show' is
+-- written out below: a program built against an older crypton still says
+-- something useful about a value from a newer one.
+--
+-- The names are the processor's, not the operation's.  'AESNI' is x86's
+-- and 'ARMAES' is AArch64's, and a machine reports only the ones it has;
+-- ask 'hasAESAcceleration' if the question is whether AES is fast here.
+--
+-- They are bundled with the type in the export list, so @ProcessorOption
+-- (..)@ brings in all of them and naming one brings in that one, as for a
+-- type with constructors.  The constructor underneath is not exported:
+-- these values say what the processor was found to have, and a caller has
+-- nothing to build.
+newtype ProcessorOption = ProcessorOption Word16
+    deriving (Eq, Ord)
 
--- | CPU options impacting cryptography implementation and library performance.
-data ProcessorOption
-    = -- | Support for AES instructions, with flag @support_aesni@
-      AESNI
-    | -- | Support for CLMUL instructions, with flag @support_pclmuldq@
-      PCLMUL
-    | -- | Support for RDRAND instruction, with flag @support_rdrand@
-      RDRAND
-    deriving (Show, Eq, Enum, Data)
+-- | Support for AES instructions, with flag @support_aesni@.
+pattern AESNI :: ProcessorOption
+pattern AESNI = ProcessorOption 0
 
+-- | Support for CLMUL instructions, with flag @support_pclmuldq@.
+pattern PCLMUL :: ProcessorOption
+pattern PCLMUL = ProcessorOption 1
+
+-- | Supplemental SSE3.
+pattern SSSE3 :: ProcessorOption
+pattern SSSE3 = ProcessorOption 3
+
+-- | AVX, and an operating system that saves its registers.
+pattern AVX :: ProcessorOption
+pattern AVX = ProcessorOption 4
+
+-- | AVX2, and an operating system that saves its registers.
+pattern AVX2 :: ProcessorOption
+pattern AVX2 = ProcessorOption 5
+
+-- | The SHA extensions, @sha1rnds4@ and @sha256rnds2@ and their neighbours.
+pattern SHANI :: ProcessorOption
+pattern SHANI = ProcessorOption 6
+
+-- | The byte-swapping load.
+pattern MOVBE :: ProcessorOption
+pattern MOVBE = ProcessorOption 7
+
+-- | @MULX@, @ADCX@ and @ADOX@: the two independent carry chains.
+pattern ADX :: ProcessorOption
+pattern ADX = ProcessorOption 8
+
+-- | The AES and carry-less multiply instructions in their 256-bit form.
+pattern VAES :: ProcessorOption
+pattern VAES = ProcessorOption 9
+
+-- | The same pair in their 512-bit form.
+pattern VAES512 :: ProcessorOption
+pattern VAES512 = ProcessorOption 10
+
+-- | Advanced SIMD, which is not optional on AArch64.
+pattern NEON :: ProcessorOption
+pattern NEON = ProcessorOption 11
+
+-- | The ARMv8 AES instructions.
+pattern ARMAES :: ProcessorOption
+pattern ARMAES = ProcessorOption 12
+
+-- | @PMULL@, the ARMv8 carry-less multiply.
+pattern ARMPMULL :: ProcessorOption
+pattern ARMPMULL = ProcessorOption 13
+
+-- | The ARMv8 SHA-1 instructions.
+pattern ARMSHA1 :: ProcessorOption
+pattern ARMSHA1 = ProcessorOption 14
+
+-- | The ARMv8 SHA-256 instructions.
+pattern ARMSHA2 :: ProcessorOption
+pattern ARMSHA2 = ProcessorOption 15
+
+-- | The ARMv8.2 SHA-512 instructions, which are optional where SHA-256's
+-- are not.
+pattern ARMSHA512 :: ProcessorOption
+pattern ARMSHA512 = ProcessorOption 16
+
+-- | Support for the PowerISA 2.07 vector AES instructions, which POWER8 was
+-- the first to implement.
+pattern PPCAES :: ProcessorOption
+pattern PPCAES = ProcessorOption 17
+
+-- | Support for @vpmsumd@, the vector carry-less multiply that came with
+-- them, which is what makes GHASH fast.
+pattern PPCVPMSUM :: ProcessorOption
+pattern PPCVPMSUM = ProcessorOption 18
+
+-- | Named where the name is known, numbered where it is not, so that a
+-- binary built against an older crypton can still print a value a newer one
+-- produced.
+instance Show ProcessorOption where
+    show AESNI = "AESNI"
+    show PCLMUL = "PCLMUL"
+    show SSSE3 = "SSSE3"
+    show AVX = "AVX"
+    show AVX2 = "AVX2"
+    show SHANI = "SHANI"
+    show MOVBE = "MOVBE"
+    show ADX = "ADX"
+    show VAES = "VAES"
+    show VAES512 = "VAES512"
+    show NEON = "NEON"
+    show ARMAES = "ARMAES"
+    show ARMPMULL = "ARMPMULL"
+    show ARMSHA1 = "ARMSHA1"
+    show ARMSHA2 = "ARMSHA2"
+    show ARMSHA512 = "ARMSHA512"
+    show PPCAES = "PPCAES"
+    show PPCVPMSUM = "PPCVPMSUM"
+    show (ProcessorOption n) = "ProcessorOption " ++ show n
+
 -- | Options which have been enabled at compile time and are supported by the
 -- current CPU.
+--
+-- Sorted, and without repeats.  A machine reports the names of its own
+-- architecture only: an AArch64 processor with AES says 'ARMAES', not
+-- 'AESNI', which it does not have.
 processorOptions :: [ProcessorOption]
-processorOptions = unsafeDoIO $ do
-    p <- crypton_aes_cpu_init
-    options <- traverse (getOption p) aesOptions
-    rdrand <- hasRDRand
-    return (decodeOptions options ++ [RDRAND | rdrand])
+processorOptions = unsafeDoIO (sort <$> filterM askC allOptions)
   where
-    aesOptions = [AESNI .. PCLMUL]
-    getOption p = peekElemOff p . fromEnum
-    decodeOptions = map toEnum . findIndices (> 0)
+    -- 2 is not asked for and has no name: it was RDRAND, which crypton no
+    -- longer dispatches on.  The number is left out rather than reused, so
+    -- that the others keep the values they had.
+    allOptions = [ProcessorOption n | n <- [0 .. 18], n /= 2]
+    askC (ProcessorOption n) =
+        (/= 0) <$> crypton_cpu_option (fromIntegral n)
 {-# NOINLINE processorOptions #-}
 
-hasRDRand :: IO Bool
-#ifdef SUPPORT_RDRAND
-hasRDRand = fmap isJust getRDRand
-  where getRDRand = entropyOpen :: IO (Maybe RDRand)
-#else
-hasRDRand = return False
-#endif
+-- | Is there hardware AES on this machine?
+--
+-- The instructions have different names on different architectures, and a
+-- caller that wants to know whether AES-GCM will be fast wants this rather
+-- than either name.
+hasAESAcceleration :: Bool
+hasAESAcceleration =
+    any
+        (`elem` processorOptions)
+        [AESNI, ARMAES, PPCAES]
 
-foreign import ccall unsafe "crypton_aes_cpu_init"
-    crypton_aes_cpu_init :: IO (Ptr Word8)
+-- | Is there a hardware carry-less multiply, which is what GHASH, and so
+-- AES-GCM, spends its time in once AES itself is fast?
+hasGHASHAcceleration :: Bool
+hasGHASHAcceleration =
+    any
+        (`elem` processorOptions)
+        [PCLMUL, ARMPMULL, PPCVPMSUM]
+
+foreign import ccall unsafe "crypton_cpu_option"
+    crypton_cpu_option :: CUInt -> IO CInt
diff --git a/Crypto/Tutorial.hs b/Crypto/Tutorial.hs
--- a/Crypto/Tutorial.hs
+++ b/Crypto/Tutorial.hs
@@ -1,4 +1,7 @@
 -- | Examples of how to use @crypton@.
+--
+-- Every code block here is extracted and compiled against this version of
+-- the library by @tests\/tutorial\/run.sh@, so what is written below builds.
 module Crypto.Tutorial (
     -- * API design
     -- $api_design
@@ -6,9 +9,21 @@
     -- * Hash algorithms
     -- $hash_algorithms
 
-    -- * Symmetric block ciphers
-    -- $symmetric_block_ciphers
+    -- * Authenticated encryption
+    -- $authenticated_encryption
 
+    -- * Comparing secrets
+    -- $comparing_secrets
+
+    -- * Password storage
+    -- $password_storage
+
+    -- * Key derivation
+    -- $key_derivation
+
+    -- * Digital signatures
+    -- $digital_signatures
+
     -- * Combining primitives
     -- $combining_primitives
 ) where
@@ -31,6 +46,14 @@
 -- Error conditions are returned with data type 'Crypto.Error.CryptoFailable'.
 -- Functions in module "Crypto.Error" can convert those values to runtime
 -- exceptions, 'Maybe' or 'Either' values.
+--
+-- Types that hold a secret do not print it.  A private key's 'Show' renders
+-- whatever is public and @\<secret\>@ or @\<scrubbed-bytes\>@ for the rest,
+-- because 'Show' is what @print@, @error@, an exception and a failing test
+-- all reach for, and a key arriving in a log that way is an accident nobody
+-- asked for.  "Crypto.Debug" is how one is printed when printing it is what
+-- was meant.  Several of those types keep their bytes in
+-- 'Data.ByteArray.ScrubbedBytes', which is wiped when it is collected.
 
 -- $hash_algorithms
 --
@@ -90,72 +113,258 @@
 -- >     hashMutableUpdate ctx ("dog"    :: ByteString)
 -- >     hashMutableFinalize ctx >>= print
 
--- $symmetric_block_ciphers
+-- $authenticated_encryption
 --
+-- Encrypting hides a message; it does not stop anyone changing it.  Under a
+-- counter or stream mode, flipping a bit of the ciphertext flips the same
+-- bit of the plaintext, and the receiver has no way to tell.  An AEAD mode
+-- binds the message, and anything else named as associated data, to a short
+-- authentication tag, and decrypting something that does not match that tag
+-- returns nothing at all.
+--
+-- That is the mode to use.  The unauthenticated ones are in this library
+-- for protocols that authenticate separately, not as a starting point.
+--
 -- > {-# LANGUAGE OverloadedStrings #-}
--- > {-# LANGUAGE ScopedTypeVariables #-}
--- > {-# LANGUAGE GADTs #-}
 -- >
 -- > import           Crypto.Cipher.AES (AES256)
--- > import           Crypto.Cipher.Types (BlockCipher(..), Cipher(..), nullIV, KeySizeSpecifier(..), IV, makeIV)
--- > import           Crypto.Error (CryptoFailable(..), CryptoError(..))
+-- > import           Crypto.Cipher.Types
+-- >                      ( AEADMode (AEAD_GCM)
+-- >                      , AuthTag
+-- >                      , BlockCipher (aeadInit)
+-- >                      , Cipher (cipherInit, cipherKeySize)
+-- >                      , KeySizeSpecifier (..)
+-- >                      , aeadSimpleDecrypt
+-- >                      , aeadSimpleEncrypt
+-- >                      )
+-- > import           Crypto.Error (CryptoError, eitherCryptoError)
+-- > import qualified Crypto.Random.Types as CRT
 -- >
+-- > import           Data.ByteArray (ByteArray, ByteArrayAccess, ScrubbedBytes)
+-- > import           Data.ByteString (ByteString)
+-- >
+-- > -- | A key of the length the cipher asks for, rather than a length
+-- > -- written out here.  ScrubbedBytes rather than ByteString, so that it
+-- > -- is wiped when it is collected and does not print.
+-- > genSecretKey :: (Cipher c, CRT.MonadRandom m) => c -> m ScrubbedBytes
+-- > genSecretKey c = CRT.getRandomBytes (longest (cipherKeySize c))
+-- >   where
+-- >     longest (KeySizeFixed n)   = n
+-- >     longest (KeySizeRange _ n) = n
+-- >     longest (KeySizeEnum ns)   = maximum ns
+-- >
+-- > -- | A fresh nonce for every message.  GCM must never see one twice
+-- > -- under the same key: a repeat does not just expose those two messages,
+-- > -- it hands over the key that authenticates all of them.  Twelve random
+-- > -- bytes, sent along with the ciphertext.
+-- > genNonce :: CRT.MonadRandom m => m ByteString
+-- > genNonce = CRT.getRandomBytes 12
+-- >
+-- > -- | Encrypt and authenticate.  The associated data is authenticated but
+-- > -- not encrypted: it is for what the receiver can already see and must
+-- > -- not have had altered, such as a header or an address.
+-- > encrypt
+-- >     :: (ByteArray key, ByteArrayAccess nonce, ByteArrayAccess aad, ByteArray ba)
+-- >     => key -> nonce -> aad -> ba -> Either CryptoError (AuthTag, ba)
+-- > encrypt key nonce aad plaintext = do
+-- >     cipher <- eitherCryptoError (cipherInit key) :: Either CryptoError AES256
+-- >     aead <- eitherCryptoError (aeadInit AEAD_GCM cipher nonce)
+-- >     return (aeadSimpleEncrypt aead aad plaintext 16)
+-- >
+-- > -- | And back, with two different failures.  Left is this code used
+-- > -- wrongly -- a key of the wrong length, a mode the cipher has not got.
+-- > -- Nothing is a message that is not the one that was sent; it carries no
+-- > -- plaintext and says nothing about which byte was wrong, both of which
+-- > -- are the point.
+-- > decrypt
+-- >     :: (ByteArray key, ByteArrayAccess nonce, ByteArrayAccess aad, ByteArray ba)
+-- >     => key -> nonce -> aad -> AuthTag -> ba -> Either CryptoError (Maybe ba)
+-- > decrypt key nonce aad tag ciphertext = do
+-- >     cipher <- eitherCryptoError (cipherInit key) :: Either CryptoError AES256
+-- >     aead <- eitherCryptoError (aeadInit AEAD_GCM cipher nonce)
+-- >     return (aeadSimpleDecrypt aead aad ciphertext tag)
+-- >
+-- > exampleAES256GCM :: ByteString -> IO ()
+-- > exampleAES256GCM msg = do
+-- >     key <- genSecretKey (undefined :: AES256)
+-- >     nonce <- genNonce
+-- >     let aad = "to: alice" :: ByteString
+-- >     case encrypt key nonce aad msg of
+-- >         Left err -> error (show err)
+-- >         Right (tag, ciphertext) -> do
+-- >             putStrLn $ "ciphertext: " ++ show ciphertext
+-- >             putStrLn $ "       tag: " ++ show tag
+-- >             putStrLn $ " recovered: "
+-- >                 ++ show (decrypt key nonce aad tag ciphertext)
+-- >             -- The same bytes and the same tag, with one thing changed
+-- >             -- that was never encrypted: Right Nothing.
+-- >             putStrLn $ "redirected: "
+-- >                 ++ show (decrypt key nonce ("to: eve" :: ByteString) tag ciphertext)
+--
+-- The two functions above work for any cipher that has an AEAD mode; what
+-- changes is the mode given to 'Crypto.Cipher.Types.aeadInit'.
+-- "Crypto.Cipher.ChaChaPoly1305" is the one to prefer on a machine with no
+-- AES instructions, and "Crypto.Cipher.AESGCMSIV" is the one that survives
+-- a repeated nonce, at the price of needing the whole message before it can
+-- begin.
+
+-- $comparing_secrets
+--
+-- Comparing two byte strings with '==' stops at the first byte that
+-- differs, so how long it takes says where that byte was.  Against an
+-- authentication tag that is the whole secret: someone who can send a guess
+-- and time the answer finds the first byte in a few hundred tries, then the
+-- second, and has a tag that was supposed to cost 2^128 in a few thousand.
+--
+-- crypton's own authentication types already compare in constant time, so
+-- for those there is nothing to do: 'Crypto.MAC.HMAC.HMAC',
+-- 'Crypto.MAC.CMAC.CMAC', 'Crypto.MAC.Poly1305.Auth' and
+-- 'Crypto.Cipher.Types.AuthTag' have an 'Eq' that looks at every byte
+-- whatever it finds.  What needs care is a tag that arrives as bytes, and
+-- 'Data.ByteArray.constEq' is the comparison for it.
+--
+-- > {-# LANGUAGE OverloadedStrings #-}
+-- >
+-- > import           Crypto.Hash.Algorithms (SHA256)
+-- > import           Crypto.MAC.HMAC (HMAC, hmac)
+-- >
+-- > import qualified Data.ByteArray as BA
+-- > import           Data.ByteString (ByteString)
+-- >
+-- > -- | The tag came off the wire as bytes, so it is compared as bytes.
+-- > authentic :: ByteString -> ByteString -> ByteString -> Bool
+-- > authentic key message tag =
+-- >     BA.constEq tag (hmac key message :: HMAC SHA256)
+-- >
+-- > -- | Once it has been parsed into the library's own type, (==) is
+-- > -- already the constant-time comparison.
+-- > authentic' :: ByteString -> ByteString -> HMAC SHA256 -> Bool
+-- > authentic' key message tag = tag == hmac key message
+
+-- $password_storage
+--
+-- A password is not a key.  It is short and it is guessable, and whoever
+-- takes the database can try every likely one without being watched.  What
+-- answers that is a function that is deliberately expensive to compute, and
+-- crypton has four: "Crypto.KDF.BCrypt", "Crypto.KDF.Scrypt",
+-- "Crypto.KDF.Argon2" and "Crypto.KDF.PBKDF2".  A plain hash is not one of
+-- them, however many times it is applied by hand.
+--
+-- bcrypt leaves the least to get wrong, because the record it returns
+-- carries the salt and the cost inside it:
+--
+-- > import Crypto.KDF.BCrypt (hashPassword, validatePassword)
+-- >
+-- > import Data.ByteString (ByteString)
+-- >
+-- > -- | What goes in the database.  The salt is drawn inside and ends up in
+-- > -- the result, so two accounts with the same password do not look alike.
+-- > register :: ByteString -> IO ByteString
+-- > register password = hashPassword 12 password
+-- >
+-- > -- | And what is checked against it.  The cost comes out of the stored
+-- > -- record, so raising it for new accounts leaves the old ones working.
+-- > login :: ByteString -> ByteString -> Bool
+-- > login password stored = validatePassword password stored
+--
+-- Argon2 is the stronger choice and the one to pick for something new: it
+-- asks for memory as well as time, which is what takes the advantage away
+-- from the hardware that bcrypt's small working set leaves room for.
+-- Nothing is encoded for the caller, though -- the salt and the options are
+-- theirs to store, and without all three the hash cannot be recomputed when
+-- the user comes back.
+--
+-- > import           Crypto.Error (CryptoFailable)
+-- > import qualified Crypto.KDF.Argon2 as Argon2
 -- > import qualified Crypto.Random.Types as CRT
 -- >
--- > import           Data.ByteArray (ByteArray)
 -- > import           Data.ByteString (ByteString)
 -- >
--- > -- | Not required, but most general implementation
--- > data Key c a where
--- >   Key :: (BlockCipher c, ByteArray a) => a -> Key c a
+-- > -- | Argon2id, which is the variant to prefer: it resists both a machine
+-- > -- built to guess and a process watching the cache.
+-- > options :: Argon2.Options
+-- > options = Argon2.defaultOptions{Argon2.variant = Argon2.Argon2id}
 -- >
--- > -- | Generates a string of bytes (key) of a specific length for a given block cipher
--- > genSecretKey :: forall m c a. (CRT.MonadRandom m, BlockCipher c, ByteArray a) => c -> Int -> m (Key c a)
--- > genSecretKey _ = fmap Key . CRT.getRandomBytes
+-- > newSalt :: CRT.MonadRandom m => m ByteString
+-- > newSalt = CRT.getRandomBytes 16
 -- >
--- > -- | Generate a random initialization vector for a given block cipher
--- > genRandomIV :: forall m c. (CRT.MonadRandom m, BlockCipher c) => c -> m (Maybe (IV c))
--- > genRandomIV _ = do
--- >   bytes :: ByteString <- CRT.getRandomBytes $ blockSize (undefined :: c)
--- >   return $ makeIV bytes
+-- > derive :: ByteString -> ByteString -> CryptoFailable ByteString
+-- > derive salt password = Argon2.hash options password salt 32
+
+-- $key_derivation
+--
+-- HKDF turns one secret into as many keys as a protocol needs.  It is for
+-- material that is already unguessable -- what comes out of a
+-- Diffie-Hellman, or a key already agreed -- and it is deliberately cheap,
+-- which is exactly what makes it the wrong thing for a password.  Those go
+-- to the section above.
+--
+-- The info string is what keeps the outputs independent: the same secret
+-- with a different info gives an unrelated key, so each use of a secret
+-- names itself there.
+--
+-- > {-# LANGUAGE OverloadedStrings #-}
 -- >
--- > -- | Initialize a block cipher
--- > initCipher :: (BlockCipher c, ByteArray a) => Key c a -> Either CryptoError c
--- > initCipher (Key k) = case cipherInit k of
--- >   CryptoFailed e -> Left e
--- >   CryptoPassed a -> Right a
+-- > import           Crypto.Hash.Algorithms (SHA256)
+-- > import qualified Crypto.KDF.HKDF as HKDF
 -- >
--- > encrypt :: (BlockCipher c, ByteArray a) => Key c a -> IV c -> a -> Either CryptoError a
--- > encrypt secretKey initIV msg =
--- >   case initCipher secretKey of
--- >     Left e -> Left e
--- >     Right c -> Right $ ctrCombine c initIV msg
+-- > import           Data.ByteArray (ScrubbedBytes)
+-- > import           Data.ByteString (ByteString)
 -- >
--- > decrypt :: (BlockCipher c, ByteArray a) => Key c a -> IV c -> a -> Either CryptoError a
--- > decrypt = encrypt
+-- > -- | One shared secret in, two unrelated keys out.
+-- > directionKeys :: ByteString -> ByteString -> (ScrubbedBytes, ScrubbedBytes)
+-- > directionKeys salt shared = (keyFor "client write", keyFor "server write")
+-- >   where
+-- >     prk = HKDF.extract salt shared :: HKDF.PRK SHA256
+-- >     keyFor info = HKDF.expand prk (info :: ByteString) 32
+
+-- $digital_signatures
+--
+-- Ed25519 is the one to reach for.  The keys are thirty-two bytes, there is
+-- nothing to choose and nothing to encode, and signing needs no randomness,
+-- so it cannot be ruined by a bad source of it.  "Crypto.PubKey.Ed448" is
+-- the same shape at a larger size, "Crypto.PubKey.ECDSA" and
+-- "Crypto.PubKey.RSA.PSS" are there for protocols that ask for them, and
+-- "Crypto.PubKey.MLDSA" is the post-quantum one.
+--
+-- > {-# LANGUAGE OverloadedStrings #-}
 -- >
--- > exampleAES256 :: ByteString -> IO ()
--- > exampleAES256 msg = do
--- >   -- secret key needs 256 bits (32 * 8)
--- >   secretKey <- genSecretKey (undefined :: AES256) 32
--- >   mInitIV <- genRandomIV (undefined :: AES256)
--- >   case mInitIV of
--- >     Nothing -> error "Failed to generate and initialization vector."
--- >     Just initIV -> do
--- >       let encryptedMsg = encrypt secretKey initIV msg
--- >           decryptedMsg = decrypt secretKey initIV =<< encryptedMsg
--- >       case (,) <$> encryptedMsg <*> decryptedMsg of
--- >         Left err -> error $ show err
--- >         Right (eMsg, dMsg) -> do
--- >           putStrLn $ "Original Message: " ++ show msg
--- >           putStrLn $ "Message after encryption: " ++ show eMsg
--- >           putStrLn $ "Message after decryption: " ++ show dMsg
+-- > import           Crypto.Error (throwCryptoError)
+-- > import qualified Crypto.PubKey.Ed25519 as Ed25519
+-- >
+-- > import qualified Data.ByteArray as BA
+-- > import           Data.ByteString (ByteString)
+-- >
+-- > exampleEd25519 :: ByteString -> IO ()
+-- > exampleEd25519 msg = do
+-- >     sk <- Ed25519.generateSecretKey
+-- >     let pk = Ed25519.toPublic sk
+-- >         sig = Ed25519.sign sk pk msg
+-- >     print (Ed25519.verify pk msg sig)
+-- >     -- The same signature against a message one byte longer: False.
+-- >     print (Ed25519.verify pk (msg <> "!") sig)
+-- >
+-- > -- | A secret key on its way to storage.  Printing one does not reveal
+-- > -- it -- see the first section -- so this is the way out.
+-- > store :: Ed25519.SecretKey -> ByteString
+-- > store = BA.convert
+-- >
+-- > -- | And the way back in, which is checked, because the bytes read from
+-- > -- a file may be anything at all.
+-- > load :: ByteString -> Ed25519.SecretKey
+-- > load = throwCryptoError . Ed25519.secretKey
 
 -- $combining_primitives
 --
 -- This example shows how to use Curve25519, XSalsa and Poly1305 primitives to
 -- emulate NaCl's @crypto_box@ construct.
 --
+-- It is here to show how the pieces fit together, not as something to
+-- deploy.  An authenticated encryption scheme assembled by hand is the kind
+-- of code that is wrong in ways no test notices; for actual use,
+-- "Crypto.Cipher.ChaChaPoly1305" does this job with the mistakes already
+-- made.
+--
 -- > import qualified Data.ByteArray as BA
 -- > import           Data.ByteString (ByteString)
 -- > import qualified Data.ByteString as B
@@ -167,6 +376,9 @@
 -- >
 -- > -- | Build a @crypto_box@ packet encrypting the specified content with a
 -- > -- 192-bit nonce, receiver public key and sender private key.
+-- > crypto_box
+-- >     :: ByteString -> ByteString -> X25519.PublicKey -> X25519.SecretKey
+-- >     -> ByteString
 -- > crypto_box content nonce pk sk = BA.convert tag `B.append` c
 -- >   where
 -- >     zero         = B.replicate 16 0
@@ -181,6 +393,9 @@
 -- >
 -- > -- | Try to open a @crypto_box@ packet and recover the content using the
 -- > -- 192-bit nonce, sender public key and receiver private key.
+-- > crypto_box_open
+-- >     :: ByteString -> ByteString -> X25519.PublicKey -> X25519.SecretKey
+-- >     -> Maybe ByteString
 -- > crypto_box_open packet nonce pk sk
 -- >     | B.length packet < 16 = Nothing
 -- >     | BA.constEq tag' tag  = Just content
diff --git a/README.md b/README.md
--- a/README.md
+++ b/README.md
@@ -14,33 +14,49 @@
 Side channels
 -------------
 
+### AES
+
 AES is where this matters most, and which implementation runs is decided at
 runtime from what the processor has.
 
-On x86-64 with AES-NI and carry-less multiply, and on AArch64 with the ARMv8
-cryptographic extension, AES and GHASH are instructions rather than tables.
-crypton's AES and AES-GCM then make no branch and no memory access that
-depends on the key or on the data: the secrets stay in vector registers and
-never reach one a branch can test, which the generated code is checked
-against.  Every x86-64 part since about 2010 and every AArch64 part in
-ordinary use has these.
+On x86-64 with AES-NI and carry-less multiply, on AArch64 and on 32-bit ARM
+with the ARMv8 cryptographic extension, and on ppc64le with the vector
+instructions PowerISA 2.07 brought to POWER8, AES and GHASH are instructions
+rather than software.  crypton's AES and AES-GCM then make no branch and no
+memory access that depends on the key or on the data: the secrets stay in
+vector registers and never reach one a branch can test.  Every x86-64 part
+since about 2010 and every AArch64 part in ordinary use has these, and the
+ppc64le path is little-endian only.
 
-Where neither is present crypton falls back to a table-driven AES, which
-indexes a 256-byte substitution table with data derived from the key and the
-input.  **That is not constant time**, and on a machine where an attacker can
-observe the cache it is open to a timing attack.  The fallback exists so that
-the library builds and runs everywhere; it is not meant for a setting where
-that matters.
+Where none of them is there, crypton computes AES and GHASH without a table.
+The portable AES is bitsliced -- the S-box is boolean algebra over four
+blocks held across eight 64-bit words -- and the GF(2^128) multiply is built
+from shifts, masks and integer multiplies.  Neither looks anything up, so
+neither derives an address from a secret, and **the portable path is constant
+time as well**.  The code is BearSSL's, under MIT, in `cbits/bearssl`.
 
-`Crypto.System.CPU.processorOptions` says which is in use.  `AESNI` in that
-list means the instruction path, and `PCLMUL` that GHASH has its instruction
-too; without `AESNI` it is the tables.  The list also reports `RDRAND`, which
-is unrelated to this.
+The constant-time harness in `cbits/tests/ct` runs the same driver against
+the portable implementation and against each architecture's instructions, and
+every one of them is required to report nothing at all.
 
+So the answer does not depend on the machine any more, and what is left to
+ask is whether AES is fast on it.  `hasAESAcceleration` and
+`hasGHASHAcceleration` answer that without naming an architecture.
+`processorOptions` says what was found, in the processor's own names --
+`AESNI` and `PCLMUL` on x86, `ARMAES` and `ARMPMULL` on AArch64 and 32-bit
+ARM, `PPCAES` and `PPCVPMSUM` on ppc64le -- along with everything else it was
+asked about, which is unrelated to this.
+
     ghci> import Crypto.System.CPU
-    ghci> processorOptions
-    [AESNI,PCLMUL]
+    ghci> hasAESAcceleration
+    True
+    ghci> processorOptions           -- on an x86-64 machine
+    [AESNI,PCLMUL,SSSE3,AVX,AVX2,SHANI,MOVBE,ADX,VAES]
+    ghci> processorOptions           -- and on an AArch64 one
+    [NEON,ARMAES,ARMPMULL,ARMSHA1,ARMSHA2,ARMSHA512]
 
+### RSA
+
 RSA is the other place to know about, and there the choice is the caller's.
 The private key operations in `Crypto.PubKey.RSA.PKCS15`, `.OAEP` and `.PSS`
 take a `Maybe Blinder`, and `Nothing` is no harder to write than the safe
@@ -252,3 +268,8 @@
 ask for it, not because it is a good choice for anything new.  The algorithms
 that nothing should ask for any more -- MD5, 3DES, RC4, CBC mode -- are left
 out.
+
+ML-KEM and ML-DSA are left out too, though 2.1.8 brought both and they are
+fast -- `mlkem-native` and `mldsa-native`, from the PQ Code Package.  No
+comparison is reported here: the implementations to measure against are
+still developing.
diff --git a/cbits/aes/generic.c b/cbits/aes/generic.c
--- a/cbits/aes/generic.c
+++ b/cbits/aes/generic.c
@@ -27,420 +27,191 @@
  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  * SUCH DAMAGE.
  *
- * AES implementation
+ * AES, portable, for machines whose processor has no AES instructions or
+ * whose build was not given them.
+ *
+ * This was a table-driven implementation: an S-box indexed by a byte of the
+ * state, in every round and in the key expansion, which is what made it fast
+ * and what made it variable-time.  It is now BearSSL's bitsliced aes_ct64,
+ * which holds four blocks interleaved across eight 64-bit words and computes
+ * the S-box as boolean algebra -- no table, and so no address derived from a
+ * secret.  See cbits/bearssl/README.md.
+ *
+ * What is here is only the glue.  Nothing in this file branches or indexes on
+ * the key or the data.
  */
 
 #include <stdint.h>
-#include <stdlib.h>
+#include <string.h>
 #include <crypton_aes.h>
-#include <crypton_bitfn.h>
-
-static uint8_t sbox[256] = {
-	0x63, 0x7c, 0x77, 0x7b, 0xf2, 0x6b, 0x6f, 0xc5, 0x30, 0x01, 0x67, 0x2b, 0xfe,
-	0xd7, 0xab, 0x76, 0xca, 0x82, 0xc9, 0x7d, 0xfa, 0x59, 0x47, 0xf0, 0xad, 0xd4,
-	0xa2, 0xaf, 0x9c, 0xa4, 0x72, 0xc0, 0xb7, 0xfd, 0x93, 0x26, 0x36, 0x3f, 0xf7,
-	0xcc, 0x34, 0xa5, 0xe5, 0xf1, 0x71, 0xd8, 0x31, 0x15, 0x04, 0xc7, 0x23, 0xc3,
-	0x18, 0x96, 0x05, 0x9a, 0x07, 0x12, 0x80, 0xe2, 0xeb, 0x27, 0xb2, 0x75, 0x09,
-	0x83, 0x2c, 0x1a, 0x1b, 0x6e, 0x5a, 0xa0, 0x52, 0x3b, 0xd6, 0xb3, 0x29, 0xe3,
-	0x2f, 0x84, 0x53, 0xd1, 0x00, 0xed, 0x20, 0xfc, 0xb1, 0x5b, 0x6a, 0xcb, 0xbe,
-	0x39, 0x4a, 0x4c, 0x58, 0xcf, 0xd0, 0xef, 0xaa, 0xfb, 0x43, 0x4d, 0x33, 0x85,
-	0x45, 0xf9, 0x02, 0x7f, 0x50, 0x3c, 0x9f, 0xa8, 0x51, 0xa3, 0x40, 0x8f, 0x92,
-	0x9d, 0x38, 0xf5, 0xbc, 0xb6, 0xda, 0x21, 0x10, 0xff, 0xf3, 0xd2, 0xcd, 0x0c,
-	0x13, 0xec, 0x5f, 0x97, 0x44, 0x17, 0xc4, 0xa7, 0x7e, 0x3d, 0x64, 0x5d, 0x19,
-	0x73, 0x60, 0x81, 0x4f, 0xdc, 0x22, 0x2a, 0x90, 0x88, 0x46, 0xee, 0xb8, 0x14,
-	0xde, 0x5e, 0x0b, 0xdb, 0xe0, 0x32, 0x3a, 0x0a, 0x49, 0x06, 0x24, 0x5c, 0xc2,
-	0xd3, 0xac, 0x62, 0x91, 0x95, 0xe4, 0x79, 0xe7, 0xc8, 0x37, 0x6d, 0x8d, 0xd5,
-	0x4e, 0xa9, 0x6c, 0x56, 0xf4, 0xea, 0x65, 0x7a, 0xae, 0x08, 0xba, 0x78, 0x25,
-	0x2e, 0x1c, 0xa6, 0xb4, 0xc6, 0xe8, 0xdd, 0x74, 0x1f, 0x4b, 0xbd, 0x8b, 0x8a,
-	0x70, 0x3e, 0xb5, 0x66, 0x48, 0x03, 0xf6, 0x0e, 0x61, 0x35, 0x57, 0xb9, 0x86,
-	0xc1, 0x1d, 0x9e, 0xe1, 0xf8, 0x98, 0x11, 0x69, 0xd9, 0x8e, 0x94, 0x9b, 0x1e,
-	0x87, 0xe9, 0xce, 0x55, 0x28, 0xdf, 0x8c, 0xa1, 0x89, 0x0d, 0xbf, 0xe6, 0x42,
-	0x68, 0x41, 0x99, 0x2d, 0x0f, 0xb0, 0x54, 0xbb, 0x16
-};
-
-static uint8_t rsbox[256] = {
-	0x52, 0x09, 0x6a, 0xd5, 0x30, 0x36, 0xa5, 0x38, 0xbf, 0x40, 0xa3, 0x9e, 0x81,
-	0xf3, 0xd7, 0xfb, 0x7c, 0xe3, 0x39, 0x82, 0x9b, 0x2f, 0xff, 0x87, 0x34, 0x8e,
-	0x43, 0x44, 0xc4, 0xde, 0xe9, 0xcb, 0x54, 0x7b, 0x94, 0x32, 0xa6, 0xc2, 0x23,
-	0x3d, 0xee, 0x4c, 0x95, 0x0b, 0x42, 0xfa, 0xc3, 0x4e, 0x08, 0x2e, 0xa1, 0x66,
-	0x28, 0xd9, 0x24, 0xb2, 0x76, 0x5b, 0xa2, 0x49, 0x6d, 0x8b, 0xd1, 0x25, 0x72,
-	0xf8, 0xf6, 0x64, 0x86, 0x68, 0x98, 0x16, 0xd4, 0xa4, 0x5c, 0xcc, 0x5d, 0x65,
-	0xb6, 0x92, 0x6c, 0x70, 0x48, 0x50, 0xfd, 0xed, 0xb9, 0xda, 0x5e, 0x15, 0x46,
-	0x57, 0xa7, 0x8d, 0x9d, 0x84, 0x90, 0xd8, 0xab, 0x00, 0x8c, 0xbc, 0xd3, 0x0a,
-	0xf7, 0xe4, 0x58, 0x05, 0xb8, 0xb3, 0x45, 0x06, 0xd0, 0x2c, 0x1e, 0x8f, 0xca,
-	0x3f, 0x0f, 0x02, 0xc1, 0xaf, 0xbd, 0x03, 0x01, 0x13, 0x8a, 0x6b, 0x3a, 0x91,
-	0x11, 0x41, 0x4f, 0x67, 0xdc, 0xea, 0x97, 0xf2, 0xcf, 0xce, 0xf0, 0xb4, 0xe6,
-	0x73, 0x96, 0xac, 0x74, 0x22, 0xe7, 0xad, 0x35, 0x85, 0xe2, 0xf9, 0x37, 0xe8,
-	0x1c, 0x75, 0xdf, 0x6e, 0x47, 0xf1, 0x1a, 0x71, 0x1d, 0x29, 0xc5, 0x89, 0x6f,
-	0xb7, 0x62, 0x0e, 0xaa, 0x18, 0xbe, 0x1b, 0xfc, 0x56, 0x3e, 0x4b, 0xc6, 0xd2,
-	0x79, 0x20, 0x9a, 0xdb, 0xc0, 0xfe, 0x78, 0xcd, 0x5a, 0xf4, 0x1f, 0xdd, 0xa8,
-	0x33, 0x88, 0x07, 0xc7, 0x31, 0xb1, 0x12, 0x10, 0x59, 0x27, 0x80, 0xec, 0x5f,
-	0x60, 0x51, 0x7f, 0xa9, 0x19, 0xb5, 0x4a, 0x0d, 0x2d, 0xe5, 0x7a, 0x9f, 0x93,
-	0xc9, 0x9c, 0xef, 0xa0, 0xe0, 0x3b, 0x4d, 0xae, 0x2a, 0xf5, 0xb0, 0xc8, 0xeb,
-	0xbb, 0x3c, 0x83, 0x53, 0x99, 0x61, 0x17, 0x2b, 0x04, 0x7e, 0xba, 0x77, 0xd6,
-	0x26, 0xe1, 0x69, 0x14, 0x63, 0x55, 0x21, 0x0c, 0x7d
-};
-
-static uint8_t Rcon[] = {
-	0x8d, 0x01, 0x02, 0x04, 0x08, 0x10, 0x20, 0x40, 0x80, 0x1b, 0x36, 0x6c, 0xd8,
-	0xab, 0x4d, 0x9a, 0x2f, 0x5e, 0xbc, 0x63, 0xc6, 0x97, 0x35, 0x6a, 0xd4, 0xb3,
-	0x7d, 0xfa, 0xef, 0xc5, 0x91, 0x39, 0x72, 0xe4, 0xd3, 0xbd, 0x61, 0xc2, 0x9f,
-	0x25, 0x4a, 0x94, 0x33, 0x66, 0xcc, 0x83, 0x1d, 0x3a, 0x74, 0xe8, 0xcb,
-};
+#include "bearssl/inner.h"
+#include "aes/block128.h"
+#include "aes/generic.h"
 
-#define G(a,b,c,d,e,f) { a,b,c,d,e,f }
-static uint8_t gmtab[256][6] =
-{
-	G(0x00, 0x00, 0x00, 0x00, 0x00, 0x00), G(0x02, 0x03, 0x09, 0x0b, 0x0d, 0x0e),
-	G(0x04, 0x06, 0x12, 0x16, 0x1a, 0x1c), G(0x06, 0x05, 0x1b, 0x1d, 0x17, 0x12),
-	G(0x08, 0x0c, 0x24, 0x2c, 0x34, 0x38), G(0x0a, 0x0f, 0x2d, 0x27, 0x39, 0x36),
-	G(0x0c, 0x0a, 0x36, 0x3a, 0x2e, 0x24), G(0x0e, 0x09, 0x3f, 0x31, 0x23, 0x2a),
-	G(0x10, 0x18, 0x48, 0x58, 0x68, 0x70), G(0x12, 0x1b, 0x41, 0x53, 0x65, 0x7e),
-	G(0x14, 0x1e, 0x5a, 0x4e, 0x72, 0x6c), G(0x16, 0x1d, 0x53, 0x45, 0x7f, 0x62),
-	G(0x18, 0x14, 0x6c, 0x74, 0x5c, 0x48), G(0x1a, 0x17, 0x65, 0x7f, 0x51, 0x46),
-	G(0x1c, 0x12, 0x7e, 0x62, 0x46, 0x54), G(0x1e, 0x11, 0x77, 0x69, 0x4b, 0x5a),
-	G(0x20, 0x30, 0x90, 0xb0, 0xd0, 0xe0), G(0x22, 0x33, 0x99, 0xbb, 0xdd, 0xee),
-	G(0x24, 0x36, 0x82, 0xa6, 0xca, 0xfc), G(0x26, 0x35, 0x8b, 0xad, 0xc7, 0xf2),
-	G(0x28, 0x3c, 0xb4, 0x9c, 0xe4, 0xd8), G(0x2a, 0x3f, 0xbd, 0x97, 0xe9, 0xd6),
-	G(0x2c, 0x3a, 0xa6, 0x8a, 0xfe, 0xc4), G(0x2e, 0x39, 0xaf, 0x81, 0xf3, 0xca),
-	G(0x30, 0x28, 0xd8, 0xe8, 0xb8, 0x90), G(0x32, 0x2b, 0xd1, 0xe3, 0xb5, 0x9e),
-	G(0x34, 0x2e, 0xca, 0xfe, 0xa2, 0x8c), G(0x36, 0x2d, 0xc3, 0xf5, 0xaf, 0x82),
-	G(0x38, 0x24, 0xfc, 0xc4, 0x8c, 0xa8), G(0x3a, 0x27, 0xf5, 0xcf, 0x81, 0xa6),
-	G(0x3c, 0x22, 0xee, 0xd2, 0x96, 0xb4), G(0x3e, 0x21, 0xe7, 0xd9, 0x9b, 0xba),
-	G(0x40, 0x60, 0x3b, 0x7b, 0xbb, 0xdb), G(0x42, 0x63, 0x32, 0x70, 0xb6, 0xd5),
-	G(0x44, 0x66, 0x29, 0x6d, 0xa1, 0xc7), G(0x46, 0x65, 0x20, 0x66, 0xac, 0xc9),
-	G(0x48, 0x6c, 0x1f, 0x57, 0x8f, 0xe3), G(0x4a, 0x6f, 0x16, 0x5c, 0x82, 0xed),
-	G(0x4c, 0x6a, 0x0d, 0x41, 0x95, 0xff), G(0x4e, 0x69, 0x04, 0x4a, 0x98, 0xf1),
-	G(0x50, 0x78, 0x73, 0x23, 0xd3, 0xab), G(0x52, 0x7b, 0x7a, 0x28, 0xde, 0xa5),
-	G(0x54, 0x7e, 0x61, 0x35, 0xc9, 0xb7), G(0x56, 0x7d, 0x68, 0x3e, 0xc4, 0xb9),
-	G(0x58, 0x74, 0x57, 0x0f, 0xe7, 0x93), G(0x5a, 0x77, 0x5e, 0x04, 0xea, 0x9d),
-	G(0x5c, 0x72, 0x45, 0x19, 0xfd, 0x8f), G(0x5e, 0x71, 0x4c, 0x12, 0xf0, 0x81),
-	G(0x60, 0x50, 0xab, 0xcb, 0x6b, 0x3b), G(0x62, 0x53, 0xa2, 0xc0, 0x66, 0x35),
-	G(0x64, 0x56, 0xb9, 0xdd, 0x71, 0x27), G(0x66, 0x55, 0xb0, 0xd6, 0x7c, 0x29),
-	G(0x68, 0x5c, 0x8f, 0xe7, 0x5f, 0x03), G(0x6a, 0x5f, 0x86, 0xec, 0x52, 0x0d),
-	G(0x6c, 0x5a, 0x9d, 0xf1, 0x45, 0x1f), G(0x6e, 0x59, 0x94, 0xfa, 0x48, 0x11),
-	G(0x70, 0x48, 0xe3, 0x93, 0x03, 0x4b), G(0x72, 0x4b, 0xea, 0x98, 0x0e, 0x45),
-	G(0x74, 0x4e, 0xf1, 0x85, 0x19, 0x57), G(0x76, 0x4d, 0xf8, 0x8e, 0x14, 0x59),
-	G(0x78, 0x44, 0xc7, 0xbf, 0x37, 0x73), G(0x7a, 0x47, 0xce, 0xb4, 0x3a, 0x7d),
-	G(0x7c, 0x42, 0xd5, 0xa9, 0x2d, 0x6f), G(0x7e, 0x41, 0xdc, 0xa2, 0x20, 0x61),
-	G(0x80, 0xc0, 0x76, 0xf6, 0x6d, 0xad), G(0x82, 0xc3, 0x7f, 0xfd, 0x60, 0xa3),
-	G(0x84, 0xc6, 0x64, 0xe0, 0x77, 0xb1), G(0x86, 0xc5, 0x6d, 0xeb, 0x7a, 0xbf),
-	G(0x88, 0xcc, 0x52, 0xda, 0x59, 0x95), G(0x8a, 0xcf, 0x5b, 0xd1, 0x54, 0x9b),
-	G(0x8c, 0xca, 0x40, 0xcc, 0x43, 0x89), G(0x8e, 0xc9, 0x49, 0xc7, 0x4e, 0x87),
-	G(0x90, 0xd8, 0x3e, 0xae, 0x05, 0xdd), G(0x92, 0xdb, 0x37, 0xa5, 0x08, 0xd3),
-	G(0x94, 0xde, 0x2c, 0xb8, 0x1f, 0xc1), G(0x96, 0xdd, 0x25, 0xb3, 0x12, 0xcf),
-	G(0x98, 0xd4, 0x1a, 0x82, 0x31, 0xe5), G(0x9a, 0xd7, 0x13, 0x89, 0x3c, 0xeb),
-	G(0x9c, 0xd2, 0x08, 0x94, 0x2b, 0xf9), G(0x9e, 0xd1, 0x01, 0x9f, 0x26, 0xf7),
-	G(0xa0, 0xf0, 0xe6, 0x46, 0xbd, 0x4d), G(0xa2, 0xf3, 0xef, 0x4d, 0xb0, 0x43),
-	G(0xa4, 0xf6, 0xf4, 0x50, 0xa7, 0x51), G(0xa6, 0xf5, 0xfd, 0x5b, 0xaa, 0x5f),
-	G(0xa8, 0xfc, 0xc2, 0x6a, 0x89, 0x75), G(0xaa, 0xff, 0xcb, 0x61, 0x84, 0x7b),
-	G(0xac, 0xfa, 0xd0, 0x7c, 0x93, 0x69), G(0xae, 0xf9, 0xd9, 0x77, 0x9e, 0x67),
-	G(0xb0, 0xe8, 0xae, 0x1e, 0xd5, 0x3d), G(0xb2, 0xeb, 0xa7, 0x15, 0xd8, 0x33),
-	G(0xb4, 0xee, 0xbc, 0x08, 0xcf, 0x21), G(0xb6, 0xed, 0xb5, 0x03, 0xc2, 0x2f),
-	G(0xb8, 0xe4, 0x8a, 0x32, 0xe1, 0x05), G(0xba, 0xe7, 0x83, 0x39, 0xec, 0x0b),
-	G(0xbc, 0xe2, 0x98, 0x24, 0xfb, 0x19), G(0xbe, 0xe1, 0x91, 0x2f, 0xf6, 0x17),
-	G(0xc0, 0xa0, 0x4d, 0x8d, 0xd6, 0x76), G(0xc2, 0xa3, 0x44, 0x86, 0xdb, 0x78),
-	G(0xc4, 0xa6, 0x5f, 0x9b, 0xcc, 0x6a), G(0xc6, 0xa5, 0x56, 0x90, 0xc1, 0x64),
-	G(0xc8, 0xac, 0x69, 0xa1, 0xe2, 0x4e), G(0xca, 0xaf, 0x60, 0xaa, 0xef, 0x40),
-	G(0xcc, 0xaa, 0x7b, 0xb7, 0xf8, 0x52), G(0xce, 0xa9, 0x72, 0xbc, 0xf5, 0x5c),
-	G(0xd0, 0xb8, 0x05, 0xd5, 0xbe, 0x06), G(0xd2, 0xbb, 0x0c, 0xde, 0xb3, 0x08),
-	G(0xd4, 0xbe, 0x17, 0xc3, 0xa4, 0x1a), G(0xd6, 0xbd, 0x1e, 0xc8, 0xa9, 0x14),
-	G(0xd8, 0xb4, 0x21, 0xf9, 0x8a, 0x3e), G(0xda, 0xb7, 0x28, 0xf2, 0x87, 0x30),
-	G(0xdc, 0xb2, 0x33, 0xef, 0x90, 0x22), G(0xde, 0xb1, 0x3a, 0xe4, 0x9d, 0x2c),
-	G(0xe0, 0x90, 0xdd, 0x3d, 0x06, 0x96), G(0xe2, 0x93, 0xd4, 0x36, 0x0b, 0x98),
-	G(0xe4, 0x96, 0xcf, 0x2b, 0x1c, 0x8a), G(0xe6, 0x95, 0xc6, 0x20, 0x11, 0x84),
-	G(0xe8, 0x9c, 0xf9, 0x11, 0x32, 0xae), G(0xea, 0x9f, 0xf0, 0x1a, 0x3f, 0xa0),
-	G(0xec, 0x9a, 0xeb, 0x07, 0x28, 0xb2), G(0xee, 0x99, 0xe2, 0x0c, 0x25, 0xbc),
-	G(0xf0, 0x88, 0x95, 0x65, 0x6e, 0xe6), G(0xf2, 0x8b, 0x9c, 0x6e, 0x63, 0xe8),
-	G(0xf4, 0x8e, 0x87, 0x73, 0x74, 0xfa), G(0xf6, 0x8d, 0x8e, 0x78, 0x79, 0xf4),
-	G(0xf8, 0x84, 0xb1, 0x49, 0x5a, 0xde), G(0xfa, 0x87, 0xb8, 0x42, 0x57, 0xd0),
-	G(0xfc, 0x82, 0xa3, 0x5f, 0x40, 0xc2), G(0xfe, 0x81, 0xaa, 0x54, 0x4d, 0xcc),
-	G(0x1b, 0x9b, 0xec, 0xf7, 0xda, 0x41), G(0x19, 0x98, 0xe5, 0xfc, 0xd7, 0x4f),
-	G(0x1f, 0x9d, 0xfe, 0xe1, 0xc0, 0x5d), G(0x1d, 0x9e, 0xf7, 0xea, 0xcd, 0x53),
-	G(0x13, 0x97, 0xc8, 0xdb, 0xee, 0x79), G(0x11, 0x94, 0xc1, 0xd0, 0xe3, 0x77),
-	G(0x17, 0x91, 0xda, 0xcd, 0xf4, 0x65), G(0x15, 0x92, 0xd3, 0xc6, 0xf9, 0x6b),
-	G(0x0b, 0x83, 0xa4, 0xaf, 0xb2, 0x31), G(0x09, 0x80, 0xad, 0xa4, 0xbf, 0x3f),
-	G(0x0f, 0x85, 0xb6, 0xb9, 0xa8, 0x2d), G(0x0d, 0x86, 0xbf, 0xb2, 0xa5, 0x23),
-	G(0x03, 0x8f, 0x80, 0x83, 0x86, 0x09), G(0x01, 0x8c, 0x89, 0x88, 0x8b, 0x07),
-	G(0x07, 0x89, 0x92, 0x95, 0x9c, 0x15), G(0x05, 0x8a, 0x9b, 0x9e, 0x91, 0x1b),
-	G(0x3b, 0xab, 0x7c, 0x47, 0x0a, 0xa1), G(0x39, 0xa8, 0x75, 0x4c, 0x07, 0xaf),
-	G(0x3f, 0xad, 0x6e, 0x51, 0x10, 0xbd), G(0x3d, 0xae, 0x67, 0x5a, 0x1d, 0xb3),
-	G(0x33, 0xa7, 0x58, 0x6b, 0x3e, 0x99), G(0x31, 0xa4, 0x51, 0x60, 0x33, 0x97),
-	G(0x37, 0xa1, 0x4a, 0x7d, 0x24, 0x85), G(0x35, 0xa2, 0x43, 0x76, 0x29, 0x8b),
-	G(0x2b, 0xb3, 0x34, 0x1f, 0x62, 0xd1), G(0x29, 0xb0, 0x3d, 0x14, 0x6f, 0xdf),
-	G(0x2f, 0xb5, 0x26, 0x09, 0x78, 0xcd), G(0x2d, 0xb6, 0x2f, 0x02, 0x75, 0xc3),
-	G(0x23, 0xbf, 0x10, 0x33, 0x56, 0xe9), G(0x21, 0xbc, 0x19, 0x38, 0x5b, 0xe7),
-	G(0x27, 0xb9, 0x02, 0x25, 0x4c, 0xf5), G(0x25, 0xba, 0x0b, 0x2e, 0x41, 0xfb),
-	G(0x5b, 0xfb, 0xd7, 0x8c, 0x61, 0x9a), G(0x59, 0xf8, 0xde, 0x87, 0x6c, 0x94),
-	G(0x5f, 0xfd, 0xc5, 0x9a, 0x7b, 0x86), G(0x5d, 0xfe, 0xcc, 0x91, 0x76, 0x88),
-	G(0x53, 0xf7, 0xf3, 0xa0, 0x55, 0xa2), G(0x51, 0xf4, 0xfa, 0xab, 0x58, 0xac),
-	G(0x57, 0xf1, 0xe1, 0xb6, 0x4f, 0xbe), G(0x55, 0xf2, 0xe8, 0xbd, 0x42, 0xb0),
-	G(0x4b, 0xe3, 0x9f, 0xd4, 0x09, 0xea), G(0x49, 0xe0, 0x96, 0xdf, 0x04, 0xe4),
-	G(0x4f, 0xe5, 0x8d, 0xc2, 0x13, 0xf6), G(0x4d, 0xe6, 0x84, 0xc9, 0x1e, 0xf8),
-	G(0x43, 0xef, 0xbb, 0xf8, 0x3d, 0xd2), G(0x41, 0xec, 0xb2, 0xf3, 0x30, 0xdc),
-	G(0x47, 0xe9, 0xa9, 0xee, 0x27, 0xce), G(0x45, 0xea, 0xa0, 0xe5, 0x2a, 0xc0),
-	G(0x7b, 0xcb, 0x47, 0x3c, 0xb1, 0x7a), G(0x79, 0xc8, 0x4e, 0x37, 0xbc, 0x74),
-	G(0x7f, 0xcd, 0x55, 0x2a, 0xab, 0x66), G(0x7d, 0xce, 0x5c, 0x21, 0xa6, 0x68),
-	G(0x73, 0xc7, 0x63, 0x10, 0x85, 0x42), G(0x71, 0xc4, 0x6a, 0x1b, 0x88, 0x4c),
-	G(0x77, 0xc1, 0x71, 0x06, 0x9f, 0x5e), G(0x75, 0xc2, 0x78, 0x0d, 0x92, 0x50),
-	G(0x6b, 0xd3, 0x0f, 0x64, 0xd9, 0x0a), G(0x69, 0xd0, 0x06, 0x6f, 0xd4, 0x04),
-	G(0x6f, 0xd5, 0x1d, 0x72, 0xc3, 0x16), G(0x6d, 0xd6, 0x14, 0x79, 0xce, 0x18),
-	G(0x63, 0xdf, 0x2b, 0x48, 0xed, 0x32), G(0x61, 0xdc, 0x22, 0x43, 0xe0, 0x3c),
-	G(0x67, 0xd9, 0x39, 0x5e, 0xf7, 0x2e), G(0x65, 0xda, 0x30, 0x55, 0xfa, 0x20),
-	G(0x9b, 0x5b, 0x9a, 0x01, 0xb7, 0xec), G(0x99, 0x58, 0x93, 0x0a, 0xba, 0xe2),
-	G(0x9f, 0x5d, 0x88, 0x17, 0xad, 0xf0), G(0x9d, 0x5e, 0x81, 0x1c, 0xa0, 0xfe),
-	G(0x93, 0x57, 0xbe, 0x2d, 0x83, 0xd4), G(0x91, 0x54, 0xb7, 0x26, 0x8e, 0xda),
-	G(0x97, 0x51, 0xac, 0x3b, 0x99, 0xc8), G(0x95, 0x52, 0xa5, 0x30, 0x94, 0xc6),
-	G(0x8b, 0x43, 0xd2, 0x59, 0xdf, 0x9c), G(0x89, 0x40, 0xdb, 0x52, 0xd2, 0x92),
-	G(0x8f, 0x45, 0xc0, 0x4f, 0xc5, 0x80), G(0x8d, 0x46, 0xc9, 0x44, 0xc8, 0x8e),
-	G(0x83, 0x4f, 0xf6, 0x75, 0xeb, 0xa4), G(0x81, 0x4c, 0xff, 0x7e, 0xe6, 0xaa),
-	G(0x87, 0x49, 0xe4, 0x63, 0xf1, 0xb8), G(0x85, 0x4a, 0xed, 0x68, 0xfc, 0xb6),
-	G(0xbb, 0x6b, 0x0a, 0xb1, 0x67, 0x0c), G(0xb9, 0x68, 0x03, 0xba, 0x6a, 0x02),
-	G(0xbf, 0x6d, 0x18, 0xa7, 0x7d, 0x10), G(0xbd, 0x6e, 0x11, 0xac, 0x70, 0x1e),
-	G(0xb3, 0x67, 0x2e, 0x9d, 0x53, 0x34), G(0xb1, 0x64, 0x27, 0x96, 0x5e, 0x3a),
-	G(0xb7, 0x61, 0x3c, 0x8b, 0x49, 0x28), G(0xb5, 0x62, 0x35, 0x80, 0x44, 0x26),
-	G(0xab, 0x73, 0x42, 0xe9, 0x0f, 0x7c), G(0xa9, 0x70, 0x4b, 0xe2, 0x02, 0x72),
-	G(0xaf, 0x75, 0x50, 0xff, 0x15, 0x60), G(0xad, 0x76, 0x59, 0xf4, 0x18, 0x6e),
-	G(0xa3, 0x7f, 0x66, 0xc5, 0x3b, 0x44), G(0xa1, 0x7c, 0x6f, 0xce, 0x36, 0x4a),
-	G(0xa7, 0x79, 0x74, 0xd3, 0x21, 0x58), G(0xa5, 0x7a, 0x7d, 0xd8, 0x2c, 0x56),
-	G(0xdb, 0x3b, 0xa1, 0x7a, 0x0c, 0x37), G(0xd9, 0x38, 0xa8, 0x71, 0x01, 0x39),
-	G(0xdf, 0x3d, 0xb3, 0x6c, 0x16, 0x2b), G(0xdd, 0x3e, 0xba, 0x67, 0x1b, 0x25),
-	G(0xd3, 0x37, 0x85, 0x56, 0x38, 0x0f), G(0xd1, 0x34, 0x8c, 0x5d, 0x35, 0x01),
-	G(0xd7, 0x31, 0x97, 0x40, 0x22, 0x13), G(0xd5, 0x32, 0x9e, 0x4b, 0x2f, 0x1d),
-	G(0xcb, 0x23, 0xe9, 0x22, 0x64, 0x47), G(0xc9, 0x20, 0xe0, 0x29, 0x69, 0x49),
-	G(0xcf, 0x25, 0xfb, 0x34, 0x7e, 0x5b), G(0xcd, 0x26, 0xf2, 0x3f, 0x73, 0x55),
-	G(0xc3, 0x2f, 0xcd, 0x0e, 0x50, 0x7f), G(0xc1, 0x2c, 0xc4, 0x05, 0x5d, 0x71),
-	G(0xc7, 0x29, 0xdf, 0x18, 0x4a, 0x63), G(0xc5, 0x2a, 0xd6, 0x13, 0x47, 0x6d),
-	G(0xfb, 0x0b, 0x31, 0xca, 0xdc, 0xd7), G(0xf9, 0x08, 0x38, 0xc1, 0xd1, 0xd9),
-	G(0xff, 0x0d, 0x23, 0xdc, 0xc6, 0xcb), G(0xfd, 0x0e, 0x2a, 0xd7, 0xcb, 0xc5),
-	G(0xf3, 0x07, 0x15, 0xe6, 0xe8, 0xef), G(0xf1, 0x04, 0x1c, 0xed, 0xe5, 0xe1),
-	G(0xf7, 0x01, 0x07, 0xf0, 0xf2, 0xf3), G(0xf5, 0x02, 0x0e, 0xfb, 0xff, 0xfd),
-	G(0xeb, 0x13, 0x79, 0x92, 0xb4, 0xa7), G(0xe9, 0x10, 0x70, 0x99, 0xb9, 0xa9),
-	G(0xef, 0x15, 0x6b, 0x84, 0xae, 0xbb), G(0xed, 0x16, 0x62, 0x8f, 0xa3, 0xb5),
-	G(0xe3, 0x1f, 0x5d, 0xbe, 0x80, 0x9f), G(0xe1, 0x1c, 0x54, 0xb5, 0x8d, 0x91),
-	G(0xe7, 0x19, 0x4f, 0xa8, 0x9a, 0x83), G(0xe5, 0x1a, 0x46, 0xa3, 0x97, 0x8d),
-};
-#undef G
+/*
+ * The schedule is kept in its compressed form, 30 words of it, inside the
+ * aes_key the caller already has -- comfortably inside the 448 bytes that
+ * held the round keys before.  br_aes_ct64_skey_expand blows it up to 120
+ * words on the stack once per call, which is how BearSSL's own CTR and CBC
+ * use it.
+ *
+ * It travels through memcpy rather than a cast because aes_key is all
+ * uint8_t and so carries no alignment of its own, whatever the allocator
+ * happens to give it.
+ */
+#define COMP_SKEY_WORDS 30
 
-static void expand_key(uint8_t *expandedKey, uint8_t *key, int size, size_t expandedKeySize)
+void crypton_aes_generic_schedule(aes_sched *sched, const aes_key *key)
 {
-	int csz;
-	int i;
-	uint8_t t[4] = { 0 };
-
-	for (i = 0; i < size; i++)
-		expandedKey[i] = key[i];
-	csz = size;
+	uint64_t comp_skey[COMP_SKEY_WORDS];
 
-	i = 1;
-	while (csz < expandedKeySize) {
-		t[0] = expandedKey[(csz - 4) + 0];
-		t[1] = expandedKey[(csz - 4) + 1];
-		t[2] = expandedKey[(csz - 4) + 2];
-		t[3] = expandedKey[(csz - 4) + 3];
+	memcpy(comp_skey, key->data, sizeof comp_skey);
+	sched->nbr = key->nbr;
+	br_aes_ct64_skey_expand(sched->sk_exp, sched->nbr, comp_skey);
+}
 
-		if (csz % size == 0) {
-			uint8_t tmp;
+/*
+ * Up to four blocks through the bitsliced core at once.  Fewer than four is
+ * the same work as four -- the lanes are there whether anything is in them
+ * -- so a caller with four to offer gets them for what one used to cost.
+ */
+static void pass(uint8_t *output, const uint8_t *input, unsigned n,
+                 const aes_sched *sched, int decrypt)
+{
+	uint32_t w[16];
+	uint64_t q[8];
+	unsigned i;
 
-			tmp = t[0];
-			t[0] = sbox[t[1]] ^ Rcon[i++ % sizeof(Rcon)];
-			t[1] = sbox[t[2]];
-			t[2] = sbox[t[3]];
-			t[3] = sbox[tmp];
-		}
+	memset(w, 0, sizeof w);
+	for (i = 0; i < n; i++) {
+		w[4 * i]     = br_dec32le(input + 16 * i);
+		w[4 * i + 1] = br_dec32le(input + 16 * i + 4);
+		w[4 * i + 2] = br_dec32le(input + 16 * i + 8);
+		w[4 * i + 3] = br_dec32le(input + 16 * i + 12);
+	}
 
-		if (size == 32 && ((csz % size) == 16)) {
-			t[0] = sbox[t[0]];
-			t[1] = sbox[t[1]];
-			t[2] = sbox[t[2]];
-			t[3] = sbox[t[3]];
-		}
+	for (i = 0; i < 4; i++)
+		br_aes_ct64_interleave_in(&q[i], &q[i + 4], w + 4 * i);
+	br_aes_ct64_ortho(q);
+	if (decrypt)
+		br_aes_ct64_bitslice_decrypt(sched->nbr, sched->sk_exp, q);
+	else
+		br_aes_ct64_bitslice_encrypt(sched->nbr, sched->sk_exp, q);
+	br_aes_ct64_ortho(q);
+	for (i = 0; i < 4; i++)
+		br_aes_ct64_interleave_out(w + 4 * i, q[i], q[i + 4]);
 
-		expandedKey[csz] = expandedKey[csz - size] ^ t[0]; csz++;
-		expandedKey[csz] = expandedKey[csz - size] ^ t[1]; csz++;
-		expandedKey[csz] = expandedKey[csz - size] ^ t[2]; csz++;
-		expandedKey[csz] = expandedKey[csz - size] ^ t[3]; csz++;
+	for (i = 0; i < n; i++) {
+		br_enc32le(output + 16 * i,      w[4 * i]);
+		br_enc32le(output + 16 * i + 4,  w[4 * i + 1]);
+		br_enc32le(output + 16 * i + 8,  w[4 * i + 2]);
+		br_enc32le(output + 16 * i + 12, w[4 * i + 3]);
 	}
 }
 
-static void shift_rows(uint8_t *state)
+void crypton_aes_generic_blocks(uint8_t *output, const uint8_t *input,
+                                uint32_t nb_blocks, const aes_sched *sched,
+                                int decrypt)
 {
-	uint32_t *s32;
-	int i;
-
-	for (i = 0; i < 16; i++)
-		state[i] = sbox[state[i]];
-	s32 = (uint32_t *) state;
-	s32[1] = rol32_be(s32[1], 8);
-	s32[2] = rol32_be(s32[2], 16);
-	s32[3] = rol32_be(s32[3], 24);
+	while (nb_blocks >= 4) {
+		pass(output, input, 4, sched, decrypt);
+		output += 64;
+		input += 64;
+		nb_blocks -= 4;
+	}
+	if (nb_blocks > 0)
+		pass(output, input, (unsigned) nb_blocks, sched, decrypt);
 }
 
-static void add_round_key(uint8_t *state, uint8_t *rk)
+static void one(aes_block *output, aes_key *key, aes_block *input, int decrypt)
 {
-	uint32_t *s32, *r32;
+	aes_sched sched;
 
-	s32 = (uint32_t *) state;
-	r32 = (uint32_t *) rk;
-	s32[0] ^= r32[0];
-	s32[1] ^= r32[1];
-	s32[2] ^= r32[2];
-	s32[3] ^= r32[3];
+	crypton_aes_generic_schedule(&sched, key);
+	crypton_aes_generic_blocks((uint8_t *) output, (const uint8_t *) input,
+	                           1, &sched, decrypt);
 }
 
-#define gm1(a) (a)
-#define gm2(a) gmtab[a][0]
-#define gm3(a) gmtab[a][1]
-#define gm9(a) gmtab[a][2]
-#define gm11(a) gmtab[a][3]
-#define gm13(a) gmtab[a][4]
-#define gm14(a) gmtab[a][5]
-
-static void mix_columns(uint8_t *state)
+void crypton_aes_generic_encrypt_block(aes_block *output, aes_key *key, aes_block *input)
 {
-	int i;
-	uint8_t cpy[4];
-
-	for (i = 0; i < 4; i++) {
-		cpy[0] = state[0 * 4 + i];
-		cpy[1] = state[1 * 4 + i];
-		cpy[2] = state[2 * 4 + i];
-		cpy[3] = state[3 * 4 + i];
-		state[i] = gm2(cpy[0]) ^ gm1(cpy[3]) ^ gm1(cpy[2]) ^ gm3(cpy[1]);
-		state[4+i] = gm2(cpy[1]) ^ gm1(cpy[0]) ^ gm1(cpy[3]) ^ gm3(cpy[2]);
-		state[8+i] = gm2(cpy[2]) ^ gm1(cpy[1]) ^ gm1(cpy[0]) ^ gm3(cpy[3]);
-		state[12+i] = gm2(cpy[3]) ^ gm1(cpy[2]) ^ gm1(cpy[1]) ^ gm3(cpy[0]);
-	}
+	one(output, key, input, 0);
 }
 
-static void create_round_key(uint8_t *expandedKey, uint8_t *rk)
+void crypton_aes_generic_decrypt_block(aes_block *output, aes_key *key, aes_block *input)
 {
-	int i,j;
-	for (i = 0; i < 4; i++)
-		for (j = 0; j < 4; j++)
-			rk[i + j * 4] = expandedKey[i * 4 + j];
+	one(output, key, input, 1);
 }
 
-static void aes_main(aes_key *key, uint8_t *state)
+void crypton_aes_generic_init(aes_key *key, uint8_t *origkey, uint8_t size)
 {
-	int i = 0;
-	uint32_t rk[4];
-	uint8_t *rkptr = (uint8_t *) rk;
-
-	create_round_key(key->data, rkptr);
-	add_round_key(state, rkptr);
-
-	for (i = 1; i < key->nbr; i++) {
-		create_round_key(key->data + 16 * i, rkptr);
-		shift_rows(state);
-		mix_columns(state);
-		add_round_key(state, rkptr);
-	}
-
-	create_round_key(key->data + 16 * key->nbr, rkptr);
-	shift_rows(state);
-	add_round_key(state, rkptr);
-}
+	uint64_t comp_skey[COMP_SKEY_WORDS];
+	unsigned nbr;
 
-static void shift_rows_inv(uint8_t *state)
-{
-	uint32_t *s32;
-	int i;
+	/* 0 for a key length that is not 16, 24 or 32; the old code returned
+	 * without touching the key in that case and so does this */
+	nbr = br_aes_ct64_keysched(comp_skey, origkey, size);
+	if (nbr == 0)
+		return;
 
-	s32 = (uint32_t *) state;
-	s32[1] = ror32_be(s32[1], 8);
-	s32[2] = ror32_be(s32[2], 16);
-	s32[3] = ror32_be(s32[3], 24);
-	for (i = 0; i < 16; i++)
-		state[i] = rsbox[state[i]];
+	key->nbr = (uint8_t) nbr;
+	memcpy(key->data, comp_skey, sizeof comp_skey);
 }
 
-static void mix_columns_inv(uint8_t *state)
+/*
+ * CTR, four counter blocks at a time.  The counter itself is serial, but
+ * nothing about it depends on the keystream, so the four blocks it will
+ * reach next can be written down before any of them is encrypted.
+ */
+static void ctr(uint8_t *output, aes_key *key, aes_block *iv,
+                uint8_t *input, uint32_t len, int c32)
 {
-	int i;
-	uint8_t cpy[4];
+	aes_sched sched;
+	aes_block counter;
+	uint8_t ks[64];
+	uint32_t nb_blocks = len / 16;
+	uint32_t tail = len % 16;
+	uint32_t i;
 
-	for (i = 0; i < 4; i++) {
-		cpy[0] = state[0 * 4 + i];
-		cpy[1] = state[1 * 4 + i];
-		cpy[2] = state[2 * 4 + i];
-		cpy[3] = state[3 * 4 + i];
-		state[i] = gm14(cpy[0]) ^ gm9(cpy[3]) ^ gm13(cpy[2]) ^ gm11(cpy[1]);
-		state[4+i] = gm14(cpy[1]) ^ gm9(cpy[0]) ^ gm13(cpy[3]) ^ gm11(cpy[2]);
-		state[8+i] = gm14(cpy[2]) ^ gm9(cpy[1]) ^ gm13(cpy[0]) ^ gm11(cpy[3]);
-		state[12+i] = gm14(cpy[3]) ^ gm9(cpy[2]) ^ gm13(cpy[1]) ^ gm11(cpy[0]);
-	}
-}
+	crypton_aes_generic_schedule(&sched, key);
+	block128_copy(&counter, iv);
 
-static void aes_main_inv(aes_key *key, uint8_t *state)
-{
-	int i = 0;
-	uint32_t rk[4];
-	uint8_t *rkptr = (uint8_t *) rk;
+	while (nb_blocks > 0) {
+		uint32_t n = nb_blocks < 4 ? nb_blocks : 4;
 
-	create_round_key(key->data + 16 * key->nbr, rkptr);
-	add_round_key(state, rkptr);
+		for (i = 0; i < n; i++) {
+			block128_copy((block128 *) (ks + 16 * i), &counter);
+			if (c32)
+				block128_inc32_le(&counter);
+			else
+				block128_inc_be(&counter);
+		}
+		crypton_aes_generic_blocks(ks, ks, n, &sched, 0);
+		for (i = 0; i < n * 16; i++)
+			output[i] = ks[i] ^ input[i];
 
-	for (i = key->nbr - 1; i > 0; i--) {
-		create_round_key(key->data + 16 * i, rkptr);
-		shift_rows_inv(state);
-		add_round_key(state, rkptr);
-		mix_columns_inv(state);
+		output += n * 16;
+		input += n * 16;
+		nb_blocks -= n;
 	}
 
-	create_round_key(key->data, rkptr);
-	shift_rows_inv(state);
-	add_round_key(state, rkptr);
-}
-
-/* Set the block values, for the block:
- * a0,0 a0,1 a0,2 a0,3
- * a1,0 a1,1 a1,2 a1,3 -> a0,0 a1,0 a2,0 a3,0 a0,1 a1,1 ... a2,3 a3,3
- * a2,0 a2,1 a2,2 a2,3
- * a3,0 a3,1 a3,2 a3,3
- */
-#define swap_block(t, f) \
-	t[0] = f[0]; t[4] = f[1]; t[8] = f[2]; t[12] = f[3]; \
-	t[1] = f[4]; t[5] = f[5]; t[9] = f[6]; t[13] = f[7]; \
-	t[2] = f[8]; t[6] = f[9]; t[10] = f[10]; t[14] = f[11]; \
-	t[3] = f[12]; t[7] = f[13]; t[11] = f[14]; t[15] = f[15]
-
-void crypton_aes_generic_encrypt_block(aes_block *output, aes_key *key, aes_block *input)
-{
-	uint32_t block[4];
-	uint8_t *iptr, *optr, *bptr;
-
-	iptr = (uint8_t *) input;
-	optr = (uint8_t *) output;
-	bptr = (uint8_t *) block;
-	swap_block(bptr, iptr);
-	aes_main(key, bptr);
-	swap_block(optr, bptr);
+	if (tail != 0) {
+		block128_copy((block128 *) ks, &counter);
+		crypton_aes_generic_blocks(ks, ks, 1, &sched, 0);
+		for (i = 0; i < tail; i++)
+			output[i] = ks[i] ^ input[i];
+	}
 }
 
-void crypton_aes_generic_decrypt_block(aes_block *output, aes_key *key, aes_block *input)
+void crypton_aes_bitsliced_encrypt_ctr(uint8_t *output, aes_key *key,
+                                       aes_block *iv, uint8_t *input,
+                                       uint32_t len)
 {
-	uint32_t block[4];
-	uint8_t *iptr, *optr, *bptr;
-
-	iptr = (uint8_t *) input;
-	optr = (uint8_t *) output;
-	bptr = (uint8_t *) block;
-	swap_block(bptr, iptr);
-	aes_main_inv(key, bptr);
-	swap_block(optr, bptr);
+	ctr(output, key, iv, input, len, 0);
 }
 
-void crypton_aes_generic_init(aes_key *key, uint8_t *origkey, uint8_t size)
+void crypton_aes_bitsliced_encrypt_c32(uint8_t *output, aes_key *key,
+                                       aes_block *iv, uint8_t *input,
+                                       uint32_t len)
 {
-	int esz;
-
-	switch (size) {
-	case 16: key->nbr = 10; esz = 176; break;
-	case 24: key->nbr = 12; esz = 208; break;
-	case 32: key->nbr = 14; esz = 240; break;
-	default: return;
-	}
-	expand_key(key->data, origkey, size, esz);
-	return;
+	ctr(output, key, iv, input, len, 1);
 }
diff --git a/cbits/aes/generic.h b/cbits/aes/generic.h
--- a/cbits/aes/generic.h
+++ b/cbits/aes/generic.h
@@ -32,3 +32,35 @@
 void crypton_aes_generic_encrypt_block(aes_block *output, aes_key *key, aes_block *input);
 void crypton_aes_generic_decrypt_block(aes_block *output, aes_key *key, aes_block *input);
 void crypton_aes_generic_init(aes_key *key, uint8_t *origkey, uint8_t size);
+
+/*
+ * The bitsliced core takes four blocks at a time, and the schedule it reads
+ * is the expanded one rather than the compressed form the aes_key holds.
+ * Expanding costs about what a block costs, so a mode with more than one
+ * block to do expands once, here, and hands the result to every group.
+ */
+typedef struct {
+	uint64_t sk_exp[120];
+	unsigned nbr;
+} aes_sched;
+
+void crypton_aes_generic_schedule(aes_sched *sched, const aes_key *key);
+
+/* nb_blocks of them, four at a pass; decrypt selects the direction */
+void crypton_aes_generic_blocks(uint8_t *output, const uint8_t *input,
+                                uint32_t nb_blocks, const aes_sched *sched,
+                                int decrypt);
+
+/*
+ * CTR with the two counters crypton uses.  These are not the generic
+ * entries: those go through the branch table for the block itself and so
+ * run on accelerated machines too, where the key holds a different
+ * schedule.  crypton_aes.c installs these only when nothing was
+ * accelerated.
+ */
+void crypton_aes_bitsliced_encrypt_ctr(uint8_t *output, aes_key *key,
+                                       aes_block *iv, uint8_t *input,
+                                       uint32_t len);
+void crypton_aes_bitsliced_encrypt_c32(uint8_t *output, aes_key *key,
+                                       aes_block *iv, uint8_t *input,
+                                       uint32_t len);
diff --git a/cbits/aes/gf.c b/cbits/aes/gf.c
--- a/cbits/aes/gf.c
+++ b/cbits/aes/gf.c
@@ -33,6 +33,7 @@
 #include <crypton_cpu.h>
 #include <aes/gf.h>
 #include <aes/x86ni.h>
+#include "bearssl/inner.h"
 
 /* inplace GFMUL for xts mode */
 void crypton_aes_generic_gf_mulx(block128 *a)
@@ -45,118 +46,43 @@
 
 
 /*
- * GF multiplication with Shoup's method and 4-bit table.
+ * GHASH, without a table.
  *
- * We precompute the products of H with all 4-bit polynomials and store them in
- * a 'table_4bit' array.  To avoid unnecessary byte swapping, the 16 blocks are
- * written to the table with qwords already converted to CPU order.  Table
- * indices use the reflected bit ordering, i.e. polynomials X^0, X^1, X^2, X^3
- * map to bit positions 3, 2, 1, 0 respectively.
+ * This was Shoup's method: the products of H with all sixteen 4-bit
+ * polynomials, precomputed, and thirty-two lookups a block at indices taken
+ * from the accumulator -- which is to say, thirty-two addresses derived from
+ * a secret.  It is now BearSSL's ghash_ctmul64, which builds the GF(2^128)
+ * multiply out of shifts, masks and integer multiplies and looks nothing up.
+ * See cbits/bearssl/README.md.
  *
- * To multiply an arbitrary block with H, the input block is decomposed in 4-bit
- * segments.  We get the final result after 32 table lookups and additions, one
- * for each segment, interleaving multiplication by P(X)=X^4.
+ * The table_4bit the interface names is sixteen blocks wide because the
+ * table needed it.  This keeps H in the first and leaves the rest alone; the
+ * PMULL and PCLMUL implementations keep their own powers of H in that same
+ * space, so the width stays as it is.
  */
 
-/* convert block128 qwords between BE and CPU order */
-static inline void block128_cpu_swap_be(block128 *a, const block128 *b)
-{
-	a->q[1] = cpu_to_be64(b->q[1]);
-	a->q[0] = cpu_to_be64(b->q[0]);
-}
-
-/* multiplication by P(X)=X, assuming qwords already in CPU order */
-static inline void cpu_gf_mulx(block128 *a, const block128 *b)
-{
-	uint64_t v0 = b->q[0];
-	uint64_t v1 = b->q[1];
-	a->q[1] = v1 >> 1 | v0 << 63;
-	a->q[0] = v0 >> 1 ^ ((0-(v1 & 1)) & 0xe100000000000000ULL);
-}
-
-static const uint64_t r4_0[] =
-	{ 0x0000000000000000ULL, 0x1c20000000000000ULL
-	, 0x3840000000000000ULL, 0x2460000000000000ULL
-	, 0x7080000000000000ULL, 0x6ca0000000000000ULL
-	, 0x48c0000000000000ULL, 0x54e0000000000000ULL
-	, 0xe100000000000000ULL, 0xfd20000000000000ULL
-	, 0xd940000000000000ULL, 0xc560000000000000ULL
-	, 0x9180000000000000ULL, 0x8da0000000000000ULL
-	, 0xa9c0000000000000ULL, 0xb5e0000000000000ULL
-	};
-
-/* multiplication by P(X)=X^4, assuming qwords already in CPU order */
-static inline void cpu_gf_mulx4(block128 *a, const block128 *b)
-{
-	uint64_t v0 = b->q[0];
-	uint64_t v1 = b->q[1];
-	a->q[1] = v1 >> 4 | v0 << 60;
-	a->q[0] = v0 >> 4 ^ r4_0[v1 & 0xf];
-}
-
-/* initialize the 4-bit table given H */
+/* remember H */
 void crypton_aes_generic_hinit(table_4bit htable, const block128 *h)
 {
-	block128 v, *p;
-	int i, j;
-
-	/* multiplication by 0 is 0 */
-	block128_zero(&htable[0]);
-
-	/* at index 8=2^3 we have H.X^0 = H */
-	i = 8;
-	block128_cpu_swap_be(&htable[i], h); /* in CPU order */
-	p = &htable[i];
-
-	/* for other powers of 2, repeat multiplication by P(X)=X */
-	for (i = 4; i > 0; i >>= 1)
-	{
-		cpu_gf_mulx(&htable[i], p);
-		p = &htable[i];
-	}
-
-	/* remaining elements are linear combinations */
-	for (i = 2; i < 16; i <<= 1) {
-		p = &htable[i];
-		v = *p;
-		for (j = 1; j < i; j++) {
-			p[j] = v;
-			block128_xor_aligned(&p[j], &htable[j]);
-		}
-	}
+	block128_copy(&htable[0], h);
 }
 
-/* multiply a block with H */
+/*
+ * br_ghash_ctmul64 computes y = (y ^ x) * H for each block x it is given, so
+ * a block of zeros is the bare multiply this entry is asked for.
+ */
 void crypton_aes_generic_gf_mul(block128 *a, const table_4bit htable)
 {
-	block128 b;
-	int i;
-	block128_zero(&b);
-	for (i = 15; i >= 0; i--)
-	{
-		uint8_t v = a->b[i];
-		block128_xor_aligned(&b, &htable[v & 0xf]); /* high bits (reflected) */
-		cpu_gf_mulx4(&b, &b);
-		block128_xor_aligned(&b, &htable[v >> 4]);  /* low bits (reflected) */
-		if (i > 0)
-			cpu_gf_mulx4(&b, &b);
-		else
-			block128_cpu_swap_be(a, &b); /* restore BE order when done */
-	}
+	static const uint8_t zero[16] = { 0 };
+
+	br_ghash_ctmul64(a, &htable[0], zero, sizeof zero);
 }
 
 /*
- * Four GHASH steps at once.  The generic table-driven multiply has no cheaper
- * way to do this than one block at a time; the point of the entry is that the
- * PMULL and PCLMUL versions can fold the four products into one reduction, so
- * the GCM loops hand over four blocks whenever they have them.
+ * Four GHASH steps at once, which here is one call rather than four: the
+ * loop inside br_ghash_ctmul64 takes the blocks as they come.
  */
 void crypton_aes_generic_gf_mul4(block128 *a, const block128 *blocks, const table_4bit htable)
 {
-	int i;
-
-	for (i = 0; i < 4; i++) {
-		block128_xor(a, &blocks[i]);
-		crypton_aes_generic_gf_mul(a, htable);
-	}
+	br_ghash_ctmul64(a, &htable[0], blocks, 4 * sizeof(block128));
 }
diff --git a/cbits/aes/ppc8.c b/cbits/aes/ppc8.c
new file mode 100644
--- /dev/null
+++ b/cbits/aes/ppc8.c
@@ -0,0 +1,287 @@
+/*
+ * AES and GHASH using the PowerISA 2.07 vector instructions, first
+ * implemented by POWER8.
+ *
+ * The instructions themselves come from CRYPTOGAMS, assembled from
+ * cbits/asm/aesp8-ppc-*.S and cbits/asm/ghashp8-ppc-*.S; what is here is the
+ * glue that puts them behind crypton's branch table.  Nothing in this file
+ * branches or indexes on a key or on data.
+ *
+ * == Where the key schedule lives
+ *
+ * The assembly takes OpenSSL's AES_KEY -- sixty round-key words and a round
+ * count, 244 bytes -- and its set_decrypt_key writes a second, complete one
+ * rather than sharing ends with the first the way cbits/aes/armv8.c does.
+ * Two of those are 488 bytes and aes_key.data is 448, so both do not fit.
+ *
+ * So the forward schedule is kept there, with the key the caller gave after
+ * it, and the inverse is built on the stack by the operations that need it.
+ * Those are all bulk -- ECB, CBC and XTS decryption -- so it is one key
+ * schedule per call rather than per block.  Single-block decryption pays for
+ * one too, and is reached by nothing that runs in a loop: OCB and CCM drive
+ * the block function in the encrypting direction.
+ */
+
+#include <stdint.h>
+#include <string.h>
+#include <crypton_aes.h>
+#include <crypton_bitfn.h>
+#include "aes/block128.h"
+#include "aes/gf.h"
+#include "aes/ppc8.h"
+
+/* OpenSSL's AES_KEY, which is what the assembly was written against */
+typedef struct {
+	unsigned int rd_key[60];
+	int rounds;
+} p8_key;
+
+int  crypton_aes_p8_set_encrypt_key(const unsigned char *, int, p8_key *);
+int  crypton_aes_p8_set_decrypt_key(const unsigned char *, int, p8_key *);
+void crypton_aes_p8_encrypt(const unsigned char *, unsigned char *, const p8_key *);
+void crypton_aes_p8_decrypt(const unsigned char *, unsigned char *, const p8_key *);
+void crypton_aes_p8_cbc_encrypt(const unsigned char *, unsigned char *, size_t,
+                                const p8_key *, unsigned char *, int);
+void crypton_aes_p8_ctr32_encrypt_blocks(const unsigned char *, unsigned char *,
+                                         size_t, const p8_key *,
+                                         const unsigned char *);
+void crypton_aes_p8_xts_encrypt(const unsigned char *, unsigned char *, size_t,
+                                const p8_key *, const p8_key *,
+                                const unsigned char *);
+void crypton_aes_p8_xts_decrypt(const unsigned char *, unsigned char *, size_t,
+                                const p8_key *, const p8_key *,
+                                const unsigned char *);
+/* void * rather than uint64_t *: the assembly loads these with lvx_u and so
+ * does not want the alignment a uint64_t pointer would promise, and
+ * block128 is packed. */
+void crypton_gcm_init_p8(void *Htable, const void *H);
+void crypton_gcm_gmult_p8(void *Xi, const void *Htable);
+void crypton_gcm_ghash_p8(void *Xi, const void *Htable,
+                          const unsigned char *inp, size_t len);
+
+#define FORWARD(k)  ((p8_key *) (void *) (k)->data)
+#define USERKEY(k)  ((uint8_t *) (k)->data + sizeof(p8_key))
+#define USERLEN(k)  (*((uint8_t *) (k)->data + sizeof(p8_key) + 32))
+
+static void inverse(p8_key *dk, aes_key *key)
+{
+	crypton_aes_p8_set_decrypt_key(USERKEY(key), USERLEN(key) * 8, dk);
+}
+
+void crypton_aes_ppc8_init(aes_key *key, uint8_t *origkey, uint8_t size)
+{
+	if (size != 16 && size != 24 && size != 32)
+		return;
+	crypton_aes_p8_set_encrypt_key(origkey, size * 8, FORWARD(key));
+	memcpy(USERKEY(key), origkey, size);
+	USERLEN(key) = size;
+}
+
+void crypton_aes_ppc8_encrypt_block(aes_block *output, aes_key *key, aes_block *input)
+{
+	crypton_aes_p8_encrypt((const unsigned char *) input,
+	                       (unsigned char *) output, FORWARD(key));
+}
+
+void crypton_aes_ppc8_decrypt_block(aes_block *output, aes_key *key, aes_block *input)
+{
+	p8_key dk;
+
+	inverse(&dk, key);
+	crypton_aes_p8_decrypt((const unsigned char *) input,
+	                       (unsigned char *) output, &dk);
+	memset(&dk, 0, sizeof dk);
+}
+
+/* The assembly has no ECB entry: it is the block function in a loop, which
+ * is what the generic implementation does too. */
+void crypton_aes_ppc8_encrypt_ecb(aes_block *output, aes_key *key, aes_block *input,
+                                  uint32_t nb_blocks)
+{
+	for (; nb_blocks-- > 0; input++, output++)
+		crypton_aes_p8_encrypt((const unsigned char *) input,
+		                       (unsigned char *) output, FORWARD(key));
+}
+
+void crypton_aes_ppc8_decrypt_ecb(aes_block *output, aes_key *key, aes_block *input,
+                                  uint32_t nb_blocks)
+{
+	p8_key dk;
+
+	inverse(&dk, key);
+	for (; nb_blocks-- > 0; input++, output++)
+		crypton_aes_p8_decrypt((const unsigned char *) input,
+		                       (unsigned char *) output, &dk);
+	memset(&dk, 0, sizeof dk);
+}
+
+void crypton_aes_ppc8_encrypt_cbc(aes_block *output, aes_key *key, aes_block *iv,
+                                  aes_block *input, uint32_t nb_blocks)
+{
+	uint8_t ivbuf[16];
+
+	/* a copy, because the assembly writes the last block back through this
+	 * pointer and the callers of this entry do not expect their IV touched */
+	memcpy(ivbuf, iv, 16);
+	crypton_aes_p8_cbc_encrypt((const unsigned char *) input,
+	                           (unsigned char *) output,
+	                           (size_t) nb_blocks * 16, FORWARD(key), ivbuf, 1);
+}
+
+void crypton_aes_ppc8_decrypt_cbc(aes_block *output, aes_key *key, aes_block *iv,
+                                  aes_block *input, uint32_t nb_blocks)
+{
+	p8_key dk;
+	uint8_t ivbuf[16];
+
+	inverse(&dk, key);
+	memcpy(ivbuf, iv, 16);
+	crypton_aes_p8_cbc_encrypt((const unsigned char *) input,
+	                           (unsigned char *) output,
+	                           (size_t) nb_blocks * 16, &dk, ivbuf, 0);
+	memset(&dk, 0, sizeof dk);
+}
+
+/*
+ * CTR, which is where the two counters have to be reconciled.
+ *
+ * crypton counts with block128_inc_be, over the whole 128 bits; the assembly
+ * counts over the low 32 only.  They agree until that word wraps, so the
+ * work is handed over in runs that stop there, and the carry into the upper
+ * bits is done here.  This is the same arrangement OpenSSL makes for the
+ * same reason.
+ */
+void crypton_aes_ppc8_encrypt_ctr(uint8_t *output, aes_key *key, aes_block *iv,
+                                  uint8_t *input, uint32_t len)
+{
+	block128 ctr;
+	uint32_t nb_blocks = len / 16;
+	uint32_t tail = len % 16;
+	uint32_t i;
+
+	block128_copy(&ctr, iv);
+
+	while (nb_blocks > 0) {
+		uint64_t low = (uint64_t) be32_to_cpu(ctr.d[3]);
+		uint64_t room = 0x100000000ULL - low;   /* blocks before it wraps */
+		uint32_t n;
+
+		/* room is 2^32 when the word is at zero, which does not fit the
+		 * type the comparison would narrow it to -- and a zero n here
+		 * would not advance */
+		if (room > (uint64_t) nb_blocks)
+			room = (uint64_t) nb_blocks;
+		n = (uint32_t) room;
+
+		crypton_aes_p8_ctr32_encrypt_blocks((const unsigned char *) input,
+		                                    (unsigned char *) output,
+		                                    n, FORWARD(key),
+		                                    (const unsigned char *) &ctr);
+		/* the assembly does not write the counter back */
+		for (i = 0; i < n; i++)
+			block128_inc_be(&ctr);
+
+		output += (size_t) n * 16;
+		input += (size_t) n * 16;
+		nb_blocks -= n;
+	}
+
+	if (tail != 0) {
+		block128 ks;
+
+		crypton_aes_p8_encrypt((const unsigned char *) &ctr,
+		                       (unsigned char *) &ks, FORWARD(key));
+		for (i = 0; i < tail; i++)
+			output[i] = ks.b[i] ^ input[i];
+	}
+}
+
+/*
+ * XTS.  The assembly encrypts the tweak with its second key, unless that key
+ * is NULL, in which case it takes the tweak already encrypted -- which is
+ * what this needs, because crypton's entry also carries a starting point and
+ * the tweak has to be advanced that many doublings before any block is
+ * enciphered.
+ */
+static void xts_tweak(block128 *tweak, aes_key *k2, aes_block *dataunit,
+                      uint32_t spoint)
+{
+	block128_copy(tweak, dataunit);
+	crypton_aes_p8_encrypt((const unsigned char *) tweak,
+	                       (unsigned char *) tweak, FORWARD(k2));
+	while (spoint-- > 0)
+		crypton_aes_generic_gf_mulx(tweak);
+}
+
+void crypton_aes_ppc8_encrypt_xts(aes_block *output, aes_key *k1, aes_key *k2,
+                                  aes_block *dataunit, uint32_t spoint,
+                                  aes_block *input, uint32_t nb_blocks)
+{
+	block128 tweak;
+
+	xts_tweak(&tweak, k2, dataunit, spoint);
+	crypton_aes_p8_xts_encrypt((const unsigned char *) input,
+	                           (unsigned char *) output,
+	                           (size_t) nb_blocks * 16, FORWARD(k1), NULL,
+	                           (const unsigned char *) &tweak);
+}
+
+void crypton_aes_ppc8_decrypt_xts(aes_block *output, aes_key *k1, aes_key *k2,
+                                  aes_block *dataunit, uint32_t spoint,
+                                  aes_block *input, uint32_t nb_blocks)
+{
+	block128 tweak;
+	p8_key dk;
+
+	/* the tweak is enciphered with k2 forwards either way */
+	xts_tweak(&tweak, k2, dataunit, spoint);
+	inverse(&dk, k1);
+	crypton_aes_p8_xts_decrypt((const unsigned char *) input,
+	                           (unsigned char *) output,
+	                           (size_t) nb_blocks * 16, &dk, NULL,
+	                           (const unsigned char *) &tweak);
+	memset(&dk, 0, sizeof dk);
+}
+
+/*
+ * GHASH.  The table the assembly builds is 192 bytes, which fits the sixteen
+ * blocks the interface names.
+ *
+ * The two arguments do not take the same convention, which is worth saying
+ * because it does not look like an accident until it is checked.  H arrives
+ * here as the sixteen bytes GCM defines, and gcm_init_p8 wants those as a
+ * pair of host-order words -- so they are swapped on a little-endian machine
+ * and left alone on a big-endian one, which be64_to_cpu does.  The
+ * accumulator is the byte string throughout, in and out, and is passed as it
+ * stands.  OpenSSL makes the same pair of choices.
+ *
+ * Checked rather than assumed: against crypton's own GHASH, over two
+ * different H and with a non-zero accumulator, so that neither a symmetric
+ * value nor a zero start could hide a wrong one.
+ */
+void crypton_aes_ppc8_hinit(table_4bit htable, const block128 *h)
+{
+	uint64_t H[2];
+
+	memcpy(H, h, sizeof H);
+	H[0] = be64_to_cpu(H[0]);
+	H[1] = be64_to_cpu(H[1]);
+	crypton_gcm_init_p8(htable, H);
+}
+
+void crypton_aes_ppc8_gf_mul(block128 *a, const table_4bit htable)
+{
+	crypton_gcm_gmult_p8(a, htable);
+}
+
+void crypton_aes_ppc8_gf_mul4(block128 *a, const block128 *blocks,
+                              const table_4bit htable)
+{
+	crypton_gcm_ghash_p8(a, htable, (const unsigned char *) blocks,
+	                     4 * sizeof(block128));
+}
+
+int crypton_aes_ppc8_available(void)
+{
+	return (crypton_ppc_features() & CRYPTON_PPC_VCRYPTO) != 0;
+}
diff --git a/cbits/aes/ppc8.h b/cbits/aes/ppc8.h
new file mode 100644
--- /dev/null
+++ b/cbits/aes/ppc8.h
@@ -0,0 +1,29 @@
+/*
+ * AES and GHASH on PowerISA 2.07, as implemented by cbits/aes/ppc8.c over
+ * the CRYPTOGAMS assembly.  crypton_aes.c installs these when
+ * crypton_aes_ppc8_available says the processor has the instructions.
+ */
+#ifndef CRYPTON_AES_PPC8_H
+#define CRYPTON_AES_PPC8_H
+
+#include "crypton_aes.h"
+#include "aes/gf.h"
+#include "crypton_cpu.h"
+
+int  crypton_aes_ppc8_available(void);
+
+void crypton_aes_ppc8_init(aes_key *key, uint8_t *origkey, uint8_t size);
+void crypton_aes_ppc8_encrypt_block(aes_block *output, aes_key *key, aes_block *input);
+void crypton_aes_ppc8_decrypt_block(aes_block *output, aes_key *key, aes_block *input);
+void crypton_aes_ppc8_encrypt_ecb(aes_block *output, aes_key *key, aes_block *input, uint32_t nb_blocks);
+void crypton_aes_ppc8_decrypt_ecb(aes_block *output, aes_key *key, aes_block *input, uint32_t nb_blocks);
+void crypton_aes_ppc8_encrypt_cbc(aes_block *output, aes_key *key, aes_block *iv, aes_block *input, uint32_t nb_blocks);
+void crypton_aes_ppc8_decrypt_cbc(aes_block *output, aes_key *key, aes_block *iv, aes_block *input, uint32_t nb_blocks);
+void crypton_aes_ppc8_encrypt_ctr(uint8_t *output, aes_key *key, aes_block *iv, uint8_t *input, uint32_t len);
+void crypton_aes_ppc8_encrypt_xts(aes_block *output, aes_key *k1, aes_key *k2, aes_block *dataunit, uint32_t spoint, aes_block *input, uint32_t nb_blocks);
+void crypton_aes_ppc8_decrypt_xts(aes_block *output, aes_key *k1, aes_key *k2, aes_block *dataunit, uint32_t spoint, aes_block *input, uint32_t nb_blocks);
+void crypton_aes_ppc8_hinit(table_4bit htable, const block128 *h);
+void crypton_aes_ppc8_gf_mul(block128 *a, const table_4bit htable);
+void crypton_aes_ppc8_gf_mul4(block128 *a, const block128 *blocks, const table_4bit htable);
+
+#endif
diff --git a/cbits/asm/README.md b/cbits/asm/README.md
--- a/cbits/asm/README.md
+++ b/cbits/asm/README.md
@@ -2,7 +2,7 @@
 
 ## What is here
 
-Two modules from [CRYPTOGAMS](https://github.com/dot-asm/cryptogams), by Andy
+Modules from [CRYPTOGAMS](https://github.com/dot-asm/cryptogams), by Andy
 Polyakov, checked in unmodified together with the translators they need:
 
 | generator | what it is |
@@ -17,6 +17,8 @@
 | `sha1-armv8.pl` | SHA-1 for AArch64 |
 | `sha512-armv8.pl` | SHA-256 for AArch64 (the generator emits SHA-512 or SHA-256 according to the name it is given, and only the latter is wanted) |
 | `keccak1600-armv8.pl` | Keccak for AArch64 |
+| `aesp8-ppc.pl` | AES for PowerISA 2.07, which POWER8 was the first to implement |
+| `ghashp8-ppc.pl` | GHASH for the same, over `vpmsumd` |
 
 `x86_64-xlate.pl`, `arm-xlate.pl` and `arm_arch.h` are the machinery those
 modules use.  `generate.sh` runs the generators to produce the `.S` files, which are
diff --git a/cbits/asm/aesp8-ppc-linux64le.S b/cbits/asm/aesp8-ppc-linux64le.S
new file mode 100644
--- /dev/null
+++ b/cbits/asm/aesp8-ppc-linux64le.S
@@ -0,0 +1,3658 @@
+.machine	"any"
+
+.abiversion	2
+.text
+
+.align	7
+rcon:
+.byte	0x00,0x00,0x00,0x01,0x00,0x00,0x00,0x01,0x00,0x00,0x00,0x01,0x00,0x00,0x00,0x01
+.byte	0x00,0x00,0x00,0x1b,0x00,0x00,0x00,0x1b,0x00,0x00,0x00,0x1b,0x00,0x00,0x00,0x1b
+.byte	0x0c,0x0f,0x0e,0x0d,0x0c,0x0f,0x0e,0x0d,0x0c,0x0f,0x0e,0x0d,0x0c,0x0f,0x0e,0x0d
+.byte	0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00
+.Lconsts:
+	mflr	0
+	bcl	20,31,$+4
+	mflr	6
+	addi	6,6,-0x48
+	mtlr	0
+	blr	
+.long	0
+.byte	0,12,0x14,0,0,0,0,0
+.byte	65,69,83,32,102,111,114,32,80,111,119,101,114,73,83,65,32,50,46,48,55,44,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0
+.align	2
+
+.globl	crypton_aes_p8_set_encrypt_key
+.type	crypton_aes_p8_set_encrypt_key,@function
+.align	5
+crypton_aes_p8_set_encrypt_key:
+.localentry	crypton_aes_p8_set_encrypt_key,0
+
+.Lset_encrypt_key:
+	mflr	11
+	std	11,16(1)
+
+	li	6,-1
+	cmpldi	3,0
+	beq-	.Lenc_key_abort
+	cmpldi	5,0
+	beq-	.Lenc_key_abort
+	li	6,-2
+	cmpwi	4,128
+	blt-	.Lenc_key_abort
+	cmpwi	4,256
+	bgt-	.Lenc_key_abort
+	andi.	0,4,0x3f
+	bne-	.Lenc_key_abort
+
+	lis	0,0xfff0
+	li	12,-1
+	or	0,0,0
+
+	bl	.Lconsts
+	mtlr	11
+
+	neg	9,3
+	lvx	1,0,3
+	addi	3,3,15
+	lvsr	3,0,9
+	li	8,0x20
+	cmpwi	4,192
+	lvx	2,0,3
+	vspltisb	5,0x0f
+	lvx	4,0,6
+	vxor	3,3,5
+	lvx	5,8,6
+	addi	6,6,0x10
+	vperm	1,1,2,3
+	li	7,8
+	vxor	0,0,0
+	mtctr	7
+
+	lvsl	8,0,5
+	vspltisb	9,-1
+	lvx	10,0,5
+	vperm	9,9,0,8
+
+	blt	.Loop128
+	addi	3,3,8
+	beq	.L192
+	addi	3,3,8
+	b	.L256
+
+.align	4
+.Loop128:
+	vperm	3,1,1,5
+	vsldoi	6,0,1,12
+	vperm	11,1,1,8
+	vsel	7,10,11,9
+	vor	10,11,11
+	.long	0x10632509
+	stvx	7,0,5
+	addi	5,5,16
+
+	vxor	1,1,6
+	vsldoi	6,0,6,12
+	vxor	1,1,6
+	vsldoi	6,0,6,12
+	vxor	1,1,6
+	vadduwm	4,4,4
+	vxor	1,1,3
+	bdnz	.Loop128
+
+	lvx	4,0,6
+
+	vperm	3,1,1,5
+	vsldoi	6,0,1,12
+	vperm	11,1,1,8
+	vsel	7,10,11,9
+	vor	10,11,11
+	.long	0x10632509
+	stvx	7,0,5
+	addi	5,5,16
+
+	vxor	1,1,6
+	vsldoi	6,0,6,12
+	vxor	1,1,6
+	vsldoi	6,0,6,12
+	vxor	1,1,6
+	vadduwm	4,4,4
+	vxor	1,1,3
+
+	vperm	3,1,1,5
+	vsldoi	6,0,1,12
+	vperm	11,1,1,8
+	vsel	7,10,11,9
+	vor	10,11,11
+	.long	0x10632509
+	stvx	7,0,5
+	addi	5,5,16
+
+	vxor	1,1,6
+	vsldoi	6,0,6,12
+	vxor	1,1,6
+	vsldoi	6,0,6,12
+	vxor	1,1,6
+	vxor	1,1,3
+	vperm	11,1,1,8
+	vsel	7,10,11,9
+	vor	10,11,11
+	stvx	7,0,5
+
+	addi	3,5,15
+	addi	5,5,0x50
+
+	li	8,10
+	b	.Ldone
+
+.align	4
+.L192:
+	lvx	6,0,3
+	li	7,4
+	vperm	11,1,1,8
+	vsel	7,10,11,9
+	vor	10,11,11
+	stvx	7,0,5
+	addi	5,5,16
+	vperm	2,2,6,3
+	vspltisb	3,8
+	mtctr	7
+	vsububm	5,5,3
+
+.Loop192:
+	vperm	3,2,2,5
+	vsldoi	6,0,1,12
+	.long	0x10632509
+
+	vxor	1,1,6
+	vsldoi	6,0,6,12
+	vxor	1,1,6
+	vsldoi	6,0,6,12
+	vxor	1,1,6
+
+	vsldoi	7,0,2,8
+	vspltw	6,1,3
+	vxor	6,6,2
+	vsldoi	2,0,2,12
+	vadduwm	4,4,4
+	vxor	2,2,6
+	vxor	1,1,3
+	vxor	2,2,3
+	vsldoi	7,7,1,8
+
+	vperm	3,2,2,5
+	vsldoi	6,0,1,12
+	vperm	11,7,7,8
+	vsel	7,10,11,9
+	vor	10,11,11
+	.long	0x10632509
+	stvx	7,0,5
+	addi	5,5,16
+
+	vsldoi	7,1,2,8
+	vxor	1,1,6
+	vsldoi	6,0,6,12
+	vperm	11,7,7,8
+	vsel	7,10,11,9
+	vor	10,11,11
+	vxor	1,1,6
+	vsldoi	6,0,6,12
+	vxor	1,1,6
+	stvx	7,0,5
+	addi	5,5,16
+
+	vspltw	6,1,3
+	vxor	6,6,2
+	vsldoi	2,0,2,12
+	vadduwm	4,4,4
+	vxor	2,2,6
+	vxor	1,1,3
+	vxor	2,2,3
+	vperm	11,1,1,8
+	vsel	7,10,11,9
+	vor	10,11,11
+	stvx	7,0,5
+	addi	3,5,15
+	addi	5,5,16
+	bdnz	.Loop192
+
+	li	8,12
+	addi	5,5,0x20
+	b	.Ldone
+
+.align	4
+.L256:
+	lvx	6,0,3
+	li	7,7
+	li	8,14
+	vperm	11,1,1,8
+	vsel	7,10,11,9
+	vor	10,11,11
+	stvx	7,0,5
+	addi	5,5,16
+	vperm	2,2,6,3
+	mtctr	7
+
+.Loop256:
+	vperm	3,2,2,5
+	vsldoi	6,0,1,12
+	vperm	11,2,2,8
+	vsel	7,10,11,9
+	vor	10,11,11
+	.long	0x10632509
+	stvx	7,0,5
+	addi	5,5,16
+
+	vxor	1,1,6
+	vsldoi	6,0,6,12
+	vxor	1,1,6
+	vsldoi	6,0,6,12
+	vxor	1,1,6
+	vadduwm	4,4,4
+	vxor	1,1,3
+	vperm	11,1,1,8
+	vsel	7,10,11,9
+	vor	10,11,11
+	stvx	7,0,5
+	addi	3,5,15
+	addi	5,5,16
+	bdz	.Ldone
+
+	vspltw	3,1,3
+	vsldoi	6,0,2,12
+	.long	0x106305C8
+
+	vxor	2,2,6
+	vsldoi	6,0,6,12
+	vxor	2,2,6
+	vsldoi	6,0,6,12
+	vxor	2,2,6
+
+	vxor	2,2,3
+	b	.Loop256
+
+.align	4
+.Ldone:
+	lvx	2,0,3
+	vsel	2,10,2,9
+	stvx	2,0,3
+	li	6,0
+	or	12,12,12
+	stw	8,0(5)
+
+.Lenc_key_abort:
+	mr	3,6
+	blr	
+.long	0
+.byte	0,12,0x14,1,0,0,3,0
+.long	0
+.size	crypton_aes_p8_set_encrypt_key,.-crypton_aes_p8_set_encrypt_key
+
+.globl	crypton_aes_p8_set_decrypt_key
+.type	crypton_aes_p8_set_decrypt_key,@function
+.align	5
+crypton_aes_p8_set_decrypt_key:
+.localentry	crypton_aes_p8_set_decrypt_key,0
+
+	stdu	1,-64(1)
+	mflr	10
+	std	10,64+16(1)
+	bl	.Lset_encrypt_key
+	mtlr	10
+
+	cmpwi	3,0
+	bne-	.Ldec_key_abort
+
+	slwi	7,8,4
+	subi	3,5,240
+	srwi	8,8,1
+	add	5,3,7
+	mtctr	8
+
+.Ldeckey:
+	lwz	0, 0(3)
+	lwz	6, 4(3)
+	lwz	7, 8(3)
+	lwz	8, 12(3)
+	addi	3,3,16
+	lwz	9, 0(5)
+	lwz	10,4(5)
+	lwz	11,8(5)
+	lwz	12,12(5)
+	stw	0, 0(5)
+	stw	6, 4(5)
+	stw	7, 8(5)
+	stw	8, 12(5)
+	subi	5,5,16
+	stw	9, -16(3)
+	stw	10,-12(3)
+	stw	11,-8(3)
+	stw	12,-4(3)
+	bdnz	.Ldeckey
+
+	xor	3,3,3
+.Ldec_key_abort:
+	addi	1,1,64
+	blr	
+.long	0
+.byte	0,12,4,1,0x80,0,3,0
+.long	0
+.size	crypton_aes_p8_set_decrypt_key,.-crypton_aes_p8_set_decrypt_key
+.globl	crypton_aes_p8_encrypt
+.type	crypton_aes_p8_encrypt,@function
+.align	5
+crypton_aes_p8_encrypt:
+.localentry	crypton_aes_p8_encrypt,0
+
+	lwz	6,240(5)
+	lis	0,0xfc00
+	li	12,-1
+	li	7,15
+	or	0,0,0
+
+	lvx	0,0,3
+	neg	11,4
+	lvx	1,7,3
+	lvsl	2,0,3
+	vspltisb	4,0x0f
+	lvsr	3,0,11
+	vxor	2,2,4
+	li	7,16
+	vperm	0,0,1,2
+	lvx	1,0,5
+	lvsr	5,0,5
+	srwi	6,6,1
+	lvx	2,7,5
+	addi	7,7,16
+	subi	6,6,1
+	vperm	1,2,1,5
+
+	vxor	0,0,1
+	lvx	1,7,5
+	addi	7,7,16
+	mtctr	6
+
+.Loop_enc:
+	vperm	2,1,2,5
+	.long	0x10001508
+	lvx	2,7,5
+	addi	7,7,16
+	vperm	1,2,1,5
+	.long	0x10000D08
+	lvx	1,7,5
+	addi	7,7,16
+	bdnz	.Loop_enc
+
+	vperm	2,1,2,5
+	.long	0x10001508
+	lvx	2,7,5
+	vperm	1,2,1,5
+	.long	0x10000D09
+
+	vspltisb	2,-1
+	vxor	1,1,1
+	li	7,15
+	vperm	2,2,1,3
+	vxor	3,3,4
+	lvx	1,0,4
+	vperm	0,0,0,3
+	vsel	1,1,0,2
+	lvx	4,7,4
+	stvx	1,0,4
+	vsel	0,0,4,2
+	stvx	0,7,4
+
+	or	12,12,12
+	blr	
+.long	0
+.byte	0,12,0x14,0,0,0,3,0
+.long	0
+.size	crypton_aes_p8_encrypt,.-crypton_aes_p8_encrypt
+.globl	crypton_aes_p8_decrypt
+.type	crypton_aes_p8_decrypt,@function
+.align	5
+crypton_aes_p8_decrypt:
+.localentry	crypton_aes_p8_decrypt,0
+
+	lwz	6,240(5)
+	lis	0,0xfc00
+	li	12,-1
+	li	7,15
+	or	0,0,0
+
+	lvx	0,0,3
+	neg	11,4
+	lvx	1,7,3
+	lvsl	2,0,3
+	vspltisb	4,0x0f
+	lvsr	3,0,11
+	vxor	2,2,4
+	li	7,16
+	vperm	0,0,1,2
+	lvx	1,0,5
+	lvsr	5,0,5
+	srwi	6,6,1
+	lvx	2,7,5
+	addi	7,7,16
+	subi	6,6,1
+	vperm	1,2,1,5
+
+	vxor	0,0,1
+	lvx	1,7,5
+	addi	7,7,16
+	mtctr	6
+
+.Loop_dec:
+	vperm	2,1,2,5
+	.long	0x10001548
+	lvx	2,7,5
+	addi	7,7,16
+	vperm	1,2,1,5
+	.long	0x10000D48
+	lvx	1,7,5
+	addi	7,7,16
+	bdnz	.Loop_dec
+
+	vperm	2,1,2,5
+	.long	0x10001548
+	lvx	2,7,5
+	vperm	1,2,1,5
+	.long	0x10000D49
+
+	vspltisb	2,-1
+	vxor	1,1,1
+	li	7,15
+	vperm	2,2,1,3
+	vxor	3,3,4
+	lvx	1,0,4
+	vperm	0,0,0,3
+	vsel	1,1,0,2
+	lvx	4,7,4
+	stvx	1,0,4
+	vsel	0,0,4,2
+	stvx	0,7,4
+
+	or	12,12,12
+	blr	
+.long	0
+.byte	0,12,0x14,0,0,0,3,0
+.long	0
+.size	crypton_aes_p8_decrypt,.-crypton_aes_p8_decrypt
+.globl	crypton_aes_p8_cbc_encrypt
+.type	crypton_aes_p8_cbc_encrypt,@function
+.align	5
+crypton_aes_p8_cbc_encrypt:
+.localentry	crypton_aes_p8_cbc_encrypt,0
+
+	cmpldi	5,16
+	.long	0x4dc00020
+
+	cmpwi	8,0
+	lis	0,0xffe0
+	li	12,-1
+	or	0,0,0
+
+	li	10,15
+	vxor	0,0,0
+	vspltisb	3,0x0f
+
+	lvx	4,0,7
+	lvsl	6,0,7
+	lvx	5,10,7
+	vxor	6,6,3
+	vperm	4,4,5,6
+
+	neg	11,3
+	lvsr	10,0,6
+	lwz	9,240(6)
+
+	lvsr	6,0,11
+	lvx	5,0,3
+	addi	3,3,15
+	vxor	6,6,3
+
+	lvsl	8,0,4
+	vspltisb	9,-1
+	lvx	7,0,4
+	vperm	9,9,0,8
+	vxor	8,8,3
+
+	srwi	9,9,1
+	li	10,16
+	subi	9,9,1
+	beq	.Lcbc_dec
+
+.Lcbc_enc:
+	vor	2,5,5
+	lvx	5,0,3
+	addi	3,3,16
+	mtctr	9
+	subi	5,5,16
+
+	lvx	0,0,6
+	vperm	2,2,5,6
+	lvx	1,10,6
+	addi	10,10,16
+	vperm	0,1,0,10
+	vxor	2,2,0
+	lvx	0,10,6
+	addi	10,10,16
+	vxor	2,2,4
+
+.Loop_cbc_enc:
+	vperm	1,0,1,10
+	.long	0x10420D08
+	lvx	1,10,6
+	addi	10,10,16
+	vperm	0,1,0,10
+	.long	0x10420508
+	lvx	0,10,6
+	addi	10,10,16
+	bdnz	.Loop_cbc_enc
+
+	vperm	1,0,1,10
+	.long	0x10420D08
+	lvx	1,10,6
+	li	10,16
+	vperm	0,1,0,10
+	.long	0x10820509
+	cmpldi	5,16
+
+	vperm	3,4,4,8
+	vsel	2,7,3,9
+	vor	7,3,3
+	stvx	2,0,4
+	addi	4,4,16
+	bge	.Lcbc_enc
+
+	b	.Lcbc_done
+
+.align	4
+.Lcbc_dec:
+	cmpldi	5,128
+	bge	_aesp8_cbc_decrypt8x
+	vor	3,5,5
+	lvx	5,0,3
+	addi	3,3,16
+	mtctr	9
+	subi	5,5,16
+
+	lvx	0,0,6
+	vperm	3,3,5,6
+	lvx	1,10,6
+	addi	10,10,16
+	vperm	0,1,0,10
+	vxor	2,3,0
+	lvx	0,10,6
+	addi	10,10,16
+
+.Loop_cbc_dec:
+	vperm	1,0,1,10
+	.long	0x10420D48
+	lvx	1,10,6
+	addi	10,10,16
+	vperm	0,1,0,10
+	.long	0x10420548
+	lvx	0,10,6
+	addi	10,10,16
+	bdnz	.Loop_cbc_dec
+
+	vperm	1,0,1,10
+	.long	0x10420D48
+	lvx	1,10,6
+	li	10,16
+	vperm	0,1,0,10
+	.long	0x10420549
+	cmpldi	5,16
+
+	vxor	2,2,4
+	vor	4,3,3
+	vperm	3,2,2,8
+	vsel	2,7,3,9
+	vor	7,3,3
+	stvx	2,0,4
+	addi	4,4,16
+	bge	.Lcbc_dec
+
+.Lcbc_done:
+	addi	4,4,-1
+	lvx	2,0,4
+	vsel	2,7,2,9
+	stvx	2,0,4
+
+	neg	8,7
+	li	10,15
+	vxor	0,0,0
+	vspltisb	9,-1
+	vspltisb	3,0x0f
+	lvsr	8,0,8
+	vperm	9,9,0,8
+	vxor	8,8,3
+	lvx	7,0,7
+	vperm	4,4,4,8
+	vsel	2,7,4,9
+	lvx	5,10,7
+	stvx	2,0,7
+	vsel	2,4,5,9
+	stvx	2,10,7
+
+	or	12,12,12
+	blr	
+.long	0
+.byte	0,12,0x14,0,0,0,6,0
+.long	0
+.align	5
+_aesp8_cbc_decrypt8x:
+	stdu	1,-448(1)
+	li	10,207
+	li	11,223
+	stvx	20,10,1
+	addi	10,10,32
+	stvx	21,11,1
+	addi	11,11,32
+	stvx	22,10,1
+	addi	10,10,32
+	stvx	23,11,1
+	addi	11,11,32
+	stvx	24,10,1
+	addi	10,10,32
+	stvx	25,11,1
+	addi	11,11,32
+	stvx	26,10,1
+	addi	10,10,32
+	stvx	27,11,1
+	addi	11,11,32
+	stvx	28,10,1
+	addi	10,10,32
+	stvx	29,11,1
+	addi	11,11,32
+	stvx	30,10,1
+	stvx	31,11,1
+	li	0,-1
+	stw	12,396(1)
+	li	8,0x10
+	std	26,400(1)
+	li	26,0x20
+	std	27,408(1)
+	li	27,0x30
+	std	28,416(1)
+	li	28,0x40
+	std	29,424(1)
+	li	29,0x50
+	std	30,432(1)
+	li	30,0x60
+	std	31,440(1)
+	li	31,0x70
+	or	0,0,0
+
+	subi	9,9,3
+	subi	5,5,128
+
+	lvx	23,0,6
+	lvx	30,8,6
+	addi	6,6,0x20
+	lvx	31,0,6
+	vperm	23,30,23,10
+	addi	11,1,64+15
+	mtctr	9
+
+.Load_cbc_dec_key:
+	vperm	24,31,30,10
+	lvx	30,8,6
+	addi	6,6,0x20
+	stvx	24,0,11
+	vperm	25,30,31,10
+	lvx	31,0,6
+	stvx	25,8,11
+	addi	11,11,0x20
+	bdnz	.Load_cbc_dec_key
+
+	lvx	26,8,6
+	vperm	24,31,30,10
+	lvx	27,26,6
+	stvx	24,0,11
+	vperm	25,26,31,10
+	lvx	28,27,6
+	stvx	25,8,11
+	addi	11,1,64+15
+	vperm	26,27,26,10
+	lvx	29,28,6
+	vperm	27,28,27,10
+	lvx	30,29,6
+	vperm	28,29,28,10
+	lvx	31,30,6
+	vperm	29,30,29,10
+	lvx	14,31,6
+	vperm	30,31,30,10
+	lvx	24,0,11
+	vperm	31,14,31,10
+	lvx	25,8,11
+
+
+
+	subi	3,3,15
+
+	li	10,8
+	.long	0x7C001E99
+	lvsl	6,0,10
+	vspltisb	3,0x0f
+	.long	0x7C281E99
+	vxor	6,6,3
+	.long	0x7C5A1E99
+	vperm	0,0,0,6
+	.long	0x7C7B1E99
+	vperm	1,1,1,6
+	.long	0x7D5C1E99
+	vperm	2,2,2,6
+	vxor	14,0,23
+	.long	0x7D7D1E99
+	vperm	3,3,3,6
+	vxor	15,1,23
+	.long	0x7D9E1E99
+	vperm	10,10,10,6
+	vxor	16,2,23
+	.long	0x7DBF1E99
+	addi	3,3,0x80
+	vperm	11,11,11,6
+	vxor	17,3,23
+	vperm	12,12,12,6
+	vxor	18,10,23
+	vperm	13,13,13,6
+	vxor	19,11,23
+	vxor	20,12,23
+	vxor	21,13,23
+
+	mtctr	9
+	b	.Loop_cbc_dec8x
+.align	5
+.Loop_cbc_dec8x:
+	.long	0x11CEC548
+	.long	0x11EFC548
+	.long	0x1210C548
+	.long	0x1231C548
+	.long	0x1252C548
+	.long	0x1273C548
+	.long	0x1294C548
+	.long	0x12B5C548
+	lvx	24,26,11
+	addi	11,11,0x20
+
+	.long	0x11CECD48
+	.long	0x11EFCD48
+	.long	0x1210CD48
+	.long	0x1231CD48
+	.long	0x1252CD48
+	.long	0x1273CD48
+	.long	0x1294CD48
+	.long	0x12B5CD48
+	lvx	25,8,11
+	bdnz	.Loop_cbc_dec8x
+
+	subic	5,5,128
+	.long	0x11CEC548
+	.long	0x11EFC548
+	.long	0x1210C548
+	.long	0x1231C548
+	.long	0x1252C548
+	.long	0x1273C548
+	.long	0x1294C548
+	.long	0x12B5C548
+
+	subfe.	0,0,0
+	.long	0x11CECD48
+	.long	0x11EFCD48
+	.long	0x1210CD48
+	.long	0x1231CD48
+	.long	0x1252CD48
+	.long	0x1273CD48
+	.long	0x1294CD48
+	.long	0x12B5CD48
+
+	and	0,0,5
+	.long	0x11CED548
+	.long	0x11EFD548
+	.long	0x1210D548
+	.long	0x1231D548
+	.long	0x1252D548
+	.long	0x1273D548
+	.long	0x1294D548
+	.long	0x12B5D548
+
+	add	3,3,0
+
+
+
+	.long	0x11CEDD48
+	.long	0x11EFDD48
+	.long	0x1210DD48
+	.long	0x1231DD48
+	.long	0x1252DD48
+	.long	0x1273DD48
+	.long	0x1294DD48
+	.long	0x12B5DD48
+
+	addi	11,1,64+15
+	.long	0x11CEE548
+	.long	0x11EFE548
+	.long	0x1210E548
+	.long	0x1231E548
+	.long	0x1252E548
+	.long	0x1273E548
+	.long	0x1294E548
+	.long	0x12B5E548
+	lvx	24,0,11
+
+	.long	0x11CEED48
+	.long	0x11EFED48
+	.long	0x1210ED48
+	.long	0x1231ED48
+	.long	0x1252ED48
+	.long	0x1273ED48
+	.long	0x1294ED48
+	.long	0x12B5ED48
+	lvx	25,8,11
+
+	.long	0x11CEF548
+	vxor	4,4,31
+	.long	0x11EFF548
+	vxor	0,0,31
+	.long	0x1210F548
+	vxor	1,1,31
+	.long	0x1231F548
+	vxor	2,2,31
+	.long	0x1252F548
+	vxor	3,3,31
+	.long	0x1273F548
+	vxor	10,10,31
+	.long	0x1294F548
+	vxor	11,11,31
+	.long	0x12B5F548
+	vxor	12,12,31
+
+	.long	0x11CE2549
+	.long	0x11EF0549
+	.long	0x7C001E99
+	.long	0x12100D49
+	.long	0x7C281E99
+	.long	0x12311549
+	vperm	0,0,0,6
+	.long	0x7C5A1E99
+	.long	0x12521D49
+	vperm	1,1,1,6
+	.long	0x7C7B1E99
+	.long	0x12735549
+	vperm	2,2,2,6
+	.long	0x7D5C1E99
+	.long	0x12945D49
+	vperm	3,3,3,6
+	.long	0x7D7D1E99
+	.long	0x12B56549
+	vperm	10,10,10,6
+	.long	0x7D9E1E99
+	vor	4,13,13
+	vperm	11,11,11,6
+	.long	0x7DBF1E99
+	addi	3,3,0x80
+
+	vperm	14,14,14,6
+	vperm	15,15,15,6
+	.long	0x7DC02799
+	vperm	12,12,12,6
+	vxor	14,0,23
+	vperm	16,16,16,6
+	.long	0x7DE82799
+	vperm	13,13,13,6
+	vxor	15,1,23
+	vperm	17,17,17,6
+	.long	0x7E1A2799
+	vxor	16,2,23
+	vperm	18,18,18,6
+	.long	0x7E3B2799
+	vxor	17,3,23
+	vperm	19,19,19,6
+	.long	0x7E5C2799
+	vxor	18,10,23
+	vperm	20,20,20,6
+	.long	0x7E7D2799
+	vxor	19,11,23
+	vperm	21,21,21,6
+	.long	0x7E9E2799
+	vxor	20,12,23
+	.long	0x7EBF2799
+	addi	4,4,0x80
+	vxor	21,13,23
+
+	mtctr	9
+	beq	.Loop_cbc_dec8x
+
+	addic.	5,5,128
+	beq	.Lcbc_dec8x_done
+	nop	
+	nop	
+
+.Loop_cbc_dec8x_tail:
+	.long	0x11EFC548
+	.long	0x1210C548
+	.long	0x1231C548
+	.long	0x1252C548
+	.long	0x1273C548
+	.long	0x1294C548
+	.long	0x12B5C548
+	lvx	24,26,11
+	addi	11,11,0x20
+
+	.long	0x11EFCD48
+	.long	0x1210CD48
+	.long	0x1231CD48
+	.long	0x1252CD48
+	.long	0x1273CD48
+	.long	0x1294CD48
+	.long	0x12B5CD48
+	lvx	25,8,11
+	bdnz	.Loop_cbc_dec8x_tail
+
+	.long	0x11EFC548
+	.long	0x1210C548
+	.long	0x1231C548
+	.long	0x1252C548
+	.long	0x1273C548
+	.long	0x1294C548
+	.long	0x12B5C548
+
+	.long	0x11EFCD48
+	.long	0x1210CD48
+	.long	0x1231CD48
+	.long	0x1252CD48
+	.long	0x1273CD48
+	.long	0x1294CD48
+	.long	0x12B5CD48
+
+	.long	0x11EFD548
+	.long	0x1210D548
+	.long	0x1231D548
+	.long	0x1252D548
+	.long	0x1273D548
+	.long	0x1294D548
+	.long	0x12B5D548
+
+	.long	0x11EFDD48
+	.long	0x1210DD48
+	.long	0x1231DD48
+	.long	0x1252DD48
+	.long	0x1273DD48
+	.long	0x1294DD48
+	.long	0x12B5DD48
+
+	.long	0x11EFE548
+	.long	0x1210E548
+	.long	0x1231E548
+	.long	0x1252E548
+	.long	0x1273E548
+	.long	0x1294E548
+	.long	0x12B5E548
+
+	.long	0x11EFED48
+	.long	0x1210ED48
+	.long	0x1231ED48
+	.long	0x1252ED48
+	.long	0x1273ED48
+	.long	0x1294ED48
+	.long	0x12B5ED48
+
+	.long	0x11EFF548
+	vxor	4,4,31
+	.long	0x1210F548
+	vxor	1,1,31
+	.long	0x1231F548
+	vxor	2,2,31
+	.long	0x1252F548
+	vxor	3,3,31
+	.long	0x1273F548
+	vxor	10,10,31
+	.long	0x1294F548
+	vxor	11,11,31
+	.long	0x12B5F548
+	vxor	12,12,31
+
+	cmplwi	5,32
+	blt	.Lcbc_dec8x_one
+	nop	
+	beq	.Lcbc_dec8x_two
+	cmplwi	5,64
+	blt	.Lcbc_dec8x_three
+	nop	
+	beq	.Lcbc_dec8x_four
+	cmplwi	5,96
+	blt	.Lcbc_dec8x_five
+	nop	
+	beq	.Lcbc_dec8x_six
+
+.Lcbc_dec8x_seven:
+	.long	0x11EF2549
+	.long	0x12100D49
+	.long	0x12311549
+	.long	0x12521D49
+	.long	0x12735549
+	.long	0x12945D49
+	.long	0x12B56549
+	vor	4,13,13
+
+	vperm	15,15,15,6
+	vperm	16,16,16,6
+	.long	0x7DE02799
+	vperm	17,17,17,6
+	.long	0x7E082799
+	vperm	18,18,18,6
+	.long	0x7E3A2799
+	vperm	19,19,19,6
+	.long	0x7E5B2799
+	vperm	20,20,20,6
+	.long	0x7E7C2799
+	vperm	21,21,21,6
+	.long	0x7E9D2799
+	.long	0x7EBE2799
+	addi	4,4,0x70
+	b	.Lcbc_dec8x_done
+
+.align	5
+.Lcbc_dec8x_six:
+	.long	0x12102549
+	.long	0x12311549
+	.long	0x12521D49
+	.long	0x12735549
+	.long	0x12945D49
+	.long	0x12B56549
+	vor	4,13,13
+
+	vperm	16,16,16,6
+	vperm	17,17,17,6
+	.long	0x7E002799
+	vperm	18,18,18,6
+	.long	0x7E282799
+	vperm	19,19,19,6
+	.long	0x7E5A2799
+	vperm	20,20,20,6
+	.long	0x7E7B2799
+	vperm	21,21,21,6
+	.long	0x7E9C2799
+	.long	0x7EBD2799
+	addi	4,4,0x60
+	b	.Lcbc_dec8x_done
+
+.align	5
+.Lcbc_dec8x_five:
+	.long	0x12312549
+	.long	0x12521D49
+	.long	0x12735549
+	.long	0x12945D49
+	.long	0x12B56549
+	vor	4,13,13
+
+	vperm	17,17,17,6
+	vperm	18,18,18,6
+	.long	0x7E202799
+	vperm	19,19,19,6
+	.long	0x7E482799
+	vperm	20,20,20,6
+	.long	0x7E7A2799
+	vperm	21,21,21,6
+	.long	0x7E9B2799
+	.long	0x7EBC2799
+	addi	4,4,0x50
+	b	.Lcbc_dec8x_done
+
+.align	5
+.Lcbc_dec8x_four:
+	.long	0x12522549
+	.long	0x12735549
+	.long	0x12945D49
+	.long	0x12B56549
+	vor	4,13,13
+
+	vperm	18,18,18,6
+	vperm	19,19,19,6
+	.long	0x7E402799
+	vperm	20,20,20,6
+	.long	0x7E682799
+	vperm	21,21,21,6
+	.long	0x7E9A2799
+	.long	0x7EBB2799
+	addi	4,4,0x40
+	b	.Lcbc_dec8x_done
+
+.align	5
+.Lcbc_dec8x_three:
+	.long	0x12732549
+	.long	0x12945D49
+	.long	0x12B56549
+	vor	4,13,13
+
+	vperm	19,19,19,6
+	vperm	20,20,20,6
+	.long	0x7E602799
+	vperm	21,21,21,6
+	.long	0x7E882799
+	.long	0x7EBA2799
+	addi	4,4,0x30
+	b	.Lcbc_dec8x_done
+
+.align	5
+.Lcbc_dec8x_two:
+	.long	0x12942549
+	.long	0x12B56549
+	vor	4,13,13
+
+	vperm	20,20,20,6
+	vperm	21,21,21,6
+	.long	0x7E802799
+	.long	0x7EA82799
+	addi	4,4,0x20
+	b	.Lcbc_dec8x_done
+
+.align	5
+.Lcbc_dec8x_one:
+	.long	0x12B52549
+	vor	4,13,13
+
+	vperm	21,21,21,6
+	.long	0x7EA02799
+	addi	4,4,0x10
+
+.Lcbc_dec8x_done:
+	vperm	4,4,4,6
+	.long	0x7C803F99
+
+	li	10,79
+	li	11,95
+	stvx	6,10,1
+	addi	10,10,32
+	stvx	6,11,1
+	addi	11,11,32
+	stvx	6,10,1
+	addi	10,10,32
+	stvx	6,11,1
+	addi	11,11,32
+	stvx	6,10,1
+	addi	10,10,32
+	stvx	6,11,1
+	addi	11,11,32
+	stvx	6,10,1
+	addi	10,10,32
+	stvx	6,11,1
+	addi	11,11,32
+
+	or	12,12,12
+	lvx	20,10,1
+	addi	10,10,32
+	lvx	21,11,1
+	addi	11,11,32
+	lvx	22,10,1
+	addi	10,10,32
+	lvx	23,11,1
+	addi	11,11,32
+	lvx	24,10,1
+	addi	10,10,32
+	lvx	25,11,1
+	addi	11,11,32
+	lvx	26,10,1
+	addi	10,10,32
+	lvx	27,11,1
+	addi	11,11,32
+	lvx	28,10,1
+	addi	10,10,32
+	lvx	29,11,1
+	addi	11,11,32
+	lvx	30,10,1
+	lvx	31,11,1
+	ld	26,400(1)
+	ld	27,408(1)
+	ld	28,416(1)
+	ld	29,424(1)
+	ld	30,432(1)
+	ld	31,440(1)
+	addi	1,1,448
+	blr	
+.long	0
+.byte	0,12,0x04,0,0x80,6,6,0
+.long	0
+.size	crypton_aes_p8_cbc_encrypt,.-crypton_aes_p8_cbc_encrypt
+.globl	crypton_aes_p8_ctr32_encrypt_blocks
+.type	crypton_aes_p8_ctr32_encrypt_blocks,@function
+.align	5
+crypton_aes_p8_ctr32_encrypt_blocks:
+.localentry	crypton_aes_p8_ctr32_encrypt_blocks,0
+
+	cmpldi	5,1
+	.long	0x4dc00020
+
+	lis	0,0xfff0
+	li	12,-1
+	or	0,0,0
+
+	li	10,15
+	vxor	0,0,0
+	vspltisb	3,0x0f
+
+	lvx	4,0,7
+	lvsl	6,0,7
+	lvx	5,10,7
+	vspltisb	11,1
+	vxor	6,6,3
+	vperm	4,4,5,6
+	vsldoi	11,0,11,1
+
+	neg	11,3
+	lvsr	10,0,6
+	lwz	9,240(6)
+
+	lvsr	6,0,11
+	lvx	5,0,3
+	addi	3,3,15
+	vxor	6,6,3
+
+	srwi	9,9,1
+	li	10,16
+	subi	9,9,1
+
+	cmpldi	5,8
+	bge	_aesp8_ctr32_encrypt8x
+
+	lvsl	8,0,4
+	vspltisb	9,-1
+	lvx	7,0,4
+	vperm	9,9,0,8
+	vxor	8,8,3
+
+	lvx	0,0,6
+	mtctr	9
+	lvx	1,10,6
+	addi	10,10,16
+	vperm	0,1,0,10
+	vxor	2,4,0
+	lvx	0,10,6
+	addi	10,10,16
+	b	.Loop_ctr32_enc
+
+.align	5
+.Loop_ctr32_enc:
+	vperm	1,0,1,10
+	.long	0x10420D08
+	lvx	1,10,6
+	addi	10,10,16
+	vperm	0,1,0,10
+	.long	0x10420508
+	lvx	0,10,6
+	addi	10,10,16
+	bdnz	.Loop_ctr32_enc
+
+	vadduwm	4,4,11
+	vor	3,5,5
+	lvx	5,0,3
+	addi	3,3,16
+	subic.	5,5,1
+
+	vperm	1,0,1,10
+	.long	0x10420D08
+	lvx	1,10,6
+	vperm	3,3,5,6
+	li	10,16
+	vperm	1,1,0,10
+	lvx	0,0,6
+	vxor	3,3,1
+	.long	0x10421D09
+
+	lvx	1,10,6
+	addi	10,10,16
+	vperm	2,2,2,8
+	vsel	3,7,2,9
+	mtctr	9
+	vperm	0,1,0,10
+	vor	7,2,2
+	vxor	2,4,0
+	lvx	0,10,6
+	addi	10,10,16
+	stvx	3,0,4
+	addi	4,4,16
+	bne	.Loop_ctr32_enc
+
+	addi	4,4,-1
+	lvx	2,0,4
+	vsel	2,7,2,9
+	stvx	2,0,4
+
+	or	12,12,12
+	blr	
+.long	0
+.byte	0,12,0x14,0,0,0,6,0
+.long	0
+.align	5
+_aesp8_ctr32_encrypt8x:
+	stdu	1,-448(1)
+	li	10,207
+	li	11,223
+	stvx	20,10,1
+	addi	10,10,32
+	stvx	21,11,1
+	addi	11,11,32
+	stvx	22,10,1
+	addi	10,10,32
+	stvx	23,11,1
+	addi	11,11,32
+	stvx	24,10,1
+	addi	10,10,32
+	stvx	25,11,1
+	addi	11,11,32
+	stvx	26,10,1
+	addi	10,10,32
+	stvx	27,11,1
+	addi	11,11,32
+	stvx	28,10,1
+	addi	10,10,32
+	stvx	29,11,1
+	addi	11,11,32
+	stvx	30,10,1
+	stvx	31,11,1
+	li	0,-1
+	stw	12,396(1)
+	li	8,0x10
+	std	26,400(1)
+	li	26,0x20
+	std	27,408(1)
+	li	27,0x30
+	std	28,416(1)
+	li	28,0x40
+	std	29,424(1)
+	li	29,0x50
+	std	30,432(1)
+	li	30,0x60
+	std	31,440(1)
+	li	31,0x70
+	or	0,0,0
+
+	subi	9,9,3
+
+	lvx	23,0,6
+	lvx	30,8,6
+	addi	6,6,0x20
+	lvx	31,0,6
+	vperm	23,30,23,10
+	addi	11,1,64+15
+	mtctr	9
+
+.Load_ctr32_enc_key:
+	vperm	24,31,30,10
+	lvx	30,8,6
+	addi	6,6,0x20
+	stvx	24,0,11
+	vperm	25,30,31,10
+	lvx	31,0,6
+	stvx	25,8,11
+	addi	11,11,0x20
+	bdnz	.Load_ctr32_enc_key
+
+	lvx	26,8,6
+	vperm	24,31,30,10
+	lvx	27,26,6
+	stvx	24,0,11
+	vperm	25,26,31,10
+	lvx	28,27,6
+	stvx	25,8,11
+	addi	11,1,64+15
+	vperm	26,27,26,10
+	lvx	29,28,6
+	vperm	27,28,27,10
+	lvx	30,29,6
+	vperm	28,29,28,10
+	lvx	31,30,6
+	vperm	29,30,29,10
+	lvx	15,31,6
+	vperm	30,31,30,10
+	lvx	24,0,11
+	vperm	31,15,31,10
+	lvx	25,8,11
+
+	vadduwm	7,11,11
+	subi	3,3,15
+	sldi	5,5,4
+
+	vadduwm	16,4,11
+	vadduwm	17,4,7
+	vxor	15,4,23
+	li	10,8
+	vadduwm	18,16,7
+	vxor	16,16,23
+	lvsl	6,0,10
+	vadduwm	19,17,7
+	vxor	17,17,23
+	vspltisb	3,0x0f
+	vadduwm	20,18,7
+	vxor	18,18,23
+	vxor	6,6,3
+	vadduwm	21,19,7
+	vxor	19,19,23
+	vadduwm	22,20,7
+	vxor	20,20,23
+	vadduwm	4,21,7
+	vxor	21,21,23
+	vxor	22,22,23
+
+	mtctr	9
+	b	.Loop_ctr32_enc8x
+.align	5
+.Loop_ctr32_enc8x:
+	.long	0x11EFC508
+	.long	0x1210C508
+	.long	0x1231C508
+	.long	0x1252C508
+	.long	0x1273C508
+	.long	0x1294C508
+	.long	0x12B5C508
+	.long	0x12D6C508
+.Loop_ctr32_enc8x_middle:
+	lvx	24,26,11
+	addi	11,11,0x20
+
+	.long	0x11EFCD08
+	.long	0x1210CD08
+	.long	0x1231CD08
+	.long	0x1252CD08
+	.long	0x1273CD08
+	.long	0x1294CD08
+	.long	0x12B5CD08
+	.long	0x12D6CD08
+	lvx	25,8,11
+	bdnz	.Loop_ctr32_enc8x
+
+	subic	11,5,256
+	.long	0x11EFC508
+	.long	0x1210C508
+	.long	0x1231C508
+	.long	0x1252C508
+	.long	0x1273C508
+	.long	0x1294C508
+	.long	0x12B5C508
+	.long	0x12D6C508
+
+	subfe	0,0,0
+	.long	0x11EFCD08
+	.long	0x1210CD08
+	.long	0x1231CD08
+	.long	0x1252CD08
+	.long	0x1273CD08
+	.long	0x1294CD08
+	.long	0x12B5CD08
+	.long	0x12D6CD08
+
+	and	0,0,11
+	addi	11,1,64+15
+	.long	0x11EFD508
+	.long	0x1210D508
+	.long	0x1231D508
+	.long	0x1252D508
+	.long	0x1273D508
+	.long	0x1294D508
+	.long	0x12B5D508
+	.long	0x12D6D508
+	lvx	24,0,11
+
+	subic	5,5,129
+	.long	0x11EFDD08
+	addi	5,5,1
+	.long	0x1210DD08
+	.long	0x1231DD08
+	.long	0x1252DD08
+	.long	0x1273DD08
+	.long	0x1294DD08
+	.long	0x12B5DD08
+	.long	0x12D6DD08
+	lvx	25,8,11
+
+	.long	0x11EFE508
+	.long	0x7C001E99
+	.long	0x1210E508
+	.long	0x7C281E99
+	.long	0x1231E508
+	.long	0x7C5A1E99
+	.long	0x1252E508
+	.long	0x7C7B1E99
+	.long	0x1273E508
+	.long	0x7D5C1E99
+	.long	0x1294E508
+	.long	0x7D9D1E99
+	.long	0x12B5E508
+	.long	0x7DBE1E99
+	.long	0x12D6E508
+	.long	0x7DDF1E99
+	addi	3,3,0x80
+
+	.long	0x11EFED08
+	vperm	0,0,0,6
+	.long	0x1210ED08
+	vperm	1,1,1,6
+	.long	0x1231ED08
+	vperm	2,2,2,6
+	.long	0x1252ED08
+	vperm	3,3,3,6
+	.long	0x1273ED08
+	vperm	10,10,10,6
+	.long	0x1294ED08
+	vperm	12,12,12,6
+	.long	0x12B5ED08
+	vperm	13,13,13,6
+	.long	0x12D6ED08
+	vperm	14,14,14,6
+
+	add	3,3,0
+
+
+
+	subfe.	0,0,0
+	.long	0x11EFF508
+	vxor	0,0,31
+	.long	0x1210F508
+	vxor	1,1,31
+	.long	0x1231F508
+	vxor	2,2,31
+	.long	0x1252F508
+	vxor	3,3,31
+	.long	0x1273F508
+	vxor	10,10,31
+	.long	0x1294F508
+	vxor	12,12,31
+	.long	0x12B5F508
+	vxor	13,13,31
+	.long	0x12D6F508
+	vxor	14,14,31
+
+	bne	.Lctr32_enc8x_break
+
+	.long	0x100F0509
+	.long	0x10300D09
+	vadduwm	16,4,11
+	.long	0x10511509
+	vadduwm	17,4,7
+	vxor	15,4,23
+	.long	0x10721D09
+	vadduwm	18,16,7
+	vxor	16,16,23
+	.long	0x11535509
+	vadduwm	19,17,7
+	vxor	17,17,23
+	.long	0x11946509
+	vadduwm	20,18,7
+	vxor	18,18,23
+	.long	0x11B56D09
+	vadduwm	21,19,7
+	vxor	19,19,23
+	.long	0x11D67509
+	vadduwm	22,20,7
+	vxor	20,20,23
+	vperm	0,0,0,6
+	vadduwm	4,21,7
+	vxor	21,21,23
+	vperm	1,1,1,6
+	vxor	22,22,23
+	mtctr	9
+
+	.long	0x11EFC508
+	.long	0x7C002799
+	vperm	2,2,2,6
+	.long	0x1210C508
+	.long	0x7C282799
+	vperm	3,3,3,6
+	.long	0x1231C508
+	.long	0x7C5A2799
+	vperm	10,10,10,6
+	.long	0x1252C508
+	.long	0x7C7B2799
+	vperm	12,12,12,6
+	.long	0x1273C508
+	.long	0x7D5C2799
+	vperm	13,13,13,6
+	.long	0x1294C508
+	.long	0x7D9D2799
+	vperm	14,14,14,6
+	.long	0x12B5C508
+	.long	0x7DBE2799
+	.long	0x12D6C508
+	.long	0x7DDF2799
+	addi	4,4,0x80
+
+	b	.Loop_ctr32_enc8x_middle
+
+.align	5
+.Lctr32_enc8x_break:
+	cmpwi	5,-0x60
+	blt	.Lctr32_enc8x_one
+	nop	
+	beq	.Lctr32_enc8x_two
+	cmpwi	5,-0x40
+	blt	.Lctr32_enc8x_three
+	nop	
+	beq	.Lctr32_enc8x_four
+	cmpwi	5,-0x20
+	blt	.Lctr32_enc8x_five
+	nop	
+	beq	.Lctr32_enc8x_six
+	cmpwi	5,0x00
+	blt	.Lctr32_enc8x_seven
+
+.Lctr32_enc8x_eight:
+	.long	0x11EF0509
+	.long	0x12100D09
+	.long	0x12311509
+	.long	0x12521D09
+	.long	0x12735509
+	.long	0x12946509
+	.long	0x12B56D09
+	.long	0x12D67509
+
+	vperm	15,15,15,6
+	vperm	16,16,16,6
+	.long	0x7DE02799
+	vperm	17,17,17,6
+	.long	0x7E082799
+	vperm	18,18,18,6
+	.long	0x7E3A2799
+	vperm	19,19,19,6
+	.long	0x7E5B2799
+	vperm	20,20,20,6
+	.long	0x7E7C2799
+	vperm	21,21,21,6
+	.long	0x7E9D2799
+	vperm	22,22,22,6
+	.long	0x7EBE2799
+	.long	0x7EDF2799
+	addi	4,4,0x80
+	b	.Lctr32_enc8x_done
+
+.align	5
+.Lctr32_enc8x_seven:
+	.long	0x11EF0D09
+	.long	0x12101509
+	.long	0x12311D09
+	.long	0x12525509
+	.long	0x12736509
+	.long	0x12946D09
+	.long	0x12B57509
+
+	vperm	15,15,15,6
+	vperm	16,16,16,6
+	.long	0x7DE02799
+	vperm	17,17,17,6
+	.long	0x7E082799
+	vperm	18,18,18,6
+	.long	0x7E3A2799
+	vperm	19,19,19,6
+	.long	0x7E5B2799
+	vperm	20,20,20,6
+	.long	0x7E7C2799
+	vperm	21,21,21,6
+	.long	0x7E9D2799
+	.long	0x7EBE2799
+	addi	4,4,0x70
+	b	.Lctr32_enc8x_done
+
+.align	5
+.Lctr32_enc8x_six:
+	.long	0x11EF1509
+	.long	0x12101D09
+	.long	0x12315509
+	.long	0x12526509
+	.long	0x12736D09
+	.long	0x12947509
+
+	vperm	15,15,15,6
+	vperm	16,16,16,6
+	.long	0x7DE02799
+	vperm	17,17,17,6
+	.long	0x7E082799
+	vperm	18,18,18,6
+	.long	0x7E3A2799
+	vperm	19,19,19,6
+	.long	0x7E5B2799
+	vperm	20,20,20,6
+	.long	0x7E7C2799
+	.long	0x7E9D2799
+	addi	4,4,0x60
+	b	.Lctr32_enc8x_done
+
+.align	5
+.Lctr32_enc8x_five:
+	.long	0x11EF1D09
+	.long	0x12105509
+	.long	0x12316509
+	.long	0x12526D09
+	.long	0x12737509
+
+	vperm	15,15,15,6
+	vperm	16,16,16,6
+	.long	0x7DE02799
+	vperm	17,17,17,6
+	.long	0x7E082799
+	vperm	18,18,18,6
+	.long	0x7E3A2799
+	vperm	19,19,19,6
+	.long	0x7E5B2799
+	.long	0x7E7C2799
+	addi	4,4,0x50
+	b	.Lctr32_enc8x_done
+
+.align	5
+.Lctr32_enc8x_four:
+	.long	0x11EF5509
+	.long	0x12106509
+	.long	0x12316D09
+	.long	0x12527509
+
+	vperm	15,15,15,6
+	vperm	16,16,16,6
+	.long	0x7DE02799
+	vperm	17,17,17,6
+	.long	0x7E082799
+	vperm	18,18,18,6
+	.long	0x7E3A2799
+	.long	0x7E5B2799
+	addi	4,4,0x40
+	b	.Lctr32_enc8x_done
+
+.align	5
+.Lctr32_enc8x_three:
+	.long	0x11EF6509
+	.long	0x12106D09
+	.long	0x12317509
+
+	vperm	15,15,15,6
+	vperm	16,16,16,6
+	.long	0x7DE02799
+	vperm	17,17,17,6
+	.long	0x7E082799
+	.long	0x7E3A2799
+	addi	4,4,0x30
+	b	.Lctr32_enc8x_done
+
+.align	5
+.Lctr32_enc8x_two:
+	.long	0x11EF6D09
+	.long	0x12107509
+
+	vperm	15,15,15,6
+	vperm	16,16,16,6
+	.long	0x7DE02799
+	.long	0x7E082799
+	addi	4,4,0x20
+	b	.Lctr32_enc8x_done
+
+.align	5
+.Lctr32_enc8x_one:
+	.long	0x11EF7509
+
+	vperm	15,15,15,6
+	.long	0x7DE02799
+	addi	4,4,0x10
+
+.Lctr32_enc8x_done:
+	li	10,79
+	li	11,95
+	stvx	6,10,1
+	addi	10,10,32
+	stvx	6,11,1
+	addi	11,11,32
+	stvx	6,10,1
+	addi	10,10,32
+	stvx	6,11,1
+	addi	11,11,32
+	stvx	6,10,1
+	addi	10,10,32
+	stvx	6,11,1
+	addi	11,11,32
+	stvx	6,10,1
+	addi	10,10,32
+	stvx	6,11,1
+	addi	11,11,32
+
+	or	12,12,12
+	lvx	20,10,1
+	addi	10,10,32
+	lvx	21,11,1
+	addi	11,11,32
+	lvx	22,10,1
+	addi	10,10,32
+	lvx	23,11,1
+	addi	11,11,32
+	lvx	24,10,1
+	addi	10,10,32
+	lvx	25,11,1
+	addi	11,11,32
+	lvx	26,10,1
+	addi	10,10,32
+	lvx	27,11,1
+	addi	11,11,32
+	lvx	28,10,1
+	addi	10,10,32
+	lvx	29,11,1
+	addi	11,11,32
+	lvx	30,10,1
+	lvx	31,11,1
+	ld	26,400(1)
+	ld	27,408(1)
+	ld	28,416(1)
+	ld	29,424(1)
+	ld	30,432(1)
+	ld	31,440(1)
+	addi	1,1,448
+	blr	
+.long	0
+.byte	0,12,0x04,0,0x80,6,6,0
+.long	0
+.size	crypton_aes_p8_ctr32_encrypt_blocks,.-crypton_aes_p8_ctr32_encrypt_blocks
+.globl	crypton_aes_p8_xts_encrypt
+.type	crypton_aes_p8_xts_encrypt,@function
+.align	5
+crypton_aes_p8_xts_encrypt:
+.localentry	crypton_aes_p8_xts_encrypt,0
+
+	mr	10,3
+	li	3,-1
+	cmpldi	5,16
+	.long	0x4dc00020
+
+	lis	0,0xfff0
+	li	12,-1
+	li	11,0
+	or	0,0,0
+
+	vspltisb	9,0x07
+	lvsl	6,11,11
+	vspltisb	11,0x0f
+	vxor	6,6,9
+
+	li	3,15
+	lvx	8,0,8
+	lvsl	5,0,8
+	lvx	4,3,8
+	vxor	5,5,11
+	vperm	8,8,4,5
+
+	neg	11,10
+	lvsr	5,0,11
+	lvx	2,0,10
+	addi	10,10,15
+	vxor	5,5,11
+
+	cmpldi	7,0
+	beq	.Lxts_enc_no_key2
+
+	lvsr	7,0,7
+	lwz	9,240(7)
+	srwi	9,9,1
+	subi	9,9,1
+	li	3,16
+
+	lvx	0,0,7
+	lvx	1,3,7
+	addi	3,3,16
+	vperm	0,1,0,7
+	vxor	8,8,0
+	lvx	0,3,7
+	addi	3,3,16
+	mtctr	9
+
+.Ltweak_xts_enc:
+	vperm	1,0,1,7
+	.long	0x11080D08
+	lvx	1,3,7
+	addi	3,3,16
+	vperm	0,1,0,7
+	.long	0x11080508
+	lvx	0,3,7
+	addi	3,3,16
+	bdnz	.Ltweak_xts_enc
+
+	vperm	1,0,1,7
+	.long	0x11080D08
+	lvx	1,3,7
+	vperm	0,1,0,7
+	.long	0x11080509
+
+	li	8,0
+	b	.Lxts_enc
+
+.Lxts_enc_no_key2:
+	li	3,-16
+	and	5,5,3
+
+
+.Lxts_enc:
+	lvx	4,0,10
+	addi	10,10,16
+
+	lvsr	7,0,6
+	lwz	9,240(6)
+	srwi	9,9,1
+	subi	9,9,1
+	li	3,16
+
+	vslb	10,9,9
+	vor	10,10,9
+	vspltisb	11,1
+	vsldoi	10,10,11,15
+
+	cmpldi	5,96
+	bge	_aesp8_xts_encrypt6x
+
+	andi.	7,5,15
+	subic	0,5,32
+	subi	7,7,16
+	subfe	0,0,0
+	and	0,0,7
+	add	10,10,0
+
+	lvx	0,0,6
+	lvx	1,3,6
+	addi	3,3,16
+	vperm	2,2,4,5
+	vperm	0,1,0,7
+	vxor	2,2,8
+	vxor	2,2,0
+	lvx	0,3,6
+	addi	3,3,16
+	mtctr	9
+	b	.Loop_xts_enc
+
+.align	5
+.Loop_xts_enc:
+	vperm	1,0,1,7
+	.long	0x10420D08
+	lvx	1,3,6
+	addi	3,3,16
+	vperm	0,1,0,7
+	.long	0x10420508
+	lvx	0,3,6
+	addi	3,3,16
+	bdnz	.Loop_xts_enc
+
+	vperm	1,0,1,7
+	.long	0x10420D08
+	lvx	1,3,6
+	li	3,16
+	vperm	0,1,0,7
+	vxor	0,0,8
+	.long	0x10620509
+
+	vperm	11,3,3,6
+
+	.long	0x7D602799
+
+	addi	4,4,16
+
+	subic.	5,5,16
+	beq	.Lxts_enc_done
+
+	vor	2,4,4
+	lvx	4,0,10
+	addi	10,10,16
+	lvx	0,0,6
+	lvx	1,3,6
+	addi	3,3,16
+
+	subic	0,5,32
+	subfe	0,0,0
+	and	0,0,7
+	add	10,10,0
+
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vand	11,11,10
+	vxor	8,8,11
+
+	vperm	2,2,4,5
+	vperm	0,1,0,7
+	vxor	2,2,8
+	vxor	3,3,0
+	vxor	2,2,0
+	lvx	0,3,6
+	addi	3,3,16
+
+	mtctr	9
+	cmpldi	5,16
+	bge	.Loop_xts_enc
+
+	vxor	3,3,8
+	lvsr	5,0,5
+	vxor	4,4,4
+	vspltisb	11,-1
+	vperm	4,4,11,5
+	vsel	2,2,3,4
+
+	subi	11,4,17
+	subi	4,4,16
+	mtctr	5
+	li	5,16
+.Loop_xts_enc_steal:
+	lbzu	0,1(11)
+	stb	0,16(11)
+	bdnz	.Loop_xts_enc_steal
+
+	mtctr	9
+	b	.Loop_xts_enc
+
+.Lxts_enc_done:
+	cmpldi	8,0
+	beq	.Lxts_enc_ret
+
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vand	11,11,10
+	vxor	8,8,11
+
+	vperm	8,8,8,6
+	.long	0x7D004799
+
+.Lxts_enc_ret:
+	or	12,12,12
+	li	3,0
+	blr	
+.long	0
+.byte	0,12,0x04,0,0x80,6,6,0
+.long	0
+.size	crypton_aes_p8_xts_encrypt,.-crypton_aes_p8_xts_encrypt
+
+.globl	crypton_aes_p8_xts_decrypt
+.type	crypton_aes_p8_xts_decrypt,@function
+.align	5
+crypton_aes_p8_xts_decrypt:
+.localentry	crypton_aes_p8_xts_decrypt,0
+
+	mr	10,3
+	li	3,-1
+	cmpldi	5,16
+	.long	0x4dc00020
+
+	lis	0,0xfff8
+	li	12,-1
+	li	11,0
+	or	0,0,0
+
+	andi.	0,5,15
+	neg	0,0
+	andi.	0,0,16
+	sub	5,5,0
+
+	vspltisb	9,0x07
+	lvsl	6,11,11
+	vspltisb	11,0x0f
+	vxor	6,6,9
+
+	li	3,15
+	lvx	8,0,8
+	lvsl	5,0,8
+	lvx	4,3,8
+	vxor	5,5,11
+	vperm	8,8,4,5
+
+	neg	11,10
+	lvsr	5,0,11
+	lvx	2,0,10
+	addi	10,10,15
+	vxor	5,5,11
+
+	cmpldi	7,0
+	beq	.Lxts_dec_no_key2
+
+	lvsr	7,0,7
+	lwz	9,240(7)
+	srwi	9,9,1
+	subi	9,9,1
+	li	3,16
+
+	lvx	0,0,7
+	lvx	1,3,7
+	addi	3,3,16
+	vperm	0,1,0,7
+	vxor	8,8,0
+	lvx	0,3,7
+	addi	3,3,16
+	mtctr	9
+
+.Ltweak_xts_dec:
+	vperm	1,0,1,7
+	.long	0x11080D08
+	lvx	1,3,7
+	addi	3,3,16
+	vperm	0,1,0,7
+	.long	0x11080508
+	lvx	0,3,7
+	addi	3,3,16
+	bdnz	.Ltweak_xts_dec
+
+	vperm	1,0,1,7
+	.long	0x11080D08
+	lvx	1,3,7
+	vperm	0,1,0,7
+	.long	0x11080509
+
+	li	8,0
+	b	.Lxts_dec
+
+.Lxts_dec_no_key2:
+	neg	3,5
+	andi.	3,3,15
+	add	5,5,3
+
+
+.Lxts_dec:
+	lvx	4,0,10
+	addi	10,10,16
+
+	lvsr	7,0,6
+	lwz	9,240(6)
+	srwi	9,9,1
+	subi	9,9,1
+	li	3,16
+
+	vslb	10,9,9
+	vor	10,10,9
+	vspltisb	11,1
+	vsldoi	10,10,11,15
+
+	cmpldi	5,96
+	bge	_aesp8_xts_decrypt6x
+
+	lvx	0,0,6
+	lvx	1,3,6
+	addi	3,3,16
+	vperm	2,2,4,5
+	vperm	0,1,0,7
+	vxor	2,2,8
+	vxor	2,2,0
+	lvx	0,3,6
+	addi	3,3,16
+	mtctr	9
+
+	cmpldi	5,16
+	blt	.Ltail_xts_dec
+
+
+.align	5
+.Loop_xts_dec:
+	vperm	1,0,1,7
+	.long	0x10420D48
+	lvx	1,3,6
+	addi	3,3,16
+	vperm	0,1,0,7
+	.long	0x10420548
+	lvx	0,3,6
+	addi	3,3,16
+	bdnz	.Loop_xts_dec
+
+	vperm	1,0,1,7
+	.long	0x10420D48
+	lvx	1,3,6
+	li	3,16
+	vperm	0,1,0,7
+	vxor	0,0,8
+	.long	0x10620549
+
+	vperm	11,3,3,6
+
+	.long	0x7D602799
+
+	addi	4,4,16
+
+	subic.	5,5,16
+	beq	.Lxts_dec_done
+
+	vor	2,4,4
+	lvx	4,0,10
+	addi	10,10,16
+	lvx	0,0,6
+	lvx	1,3,6
+	addi	3,3,16
+
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vand	11,11,10
+	vxor	8,8,11
+
+	vperm	2,2,4,5
+	vperm	0,1,0,7
+	vxor	2,2,8
+	vxor	2,2,0
+	lvx	0,3,6
+	addi	3,3,16
+
+	mtctr	9
+	cmpldi	5,16
+	bge	.Loop_xts_dec
+
+.Ltail_xts_dec:
+	vsrab	11,8,9
+	vaddubm	12,8,8
+	vsldoi	11,11,11,15
+	vand	11,11,10
+	vxor	12,12,11
+
+	subi	10,10,16
+	add	10,10,5
+
+	vxor	2,2,8
+	vxor	2,2,12
+
+.Loop_xts_dec_short:
+	vperm	1,0,1,7
+	.long	0x10420D48
+	lvx	1,3,6
+	addi	3,3,16
+	vperm	0,1,0,7
+	.long	0x10420548
+	lvx	0,3,6
+	addi	3,3,16
+	bdnz	.Loop_xts_dec_short
+
+	vperm	1,0,1,7
+	.long	0x10420D48
+	lvx	1,3,6
+	li	3,16
+	vperm	0,1,0,7
+	vxor	0,0,12
+	.long	0x10620549
+
+	vperm	11,3,3,6
+
+	.long	0x7D602799
+
+
+	vor	2,4,4
+	lvx	4,0,10
+
+	lvx	0,0,6
+	lvx	1,3,6
+	addi	3,3,16
+	vperm	2,2,4,5
+	vperm	0,1,0,7
+
+	lvsr	5,0,5
+	vxor	4,4,4
+	vspltisb	11,-1
+	vperm	4,4,11,5
+	vsel	2,2,3,4
+
+	vxor	0,0,8
+	vxor	2,2,0
+	lvx	0,3,6
+	addi	3,3,16
+
+	subi	11,4,1
+	mtctr	5
+	li	5,16
+.Loop_xts_dec_steal:
+	lbzu	0,1(11)
+	stb	0,16(11)
+	bdnz	.Loop_xts_dec_steal
+
+	mtctr	9
+	b	.Loop_xts_dec
+
+.Lxts_dec_done:
+	cmpldi	8,0
+	beq	.Lxts_dec_ret
+
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vand	11,11,10
+	vxor	8,8,11
+
+	vperm	8,8,8,6
+	.long	0x7D004799
+
+.Lxts_dec_ret:
+	or	12,12,12
+	li	3,0
+	blr	
+.long	0
+.byte	0,12,0x04,0,0x80,6,6,0
+.long	0
+.size	crypton_aes_p8_xts_decrypt,.-crypton_aes_p8_xts_decrypt
+.align	5
+_aesp8_xts_encrypt6x:
+	stdu	1,-448(1)
+	mflr	11
+	li	7,207
+	li	3,223
+	std	11,464(1)
+	stvx	20,7,1
+	addi	7,7,32
+	stvx	21,3,1
+	addi	3,3,32
+	stvx	22,7,1
+	addi	7,7,32
+	stvx	23,3,1
+	addi	3,3,32
+	stvx	24,7,1
+	addi	7,7,32
+	stvx	25,3,1
+	addi	3,3,32
+	stvx	26,7,1
+	addi	7,7,32
+	stvx	27,3,1
+	addi	3,3,32
+	stvx	28,7,1
+	addi	7,7,32
+	stvx	29,3,1
+	addi	3,3,32
+	stvx	30,7,1
+	stvx	31,3,1
+	li	0,-1
+	stw	12,396(1)
+	li	3,0x10
+	std	26,400(1)
+	li	26,0x20
+	std	27,408(1)
+	li	27,0x30
+	std	28,416(1)
+	li	28,0x40
+	std	29,424(1)
+	li	29,0x50
+	std	30,432(1)
+	li	30,0x60
+	std	31,440(1)
+	li	31,0x70
+	or	0,0,0
+
+	subi	9,9,3
+
+	lvx	23,0,6
+	lvx	30,3,6
+	addi	6,6,0x20
+	lvx	31,0,6
+	vperm	23,30,23,7
+	addi	7,1,64+15
+	mtctr	9
+
+.Load_xts_enc_key:
+	vperm	24,31,30,7
+	lvx	30,3,6
+	addi	6,6,0x20
+	stvx	24,0,7
+	vperm	25,30,31,7
+	lvx	31,0,6
+	stvx	25,3,7
+	addi	7,7,0x20
+	bdnz	.Load_xts_enc_key
+
+	lvx	26,3,6
+	vperm	24,31,30,7
+	lvx	27,26,6
+	stvx	24,0,7
+	vperm	25,26,31,7
+	lvx	28,27,6
+	stvx	25,3,7
+	addi	7,1,64+15
+	vperm	26,27,26,7
+	lvx	29,28,6
+	vperm	27,28,27,7
+	lvx	30,29,6
+	vperm	28,29,28,7
+	lvx	31,30,6
+	vperm	29,30,29,7
+	lvx	22,31,6
+	vperm	30,31,30,7
+	lvx	24,0,7
+	vperm	31,22,31,7
+	lvx	25,3,7
+
+	vperm	0,2,4,5
+	subi	10,10,31
+	vxor	17,8,23
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vand	11,11,10
+	vxor	7,0,17
+	vxor	8,8,11
+
+	.long	0x7C235699
+	vxor	18,8,23
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vperm	1,1,1,6
+	vand	11,11,10
+	vxor	12,1,18
+	vxor	8,8,11
+
+	.long	0x7C5A5699
+	andi.	31,5,15
+	vxor	19,8,23
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vperm	2,2,2,6
+	vand	11,11,10
+	vxor	13,2,19
+	vxor	8,8,11
+
+	.long	0x7C7B5699
+	sub	5,5,31
+	vxor	20,8,23
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vperm	3,3,3,6
+	vand	11,11,10
+	vxor	14,3,20
+	vxor	8,8,11
+
+	.long	0x7C9C5699
+	subi	5,5,0x60
+	vxor	21,8,23
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vperm	4,4,4,6
+	vand	11,11,10
+	vxor	15,4,21
+	vxor	8,8,11
+
+	.long	0x7CBD5699
+	addi	10,10,0x60
+	vxor	22,8,23
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vperm	5,5,5,6
+	vand	11,11,10
+	vxor	16,5,22
+	vxor	8,8,11
+
+	vxor	31,31,23
+	mtctr	9
+	b	.Loop_xts_enc6x
+
+.align	5
+.Loop_xts_enc6x:
+	.long	0x10E7C508
+	.long	0x118CC508
+	.long	0x11ADC508
+	.long	0x11CEC508
+	.long	0x11EFC508
+	.long	0x1210C508
+	lvx	24,26,7
+	addi	7,7,0x20
+
+	.long	0x10E7CD08
+	.long	0x118CCD08
+	.long	0x11ADCD08
+	.long	0x11CECD08
+	.long	0x11EFCD08
+	.long	0x1210CD08
+	lvx	25,3,7
+	bdnz	.Loop_xts_enc6x
+
+	subic	5,5,96
+	vxor	0,17,31
+	.long	0x10E7C508
+	.long	0x118CC508
+	vsrab	11,8,9
+	vxor	17,8,23
+	vaddubm	8,8,8
+	.long	0x11ADC508
+	.long	0x11CEC508
+	vsldoi	11,11,11,15
+	.long	0x11EFC508
+	.long	0x1210C508
+
+	subfe.	0,0,0
+	vand	11,11,10
+	.long	0x10E7CD08
+	.long	0x118CCD08
+	vxor	8,8,11
+	.long	0x11ADCD08
+	.long	0x11CECD08
+	vxor	1,18,31
+	vsrab	11,8,9
+	vxor	18,8,23
+	.long	0x11EFCD08
+	.long	0x1210CD08
+
+	and	0,0,5
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	.long	0x10E7D508
+	.long	0x118CD508
+	vand	11,11,10
+	.long	0x11ADD508
+	.long	0x11CED508
+	vxor	8,8,11
+	.long	0x11EFD508
+	.long	0x1210D508
+
+	add	10,10,0
+
+
+
+	vxor	2,19,31
+	vsrab	11,8,9
+	vxor	19,8,23
+	vaddubm	8,8,8
+	.long	0x10E7DD08
+	.long	0x118CDD08
+	vsldoi	11,11,11,15
+	.long	0x11ADDD08
+	.long	0x11CEDD08
+	vand	11,11,10
+	.long	0x11EFDD08
+	.long	0x1210DD08
+
+	addi	7,1,64+15
+	vxor	8,8,11
+	.long	0x10E7E508
+	.long	0x118CE508
+	vxor	3,20,31
+	vsrab	11,8,9
+	vxor	20,8,23
+	.long	0x11ADE508
+	.long	0x11CEE508
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	.long	0x11EFE508
+	.long	0x1210E508
+	lvx	24,0,7
+	vand	11,11,10
+
+	.long	0x10E7ED08
+	.long	0x118CED08
+	vxor	8,8,11
+	.long	0x11ADED08
+	.long	0x11CEED08
+	vxor	4,21,31
+	vsrab	11,8,9
+	vxor	21,8,23
+	.long	0x11EFED08
+	.long	0x1210ED08
+	lvx	25,3,7
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+
+	.long	0x10E7F508
+	.long	0x118CF508
+	vand	11,11,10
+	.long	0x11ADF508
+	.long	0x11CEF508
+	vxor	8,8,11
+	.long	0x11EFF508
+	.long	0x1210F508
+	vxor	5,22,31
+	vsrab	11,8,9
+	vxor	22,8,23
+
+	.long	0x10E70509
+	.long	0x7C005699
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	.long	0x118C0D09
+	.long	0x7C235699
+	.long	0x11AD1509
+	vperm	0,0,0,6
+	.long	0x7C5A5699
+	vand	11,11,10
+	.long	0x11CE1D09
+	vperm	1,1,1,6
+	.long	0x7C7B5699
+	.long	0x11EF2509
+	vperm	2,2,2,6
+	.long	0x7C9C5699
+	vxor	8,8,11
+	.long	0x11702D09
+
+	vperm	3,3,3,6
+	.long	0x7CBD5699
+	addi	10,10,0x60
+	vperm	4,4,4,6
+	vperm	5,5,5,6
+
+	vperm	7,7,7,6
+	vperm	12,12,12,6
+	.long	0x7CE02799
+	vxor	7,0,17
+	vperm	13,13,13,6
+	.long	0x7D832799
+	vxor	12,1,18
+	vperm	14,14,14,6
+	.long	0x7DBA2799
+	vxor	13,2,19
+	vperm	15,15,15,6
+	.long	0x7DDB2799
+	vxor	14,3,20
+	vperm	16,11,11,6
+	.long	0x7DFC2799
+	vxor	15,4,21
+	.long	0x7E1D2799
+
+	vxor	16,5,22
+	addi	4,4,0x60
+
+	mtctr	9
+	beq	.Loop_xts_enc6x
+
+	addic.	5,5,0x60
+	beq	.Lxts_enc6x_zero
+	cmpwi	5,0x20
+	blt	.Lxts_enc6x_one
+	nop	
+	beq	.Lxts_enc6x_two
+	cmpwi	5,0x40
+	blt	.Lxts_enc6x_three
+	nop	
+	beq	.Lxts_enc6x_four
+
+.Lxts_enc6x_five:
+	vxor	7,1,17
+	vxor	12,2,18
+	vxor	13,3,19
+	vxor	14,4,20
+	vxor	15,5,21
+
+	bl	_aesp8_xts_enc5x
+
+	vperm	7,7,7,6
+	vor	17,22,22
+	vperm	12,12,12,6
+	.long	0x7CE02799
+	vperm	13,13,13,6
+	.long	0x7D832799
+	vperm	14,14,14,6
+	.long	0x7DBA2799
+	vxor	11,15,22
+	vperm	15,15,15,6
+	.long	0x7DDB2799
+	.long	0x7DFC2799
+	addi	4,4,0x50
+	bne	.Lxts_enc6x_steal
+	b	.Lxts_enc6x_done
+
+.align	4
+.Lxts_enc6x_four:
+	vxor	7,2,17
+	vxor	12,3,18
+	vxor	13,4,19
+	vxor	14,5,20
+	vxor	15,15,15
+
+	bl	_aesp8_xts_enc5x
+
+	vperm	7,7,7,6
+	vor	17,21,21
+	vperm	12,12,12,6
+	.long	0x7CE02799
+	vperm	13,13,13,6
+	.long	0x7D832799
+	vxor	11,14,21
+	vperm	14,14,14,6
+	.long	0x7DBA2799
+	.long	0x7DDB2799
+	addi	4,4,0x40
+	bne	.Lxts_enc6x_steal
+	b	.Lxts_enc6x_done
+
+.align	4
+.Lxts_enc6x_three:
+	vxor	7,3,17
+	vxor	12,4,18
+	vxor	13,5,19
+	vxor	14,14,14
+	vxor	15,15,15
+
+	bl	_aesp8_xts_enc5x
+
+	vperm	7,7,7,6
+	vor	17,20,20
+	vperm	12,12,12,6
+	.long	0x7CE02799
+	vxor	11,13,20
+	vperm	13,13,13,6
+	.long	0x7D832799
+	.long	0x7DBA2799
+	addi	4,4,0x30
+	bne	.Lxts_enc6x_steal
+	b	.Lxts_enc6x_done
+
+.align	4
+.Lxts_enc6x_two:
+	vxor	7,4,17
+	vxor	12,5,18
+	vxor	13,13,13
+	vxor	14,14,14
+	vxor	15,15,15
+
+	bl	_aesp8_xts_enc5x
+
+	vperm	7,7,7,6
+	vor	17,19,19
+	vxor	11,12,19
+	vperm	12,12,12,6
+	.long	0x7CE02799
+	.long	0x7D832799
+	addi	4,4,0x20
+	bne	.Lxts_enc6x_steal
+	b	.Lxts_enc6x_done
+
+.align	4
+.Lxts_enc6x_one:
+	vxor	7,5,17
+	nop	
+.Loop_xts_enc1x:
+	.long	0x10E7C508
+	lvx	24,26,7
+	addi	7,7,0x20
+
+	.long	0x10E7CD08
+	lvx	25,3,7
+	bdnz	.Loop_xts_enc1x
+
+	add	10,10,31
+	cmpwi	31,0
+	.long	0x10E7C508
+
+	subi	10,10,16
+	.long	0x10E7CD08
+
+	lvsr	5,0,31
+	.long	0x10E7D508
+
+	.long	0x7C005699
+	.long	0x10E7DD08
+
+	addi	7,1,64+15
+	.long	0x10E7E508
+	lvx	24,0,7
+
+	.long	0x10E7ED08
+	lvx	25,3,7
+	vxor	17,17,31
+
+	vperm	0,0,0,6
+	.long	0x10E7F508
+
+	vperm	0,0,0,5
+	.long	0x10E78D09
+
+	vor	17,18,18
+	vxor	11,7,18
+	vperm	7,7,7,6
+	.long	0x7CE02799
+	addi	4,4,0x10
+	bne	.Lxts_enc6x_steal
+	b	.Lxts_enc6x_done
+
+.align	4
+.Lxts_enc6x_zero:
+	cmpwi	31,0
+	beq	.Lxts_enc6x_done
+
+	add	10,10,31
+	subi	10,10,16
+	.long	0x7C005699
+	lvsr	5,0,31
+	vperm	0,0,0,6
+	vperm	0,0,0,5
+	vxor	11,11,17
+.Lxts_enc6x_steal:
+	vxor	0,0,17
+	vxor	7,7,7
+	vspltisb	12,-1
+	vperm	7,7,12,5
+	vsel	7,0,11,7
+
+	subi	30,4,17
+	subi	4,4,16
+	mtctr	31
+.Loop_xts_enc6x_steal:
+	lbzu	0,1(30)
+	stb	0,16(30)
+	bdnz	.Loop_xts_enc6x_steal
+
+	li	31,0
+	mtctr	9
+	b	.Loop_xts_enc1x
+
+.align	4
+.Lxts_enc6x_done:
+	cmpldi	8,0
+	beq	.Lxts_enc6x_ret
+
+	vxor	8,17,23
+	vperm	8,8,8,6
+	.long	0x7D004799
+
+.Lxts_enc6x_ret:
+	mtlr	11
+	li	10,79
+	li	11,95
+	stvx	9,10,1
+	addi	10,10,32
+	stvx	9,11,1
+	addi	11,11,32
+	stvx	9,10,1
+	addi	10,10,32
+	stvx	9,11,1
+	addi	11,11,32
+	stvx	9,10,1
+	addi	10,10,32
+	stvx	9,11,1
+	addi	11,11,32
+	stvx	9,10,1
+	addi	10,10,32
+	stvx	9,11,1
+	addi	11,11,32
+
+	or	12,12,12
+	lvx	20,10,1
+	addi	10,10,32
+	lvx	21,11,1
+	addi	11,11,32
+	lvx	22,10,1
+	addi	10,10,32
+	lvx	23,11,1
+	addi	11,11,32
+	lvx	24,10,1
+	addi	10,10,32
+	lvx	25,11,1
+	addi	11,11,32
+	lvx	26,10,1
+	addi	10,10,32
+	lvx	27,11,1
+	addi	11,11,32
+	lvx	28,10,1
+	addi	10,10,32
+	lvx	29,11,1
+	addi	11,11,32
+	lvx	30,10,1
+	lvx	31,11,1
+	ld	26,400(1)
+	ld	27,408(1)
+	ld	28,416(1)
+	ld	29,424(1)
+	ld	30,432(1)
+	ld	31,440(1)
+	addi	1,1,448
+	blr	
+.long	0
+.byte	0,12,0x04,1,0x80,6,6,0
+.long	0
+
+.align	5
+_aesp8_xts_enc5x:
+	.long	0x10E7C508
+	.long	0x118CC508
+	.long	0x11ADC508
+	.long	0x11CEC508
+	.long	0x11EFC508
+	lvx	24,26,7
+	addi	7,7,0x20
+
+	.long	0x10E7CD08
+	.long	0x118CCD08
+	.long	0x11ADCD08
+	.long	0x11CECD08
+	.long	0x11EFCD08
+	lvx	25,3,7
+	bdnz	_aesp8_xts_enc5x
+
+	add	10,10,31
+	cmpwi	31,0
+	.long	0x10E7C508
+	.long	0x118CC508
+	.long	0x11ADC508
+	.long	0x11CEC508
+	.long	0x11EFC508
+
+	subi	10,10,16
+	.long	0x10E7CD08
+	.long	0x118CCD08
+	.long	0x11ADCD08
+	.long	0x11CECD08
+	.long	0x11EFCD08
+	vxor	17,17,31
+
+	.long	0x10E7D508
+	lvsr	5,0,31
+	.long	0x118CD508
+	.long	0x11ADD508
+	.long	0x11CED508
+	.long	0x11EFD508
+	vxor	1,18,31
+
+	.long	0x10E7DD08
+	.long	0x7C005699
+	.long	0x118CDD08
+	.long	0x11ADDD08
+	.long	0x11CEDD08
+	.long	0x11EFDD08
+	vxor	2,19,31
+
+	addi	7,1,64+15
+	.long	0x10E7E508
+	.long	0x118CE508
+	.long	0x11ADE508
+	.long	0x11CEE508
+	.long	0x11EFE508
+	lvx	24,0,7
+	vxor	3,20,31
+
+	.long	0x10E7ED08
+	vperm	0,0,0,6
+	.long	0x118CED08
+	.long	0x11ADED08
+	.long	0x11CEED08
+	.long	0x11EFED08
+	lvx	25,3,7
+	vxor	4,21,31
+
+	.long	0x10E7F508
+	vperm	0,0,0,5
+	.long	0x118CF508
+	.long	0x11ADF508
+	.long	0x11CEF508
+	.long	0x11EFF508
+
+	.long	0x10E78D09
+	.long	0x118C0D09
+	.long	0x11AD1509
+	.long	0x11CE1D09
+	.long	0x11EF2509
+	blr	
+.long	0
+.byte	0,12,0x14,0,0,0,0,0
+
+.align	5
+_aesp8_xts_decrypt6x:
+	stdu	1,-448(1)
+	mflr	11
+	li	7,207
+	li	3,223
+	std	11,464(1)
+	stvx	20,7,1
+	addi	7,7,32
+	stvx	21,3,1
+	addi	3,3,32
+	stvx	22,7,1
+	addi	7,7,32
+	stvx	23,3,1
+	addi	3,3,32
+	stvx	24,7,1
+	addi	7,7,32
+	stvx	25,3,1
+	addi	3,3,32
+	stvx	26,7,1
+	addi	7,7,32
+	stvx	27,3,1
+	addi	3,3,32
+	stvx	28,7,1
+	addi	7,7,32
+	stvx	29,3,1
+	addi	3,3,32
+	stvx	30,7,1
+	stvx	31,3,1
+	li	0,-1
+	stw	12,396(1)
+	li	3,0x10
+	std	26,400(1)
+	li	26,0x20
+	std	27,408(1)
+	li	27,0x30
+	std	28,416(1)
+	li	28,0x40
+	std	29,424(1)
+	li	29,0x50
+	std	30,432(1)
+	li	30,0x60
+	std	31,440(1)
+	li	31,0x70
+	or	0,0,0
+
+	subi	9,9,3
+
+	lvx	23,0,6
+	lvx	30,3,6
+	addi	6,6,0x20
+	lvx	31,0,6
+	vperm	23,30,23,7
+	addi	7,1,64+15
+	mtctr	9
+
+.Load_xts_dec_key:
+	vperm	24,31,30,7
+	lvx	30,3,6
+	addi	6,6,0x20
+	stvx	24,0,7
+	vperm	25,30,31,7
+	lvx	31,0,6
+	stvx	25,3,7
+	addi	7,7,0x20
+	bdnz	.Load_xts_dec_key
+
+	lvx	26,3,6
+	vperm	24,31,30,7
+	lvx	27,26,6
+	stvx	24,0,7
+	vperm	25,26,31,7
+	lvx	28,27,6
+	stvx	25,3,7
+	addi	7,1,64+15
+	vperm	26,27,26,7
+	lvx	29,28,6
+	vperm	27,28,27,7
+	lvx	30,29,6
+	vperm	28,29,28,7
+	lvx	31,30,6
+	vperm	29,30,29,7
+	lvx	22,31,6
+	vperm	30,31,30,7
+	lvx	24,0,7
+	vperm	31,22,31,7
+	lvx	25,3,7
+
+	vperm	0,2,4,5
+	subi	10,10,31
+	vxor	17,8,23
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vand	11,11,10
+	vxor	7,0,17
+	vxor	8,8,11
+
+	.long	0x7C235699
+	vxor	18,8,23
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vperm	1,1,1,6
+	vand	11,11,10
+	vxor	12,1,18
+	vxor	8,8,11
+
+	.long	0x7C5A5699
+	andi.	31,5,15
+	vxor	19,8,23
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vperm	2,2,2,6
+	vand	11,11,10
+	vxor	13,2,19
+	vxor	8,8,11
+
+	.long	0x7C7B5699
+	sub	5,5,31
+	vxor	20,8,23
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vperm	3,3,3,6
+	vand	11,11,10
+	vxor	14,3,20
+	vxor	8,8,11
+
+	.long	0x7C9C5699
+	subi	5,5,0x60
+	vxor	21,8,23
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vperm	4,4,4,6
+	vand	11,11,10
+	vxor	15,4,21
+	vxor	8,8,11
+
+	.long	0x7CBD5699
+	addi	10,10,0x60
+	vxor	22,8,23
+	vsrab	11,8,9
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	vperm	5,5,5,6
+	vand	11,11,10
+	vxor	16,5,22
+	vxor	8,8,11
+
+	vxor	31,31,23
+	mtctr	9
+	b	.Loop_xts_dec6x
+
+.align	5
+.Loop_xts_dec6x:
+	.long	0x10E7C548
+	.long	0x118CC548
+	.long	0x11ADC548
+	.long	0x11CEC548
+	.long	0x11EFC548
+	.long	0x1210C548
+	lvx	24,26,7
+	addi	7,7,0x20
+
+	.long	0x10E7CD48
+	.long	0x118CCD48
+	.long	0x11ADCD48
+	.long	0x11CECD48
+	.long	0x11EFCD48
+	.long	0x1210CD48
+	lvx	25,3,7
+	bdnz	.Loop_xts_dec6x
+
+	subic	5,5,96
+	vxor	0,17,31
+	.long	0x10E7C548
+	.long	0x118CC548
+	vsrab	11,8,9
+	vxor	17,8,23
+	vaddubm	8,8,8
+	.long	0x11ADC548
+	.long	0x11CEC548
+	vsldoi	11,11,11,15
+	.long	0x11EFC548
+	.long	0x1210C548
+
+	subfe.	0,0,0
+	vand	11,11,10
+	.long	0x10E7CD48
+	.long	0x118CCD48
+	vxor	8,8,11
+	.long	0x11ADCD48
+	.long	0x11CECD48
+	vxor	1,18,31
+	vsrab	11,8,9
+	vxor	18,8,23
+	.long	0x11EFCD48
+	.long	0x1210CD48
+
+	and	0,0,5
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	.long	0x10E7D548
+	.long	0x118CD548
+	vand	11,11,10
+	.long	0x11ADD548
+	.long	0x11CED548
+	vxor	8,8,11
+	.long	0x11EFD548
+	.long	0x1210D548
+
+	add	10,10,0
+
+
+
+	vxor	2,19,31
+	vsrab	11,8,9
+	vxor	19,8,23
+	vaddubm	8,8,8
+	.long	0x10E7DD48
+	.long	0x118CDD48
+	vsldoi	11,11,11,15
+	.long	0x11ADDD48
+	.long	0x11CEDD48
+	vand	11,11,10
+	.long	0x11EFDD48
+	.long	0x1210DD48
+
+	addi	7,1,64+15
+	vxor	8,8,11
+	.long	0x10E7E548
+	.long	0x118CE548
+	vxor	3,20,31
+	vsrab	11,8,9
+	vxor	20,8,23
+	.long	0x11ADE548
+	.long	0x11CEE548
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	.long	0x11EFE548
+	.long	0x1210E548
+	lvx	24,0,7
+	vand	11,11,10
+
+	.long	0x10E7ED48
+	.long	0x118CED48
+	vxor	8,8,11
+	.long	0x11ADED48
+	.long	0x11CEED48
+	vxor	4,21,31
+	vsrab	11,8,9
+	vxor	21,8,23
+	.long	0x11EFED48
+	.long	0x1210ED48
+	lvx	25,3,7
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+
+	.long	0x10E7F548
+	.long	0x118CF548
+	vand	11,11,10
+	.long	0x11ADF548
+	.long	0x11CEF548
+	vxor	8,8,11
+	.long	0x11EFF548
+	.long	0x1210F548
+	vxor	5,22,31
+	vsrab	11,8,9
+	vxor	22,8,23
+
+	.long	0x10E70549
+	.long	0x7C005699
+	vaddubm	8,8,8
+	vsldoi	11,11,11,15
+	.long	0x118C0D49
+	.long	0x7C235699
+	.long	0x11AD1549
+	vperm	0,0,0,6
+	.long	0x7C5A5699
+	vand	11,11,10
+	.long	0x11CE1D49
+	vperm	1,1,1,6
+	.long	0x7C7B5699
+	.long	0x11EF2549
+	vperm	2,2,2,6
+	.long	0x7C9C5699
+	vxor	8,8,11
+	.long	0x12102D49
+	vperm	3,3,3,6
+	.long	0x7CBD5699
+	addi	10,10,0x60
+	vperm	4,4,4,6
+	vperm	5,5,5,6
+
+	vperm	7,7,7,6
+	vperm	12,12,12,6
+	.long	0x7CE02799
+	vxor	7,0,17
+	vperm	13,13,13,6
+	.long	0x7D832799
+	vxor	12,1,18
+	vperm	14,14,14,6
+	.long	0x7DBA2799
+	vxor	13,2,19
+	vperm	15,15,15,6
+	.long	0x7DDB2799
+	vxor	14,3,20
+	vperm	16,16,16,6
+	.long	0x7DFC2799
+	vxor	15,4,21
+	.long	0x7E1D2799
+	vxor	16,5,22
+	addi	4,4,0x60
+
+	mtctr	9
+	beq	.Loop_xts_dec6x
+
+	addic.	5,5,0x60
+	beq	.Lxts_dec6x_zero
+	cmpwi	5,0x20
+	blt	.Lxts_dec6x_one
+	nop	
+	beq	.Lxts_dec6x_two
+	cmpwi	5,0x40
+	blt	.Lxts_dec6x_three
+	nop	
+	beq	.Lxts_dec6x_four
+
+.Lxts_dec6x_five:
+	vxor	7,1,17
+	vxor	12,2,18
+	vxor	13,3,19
+	vxor	14,4,20
+	vxor	15,5,21
+
+	bl	_aesp8_xts_dec5x
+
+	vperm	7,7,7,6
+	vor	17,22,22
+	vxor	18,8,23
+	vperm	12,12,12,6
+	.long	0x7CE02799
+	vxor	7,0,18
+	vperm	13,13,13,6
+	.long	0x7D832799
+	vperm	14,14,14,6
+	.long	0x7DBA2799
+	vperm	15,15,15,6
+	.long	0x7DDB2799
+	.long	0x7DFC2799
+	addi	4,4,0x50
+	bne	.Lxts_dec6x_steal
+	b	.Lxts_dec6x_done
+
+.align	4
+.Lxts_dec6x_four:
+	vxor	7,2,17
+	vxor	12,3,18
+	vxor	13,4,19
+	vxor	14,5,20
+	vxor	15,15,15
+
+	bl	_aesp8_xts_dec5x
+
+	vperm	7,7,7,6
+	vor	17,21,21
+	vor	18,22,22
+	vperm	12,12,12,6
+	.long	0x7CE02799
+	vxor	7,0,22
+	vperm	13,13,13,6
+	.long	0x7D832799
+	vperm	14,14,14,6
+	.long	0x7DBA2799
+	.long	0x7DDB2799
+	addi	4,4,0x40
+	bne	.Lxts_dec6x_steal
+	b	.Lxts_dec6x_done
+
+.align	4
+.Lxts_dec6x_three:
+	vxor	7,3,17
+	vxor	12,4,18
+	vxor	13,5,19
+	vxor	14,14,14
+	vxor	15,15,15
+
+	bl	_aesp8_xts_dec5x
+
+	vperm	7,7,7,6
+	vor	17,20,20
+	vor	18,21,21
+	vperm	12,12,12,6
+	.long	0x7CE02799
+	vxor	7,0,21
+	vperm	13,13,13,6
+	.long	0x7D832799
+	.long	0x7DBA2799
+	addi	4,4,0x30
+	bne	.Lxts_dec6x_steal
+	b	.Lxts_dec6x_done
+
+.align	4
+.Lxts_dec6x_two:
+	vxor	7,4,17
+	vxor	12,5,18
+	vxor	13,13,13
+	vxor	14,14,14
+	vxor	15,15,15
+
+	bl	_aesp8_xts_dec5x
+
+	vperm	7,7,7,6
+	vor	17,19,19
+	vor	18,20,20
+	vperm	12,12,12,6
+	.long	0x7CE02799
+	vxor	7,0,20
+	.long	0x7D832799
+	addi	4,4,0x20
+	bne	.Lxts_dec6x_steal
+	b	.Lxts_dec6x_done
+
+.align	4
+.Lxts_dec6x_one:
+	vxor	7,5,17
+	nop	
+.Loop_xts_dec1x:
+	.long	0x10E7C548
+	lvx	24,26,7
+	addi	7,7,0x20
+
+	.long	0x10E7CD48
+	lvx	25,3,7
+	bdnz	.Loop_xts_dec1x
+
+	subi	0,31,1
+	.long	0x10E7C548
+
+	andi.	0,0,16
+	cmpwi	31,0
+	.long	0x10E7CD48
+
+	sub	10,10,0
+	.long	0x10E7D548
+
+	.long	0x7C005699
+	.long	0x10E7DD48
+
+	addi	7,1,64+15
+	.long	0x10E7E548
+	lvx	24,0,7
+
+	.long	0x10E7ED48
+	lvx	25,3,7
+	vxor	17,17,31
+
+	vperm	0,0,0,6
+	.long	0x10E7F548
+
+	mtctr	9
+	.long	0x10E78D49
+
+	vor	17,18,18
+	vor	18,19,19
+	vperm	7,7,7,6
+	.long	0x7CE02799
+	addi	4,4,0x10
+	vxor	7,0,19
+	bne	.Lxts_dec6x_steal
+	b	.Lxts_dec6x_done
+
+.align	4
+.Lxts_dec6x_zero:
+	cmpwi	31,0
+	beq	.Lxts_dec6x_done
+
+	.long	0x7C005699
+	vperm	0,0,0,6
+	vxor	7,0,18
+.Lxts_dec6x_steal:
+	.long	0x10E7C548
+	lvx	24,26,7
+	addi	7,7,0x20
+
+	.long	0x10E7CD48
+	lvx	25,3,7
+	bdnz	.Lxts_dec6x_steal
+
+	add	10,10,31
+	.long	0x10E7C548
+
+	cmpwi	31,0
+	.long	0x10E7CD48
+
+	.long	0x7C005699
+	.long	0x10E7D548
+
+	lvsr	5,0,31
+	.long	0x10E7DD48
+
+	addi	7,1,64+15
+	.long	0x10E7E548
+	lvx	24,0,7
+
+	.long	0x10E7ED48
+	lvx	25,3,7
+	vxor	18,18,31
+
+	vperm	0,0,0,6
+	.long	0x10E7F548
+
+	vperm	0,0,0,5
+	.long	0x11679549
+
+	vperm	7,11,11,6
+	.long	0x7CE02799
+
+
+	vxor	7,7,7
+	vspltisb	12,-1
+	vperm	7,7,12,5
+	vsel	7,0,11,7
+	vxor	7,7,17
+
+	subi	30,4,1
+	mtctr	31
+.Loop_xts_dec6x_steal:
+	lbzu	0,1(30)
+	stb	0,16(30)
+	bdnz	.Loop_xts_dec6x_steal
+
+	li	31,0
+	mtctr	9
+	b	.Loop_xts_dec1x
+
+.align	4
+.Lxts_dec6x_done:
+	cmpldi	8,0
+	beq	.Lxts_dec6x_ret
+
+	vxor	8,17,23
+	vperm	8,8,8,6
+	.long	0x7D004799
+
+.Lxts_dec6x_ret:
+	mtlr	11
+	li	10,79
+	li	11,95
+	stvx	9,10,1
+	addi	10,10,32
+	stvx	9,11,1
+	addi	11,11,32
+	stvx	9,10,1
+	addi	10,10,32
+	stvx	9,11,1
+	addi	11,11,32
+	stvx	9,10,1
+	addi	10,10,32
+	stvx	9,11,1
+	addi	11,11,32
+	stvx	9,10,1
+	addi	10,10,32
+	stvx	9,11,1
+	addi	11,11,32
+
+	or	12,12,12
+	lvx	20,10,1
+	addi	10,10,32
+	lvx	21,11,1
+	addi	11,11,32
+	lvx	22,10,1
+	addi	10,10,32
+	lvx	23,11,1
+	addi	11,11,32
+	lvx	24,10,1
+	addi	10,10,32
+	lvx	25,11,1
+	addi	11,11,32
+	lvx	26,10,1
+	addi	10,10,32
+	lvx	27,11,1
+	addi	11,11,32
+	lvx	28,10,1
+	addi	10,10,32
+	lvx	29,11,1
+	addi	11,11,32
+	lvx	30,10,1
+	lvx	31,11,1
+	ld	26,400(1)
+	ld	27,408(1)
+	ld	28,416(1)
+	ld	29,424(1)
+	ld	30,432(1)
+	ld	31,440(1)
+	addi	1,1,448
+	blr	
+.long	0
+.byte	0,12,0x04,1,0x80,6,6,0
+.long	0
+
+.align	5
+_aesp8_xts_dec5x:
+	.long	0x10E7C548
+	.long	0x118CC548
+	.long	0x11ADC548
+	.long	0x11CEC548
+	.long	0x11EFC548
+	lvx	24,26,7
+	addi	7,7,0x20
+
+	.long	0x10E7CD48
+	.long	0x118CCD48
+	.long	0x11ADCD48
+	.long	0x11CECD48
+	.long	0x11EFCD48
+	lvx	25,3,7
+	bdnz	_aesp8_xts_dec5x
+
+	subi	0,31,1
+	.long	0x10E7C548
+	.long	0x118CC548
+	.long	0x11ADC548
+	.long	0x11CEC548
+	.long	0x11EFC548
+
+	andi.	0,0,16
+	cmpwi	31,0
+	.long	0x10E7CD48
+	.long	0x118CCD48
+	.long	0x11ADCD48
+	.long	0x11CECD48
+	.long	0x11EFCD48
+	vxor	17,17,31
+
+	sub	10,10,0
+	.long	0x10E7D548
+	.long	0x118CD548
+	.long	0x11ADD548
+	.long	0x11CED548
+	.long	0x11EFD548
+	vxor	1,18,31
+
+	.long	0x10E7DD48
+	.long	0x7C005699
+	.long	0x118CDD48
+	.long	0x11ADDD48
+	.long	0x11CEDD48
+	.long	0x11EFDD48
+	vxor	2,19,31
+
+	addi	7,1,64+15
+	.long	0x10E7E548
+	.long	0x118CE548
+	.long	0x11ADE548
+	.long	0x11CEE548
+	.long	0x11EFE548
+	lvx	24,0,7
+	vxor	3,20,31
+
+	.long	0x10E7ED48
+	vperm	0,0,0,6
+	.long	0x118CED48
+	.long	0x11ADED48
+	.long	0x11CEED48
+	.long	0x11EFED48
+	lvx	25,3,7
+	vxor	4,21,31
+
+	.long	0x10E7F548
+	.long	0x118CF548
+	.long	0x11ADF548
+	.long	0x11CEF548
+	.long	0x11EFF548
+
+	.long	0x10E78D49
+	.long	0x118C0D49
+	.long	0x11AD1549
+	.long	0x11CE1D49
+	.long	0x11EF2549
+	mtctr	9
+	blr	
+.long	0
+.byte	0,12,0x14,0,0,0,0,0
diff --git a/cbits/asm/aesp8-ppc.pl b/cbits/asm/aesp8-ppc.pl
new file mode 100644
--- /dev/null
+++ b/cbits/asm/aesp8-ppc.pl
@@ -0,0 +1,3801 @@
+#!/usr/bin/env perl
+#
+# ====================================================================
+# Written by Andy Polyakov <appro@openssl.org> for the OpenSSL
+# project. The module is, however, dual licensed under OpenSSL and
+# CRYPTOGAMS licenses depending on where you obtain it. For further
+# details see http://www.openssl.org/~appro/cryptogams/.
+# ====================================================================
+#
+# This module implements support for AES instructions as per PowerISA
+# specification version 2.07, first implemented by POWER8 processor.
+# The module is endian-agnostic in sense that it supports both big-
+# and little-endian cases. Data alignment in parallelizable modes is
+# handled with VSX loads and stores, which implies MSR.VSX flag being
+# set. It should also be noted that ISA specification doesn't prohibit
+# alignment exceptions for these instructions on page boundaries.
+# Initially alignment was handled in pure AltiVec/VMX way [when data
+# is aligned programmatically, which in turn guarantees exception-
+# free execution], but it turned to hamper performance when vcipher
+# instructions are interleaved. It's reckoned that eventual
+# misalignment penalties at page boundaries are in average lower
+# than additional overhead in pure AltiVec approach.
+#
+# May 2016
+#
+# Add XTS subroutine, 9x on little- and 12x improvement on big-endian
+# systems were measured.
+#
+######################################################################
+# Current large-block performance in cycles per byte processed with
+# 128-bit key (less is better).
+#
+#		CBC en-/decrypt	CTR	XTS
+# POWER8[le]	3.96/0.72	0.74	1.1
+# POWER8[be]	3.75/0.65	0.66	1.0
+# POWER9[le]	4.02/0.86	0.84	1.05
+# POWER9[be]	3.99/0.78	0.79	0.97
+
+
+$flavour = shift;
+
+if ($flavour =~ /64/) {
+	$SIZE_T	=8;
+	$LRSAVE	=2*$SIZE_T;
+	$STU	="stdu";
+	$POP	="ld";
+	$PUSH	="std";
+	$UCMP	="cmpld";
+	$SHL	="sldi";
+} elsif ($flavour =~ /32/) {
+	$SIZE_T	=4;
+	$LRSAVE	=$SIZE_T;
+	$STU	="stwu";
+	$POP	="lwz";
+	$PUSH	="stw";
+	$UCMP	="cmplw";
+	$SHL	="slwi";
+} else { die "nonsense $flavour"; }
+
+$LITTLE_ENDIAN = ($flavour=~/le$/) ? $SIZE_T : 0;
+
+$0 =~ m/(.*[\/\\])[^\/\\]+$/; $dir=$1;
+( $xlate="${dir}ppc-xlate.pl" and -f $xlate ) or
+( $xlate="${dir}../../perlasm/ppc-xlate.pl" and -f $xlate) or
+die "can't locate ppc-xlate.pl";
+
+open STDOUT,"| $^X $xlate $flavour ".shift || die "can't call $xlate: $!";
+
+$FRAME=8*$SIZE_T;
+$prefix="aes_p8";
+
+$sp="r1";
+$vrsave="r12";
+
+#########################################################################
+{{{	# Key setup procedures						#
+my ($inp,$bits,$out,$ptr,$cnt,$rounds)=map("r$_",(3..8));
+my ($zero,$in0,$in1,$key,$rcon,$mask,$tmp)=map("v$_",(0..6));
+my ($stage,$outperm,$outmask,$outhead,$outtail)=map("v$_",(7..11));
+
+$code.=<<___;
+.machine	"any"
+
+.text
+
+.align	7
+rcon:
+.long	0x01000000, 0x01000000, 0x01000000, 0x01000000	?rev
+.long	0x1b000000, 0x1b000000, 0x1b000000, 0x1b000000	?rev
+.long	0x0d0e0f0c, 0x0d0e0f0c, 0x0d0e0f0c, 0x0d0e0f0c	?rev
+.long	0,0,0,0						?asis
+Lconsts:
+	mflr	r0
+	bcl	20,31,\$+4
+	mflr	$ptr	 #vvvvv "distance between . and rcon
+	addi	$ptr,$ptr,-0x48
+	mtlr	r0
+	blr
+	.long	0
+	.byte	0,12,0x14,0,0,0,0,0
+.asciz	"AES for PowerISA 2.07, CRYPTOGAMS by <appro\@openssl.org>"
+
+.globl	.${prefix}_set_encrypt_key
+.align	5
+.${prefix}_set_encrypt_key:
+Lset_encrypt_key:
+	mflr		r11
+	$PUSH		r11,$LRSAVE($sp)
+
+	li		$ptr,-1
+	${UCMP}i	$inp,0
+	beq-		Lenc_key_abort		# if ($inp==0) return -1;
+	${UCMP}i	$out,0
+	beq-		Lenc_key_abort		# if ($out==0) return -1;
+	li		$ptr,-2
+	cmpwi		$bits,128
+	blt-		Lenc_key_abort
+	cmpwi		$bits,256
+	bgt-		Lenc_key_abort
+	andi.		r0,$bits,0x3f
+	bne-		Lenc_key_abort
+
+	lis		r0,0xfff0
+	mfspr		$vrsave,256
+	mtspr		256,r0
+
+	bl		Lconsts
+	mtlr		r11
+
+	neg		r9,$inp
+	lvx		$in0,0,$inp
+	addi		$inp,$inp,15		# 15 is not typo
+	lvsr		$key,0,r9		# borrow $key
+	li		r8,0x20
+	cmpwi		$bits,192
+	lvx		$in1,0,$inp
+	le?vspltisb	$mask,0x0f		# borrow $mask
+	lvx		$rcon,0,$ptr
+	le?vxor		$key,$key,$mask		# adjust for byte swap
+	lvx		$mask,r8,$ptr
+	addi		$ptr,$ptr,0x10
+	vperm		$in0,$in0,$in1,$key	# align [and byte swap in LE]
+	li		$cnt,8
+	vxor		$zero,$zero,$zero
+	mtctr		$cnt
+
+	?lvsr		$outperm,0,$out
+	vspltisb	$outmask,-1
+	lvx		$outhead,0,$out
+	?vperm		$outmask,$zero,$outmask,$outperm
+
+	blt		Loop128
+	addi		$inp,$inp,8
+	beq		L192
+	addi		$inp,$inp,8
+	b		L256
+
+.align	4
+Loop128:
+	vperm		$key,$in0,$in0,$mask	# rotate-n-splat
+	vsldoi		$tmp,$zero,$in0,12	# >>32
+	 vperm		$outtail,$in0,$in0,$outperm	# rotate
+	 vsel		$stage,$outhead,$outtail,$outmask
+	 vmr		$outhead,$outtail
+	vcipherlast	$key,$key,$rcon
+	 stvx		$stage,0,$out
+	 addi		$out,$out,16
+
+	vxor		$in0,$in0,$tmp
+	vsldoi		$tmp,$zero,$tmp,12	# >>32
+	vxor		$in0,$in0,$tmp
+	vsldoi		$tmp,$zero,$tmp,12	# >>32
+	vxor		$in0,$in0,$tmp
+	 vadduwm	$rcon,$rcon,$rcon
+	vxor		$in0,$in0,$key
+	bdnz		Loop128
+
+	lvx		$rcon,0,$ptr		# last two round keys
+
+	vperm		$key,$in0,$in0,$mask	# rotate-n-splat
+	vsldoi		$tmp,$zero,$in0,12	# >>32
+	 vperm		$outtail,$in0,$in0,$outperm	# rotate
+	 vsel		$stage,$outhead,$outtail,$outmask
+	 vmr		$outhead,$outtail
+	vcipherlast	$key,$key,$rcon
+	 stvx		$stage,0,$out
+	 addi		$out,$out,16
+
+	vxor		$in0,$in0,$tmp
+	vsldoi		$tmp,$zero,$tmp,12	# >>32
+	vxor		$in0,$in0,$tmp
+	vsldoi		$tmp,$zero,$tmp,12	# >>32
+	vxor		$in0,$in0,$tmp
+	 vadduwm	$rcon,$rcon,$rcon
+	vxor		$in0,$in0,$key
+
+	vperm		$key,$in0,$in0,$mask	# rotate-n-splat
+	vsldoi		$tmp,$zero,$in0,12	# >>32
+	 vperm		$outtail,$in0,$in0,$outperm	# rotate
+	 vsel		$stage,$outhead,$outtail,$outmask
+	 vmr		$outhead,$outtail
+	vcipherlast	$key,$key,$rcon
+	 stvx		$stage,0,$out
+	 addi		$out,$out,16
+
+	vxor		$in0,$in0,$tmp
+	vsldoi		$tmp,$zero,$tmp,12	# >>32
+	vxor		$in0,$in0,$tmp
+	vsldoi		$tmp,$zero,$tmp,12	# >>32
+	vxor		$in0,$in0,$tmp
+	vxor		$in0,$in0,$key
+	 vperm		$outtail,$in0,$in0,$outperm	# rotate
+	 vsel		$stage,$outhead,$outtail,$outmask
+	 vmr		$outhead,$outtail
+	 stvx		$stage,0,$out
+
+	addi		$inp,$out,15		# 15 is not typo
+	addi		$out,$out,0x50
+
+	li		$rounds,10
+	b		Ldone
+
+.align	4
+L192:
+	lvx		$tmp,0,$inp
+	li		$cnt,4
+	 vperm		$outtail,$in0,$in0,$outperm	# rotate
+	 vsel		$stage,$outhead,$outtail,$outmask
+	 vmr		$outhead,$outtail
+	 stvx		$stage,0,$out
+	 addi		$out,$out,16
+	vperm		$in1,$in1,$tmp,$key	# align [and byte swap in LE]
+	vspltisb	$key,8			# borrow $key
+	mtctr		$cnt
+	vsububm		$mask,$mask,$key	# adjust the mask
+
+Loop192:
+	vperm		$key,$in1,$in1,$mask	# roate-n-splat
+	vsldoi		$tmp,$zero,$in0,12	# >>32
+	vcipherlast	$key,$key,$rcon
+
+	vxor		$in0,$in0,$tmp
+	vsldoi		$tmp,$zero,$tmp,12	# >>32
+	vxor		$in0,$in0,$tmp
+	vsldoi		$tmp,$zero,$tmp,12	# >>32
+	vxor		$in0,$in0,$tmp
+
+	 vsldoi		$stage,$zero,$in1,8
+	vspltw		$tmp,$in0,3
+	vxor		$tmp,$tmp,$in1
+	vsldoi		$in1,$zero,$in1,12	# >>32
+	 vadduwm	$rcon,$rcon,$rcon
+	vxor		$in1,$in1,$tmp
+	vxor		$in0,$in0,$key
+	vxor		$in1,$in1,$key
+	 vsldoi		$stage,$stage,$in0,8
+
+	vperm		$key,$in1,$in1,$mask	# rotate-n-splat
+	vsldoi		$tmp,$zero,$in0,12	# >>32
+	 vperm		$outtail,$stage,$stage,$outperm	# rotate
+	 vsel		$stage,$outhead,$outtail,$outmask
+	 vmr		$outhead,$outtail
+	vcipherlast	$key,$key,$rcon
+	 stvx		$stage,0,$out
+	 addi		$out,$out,16
+
+	 vsldoi		$stage,$in0,$in1,8
+	vxor		$in0,$in0,$tmp
+	vsldoi		$tmp,$zero,$tmp,12	# >>32
+	 vperm		$outtail,$stage,$stage,$outperm	# rotate
+	 vsel		$stage,$outhead,$outtail,$outmask
+	 vmr		$outhead,$outtail
+	vxor		$in0,$in0,$tmp
+	vsldoi		$tmp,$zero,$tmp,12	# >>32
+	vxor		$in0,$in0,$tmp
+	 stvx		$stage,0,$out
+	 addi		$out,$out,16
+
+	vspltw		$tmp,$in0,3
+	vxor		$tmp,$tmp,$in1
+	vsldoi		$in1,$zero,$in1,12	# >>32
+	 vadduwm	$rcon,$rcon,$rcon
+	vxor		$in1,$in1,$tmp
+	vxor		$in0,$in0,$key
+	vxor		$in1,$in1,$key
+	 vperm		$outtail,$in0,$in0,$outperm	# rotate
+	 vsel		$stage,$outhead,$outtail,$outmask
+	 vmr		$outhead,$outtail
+	 stvx		$stage,0,$out
+	 addi		$inp,$out,15		# 15 is not typo
+	 addi		$out,$out,16
+	bdnz		Loop192
+
+	li		$rounds,12
+	addi		$out,$out,0x20
+	b		Ldone
+
+.align	4
+L256:
+	lvx		$tmp,0,$inp
+	li		$cnt,7
+	li		$rounds,14
+	 vperm		$outtail,$in0,$in0,$outperm	# rotate
+	 vsel		$stage,$outhead,$outtail,$outmask
+	 vmr		$outhead,$outtail
+	 stvx		$stage,0,$out
+	 addi		$out,$out,16
+	vperm		$in1,$in1,$tmp,$key	# align [and byte swap in LE]
+	mtctr		$cnt
+
+Loop256:
+	vperm		$key,$in1,$in1,$mask	# rotate-n-splat
+	vsldoi		$tmp,$zero,$in0,12	# >>32
+	 vperm		$outtail,$in1,$in1,$outperm	# rotate
+	 vsel		$stage,$outhead,$outtail,$outmask
+	 vmr		$outhead,$outtail
+	vcipherlast	$key,$key,$rcon
+	 stvx		$stage,0,$out
+	 addi		$out,$out,16
+
+	vxor		$in0,$in0,$tmp
+	vsldoi		$tmp,$zero,$tmp,12	# >>32
+	vxor		$in0,$in0,$tmp
+	vsldoi		$tmp,$zero,$tmp,12	# >>32
+	vxor		$in0,$in0,$tmp
+	 vadduwm	$rcon,$rcon,$rcon
+	vxor		$in0,$in0,$key
+	 vperm		$outtail,$in0,$in0,$outperm	# rotate
+	 vsel		$stage,$outhead,$outtail,$outmask
+	 vmr		$outhead,$outtail
+	 stvx		$stage,0,$out
+	 addi		$inp,$out,15		# 15 is not typo
+	 addi		$out,$out,16
+	bdz		Ldone
+
+	vspltw		$key,$in0,3		# just splat
+	vsldoi		$tmp,$zero,$in1,12	# >>32
+	vsbox		$key,$key
+
+	vxor		$in1,$in1,$tmp
+	vsldoi		$tmp,$zero,$tmp,12	# >>32
+	vxor		$in1,$in1,$tmp
+	vsldoi		$tmp,$zero,$tmp,12	# >>32
+	vxor		$in1,$in1,$tmp
+
+	vxor		$in1,$in1,$key
+	b		Loop256
+
+.align	4
+Ldone:
+	lvx		$in1,0,$inp		# redundant in aligned case
+	vsel		$in1,$outhead,$in1,$outmask
+	stvx		$in1,0,$inp
+	li		$ptr,0
+	mtspr		256,$vrsave
+	stw		$rounds,0($out)
+
+Lenc_key_abort:
+	mr		r3,$ptr
+	blr
+	.long		0
+	.byte		0,12,0x14,1,0,0,3,0
+	.long		0
+.size	.${prefix}_set_encrypt_key,.-.${prefix}_set_encrypt_key
+
+.globl	.${prefix}_set_decrypt_key
+.align	5
+.${prefix}_set_decrypt_key:
+	$STU		$sp,-$FRAME($sp)
+	mflr		r10
+	$PUSH		r10,$FRAME+$LRSAVE($sp)
+	bl		Lset_encrypt_key
+	mtlr		r10
+
+	cmpwi		r3,0
+	bne-		Ldec_key_abort
+
+	slwi		$cnt,$rounds,4
+	subi		$inp,$out,240		# first round key
+	srwi		$rounds,$rounds,1
+	add		$out,$inp,$cnt		# last round key
+	mtctr		$rounds
+
+Ldeckey:
+	lwz		r0, 0($inp)
+	lwz		r6, 4($inp)
+	lwz		r7, 8($inp)
+	lwz		r8, 12($inp)
+	addi		$inp,$inp,16
+	lwz		r9, 0($out)
+	lwz		r10,4($out)
+	lwz		r11,8($out)
+	lwz		r12,12($out)
+	stw		r0, 0($out)
+	stw		r6, 4($out)
+	stw		r7, 8($out)
+	stw		r8, 12($out)
+	subi		$out,$out,16
+	stw		r9, -16($inp)
+	stw		r10,-12($inp)
+	stw		r11,-8($inp)
+	stw		r12,-4($inp)
+	bdnz		Ldeckey
+
+	xor		r3,r3,r3		# return value
+Ldec_key_abort:
+	addi		$sp,$sp,$FRAME
+	blr
+	.long		0
+	.byte		0,12,4,1,0x80,0,3,0
+	.long		0
+.size	.${prefix}_set_decrypt_key,.-.${prefix}_set_decrypt_key
+___
+}}}
+#########################################################################
+{{{	# Single block en- and decrypt procedures			#
+sub gen_block () {
+my $dir = shift;
+my $n   = $dir eq "de" ? "n" : "";
+my ($inp,$out,$key,$rounds,$idx)=map("r$_",(3..7));
+
+$code.=<<___;
+.globl	.${prefix}_${dir}crypt
+.align	5
+.${prefix}_${dir}crypt:
+	lwz		$rounds,240($key)
+	lis		r0,0xfc00
+	mfspr		$vrsave,256
+	li		$idx,15			# 15 is not typo
+	mtspr		256,r0
+
+	lvx		v0,0,$inp
+	neg		r11,$out
+	lvx		v1,$idx,$inp
+	lvsl		v2,0,$inp		# inpperm
+	le?vspltisb	v4,0x0f
+	?lvsl		v3,0,r11		# outperm
+	le?vxor		v2,v2,v4
+	li		$idx,16
+	vperm		v0,v0,v1,v2		# align [and byte swap in LE]
+	lvx		v1,0,$key
+	?lvsl		v5,0,$key		# keyperm
+	srwi		$rounds,$rounds,1
+	lvx		v2,$idx,$key
+	addi		$idx,$idx,16
+	subi		$rounds,$rounds,1
+	?vperm		v1,v1,v2,v5		# align round key
+
+	vxor		v0,v0,v1
+	lvx		v1,$idx,$key
+	addi		$idx,$idx,16
+	mtctr		$rounds
+
+Loop_${dir}c:
+	?vperm		v2,v2,v1,v5
+	v${n}cipher	v0,v0,v2
+	lvx		v2,$idx,$key
+	addi		$idx,$idx,16
+	?vperm		v1,v1,v2,v5
+	v${n}cipher	v0,v0,v1
+	lvx		v1,$idx,$key
+	addi		$idx,$idx,16
+	bdnz		Loop_${dir}c
+
+	?vperm		v2,v2,v1,v5
+	v${n}cipher	v0,v0,v2
+	lvx		v2,$idx,$key
+	?vperm		v1,v1,v2,v5
+	v${n}cipherlast	v0,v0,v1
+
+	vspltisb	v2,-1
+	vxor		v1,v1,v1
+	li		$idx,15			# 15 is not typo
+	?vperm		v2,v1,v2,v3		# outmask
+	le?vxor		v3,v3,v4
+	lvx		v1,0,$out		# outhead
+	vperm		v0,v0,v0,v3		# rotate [and byte swap in LE]
+	vsel		v1,v1,v0,v2
+	lvx		v4,$idx,$out
+	stvx		v1,0,$out
+	vsel		v0,v0,v4,v2
+	stvx		v0,$idx,$out
+
+	mtspr		256,$vrsave
+	blr
+	.long		0
+	.byte		0,12,0x14,0,0,0,3,0
+	.long		0
+.size	.${prefix}_${dir}crypt,.-.${prefix}_${dir}crypt
+___
+}
+&gen_block("en");
+&gen_block("de");
+}}}
+#########################################################################
+{{{	# CBC en- and decrypt procedures				#
+my ($inp,$out,$len,$key,$ivp,$enc,$rounds,$idx)=map("r$_",(3..10));
+my ($rndkey0,$rndkey1,$inout,$tmp)=		map("v$_",(0..3));
+my ($ivec,$inptail,$inpperm,$outhead,$outperm,$outmask,$keyperm)=
+						map("v$_",(4..10));
+$code.=<<___;
+.globl	.${prefix}_cbc_encrypt
+.align	5
+.${prefix}_cbc_encrypt:
+	${UCMP}i	$len,16
+	bltlr-
+
+	cmpwi		$enc,0			# test direction
+	lis		r0,0xffe0
+	mfspr		$vrsave,256
+	mtspr		256,r0
+
+	li		$idx,15
+	vxor		$rndkey0,$rndkey0,$rndkey0
+	le?vspltisb	$tmp,0x0f
+
+	lvx		$ivec,0,$ivp		# load [unaligned] iv
+	lvsl		$inpperm,0,$ivp
+	lvx		$inptail,$idx,$ivp
+	le?vxor		$inpperm,$inpperm,$tmp
+	vperm		$ivec,$ivec,$inptail,$inpperm
+
+	neg		r11,$inp
+	?lvsl		$keyperm,0,$key		# prepare for unaligned key
+	lwz		$rounds,240($key)
+
+	lvsr		$inpperm,0,r11		# prepare for unaligned load
+	lvx		$inptail,0,$inp
+	addi		$inp,$inp,15		# 15 is not typo
+	le?vxor		$inpperm,$inpperm,$tmp
+
+	?lvsr		$outperm,0,$out		# prepare for unaligned store
+	vspltisb	$outmask,-1
+	lvx		$outhead,0,$out
+	?vperm		$outmask,$rndkey0,$outmask,$outperm
+	le?vxor		$outperm,$outperm,$tmp
+
+	srwi		$rounds,$rounds,1
+	li		$idx,16
+	subi		$rounds,$rounds,1
+	beq		Lcbc_dec
+
+Lcbc_enc:
+	vmr		$inout,$inptail
+	lvx		$inptail,0,$inp
+	addi		$inp,$inp,16
+	mtctr		$rounds
+	subi		$len,$len,16		# len-=16
+
+	lvx		$rndkey0,0,$key
+	 vperm		$inout,$inout,$inptail,$inpperm
+	lvx		$rndkey1,$idx,$key
+	addi		$idx,$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vxor		$inout,$inout,$rndkey0
+	lvx		$rndkey0,$idx,$key
+	addi		$idx,$idx,16
+	vxor		$inout,$inout,$ivec
+
+Loop_cbc_enc:
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vcipher		$inout,$inout,$rndkey1
+	lvx		$rndkey1,$idx,$key
+	addi		$idx,$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vcipher		$inout,$inout,$rndkey0
+	lvx		$rndkey0,$idx,$key
+	addi		$idx,$idx,16
+	bdnz		Loop_cbc_enc
+
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vcipher		$inout,$inout,$rndkey1
+	lvx		$rndkey1,$idx,$key
+	li		$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vcipherlast	$ivec,$inout,$rndkey0
+	${UCMP}i	$len,16
+
+	vperm		$tmp,$ivec,$ivec,$outperm
+	vsel		$inout,$outhead,$tmp,$outmask
+	vmr		$outhead,$tmp
+	stvx		$inout,0,$out
+	addi		$out,$out,16
+	bge		Lcbc_enc
+
+	b		Lcbc_done
+
+.align	4
+Lcbc_dec:
+	${UCMP}i	$len,128
+	bge		_aesp8_cbc_decrypt8x
+	vmr		$tmp,$inptail
+	lvx		$inptail,0,$inp
+	addi		$inp,$inp,16
+	mtctr		$rounds
+	subi		$len,$len,16		# len-=16
+
+	lvx		$rndkey0,0,$key
+	 vperm		$tmp,$tmp,$inptail,$inpperm
+	lvx		$rndkey1,$idx,$key
+	addi		$idx,$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vxor		$inout,$tmp,$rndkey0
+	lvx		$rndkey0,$idx,$key
+	addi		$idx,$idx,16
+
+Loop_cbc_dec:
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vncipher	$inout,$inout,$rndkey1
+	lvx		$rndkey1,$idx,$key
+	addi		$idx,$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vncipher	$inout,$inout,$rndkey0
+	lvx		$rndkey0,$idx,$key
+	addi		$idx,$idx,16
+	bdnz		Loop_cbc_dec
+
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vncipher	$inout,$inout,$rndkey1
+	lvx		$rndkey1,$idx,$key
+	li		$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vncipherlast	$inout,$inout,$rndkey0
+	${UCMP}i	$len,16
+
+	vxor		$inout,$inout,$ivec
+	vmr		$ivec,$tmp
+	vperm		$tmp,$inout,$inout,$outperm
+	vsel		$inout,$outhead,$tmp,$outmask
+	vmr		$outhead,$tmp
+	stvx		$inout,0,$out
+	addi		$out,$out,16
+	bge		Lcbc_dec
+
+Lcbc_done:
+	addi		$out,$out,-1
+	lvx		$inout,0,$out		# redundant in aligned case
+	vsel		$inout,$outhead,$inout,$outmask
+	stvx		$inout,0,$out
+
+	neg		$enc,$ivp		# write [unaligned] iv
+	li		$idx,15			# 15 is not typo
+	vxor		$rndkey0,$rndkey0,$rndkey0
+	vspltisb	$outmask,-1
+	le?vspltisb	$tmp,0x0f
+	?lvsl		$outperm,0,$enc
+	?vperm		$outmask,$rndkey0,$outmask,$outperm
+	le?vxor		$outperm,$outperm,$tmp
+	lvx		$outhead,0,$ivp
+	vperm		$ivec,$ivec,$ivec,$outperm
+	vsel		$inout,$outhead,$ivec,$outmask
+	lvx		$inptail,$idx,$ivp
+	stvx		$inout,0,$ivp
+	vsel		$inout,$ivec,$inptail,$outmask
+	stvx		$inout,$idx,$ivp
+
+	mtspr		256,$vrsave
+	blr
+	.long		0
+	.byte		0,12,0x14,0,0,0,6,0
+	.long		0
+___
+#########################################################################
+{{	# Optimized CBC decrypt procedure				#
+my $key_="r11";
+my ($x00,$x10,$x20,$x30,$x40,$x50,$x60,$x70)=map("r$_",(0,8,26..31));
+    $x00=0 if ($flavour =~ /osx/);
+my ($in0, $in1, $in2, $in3, $in4, $in5, $in6, $in7 )=map("v$_",(0..3,10..13));
+my ($out0,$out1,$out2,$out3,$out4,$out5,$out6,$out7)=map("v$_",(14..21));
+my $rndkey0="v23";	# v24-v25 rotating buffer for first found keys
+			# v26-v31 last 6 round keys
+my ($tmp,$keyperm)=($in3,$in4);	# aliases with "caller", redundant assignment
+
+$code.=<<___;
+.align	5
+_aesp8_cbc_decrypt8x:
+	$STU		$sp,-`($FRAME+21*16+6*$SIZE_T)`($sp)
+	li		r10,`$FRAME+8*16+15`
+	li		r11,`$FRAME+8*16+31`
+	stvx		v20,r10,$sp		# ABI says so
+	addi		r10,r10,32
+	stvx		v21,r11,$sp
+	addi		r11,r11,32
+	stvx		v22,r10,$sp
+	addi		r10,r10,32
+	stvx		v23,r11,$sp
+	addi		r11,r11,32
+	stvx		v24,r10,$sp
+	addi		r10,r10,32
+	stvx		v25,r11,$sp
+	addi		r11,r11,32
+	stvx		v26,r10,$sp
+	addi		r10,r10,32
+	stvx		v27,r11,$sp
+	addi		r11,r11,32
+	stvx		v28,r10,$sp
+	addi		r10,r10,32
+	stvx		v29,r11,$sp
+	addi		r11,r11,32
+	stvx		v30,r10,$sp
+	stvx		v31,r11,$sp
+	li		r0,-1
+	stw		$vrsave,`$FRAME+21*16-4`($sp)	# save vrsave
+	li		$x10,0x10
+	$PUSH		r26,`$FRAME+21*16+0*$SIZE_T`($sp)
+	li		$x20,0x20
+	$PUSH		r27,`$FRAME+21*16+1*$SIZE_T`($sp)
+	li		$x30,0x30
+	$PUSH		r28,`$FRAME+21*16+2*$SIZE_T`($sp)
+	li		$x40,0x40
+	$PUSH		r29,`$FRAME+21*16+3*$SIZE_T`($sp)
+	li		$x50,0x50
+	$PUSH		r30,`$FRAME+21*16+4*$SIZE_T`($sp)
+	li		$x60,0x60
+	$PUSH		r31,`$FRAME+21*16+5*$SIZE_T`($sp)
+	li		$x70,0x70
+	mtspr		256,r0
+
+	subi		$rounds,$rounds,3	# -4 in total
+	subi		$len,$len,128		# bias
+
+	lvx		$rndkey0,$x00,$key	# load key schedule
+	lvx		v30,$x10,$key
+	addi		$key,$key,0x20
+	lvx		v31,$x00,$key
+	?vperm		$rndkey0,$rndkey0,v30,$keyperm
+	addi		$key_,$sp,$FRAME+15
+	mtctr		$rounds
+
+Load_cbc_dec_key:
+	?vperm		v24,v30,v31,$keyperm
+	lvx		v30,$x10,$key
+	addi		$key,$key,0x20
+	stvx		v24,$x00,$key_		# off-load round[1]
+	?vperm		v25,v31,v30,$keyperm
+	lvx		v31,$x00,$key
+	stvx		v25,$x10,$key_		# off-load round[2]
+	addi		$key_,$key_,0x20
+	bdnz		Load_cbc_dec_key
+
+	lvx		v26,$x10,$key
+	?vperm		v24,v30,v31,$keyperm
+	lvx		v27,$x20,$key
+	stvx		v24,$x00,$key_		# off-load round[3]
+	?vperm		v25,v31,v26,$keyperm
+	lvx		v28,$x30,$key
+	stvx		v25,$x10,$key_		# off-load round[4]
+	addi		$key_,$sp,$FRAME+15	# rewind $key_
+	?vperm		v26,v26,v27,$keyperm
+	lvx		v29,$x40,$key
+	?vperm		v27,v27,v28,$keyperm
+	lvx		v30,$x50,$key
+	?vperm		v28,v28,v29,$keyperm
+	lvx		v31,$x60,$key
+	?vperm		v29,v29,v30,$keyperm
+	lvx		$out0,$x70,$key		# borrow $out0
+	?vperm		v30,v30,v31,$keyperm
+	lvx		v24,$x00,$key_		# pre-load round[1]
+	?vperm		v31,v31,$out0,$keyperm
+	lvx		v25,$x10,$key_		# pre-load round[2]
+
+	#lvx		$inptail,0,$inp		# "caller" already did this
+	#addi		$inp,$inp,15		# 15 is not typo
+	subi		$inp,$inp,15		# undo "caller"
+
+	 le?li		$idx,8
+	lvx_u		$in0,$x00,$inp		# load first 8 "words"
+	 le?lvsl	$inpperm,0,$idx
+	 le?vspltisb	$tmp,0x0f
+	lvx_u		$in1,$x10,$inp
+	 le?vxor	$inpperm,$inpperm,$tmp	# transform for lvx_u/stvx_u
+	lvx_u		$in2,$x20,$inp
+	 le?vperm	$in0,$in0,$in0,$inpperm
+	lvx_u		$in3,$x30,$inp
+	 le?vperm	$in1,$in1,$in1,$inpperm
+	lvx_u		$in4,$x40,$inp
+	 le?vperm	$in2,$in2,$in2,$inpperm
+	vxor		$out0,$in0,$rndkey0
+	lvx_u		$in5,$x50,$inp
+	 le?vperm	$in3,$in3,$in3,$inpperm
+	vxor		$out1,$in1,$rndkey0
+	lvx_u		$in6,$x60,$inp
+	 le?vperm	$in4,$in4,$in4,$inpperm
+	vxor		$out2,$in2,$rndkey0
+	lvx_u		$in7,$x70,$inp
+	addi		$inp,$inp,0x80
+	 le?vperm	$in5,$in5,$in5,$inpperm
+	vxor		$out3,$in3,$rndkey0
+	 le?vperm	$in6,$in6,$in6,$inpperm
+	vxor		$out4,$in4,$rndkey0
+	 le?vperm	$in7,$in7,$in7,$inpperm
+	vxor		$out5,$in5,$rndkey0
+	vxor		$out6,$in6,$rndkey0
+	vxor		$out7,$in7,$rndkey0
+
+	mtctr		$rounds
+	b		Loop_cbc_dec8x
+.align	5
+Loop_cbc_dec8x:
+	vncipher	$out0,$out0,v24
+	vncipher	$out1,$out1,v24
+	vncipher	$out2,$out2,v24
+	vncipher	$out3,$out3,v24
+	vncipher	$out4,$out4,v24
+	vncipher	$out5,$out5,v24
+	vncipher	$out6,$out6,v24
+	vncipher	$out7,$out7,v24
+	lvx		v24,$x20,$key_		# round[3]
+	addi		$key_,$key_,0x20
+
+	vncipher	$out0,$out0,v25
+	vncipher	$out1,$out1,v25
+	vncipher	$out2,$out2,v25
+	vncipher	$out3,$out3,v25
+	vncipher	$out4,$out4,v25
+	vncipher	$out5,$out5,v25
+	vncipher	$out6,$out6,v25
+	vncipher	$out7,$out7,v25
+	lvx		v25,$x10,$key_		# round[4]
+	bdnz		Loop_cbc_dec8x
+
+	subic		$len,$len,128		# $len-=128
+	vncipher	$out0,$out0,v24
+	vncipher	$out1,$out1,v24
+	vncipher	$out2,$out2,v24
+	vncipher	$out3,$out3,v24
+	vncipher	$out4,$out4,v24
+	vncipher	$out5,$out5,v24
+	vncipher	$out6,$out6,v24
+	vncipher	$out7,$out7,v24
+
+	subfe.		r0,r0,r0		# borrow?-1:0
+	vncipher	$out0,$out0,v25
+	vncipher	$out1,$out1,v25
+	vncipher	$out2,$out2,v25
+	vncipher	$out3,$out3,v25
+	vncipher	$out4,$out4,v25
+	vncipher	$out5,$out5,v25
+	vncipher	$out6,$out6,v25
+	vncipher	$out7,$out7,v25
+
+	and		r0,r0,$len
+	vncipher	$out0,$out0,v26
+	vncipher	$out1,$out1,v26
+	vncipher	$out2,$out2,v26
+	vncipher	$out3,$out3,v26
+	vncipher	$out4,$out4,v26
+	vncipher	$out5,$out5,v26
+	vncipher	$out6,$out6,v26
+	vncipher	$out7,$out7,v26
+
+	add		$inp,$inp,r0		# $inp is adjusted in such
+						# way that at exit from the
+						# loop inX-in7 are loaded
+						# with last "words"
+	vncipher	$out0,$out0,v27
+	vncipher	$out1,$out1,v27
+	vncipher	$out2,$out2,v27
+	vncipher	$out3,$out3,v27
+	vncipher	$out4,$out4,v27
+	vncipher	$out5,$out5,v27
+	vncipher	$out6,$out6,v27
+	vncipher	$out7,$out7,v27
+
+	addi		$key_,$sp,$FRAME+15	# rewind $key_
+	vncipher	$out0,$out0,v28
+	vncipher	$out1,$out1,v28
+	vncipher	$out2,$out2,v28
+	vncipher	$out3,$out3,v28
+	vncipher	$out4,$out4,v28
+	vncipher	$out5,$out5,v28
+	vncipher	$out6,$out6,v28
+	vncipher	$out7,$out7,v28
+	lvx		v24,$x00,$key_		# re-pre-load round[1]
+
+	vncipher	$out0,$out0,v29
+	vncipher	$out1,$out1,v29
+	vncipher	$out2,$out2,v29
+	vncipher	$out3,$out3,v29
+	vncipher	$out4,$out4,v29
+	vncipher	$out5,$out5,v29
+	vncipher	$out6,$out6,v29
+	vncipher	$out7,$out7,v29
+	lvx		v25,$x10,$key_		# re-pre-load round[2]
+
+	vncipher	$out0,$out0,v30
+	 vxor		$ivec,$ivec,v31		# xor with last round key
+	vncipher	$out1,$out1,v30
+	 vxor		$in0,$in0,v31
+	vncipher	$out2,$out2,v30
+	 vxor		$in1,$in1,v31
+	vncipher	$out3,$out3,v30
+	 vxor		$in2,$in2,v31
+	vncipher	$out4,$out4,v30
+	 vxor		$in3,$in3,v31
+	vncipher	$out5,$out5,v30
+	 vxor		$in4,$in4,v31
+	vncipher	$out6,$out6,v30
+	 vxor		$in5,$in5,v31
+	vncipher	$out7,$out7,v30
+	 vxor		$in6,$in6,v31
+
+	vncipherlast	$out0,$out0,$ivec
+	vncipherlast	$out1,$out1,$in0
+	 lvx_u		$in0,$x00,$inp		# load next input block
+	vncipherlast	$out2,$out2,$in1
+	 lvx_u		$in1,$x10,$inp
+	vncipherlast	$out3,$out3,$in2
+	 le?vperm	$in0,$in0,$in0,$inpperm
+	 lvx_u		$in2,$x20,$inp
+	vncipherlast	$out4,$out4,$in3
+	 le?vperm	$in1,$in1,$in1,$inpperm
+	 lvx_u		$in3,$x30,$inp
+	vncipherlast	$out5,$out5,$in4
+	 le?vperm	$in2,$in2,$in2,$inpperm
+	 lvx_u		$in4,$x40,$inp
+	vncipherlast	$out6,$out6,$in5
+	 le?vperm	$in3,$in3,$in3,$inpperm
+	 lvx_u		$in5,$x50,$inp
+	vncipherlast	$out7,$out7,$in6
+	 le?vperm	$in4,$in4,$in4,$inpperm
+	 lvx_u		$in6,$x60,$inp
+	vmr		$ivec,$in7
+	 le?vperm	$in5,$in5,$in5,$inpperm
+	 lvx_u		$in7,$x70,$inp
+	 addi		$inp,$inp,0x80
+
+	le?vperm	$out0,$out0,$out0,$inpperm
+	le?vperm	$out1,$out1,$out1,$inpperm
+	stvx_u		$out0,$x00,$out
+	 le?vperm	$in6,$in6,$in6,$inpperm
+	 vxor		$out0,$in0,$rndkey0
+	le?vperm	$out2,$out2,$out2,$inpperm
+	stvx_u		$out1,$x10,$out
+	 le?vperm	$in7,$in7,$in7,$inpperm
+	 vxor		$out1,$in1,$rndkey0
+	le?vperm	$out3,$out3,$out3,$inpperm
+	stvx_u		$out2,$x20,$out
+	 vxor		$out2,$in2,$rndkey0
+	le?vperm	$out4,$out4,$out4,$inpperm
+	stvx_u		$out3,$x30,$out
+	 vxor		$out3,$in3,$rndkey0
+	le?vperm	$out5,$out5,$out5,$inpperm
+	stvx_u		$out4,$x40,$out
+	 vxor		$out4,$in4,$rndkey0
+	le?vperm	$out6,$out6,$out6,$inpperm
+	stvx_u		$out5,$x50,$out
+	 vxor		$out5,$in5,$rndkey0
+	le?vperm	$out7,$out7,$out7,$inpperm
+	stvx_u		$out6,$x60,$out
+	 vxor		$out6,$in6,$rndkey0
+	stvx_u		$out7,$x70,$out
+	addi		$out,$out,0x80
+	 vxor		$out7,$in7,$rndkey0
+
+	mtctr		$rounds
+	beq		Loop_cbc_dec8x		# did $len-=128 borrow?
+
+	addic.		$len,$len,128
+	beq		Lcbc_dec8x_done
+	nop
+	nop
+
+Loop_cbc_dec8x_tail:				# up to 7 "words" tail...
+	vncipher	$out1,$out1,v24
+	vncipher	$out2,$out2,v24
+	vncipher	$out3,$out3,v24
+	vncipher	$out4,$out4,v24
+	vncipher	$out5,$out5,v24
+	vncipher	$out6,$out6,v24
+	vncipher	$out7,$out7,v24
+	lvx		v24,$x20,$key_		# round[3]
+	addi		$key_,$key_,0x20
+
+	vncipher	$out1,$out1,v25
+	vncipher	$out2,$out2,v25
+	vncipher	$out3,$out3,v25
+	vncipher	$out4,$out4,v25
+	vncipher	$out5,$out5,v25
+	vncipher	$out6,$out6,v25
+	vncipher	$out7,$out7,v25
+	lvx		v25,$x10,$key_		# round[4]
+	bdnz		Loop_cbc_dec8x_tail
+
+	vncipher	$out1,$out1,v24
+	vncipher	$out2,$out2,v24
+	vncipher	$out3,$out3,v24
+	vncipher	$out4,$out4,v24
+	vncipher	$out5,$out5,v24
+	vncipher	$out6,$out6,v24
+	vncipher	$out7,$out7,v24
+
+	vncipher	$out1,$out1,v25
+	vncipher	$out2,$out2,v25
+	vncipher	$out3,$out3,v25
+	vncipher	$out4,$out4,v25
+	vncipher	$out5,$out5,v25
+	vncipher	$out6,$out6,v25
+	vncipher	$out7,$out7,v25
+
+	vncipher	$out1,$out1,v26
+	vncipher	$out2,$out2,v26
+	vncipher	$out3,$out3,v26
+	vncipher	$out4,$out4,v26
+	vncipher	$out5,$out5,v26
+	vncipher	$out6,$out6,v26
+	vncipher	$out7,$out7,v26
+
+	vncipher	$out1,$out1,v27
+	vncipher	$out2,$out2,v27
+	vncipher	$out3,$out3,v27
+	vncipher	$out4,$out4,v27
+	vncipher	$out5,$out5,v27
+	vncipher	$out6,$out6,v27
+	vncipher	$out7,$out7,v27
+
+	vncipher	$out1,$out1,v28
+	vncipher	$out2,$out2,v28
+	vncipher	$out3,$out3,v28
+	vncipher	$out4,$out4,v28
+	vncipher	$out5,$out5,v28
+	vncipher	$out6,$out6,v28
+	vncipher	$out7,$out7,v28
+
+	vncipher	$out1,$out1,v29
+	vncipher	$out2,$out2,v29
+	vncipher	$out3,$out3,v29
+	vncipher	$out4,$out4,v29
+	vncipher	$out5,$out5,v29
+	vncipher	$out6,$out6,v29
+	vncipher	$out7,$out7,v29
+
+	vncipher	$out1,$out1,v30
+	 vxor		$ivec,$ivec,v31		# last round key
+	vncipher	$out2,$out2,v30
+	 vxor		$in1,$in1,v31
+	vncipher	$out3,$out3,v30
+	 vxor		$in2,$in2,v31
+	vncipher	$out4,$out4,v30
+	 vxor		$in3,$in3,v31
+	vncipher	$out5,$out5,v30
+	 vxor		$in4,$in4,v31
+	vncipher	$out6,$out6,v30
+	 vxor		$in5,$in5,v31
+	vncipher	$out7,$out7,v30
+	 vxor		$in6,$in6,v31
+
+	cmplwi		$len,32			# switch($len)
+	blt		Lcbc_dec8x_one
+	nop
+	beq		Lcbc_dec8x_two
+	cmplwi		$len,64
+	blt		Lcbc_dec8x_three
+	nop
+	beq		Lcbc_dec8x_four
+	cmplwi		$len,96
+	blt		Lcbc_dec8x_five
+	nop
+	beq		Lcbc_dec8x_six
+
+Lcbc_dec8x_seven:
+	vncipherlast	$out1,$out1,$ivec
+	vncipherlast	$out2,$out2,$in1
+	vncipherlast	$out3,$out3,$in2
+	vncipherlast	$out4,$out4,$in3
+	vncipherlast	$out5,$out5,$in4
+	vncipherlast	$out6,$out6,$in5
+	vncipherlast	$out7,$out7,$in6
+	vmr		$ivec,$in7
+
+	le?vperm	$out1,$out1,$out1,$inpperm
+	le?vperm	$out2,$out2,$out2,$inpperm
+	stvx_u		$out1,$x00,$out
+	le?vperm	$out3,$out3,$out3,$inpperm
+	stvx_u		$out2,$x10,$out
+	le?vperm	$out4,$out4,$out4,$inpperm
+	stvx_u		$out3,$x20,$out
+	le?vperm	$out5,$out5,$out5,$inpperm
+	stvx_u		$out4,$x30,$out
+	le?vperm	$out6,$out6,$out6,$inpperm
+	stvx_u		$out5,$x40,$out
+	le?vperm	$out7,$out7,$out7,$inpperm
+	stvx_u		$out6,$x50,$out
+	stvx_u		$out7,$x60,$out
+	addi		$out,$out,0x70
+	b		Lcbc_dec8x_done
+
+.align	5
+Lcbc_dec8x_six:
+	vncipherlast	$out2,$out2,$ivec
+	vncipherlast	$out3,$out3,$in2
+	vncipherlast	$out4,$out4,$in3
+	vncipherlast	$out5,$out5,$in4
+	vncipherlast	$out6,$out6,$in5
+	vncipherlast	$out7,$out7,$in6
+	vmr		$ivec,$in7
+
+	le?vperm	$out2,$out2,$out2,$inpperm
+	le?vperm	$out3,$out3,$out3,$inpperm
+	stvx_u		$out2,$x00,$out
+	le?vperm	$out4,$out4,$out4,$inpperm
+	stvx_u		$out3,$x10,$out
+	le?vperm	$out5,$out5,$out5,$inpperm
+	stvx_u		$out4,$x20,$out
+	le?vperm	$out6,$out6,$out6,$inpperm
+	stvx_u		$out5,$x30,$out
+	le?vperm	$out7,$out7,$out7,$inpperm
+	stvx_u		$out6,$x40,$out
+	stvx_u		$out7,$x50,$out
+	addi		$out,$out,0x60
+	b		Lcbc_dec8x_done
+
+.align	5
+Lcbc_dec8x_five:
+	vncipherlast	$out3,$out3,$ivec
+	vncipherlast	$out4,$out4,$in3
+	vncipherlast	$out5,$out5,$in4
+	vncipherlast	$out6,$out6,$in5
+	vncipherlast	$out7,$out7,$in6
+	vmr		$ivec,$in7
+
+	le?vperm	$out3,$out3,$out3,$inpperm
+	le?vperm	$out4,$out4,$out4,$inpperm
+	stvx_u		$out3,$x00,$out
+	le?vperm	$out5,$out5,$out5,$inpperm
+	stvx_u		$out4,$x10,$out
+	le?vperm	$out6,$out6,$out6,$inpperm
+	stvx_u		$out5,$x20,$out
+	le?vperm	$out7,$out7,$out7,$inpperm
+	stvx_u		$out6,$x30,$out
+	stvx_u		$out7,$x40,$out
+	addi		$out,$out,0x50
+	b		Lcbc_dec8x_done
+
+.align	5
+Lcbc_dec8x_four:
+	vncipherlast	$out4,$out4,$ivec
+	vncipherlast	$out5,$out5,$in4
+	vncipherlast	$out6,$out6,$in5
+	vncipherlast	$out7,$out7,$in6
+	vmr		$ivec,$in7
+
+	le?vperm	$out4,$out4,$out4,$inpperm
+	le?vperm	$out5,$out5,$out5,$inpperm
+	stvx_u		$out4,$x00,$out
+	le?vperm	$out6,$out6,$out6,$inpperm
+	stvx_u		$out5,$x10,$out
+	le?vperm	$out7,$out7,$out7,$inpperm
+	stvx_u		$out6,$x20,$out
+	stvx_u		$out7,$x30,$out
+	addi		$out,$out,0x40
+	b		Lcbc_dec8x_done
+
+.align	5
+Lcbc_dec8x_three:
+	vncipherlast	$out5,$out5,$ivec
+	vncipherlast	$out6,$out6,$in5
+	vncipherlast	$out7,$out7,$in6
+	vmr		$ivec,$in7
+
+	le?vperm	$out5,$out5,$out5,$inpperm
+	le?vperm	$out6,$out6,$out6,$inpperm
+	stvx_u		$out5,$x00,$out
+	le?vperm	$out7,$out7,$out7,$inpperm
+	stvx_u		$out6,$x10,$out
+	stvx_u		$out7,$x20,$out
+	addi		$out,$out,0x30
+	b		Lcbc_dec8x_done
+
+.align	5
+Lcbc_dec8x_two:
+	vncipherlast	$out6,$out6,$ivec
+	vncipherlast	$out7,$out7,$in6
+	vmr		$ivec,$in7
+
+	le?vperm	$out6,$out6,$out6,$inpperm
+	le?vperm	$out7,$out7,$out7,$inpperm
+	stvx_u		$out6,$x00,$out
+	stvx_u		$out7,$x10,$out
+	addi		$out,$out,0x20
+	b		Lcbc_dec8x_done
+
+.align	5
+Lcbc_dec8x_one:
+	vncipherlast	$out7,$out7,$ivec
+	vmr		$ivec,$in7
+
+	le?vperm	$out7,$out7,$out7,$inpperm
+	stvx_u		$out7,0,$out
+	addi		$out,$out,0x10
+
+Lcbc_dec8x_done:
+	le?vperm	$ivec,$ivec,$ivec,$inpperm
+	stvx_u		$ivec,0,$ivp		# write [unaligned] iv
+
+	li		r10,`$FRAME+15`
+	li		r11,`$FRAME+31`
+	stvx		$inpperm,r10,$sp	# wipe copies of round keys
+	addi		r10,r10,32
+	stvx		$inpperm,r11,$sp
+	addi		r11,r11,32
+	stvx		$inpperm,r10,$sp
+	addi		r10,r10,32
+	stvx		$inpperm,r11,$sp
+	addi		r11,r11,32
+	stvx		$inpperm,r10,$sp
+	addi		r10,r10,32
+	stvx		$inpperm,r11,$sp
+	addi		r11,r11,32
+	stvx		$inpperm,r10,$sp
+	addi		r10,r10,32
+	stvx		$inpperm,r11,$sp
+	addi		r11,r11,32
+
+	mtspr		256,$vrsave
+	lvx		v20,r10,$sp		# ABI says so
+	addi		r10,r10,32
+	lvx		v21,r11,$sp
+	addi		r11,r11,32
+	lvx		v22,r10,$sp
+	addi		r10,r10,32
+	lvx		v23,r11,$sp
+	addi		r11,r11,32
+	lvx		v24,r10,$sp
+	addi		r10,r10,32
+	lvx		v25,r11,$sp
+	addi		r11,r11,32
+	lvx		v26,r10,$sp
+	addi		r10,r10,32
+	lvx		v27,r11,$sp
+	addi		r11,r11,32
+	lvx		v28,r10,$sp
+	addi		r10,r10,32
+	lvx		v29,r11,$sp
+	addi		r11,r11,32
+	lvx		v30,r10,$sp
+	lvx		v31,r11,$sp
+	$POP		r26,`$FRAME+21*16+0*$SIZE_T`($sp)
+	$POP		r27,`$FRAME+21*16+1*$SIZE_T`($sp)
+	$POP		r28,`$FRAME+21*16+2*$SIZE_T`($sp)
+	$POP		r29,`$FRAME+21*16+3*$SIZE_T`($sp)
+	$POP		r30,`$FRAME+21*16+4*$SIZE_T`($sp)
+	$POP		r31,`$FRAME+21*16+5*$SIZE_T`($sp)
+	addi		$sp,$sp,`$FRAME+21*16+6*$SIZE_T`
+	blr
+	.long		0
+	.byte		0,12,0x04,0,0x80,6,6,0
+	.long		0
+.size	.${prefix}_cbc_encrypt,.-.${prefix}_cbc_encrypt
+___
+}}	}}}
+
+#########################################################################
+{{{	# CTR procedure[s]						#
+my ($inp,$out,$len,$key,$ivp,$x10,$rounds,$idx)=map("r$_",(3..10));
+my ($rndkey0,$rndkey1,$inout,$tmp)=		map("v$_",(0..3));
+my ($ivec,$inptail,$inpperm,$outhead,$outperm,$outmask,$keyperm,$one)=
+						map("v$_",(4..11));
+my $dat=$tmp;
+
+$code.=<<___;
+.globl	.${prefix}_ctr32_encrypt_blocks
+.align	5
+.${prefix}_ctr32_encrypt_blocks:
+	${UCMP}i	$len,1
+	bltlr-
+
+	lis		r0,0xfff0
+	mfspr		$vrsave,256
+	mtspr		256,r0
+
+	li		$idx,15
+	vxor		$rndkey0,$rndkey0,$rndkey0
+	le?vspltisb	$tmp,0x0f
+
+	lvx		$ivec,0,$ivp		# load [unaligned] iv
+	lvsl		$inpperm,0,$ivp
+	lvx		$inptail,$idx,$ivp
+	 vspltisb	$one,1
+	le?vxor		$inpperm,$inpperm,$tmp
+	vperm		$ivec,$ivec,$inptail,$inpperm
+	 vsldoi		$one,$rndkey0,$one,1
+
+	neg		r11,$inp
+	?lvsl		$keyperm,0,$key		# prepare for unaligned key
+	lwz		$rounds,240($key)
+
+	lvsr		$inpperm,0,r11		# prepare for unaligned load
+	lvx		$inptail,0,$inp
+	addi		$inp,$inp,15		# 15 is not typo
+	le?vxor		$inpperm,$inpperm,$tmp
+
+	srwi		$rounds,$rounds,1
+	li		$idx,16
+	subi		$rounds,$rounds,1
+
+	${UCMP}i	$len,8
+	bge		_aesp8_ctr32_encrypt8x
+
+	?lvsr		$outperm,0,$out		# prepare for unaligned store
+	vspltisb	$outmask,-1
+	lvx		$outhead,0,$out
+	?vperm		$outmask,$rndkey0,$outmask,$outperm
+	le?vxor		$outperm,$outperm,$tmp
+
+	lvx		$rndkey0,0,$key
+	mtctr		$rounds
+	lvx		$rndkey1,$idx,$key
+	addi		$idx,$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vxor		$inout,$ivec,$rndkey0
+	lvx		$rndkey0,$idx,$key
+	addi		$idx,$idx,16
+	b		Loop_ctr32_enc
+
+.align	5
+Loop_ctr32_enc:
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vcipher		$inout,$inout,$rndkey1
+	lvx		$rndkey1,$idx,$key
+	addi		$idx,$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vcipher		$inout,$inout,$rndkey0
+	lvx		$rndkey0,$idx,$key
+	addi		$idx,$idx,16
+	bdnz		Loop_ctr32_enc
+
+	vadduwm		$ivec,$ivec,$one
+	 vmr		$dat,$inptail
+	 lvx		$inptail,0,$inp
+	 addi		$inp,$inp,16
+	 subic.		$len,$len,1		# blocks--
+
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vcipher		$inout,$inout,$rndkey1
+	lvx		$rndkey1,$idx,$key
+	 vperm		$dat,$dat,$inptail,$inpperm
+	 li		$idx,16
+	?vperm		$rndkey1,$rndkey0,$rndkey1,$keyperm
+	 lvx		$rndkey0,0,$key
+	vxor		$dat,$dat,$rndkey1	# last round key
+	vcipherlast	$inout,$inout,$dat
+
+	 lvx		$rndkey1,$idx,$key
+	 addi		$idx,$idx,16
+	vperm		$inout,$inout,$inout,$outperm
+	vsel		$dat,$outhead,$inout,$outmask
+	 mtctr		$rounds
+	 ?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vmr		$outhead,$inout
+	 vxor		$inout,$ivec,$rndkey0
+	 lvx		$rndkey0,$idx,$key
+	 addi		$idx,$idx,16
+	stvx		$dat,0,$out
+	addi		$out,$out,16
+	bne		Loop_ctr32_enc
+
+	addi		$out,$out,-1
+	lvx		$inout,0,$out		# redundant in aligned case
+	vsel		$inout,$outhead,$inout,$outmask
+	stvx		$inout,0,$out
+
+	mtspr		256,$vrsave
+	blr
+	.long		0
+	.byte		0,12,0x14,0,0,0,6,0
+	.long		0
+___
+#########################################################################
+{{	# Optimized CTR procedure					#
+my $key_="r11";
+my ($x00,$x10,$x20,$x30,$x40,$x50,$x60,$x70)=map("r$_",(0,8,26..31));
+    $x00=0 if ($flavour =~ /osx/);
+my ($in0, $in1, $in2, $in3, $in4, $in5, $in6, $in7 )=map("v$_",(0..3,10,12..14));
+my ($out0,$out1,$out2,$out3,$out4,$out5,$out6,$out7)=map("v$_",(15..22));
+my $rndkey0="v23";	# v24-v25 rotating buffer for first found keys
+			# v26-v31 last 6 round keys
+my ($tmp,$keyperm)=($in3,$in4);	# aliases with "caller", redundant assignment
+my ($two,$three,$four)=($outhead,$outperm,$outmask);
+
+$code.=<<___;
+.align	5
+_aesp8_ctr32_encrypt8x:
+	$STU		$sp,-`($FRAME+21*16+6*$SIZE_T)`($sp)
+	li		r10,`$FRAME+8*16+15`
+	li		r11,`$FRAME+8*16+31`
+	stvx		v20,r10,$sp		# ABI says so
+	addi		r10,r10,32
+	stvx		v21,r11,$sp
+	addi		r11,r11,32
+	stvx		v22,r10,$sp
+	addi		r10,r10,32
+	stvx		v23,r11,$sp
+	addi		r11,r11,32
+	stvx		v24,r10,$sp
+	addi		r10,r10,32
+	stvx		v25,r11,$sp
+	addi		r11,r11,32
+	stvx		v26,r10,$sp
+	addi		r10,r10,32
+	stvx		v27,r11,$sp
+	addi		r11,r11,32
+	stvx		v28,r10,$sp
+	addi		r10,r10,32
+	stvx		v29,r11,$sp
+	addi		r11,r11,32
+	stvx		v30,r10,$sp
+	stvx		v31,r11,$sp
+	li		r0,-1
+	stw		$vrsave,`$FRAME+21*16-4`($sp)	# save vrsave
+	li		$x10,0x10
+	$PUSH		r26,`$FRAME+21*16+0*$SIZE_T`($sp)
+	li		$x20,0x20
+	$PUSH		r27,`$FRAME+21*16+1*$SIZE_T`($sp)
+	li		$x30,0x30
+	$PUSH		r28,`$FRAME+21*16+2*$SIZE_T`($sp)
+	li		$x40,0x40
+	$PUSH		r29,`$FRAME+21*16+3*$SIZE_T`($sp)
+	li		$x50,0x50
+	$PUSH		r30,`$FRAME+21*16+4*$SIZE_T`($sp)
+	li		$x60,0x60
+	$PUSH		r31,`$FRAME+21*16+5*$SIZE_T`($sp)
+	li		$x70,0x70
+	mtspr		256,r0
+
+	subi		$rounds,$rounds,3	# -4 in total
+
+	lvx		$rndkey0,$x00,$key	# load key schedule
+	lvx		v30,$x10,$key
+	addi		$key,$key,0x20
+	lvx		v31,$x00,$key
+	?vperm		$rndkey0,$rndkey0,v30,$keyperm
+	addi		$key_,$sp,$FRAME+15
+	mtctr		$rounds
+
+Load_ctr32_enc_key:
+	?vperm		v24,v30,v31,$keyperm
+	lvx		v30,$x10,$key
+	addi		$key,$key,0x20
+	stvx		v24,$x00,$key_		# off-load round[1]
+	?vperm		v25,v31,v30,$keyperm
+	lvx		v31,$x00,$key
+	stvx		v25,$x10,$key_		# off-load round[2]
+	addi		$key_,$key_,0x20
+	bdnz		Load_ctr32_enc_key
+
+	lvx		v26,$x10,$key
+	?vperm		v24,v30,v31,$keyperm
+	lvx		v27,$x20,$key
+	stvx		v24,$x00,$key_		# off-load round[3]
+	?vperm		v25,v31,v26,$keyperm
+	lvx		v28,$x30,$key
+	stvx		v25,$x10,$key_		# off-load round[4]
+	addi		$key_,$sp,$FRAME+15	# rewind $key_
+	?vperm		v26,v26,v27,$keyperm
+	lvx		v29,$x40,$key
+	?vperm		v27,v27,v28,$keyperm
+	lvx		v30,$x50,$key
+	?vperm		v28,v28,v29,$keyperm
+	lvx		v31,$x60,$key
+	?vperm		v29,v29,v30,$keyperm
+	lvx		$out0,$x70,$key		# borrow $out0
+	?vperm		v30,v30,v31,$keyperm
+	lvx		v24,$x00,$key_		# pre-load round[1]
+	?vperm		v31,v31,$out0,$keyperm
+	lvx		v25,$x10,$key_		# pre-load round[2]
+
+	vadduwm		$two,$one,$one
+	subi		$inp,$inp,15		# undo "caller"
+	$SHL		$len,$len,4
+
+	vadduwm		$out1,$ivec,$one	# counter values ...
+	vadduwm		$out2,$ivec,$two
+	vxor		$out0,$ivec,$rndkey0	# ... xored with rndkey[0]
+	 le?li		$idx,8
+	vadduwm		$out3,$out1,$two
+	vxor		$out1,$out1,$rndkey0
+	 le?lvsl	$inpperm,0,$idx
+	vadduwm		$out4,$out2,$two
+	vxor		$out2,$out2,$rndkey0
+	 le?vspltisb	$tmp,0x0f
+	vadduwm		$out5,$out3,$two
+	vxor		$out3,$out3,$rndkey0
+	 le?vxor	$inpperm,$inpperm,$tmp	# transform for lvx_u/stvx_u
+	vadduwm		$out6,$out4,$two
+	vxor		$out4,$out4,$rndkey0
+	vadduwm		$out7,$out5,$two
+	vxor		$out5,$out5,$rndkey0
+	vadduwm		$ivec,$out6,$two	# next counter value
+	vxor		$out6,$out6,$rndkey0
+	vxor		$out7,$out7,$rndkey0
+
+	mtctr		$rounds
+	b		Loop_ctr32_enc8x
+.align	5
+Loop_ctr32_enc8x:
+	vcipher 	$out0,$out0,v24
+	vcipher 	$out1,$out1,v24
+	vcipher 	$out2,$out2,v24
+	vcipher 	$out3,$out3,v24
+	vcipher 	$out4,$out4,v24
+	vcipher 	$out5,$out5,v24
+	vcipher 	$out6,$out6,v24
+	vcipher 	$out7,$out7,v24
+Loop_ctr32_enc8x_middle:
+	lvx		v24,$x20,$key_		# round[3]
+	addi		$key_,$key_,0x20
+
+	vcipher 	$out0,$out0,v25
+	vcipher 	$out1,$out1,v25
+	vcipher 	$out2,$out2,v25
+	vcipher 	$out3,$out3,v25
+	vcipher 	$out4,$out4,v25
+	vcipher 	$out5,$out5,v25
+	vcipher 	$out6,$out6,v25
+	vcipher 	$out7,$out7,v25
+	lvx		v25,$x10,$key_		# round[4]
+	bdnz		Loop_ctr32_enc8x
+
+	subic		r11,$len,256		# $len-256, borrow $key_
+	vcipher 	$out0,$out0,v24
+	vcipher 	$out1,$out1,v24
+	vcipher 	$out2,$out2,v24
+	vcipher 	$out3,$out3,v24
+	vcipher 	$out4,$out4,v24
+	vcipher 	$out5,$out5,v24
+	vcipher 	$out6,$out6,v24
+	vcipher 	$out7,$out7,v24
+
+	subfe		r0,r0,r0		# borrow?-1:0
+	vcipher 	$out0,$out0,v25
+	vcipher 	$out1,$out1,v25
+	vcipher 	$out2,$out2,v25
+	vcipher 	$out3,$out3,v25
+	vcipher 	$out4,$out4,v25
+	vcipher		$out5,$out5,v25
+	vcipher		$out6,$out6,v25
+	vcipher		$out7,$out7,v25
+
+	and		r0,r0,r11
+	addi		$key_,$sp,$FRAME+15	# rewind $key_
+	vcipher		$out0,$out0,v26
+	vcipher		$out1,$out1,v26
+	vcipher		$out2,$out2,v26
+	vcipher		$out3,$out3,v26
+	vcipher		$out4,$out4,v26
+	vcipher		$out5,$out5,v26
+	vcipher		$out6,$out6,v26
+	vcipher		$out7,$out7,v26
+	lvx		v24,$x00,$key_		# re-pre-load round[1]
+
+	subic		$len,$len,129		# $len-=129
+	vcipher		$out0,$out0,v27
+	addi		$len,$len,1		# $len-=128 really
+	vcipher		$out1,$out1,v27
+	vcipher		$out2,$out2,v27
+	vcipher		$out3,$out3,v27
+	vcipher		$out4,$out4,v27
+	vcipher		$out5,$out5,v27
+	vcipher		$out6,$out6,v27
+	vcipher		$out7,$out7,v27
+	lvx		v25,$x10,$key_		# re-pre-load round[2]
+
+	vcipher		$out0,$out0,v28
+	 lvx_u		$in0,$x00,$inp		# load input
+	vcipher		$out1,$out1,v28
+	 lvx_u		$in1,$x10,$inp
+	vcipher		$out2,$out2,v28
+	 lvx_u		$in2,$x20,$inp
+	vcipher		$out3,$out3,v28
+	 lvx_u		$in3,$x30,$inp
+	vcipher		$out4,$out4,v28
+	 lvx_u		$in4,$x40,$inp
+	vcipher		$out5,$out5,v28
+	 lvx_u		$in5,$x50,$inp
+	vcipher		$out6,$out6,v28
+	 lvx_u		$in6,$x60,$inp
+	vcipher		$out7,$out7,v28
+	 lvx_u		$in7,$x70,$inp
+	 addi		$inp,$inp,0x80
+
+	vcipher		$out0,$out0,v29
+	 le?vperm	$in0,$in0,$in0,$inpperm
+	vcipher		$out1,$out1,v29
+	 le?vperm	$in1,$in1,$in1,$inpperm
+	vcipher		$out2,$out2,v29
+	 le?vperm	$in2,$in2,$in2,$inpperm
+	vcipher		$out3,$out3,v29
+	 le?vperm	$in3,$in3,$in3,$inpperm
+	vcipher		$out4,$out4,v29
+	 le?vperm	$in4,$in4,$in4,$inpperm
+	vcipher		$out5,$out5,v29
+	 le?vperm	$in5,$in5,$in5,$inpperm
+	vcipher		$out6,$out6,v29
+	 le?vperm	$in6,$in6,$in6,$inpperm
+	vcipher		$out7,$out7,v29
+	 le?vperm	$in7,$in7,$in7,$inpperm
+
+	add		$inp,$inp,r0		# $inp is adjusted in such
+						# way that at exit from the
+						# loop inX-in7 are loaded
+						# with last "words"
+	subfe.		r0,r0,r0		# borrow?-1:0
+	vcipher		$out0,$out0,v30
+	 vxor		$in0,$in0,v31		# xor with last round key
+	vcipher		$out1,$out1,v30
+	 vxor		$in1,$in1,v31
+	vcipher		$out2,$out2,v30
+	 vxor		$in2,$in2,v31
+	vcipher		$out3,$out3,v30
+	 vxor		$in3,$in3,v31
+	vcipher		$out4,$out4,v30
+	 vxor		$in4,$in4,v31
+	vcipher		$out5,$out5,v30
+	 vxor		$in5,$in5,v31
+	vcipher		$out6,$out6,v30
+	 vxor		$in6,$in6,v31
+	vcipher		$out7,$out7,v30
+	 vxor		$in7,$in7,v31
+
+	bne		Lctr32_enc8x_break	# did $len-129 borrow?
+
+	vcipherlast	$in0,$out0,$in0
+	vcipherlast	$in1,$out1,$in1
+	 vadduwm	$out1,$ivec,$one	# counter values ...
+	vcipherlast	$in2,$out2,$in2
+	 vadduwm	$out2,$ivec,$two
+	 vxor		$out0,$ivec,$rndkey0	# ... xored with rndkey[0]
+	vcipherlast	$in3,$out3,$in3
+	 vadduwm	$out3,$out1,$two
+	 vxor		$out1,$out1,$rndkey0
+	vcipherlast	$in4,$out4,$in4
+	 vadduwm	$out4,$out2,$two
+	 vxor		$out2,$out2,$rndkey0
+	vcipherlast	$in5,$out5,$in5
+	 vadduwm	$out5,$out3,$two
+	 vxor		$out3,$out3,$rndkey0
+	vcipherlast	$in6,$out6,$in6
+	 vadduwm	$out6,$out4,$two
+	 vxor		$out4,$out4,$rndkey0
+	vcipherlast	$in7,$out7,$in7
+	 vadduwm	$out7,$out5,$two
+	 vxor		$out5,$out5,$rndkey0
+	le?vperm	$in0,$in0,$in0,$inpperm
+	 vadduwm	$ivec,$out6,$two	# next counter value
+	 vxor		$out6,$out6,$rndkey0
+	le?vperm	$in1,$in1,$in1,$inpperm
+	 vxor		$out7,$out7,$rndkey0
+	mtctr		$rounds
+
+	 vcipher	$out0,$out0,v24
+	stvx_u		$in0,$x00,$out
+	le?vperm	$in2,$in2,$in2,$inpperm
+	 vcipher	$out1,$out1,v24
+	stvx_u		$in1,$x10,$out
+	le?vperm	$in3,$in3,$in3,$inpperm
+	 vcipher	$out2,$out2,v24
+	stvx_u		$in2,$x20,$out
+	le?vperm	$in4,$in4,$in4,$inpperm
+	 vcipher	$out3,$out3,v24
+	stvx_u		$in3,$x30,$out
+	le?vperm	$in5,$in5,$in5,$inpperm
+	 vcipher	$out4,$out4,v24
+	stvx_u		$in4,$x40,$out
+	le?vperm	$in6,$in6,$in6,$inpperm
+	 vcipher	$out5,$out5,v24
+	stvx_u		$in5,$x50,$out
+	le?vperm	$in7,$in7,$in7,$inpperm
+	 vcipher	$out6,$out6,v24
+	stvx_u		$in6,$x60,$out
+	 vcipher	$out7,$out7,v24
+	stvx_u		$in7,$x70,$out
+	addi		$out,$out,0x80
+
+	b		Loop_ctr32_enc8x_middle
+
+.align	5
+Lctr32_enc8x_break:
+	cmpwi		$len,-0x60
+	blt		Lctr32_enc8x_one
+	nop
+	beq		Lctr32_enc8x_two
+	cmpwi		$len,-0x40
+	blt		Lctr32_enc8x_three
+	nop
+	beq		Lctr32_enc8x_four
+	cmpwi		$len,-0x20
+	blt		Lctr32_enc8x_five
+	nop
+	beq		Lctr32_enc8x_six
+	cmpwi		$len,0x00
+	blt		Lctr32_enc8x_seven
+
+Lctr32_enc8x_eight:
+	vcipherlast	$out0,$out0,$in0
+	vcipherlast	$out1,$out1,$in1
+	vcipherlast	$out2,$out2,$in2
+	vcipherlast	$out3,$out3,$in3
+	vcipherlast	$out4,$out4,$in4
+	vcipherlast	$out5,$out5,$in5
+	vcipherlast	$out6,$out6,$in6
+	vcipherlast	$out7,$out7,$in7
+
+	le?vperm	$out0,$out0,$out0,$inpperm
+	le?vperm	$out1,$out1,$out1,$inpperm
+	stvx_u		$out0,$x00,$out
+	le?vperm	$out2,$out2,$out2,$inpperm
+	stvx_u		$out1,$x10,$out
+	le?vperm	$out3,$out3,$out3,$inpperm
+	stvx_u		$out2,$x20,$out
+	le?vperm	$out4,$out4,$out4,$inpperm
+	stvx_u		$out3,$x30,$out
+	le?vperm	$out5,$out5,$out5,$inpperm
+	stvx_u		$out4,$x40,$out
+	le?vperm	$out6,$out6,$out6,$inpperm
+	stvx_u		$out5,$x50,$out
+	le?vperm	$out7,$out7,$out7,$inpperm
+	stvx_u		$out6,$x60,$out
+	stvx_u		$out7,$x70,$out
+	addi		$out,$out,0x80
+	b		Lctr32_enc8x_done
+
+.align	5
+Lctr32_enc8x_seven:
+	vcipherlast	$out0,$out0,$in1
+	vcipherlast	$out1,$out1,$in2
+	vcipherlast	$out2,$out2,$in3
+	vcipherlast	$out3,$out3,$in4
+	vcipherlast	$out4,$out4,$in5
+	vcipherlast	$out5,$out5,$in6
+	vcipherlast	$out6,$out6,$in7
+
+	le?vperm	$out0,$out0,$out0,$inpperm
+	le?vperm	$out1,$out1,$out1,$inpperm
+	stvx_u		$out0,$x00,$out
+	le?vperm	$out2,$out2,$out2,$inpperm
+	stvx_u		$out1,$x10,$out
+	le?vperm	$out3,$out3,$out3,$inpperm
+	stvx_u		$out2,$x20,$out
+	le?vperm	$out4,$out4,$out4,$inpperm
+	stvx_u		$out3,$x30,$out
+	le?vperm	$out5,$out5,$out5,$inpperm
+	stvx_u		$out4,$x40,$out
+	le?vperm	$out6,$out6,$out6,$inpperm
+	stvx_u		$out5,$x50,$out
+	stvx_u		$out6,$x60,$out
+	addi		$out,$out,0x70
+	b		Lctr32_enc8x_done
+
+.align	5
+Lctr32_enc8x_six:
+	vcipherlast	$out0,$out0,$in2
+	vcipherlast	$out1,$out1,$in3
+	vcipherlast	$out2,$out2,$in4
+	vcipherlast	$out3,$out3,$in5
+	vcipherlast	$out4,$out4,$in6
+	vcipherlast	$out5,$out5,$in7
+
+	le?vperm	$out0,$out0,$out0,$inpperm
+	le?vperm	$out1,$out1,$out1,$inpperm
+	stvx_u		$out0,$x00,$out
+	le?vperm	$out2,$out2,$out2,$inpperm
+	stvx_u		$out1,$x10,$out
+	le?vperm	$out3,$out3,$out3,$inpperm
+	stvx_u		$out2,$x20,$out
+	le?vperm	$out4,$out4,$out4,$inpperm
+	stvx_u		$out3,$x30,$out
+	le?vperm	$out5,$out5,$out5,$inpperm
+	stvx_u		$out4,$x40,$out
+	stvx_u		$out5,$x50,$out
+	addi		$out,$out,0x60
+	b		Lctr32_enc8x_done
+
+.align	5
+Lctr32_enc8x_five:
+	vcipherlast	$out0,$out0,$in3
+	vcipherlast	$out1,$out1,$in4
+	vcipherlast	$out2,$out2,$in5
+	vcipherlast	$out3,$out3,$in6
+	vcipherlast	$out4,$out4,$in7
+
+	le?vperm	$out0,$out0,$out0,$inpperm
+	le?vperm	$out1,$out1,$out1,$inpperm
+	stvx_u		$out0,$x00,$out
+	le?vperm	$out2,$out2,$out2,$inpperm
+	stvx_u		$out1,$x10,$out
+	le?vperm	$out3,$out3,$out3,$inpperm
+	stvx_u		$out2,$x20,$out
+	le?vperm	$out4,$out4,$out4,$inpperm
+	stvx_u		$out3,$x30,$out
+	stvx_u		$out4,$x40,$out
+	addi		$out,$out,0x50
+	b		Lctr32_enc8x_done
+
+.align	5
+Lctr32_enc8x_four:
+	vcipherlast	$out0,$out0,$in4
+	vcipherlast	$out1,$out1,$in5
+	vcipherlast	$out2,$out2,$in6
+	vcipherlast	$out3,$out3,$in7
+
+	le?vperm	$out0,$out0,$out0,$inpperm
+	le?vperm	$out1,$out1,$out1,$inpperm
+	stvx_u		$out0,$x00,$out
+	le?vperm	$out2,$out2,$out2,$inpperm
+	stvx_u		$out1,$x10,$out
+	le?vperm	$out3,$out3,$out3,$inpperm
+	stvx_u		$out2,$x20,$out
+	stvx_u		$out3,$x30,$out
+	addi		$out,$out,0x40
+	b		Lctr32_enc8x_done
+
+.align	5
+Lctr32_enc8x_three:
+	vcipherlast	$out0,$out0,$in5
+	vcipherlast	$out1,$out1,$in6
+	vcipherlast	$out2,$out2,$in7
+
+	le?vperm	$out0,$out0,$out0,$inpperm
+	le?vperm	$out1,$out1,$out1,$inpperm
+	stvx_u		$out0,$x00,$out
+	le?vperm	$out2,$out2,$out2,$inpperm
+	stvx_u		$out1,$x10,$out
+	stvx_u		$out2,$x20,$out
+	addi		$out,$out,0x30
+	b		Lctr32_enc8x_done
+
+.align	5
+Lctr32_enc8x_two:
+	vcipherlast	$out0,$out0,$in6
+	vcipherlast	$out1,$out1,$in7
+
+	le?vperm	$out0,$out0,$out0,$inpperm
+	le?vperm	$out1,$out1,$out1,$inpperm
+	stvx_u		$out0,$x00,$out
+	stvx_u		$out1,$x10,$out
+	addi		$out,$out,0x20
+	b		Lctr32_enc8x_done
+
+.align	5
+Lctr32_enc8x_one:
+	vcipherlast	$out0,$out0,$in7
+
+	le?vperm	$out0,$out0,$out0,$inpperm
+	stvx_u		$out0,0,$out
+	addi		$out,$out,0x10
+
+Lctr32_enc8x_done:
+	li		r10,`$FRAME+15`
+	li		r11,`$FRAME+31`
+	stvx		$inpperm,r10,$sp	# wipe copies of round keys
+	addi		r10,r10,32
+	stvx		$inpperm,r11,$sp
+	addi		r11,r11,32
+	stvx		$inpperm,r10,$sp
+	addi		r10,r10,32
+	stvx		$inpperm,r11,$sp
+	addi		r11,r11,32
+	stvx		$inpperm,r10,$sp
+	addi		r10,r10,32
+	stvx		$inpperm,r11,$sp
+	addi		r11,r11,32
+	stvx		$inpperm,r10,$sp
+	addi		r10,r10,32
+	stvx		$inpperm,r11,$sp
+	addi		r11,r11,32
+
+	mtspr		256,$vrsave
+	lvx		v20,r10,$sp		# ABI says so
+	addi		r10,r10,32
+	lvx		v21,r11,$sp
+	addi		r11,r11,32
+	lvx		v22,r10,$sp
+	addi		r10,r10,32
+	lvx		v23,r11,$sp
+	addi		r11,r11,32
+	lvx		v24,r10,$sp
+	addi		r10,r10,32
+	lvx		v25,r11,$sp
+	addi		r11,r11,32
+	lvx		v26,r10,$sp
+	addi		r10,r10,32
+	lvx		v27,r11,$sp
+	addi		r11,r11,32
+	lvx		v28,r10,$sp
+	addi		r10,r10,32
+	lvx		v29,r11,$sp
+	addi		r11,r11,32
+	lvx		v30,r10,$sp
+	lvx		v31,r11,$sp
+	$POP		r26,`$FRAME+21*16+0*$SIZE_T`($sp)
+	$POP		r27,`$FRAME+21*16+1*$SIZE_T`($sp)
+	$POP		r28,`$FRAME+21*16+2*$SIZE_T`($sp)
+	$POP		r29,`$FRAME+21*16+3*$SIZE_T`($sp)
+	$POP		r30,`$FRAME+21*16+4*$SIZE_T`($sp)
+	$POP		r31,`$FRAME+21*16+5*$SIZE_T`($sp)
+	addi		$sp,$sp,`$FRAME+21*16+6*$SIZE_T`
+	blr
+	.long		0
+	.byte		0,12,0x04,0,0x80,6,6,0
+	.long		0
+.size	.${prefix}_ctr32_encrypt_blocks,.-.${prefix}_ctr32_encrypt_blocks
+___
+}}	}}}
+
+#########################################################################
+{{{	# XTS procedures						#
+# int aes_p8_xts_[en|de]crypt(const char *inp, char *out, size_t len,	#
+#                             const AES_KEY *key1, const AES_KEY *key2,	#
+#                             [const] unsigned char iv[16]);		#
+# If $key2 is NULL, then a "tweak chaining" mode is engaged, in which	#
+# input tweak value is assumed to be encrypted already, and last tweak	#
+# value, one suitable for consecutive call on same chunk of data, is	#
+# written back to original buffer. In addition, in "tweak chaining"	#
+# mode only complete input blocks are processed.			#
+
+my ($inp,$out,$len,$key1,$key2,$ivp,$rounds,$idx) =	map("r$_",(3..10));
+my ($rndkey0,$rndkey1,$inout) =				map("v$_",(0..2));
+my ($output,$inptail,$inpperm,$leperm,$keyperm) =	map("v$_",(3..7));
+my ($tweak,$seven,$eighty7,$tmp,$tweak1) =		map("v$_",(8..12));
+my $taillen = $key2;
+
+   ($inp,$idx) = ($idx,$inp);				# reassign
+
+$code.=<<___;
+.globl	.${prefix}_xts_encrypt
+.align	5
+.${prefix}_xts_encrypt:
+	mr		$inp,r3				# reassign
+	li		r3,-1
+	${UCMP}i	$len,16
+	bltlr-
+
+	lis		r0,0xfff0
+	mfspr		r12,256				# save vrsave
+	li		r11,0
+	mtspr		256,r0
+
+	vspltisb	$seven,0x07			# 0x070707..07
+	le?lvsl		$leperm,r11,r11
+	le?vspltisb	$tmp,0x0f
+	le?vxor		$leperm,$leperm,$seven
+
+	li		$idx,15
+	lvx		$tweak,0,$ivp			# load [unaligned] iv
+	lvsl		$inpperm,0,$ivp
+	lvx		$inptail,$idx,$ivp
+	le?vxor		$inpperm,$inpperm,$tmp
+	vperm		$tweak,$tweak,$inptail,$inpperm
+
+	neg		r11,$inp
+	lvsr		$inpperm,0,r11			# prepare for unaligned load
+	lvx		$inout,0,$inp
+	addi		$inp,$inp,15			# 15 is not typo
+	le?vxor		$inpperm,$inpperm,$tmp
+
+	${UCMP}i	$key2,0				# key2==NULL?
+	beq		Lxts_enc_no_key2
+
+	?lvsl		$keyperm,0,$key2		# prepare for unaligned key
+	lwz		$rounds,240($key2)
+	srwi		$rounds,$rounds,1
+	subi		$rounds,$rounds,1
+	li		$idx,16
+
+	lvx		$rndkey0,0,$key2
+	lvx		$rndkey1,$idx,$key2
+	addi		$idx,$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vxor		$tweak,$tweak,$rndkey0
+	lvx		$rndkey0,$idx,$key2
+	addi		$idx,$idx,16
+	mtctr		$rounds
+
+Ltweak_xts_enc:
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vcipher		$tweak,$tweak,$rndkey1
+	lvx		$rndkey1,$idx,$key2
+	addi		$idx,$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vcipher		$tweak,$tweak,$rndkey0
+	lvx		$rndkey0,$idx,$key2
+	addi		$idx,$idx,16
+	bdnz		Ltweak_xts_enc
+
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vcipher		$tweak,$tweak,$rndkey1
+	lvx		$rndkey1,$idx,$key2
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vcipherlast	$tweak,$tweak,$rndkey0
+
+	li		$ivp,0				# don't chain the tweak
+	b		Lxts_enc
+
+Lxts_enc_no_key2:
+	li		$idx,-16
+	and		$len,$len,$idx			# in "tweak chaining"
+							# mode only complete
+							# blocks are processed
+Lxts_enc:
+	lvx		$inptail,0,$inp
+	addi		$inp,$inp,16
+
+	?lvsl		$keyperm,0,$key1		# prepare for unaligned key
+	lwz		$rounds,240($key1)
+	srwi		$rounds,$rounds,1
+	subi		$rounds,$rounds,1
+	li		$idx,16
+
+	vslb		$eighty7,$seven,$seven		# 0x808080..80
+	vor		$eighty7,$eighty7,$seven	# 0x878787..87
+	vspltisb	$tmp,1				# 0x010101..01
+	vsldoi		$eighty7,$eighty7,$tmp,15	# 0x870101..01
+
+	${UCMP}i	$len,96
+	bge		_aesp8_xts_encrypt6x
+
+	andi.		$taillen,$len,15
+	subic		r0,$len,32
+	subi		$taillen,$taillen,16
+	subfe		r0,r0,r0
+	and		r0,r0,$taillen
+	add		$inp,$inp,r0
+
+	lvx		$rndkey0,0,$key1
+	lvx		$rndkey1,$idx,$key1
+	addi		$idx,$idx,16
+	vperm		$inout,$inout,$inptail,$inpperm
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vxor		$inout,$inout,$tweak
+	vxor		$inout,$inout,$rndkey0
+	lvx		$rndkey0,$idx,$key1
+	addi		$idx,$idx,16
+	mtctr		$rounds
+	b		Loop_xts_enc
+
+.align	5
+Loop_xts_enc:
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vcipher		$inout,$inout,$rndkey1
+	lvx		$rndkey1,$idx,$key1
+	addi		$idx,$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vcipher		$inout,$inout,$rndkey0
+	lvx		$rndkey0,$idx,$key1
+	addi		$idx,$idx,16
+	bdnz		Loop_xts_enc
+
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vcipher		$inout,$inout,$rndkey1
+	lvx		$rndkey1,$idx,$key1
+	li		$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vxor		$rndkey0,$rndkey0,$tweak
+	vcipherlast	$output,$inout,$rndkey0
+
+	le?vperm	$tmp,$output,$output,$leperm
+	be?nop
+	le?stvx_u	$tmp,0,$out
+	be?stvx_u	$output,0,$out
+	addi		$out,$out,16
+
+	subic.		$len,$len,16
+	beq		Lxts_enc_done
+
+	vmr		$inout,$inptail
+	lvx		$inptail,0,$inp
+	addi		$inp,$inp,16
+	lvx		$rndkey0,0,$key1
+	lvx		$rndkey1,$idx,$key1
+	addi		$idx,$idx,16
+
+	subic		r0,$len,32
+	subfe		r0,r0,r0
+	and		r0,r0,$taillen
+	add		$inp,$inp,r0
+
+	vsrab		$tmp,$tweak,$seven		# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	vand		$tmp,$tmp,$eighty7
+	vxor		$tweak,$tweak,$tmp
+
+	vperm		$inout,$inout,$inptail,$inpperm
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vxor		$inout,$inout,$tweak
+	vxor		$output,$output,$rndkey0	# just in case $len<16
+	vxor		$inout,$inout,$rndkey0
+	lvx		$rndkey0,$idx,$key1
+	addi		$idx,$idx,16
+
+	mtctr		$rounds
+	${UCMP}i	$len,16
+	bge		Loop_xts_enc
+
+	vxor		$output,$output,$tweak
+	lvsr		$inpperm,0,$len			# $inpperm is no longer needed
+	vxor		$inptail,$inptail,$inptail	# $inptail is no longer needed
+	vspltisb	$tmp,-1
+	vperm		$inptail,$inptail,$tmp,$inpperm
+	vsel		$inout,$inout,$output,$inptail
+
+	subi		r11,$out,17
+	subi		$out,$out,16
+	mtctr		$len
+	li		$len,16
+Loop_xts_enc_steal:
+	lbzu		r0,1(r11)
+	stb		r0,16(r11)
+	bdnz		Loop_xts_enc_steal
+
+	mtctr		$rounds
+	b		Loop_xts_enc			# one more time...
+
+Lxts_enc_done:
+	${UCMP}i	$ivp,0
+	beq		Lxts_enc_ret
+
+	vsrab		$tmp,$tweak,$seven		# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	vand		$tmp,$tmp,$eighty7
+	vxor		$tweak,$tweak,$tmp
+
+	le?vperm	$tweak,$tweak,$tweak,$leperm
+	stvx_u		$tweak,0,$ivp
+
+Lxts_enc_ret:
+	mtspr		256,r12				# restore vrsave
+	li		r3,0
+	blr
+	.long		0
+	.byte		0,12,0x04,0,0x80,6,6,0
+	.long		0
+.size	.${prefix}_xts_encrypt,.-.${prefix}_xts_encrypt
+
+.globl	.${prefix}_xts_decrypt
+.align	5
+.${prefix}_xts_decrypt:
+	mr		$inp,r3				# reassign
+	li		r3,-1
+	${UCMP}i	$len,16
+	bltlr-
+
+	lis		r0,0xfff8
+	mfspr		r12,256				# save vrsave
+	li		r11,0
+	mtspr		256,r0
+
+	andi.		r0,$len,15
+	neg		r0,r0
+	andi.		r0,r0,16
+	sub		$len,$len,r0
+
+	vspltisb	$seven,0x07			# 0x070707..07
+	le?lvsl		$leperm,r11,r11
+	le?vspltisb	$tmp,0x0f
+	le?vxor		$leperm,$leperm,$seven
+
+	li		$idx,15
+	lvx		$tweak,0,$ivp			# load [unaligned] iv
+	lvsl		$inpperm,0,$ivp
+	lvx		$inptail,$idx,$ivp
+	le?vxor		$inpperm,$inpperm,$tmp
+	vperm		$tweak,$tweak,$inptail,$inpperm
+
+	neg		r11,$inp
+	lvsr		$inpperm,0,r11			# prepare for unaligned load
+	lvx		$inout,0,$inp
+	addi		$inp,$inp,15			# 15 is not typo
+	le?vxor		$inpperm,$inpperm,$tmp
+
+	${UCMP}i	$key2,0				# key2==NULL?
+	beq		Lxts_dec_no_key2
+
+	?lvsl		$keyperm,0,$key2		# prepare for unaligned key
+	lwz		$rounds,240($key2)
+	srwi		$rounds,$rounds,1
+	subi		$rounds,$rounds,1
+	li		$idx,16
+
+	lvx		$rndkey0,0,$key2
+	lvx		$rndkey1,$idx,$key2
+	addi		$idx,$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vxor		$tweak,$tweak,$rndkey0
+	lvx		$rndkey0,$idx,$key2
+	addi		$idx,$idx,16
+	mtctr		$rounds
+
+Ltweak_xts_dec:
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vcipher		$tweak,$tweak,$rndkey1
+	lvx		$rndkey1,$idx,$key2
+	addi		$idx,$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vcipher		$tweak,$tweak,$rndkey0
+	lvx		$rndkey0,$idx,$key2
+	addi		$idx,$idx,16
+	bdnz		Ltweak_xts_dec
+
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vcipher		$tweak,$tweak,$rndkey1
+	lvx		$rndkey1,$idx,$key2
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vcipherlast	$tweak,$tweak,$rndkey0
+
+	li		$ivp,0				# don't chain the tweak
+	b		Lxts_dec
+
+Lxts_dec_no_key2:
+	neg		$idx,$len
+	andi.		$idx,$idx,15
+	add		$len,$len,$idx			# in "tweak chaining"
+							# mode only complete
+							# blocks are processed
+Lxts_dec:
+	lvx		$inptail,0,$inp
+	addi		$inp,$inp,16
+
+	?lvsl		$keyperm,0,$key1		# prepare for unaligned key
+	lwz		$rounds,240($key1)
+	srwi		$rounds,$rounds,1
+	subi		$rounds,$rounds,1
+	li		$idx,16
+
+	vslb		$eighty7,$seven,$seven		# 0x808080..80
+	vor		$eighty7,$eighty7,$seven	# 0x878787..87
+	vspltisb	$tmp,1				# 0x010101..01
+	vsldoi		$eighty7,$eighty7,$tmp,15	# 0x870101..01
+
+	${UCMP}i	$len,96
+	bge		_aesp8_xts_decrypt6x
+
+	lvx		$rndkey0,0,$key1
+	lvx		$rndkey1,$idx,$key1
+	addi		$idx,$idx,16
+	vperm		$inout,$inout,$inptail,$inpperm
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vxor		$inout,$inout,$tweak
+	vxor		$inout,$inout,$rndkey0
+	lvx		$rndkey0,$idx,$key1
+	addi		$idx,$idx,16
+	mtctr		$rounds
+
+	${UCMP}i	$len,16
+	blt		Ltail_xts_dec
+	be?b		Loop_xts_dec
+
+.align	5
+Loop_xts_dec:
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vncipher	$inout,$inout,$rndkey1
+	lvx		$rndkey1,$idx,$key1
+	addi		$idx,$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vncipher	$inout,$inout,$rndkey0
+	lvx		$rndkey0,$idx,$key1
+	addi		$idx,$idx,16
+	bdnz		Loop_xts_dec
+
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vncipher	$inout,$inout,$rndkey1
+	lvx		$rndkey1,$idx,$key1
+	li		$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vxor		$rndkey0,$rndkey0,$tweak
+	vncipherlast	$output,$inout,$rndkey0
+
+	le?vperm	$tmp,$output,$output,$leperm
+	be?nop
+	le?stvx_u	$tmp,0,$out
+	be?stvx_u	$output,0,$out
+	addi		$out,$out,16
+
+	subic.		$len,$len,16
+	beq		Lxts_dec_done
+
+	vmr		$inout,$inptail
+	lvx		$inptail,0,$inp
+	addi		$inp,$inp,16
+	lvx		$rndkey0,0,$key1
+	lvx		$rndkey1,$idx,$key1
+	addi		$idx,$idx,16
+
+	vsrab		$tmp,$tweak,$seven		# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	vand		$tmp,$tmp,$eighty7
+	vxor		$tweak,$tweak,$tmp
+
+	vperm		$inout,$inout,$inptail,$inpperm
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vxor		$inout,$inout,$tweak
+	vxor		$inout,$inout,$rndkey0
+	lvx		$rndkey0,$idx,$key1
+	addi		$idx,$idx,16
+
+	mtctr		$rounds
+	${UCMP}i	$len,16
+	bge		Loop_xts_dec
+
+Ltail_xts_dec:
+	vsrab		$tmp,$tweak,$seven		# next tweak value
+	vaddubm		$tweak1,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	vand		$tmp,$tmp,$eighty7
+	vxor		$tweak1,$tweak1,$tmp
+
+	subi		$inp,$inp,16
+	add		$inp,$inp,$len
+
+	vxor		$inout,$inout,$tweak		# :-(
+	vxor		$inout,$inout,$tweak1		# :-)
+
+Loop_xts_dec_short:
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vncipher	$inout,$inout,$rndkey1
+	lvx		$rndkey1,$idx,$key1
+	addi		$idx,$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vncipher	$inout,$inout,$rndkey0
+	lvx		$rndkey0,$idx,$key1
+	addi		$idx,$idx,16
+	bdnz		Loop_xts_dec_short
+
+	?vperm		$rndkey1,$rndkey1,$rndkey0,$keyperm
+	vncipher	$inout,$inout,$rndkey1
+	lvx		$rndkey1,$idx,$key1
+	li		$idx,16
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+	vxor		$rndkey0,$rndkey0,$tweak1
+	vncipherlast	$output,$inout,$rndkey0
+
+	le?vperm	$tmp,$output,$output,$leperm
+	be?nop
+	le?stvx_u	$tmp,0,$out
+	be?stvx_u	$output,0,$out
+
+	vmr		$inout,$inptail
+	lvx		$inptail,0,$inp
+	#addi		$inp,$inp,16
+	lvx		$rndkey0,0,$key1
+	lvx		$rndkey1,$idx,$key1
+	addi		$idx,$idx,16
+	vperm		$inout,$inout,$inptail,$inpperm
+	?vperm		$rndkey0,$rndkey0,$rndkey1,$keyperm
+
+	lvsr		$inpperm,0,$len			# $inpperm is no longer needed
+	vxor		$inptail,$inptail,$inptail	# $inptail is no longer needed
+	vspltisb	$tmp,-1
+	vperm		$inptail,$inptail,$tmp,$inpperm
+	vsel		$inout,$inout,$output,$inptail
+
+	vxor		$rndkey0,$rndkey0,$tweak
+	vxor		$inout,$inout,$rndkey0
+	lvx		$rndkey0,$idx,$key1
+	addi		$idx,$idx,16
+
+	subi		r11,$out,1
+	mtctr		$len
+	li		$len,16
+Loop_xts_dec_steal:
+	lbzu		r0,1(r11)
+	stb		r0,16(r11)
+	bdnz		Loop_xts_dec_steal
+
+	mtctr		$rounds
+	b		Loop_xts_dec			# one more time...
+
+Lxts_dec_done:
+	${UCMP}i	$ivp,0
+	beq		Lxts_dec_ret
+
+	vsrab		$tmp,$tweak,$seven		# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	vand		$tmp,$tmp,$eighty7
+	vxor		$tweak,$tweak,$tmp
+
+	le?vperm	$tweak,$tweak,$tweak,$leperm
+	stvx_u		$tweak,0,$ivp
+
+Lxts_dec_ret:
+	mtspr		256,r12				# restore vrsave
+	li		r3,0
+	blr
+	.long		0
+	.byte		0,12,0x04,0,0x80,6,6,0
+	.long		0
+.size	.${prefix}_xts_decrypt,.-.${prefix}_xts_decrypt
+___
+#########################################################################
+{{	# Optimized XTS procedures					#
+my $key_=$key2;
+my ($x00,$x10,$x20,$x30,$x40,$x50,$x60,$x70)=map("r$_",(0,3,26..31));
+    $x00=0 if ($flavour =~ /osx/);
+my ($in0,  $in1,  $in2,  $in3,  $in4,  $in5 )=map("v$_",(0..5));
+my ($out0, $out1, $out2, $out3, $out4, $out5)=map("v$_",(7,12..16));
+my ($twk0, $twk1, $twk2, $twk3, $twk4, $twk5)=map("v$_",(17..22));
+my $rndkey0="v23";	# v24-v25 rotating buffer for first found keys
+			# v26-v31 last 6 round keys
+my ($keyperm)=($out0);	# aliases with "caller", redundant assignment
+my $taillen=$x70;
+
+$code.=<<___;
+.align	5
+_aesp8_xts_encrypt6x:
+	$STU		$sp,-`($FRAME+21*16+6*$SIZE_T)`($sp)
+	mflr		r11
+	li		r7,`$FRAME+8*16+15`
+	li		r3,`$FRAME+8*16+31`
+	$PUSH		r11,`$FRAME+21*16+6*$SIZE_T+$LRSAVE`($sp)
+	stvx		v20,r7,$sp		# ABI says so
+	addi		r7,r7,32
+	stvx		v21,r3,$sp
+	addi		r3,r3,32
+	stvx		v22,r7,$sp
+	addi		r7,r7,32
+	stvx		v23,r3,$sp
+	addi		r3,r3,32
+	stvx		v24,r7,$sp
+	addi		r7,r7,32
+	stvx		v25,r3,$sp
+	addi		r3,r3,32
+	stvx		v26,r7,$sp
+	addi		r7,r7,32
+	stvx		v27,r3,$sp
+	addi		r3,r3,32
+	stvx		v28,r7,$sp
+	addi		r7,r7,32
+	stvx		v29,r3,$sp
+	addi		r3,r3,32
+	stvx		v30,r7,$sp
+	stvx		v31,r3,$sp
+	li		r0,-1
+	stw		$vrsave,`$FRAME+21*16-4`($sp)	# save vrsave
+	li		$x10,0x10
+	$PUSH		r26,`$FRAME+21*16+0*$SIZE_T`($sp)
+	li		$x20,0x20
+	$PUSH		r27,`$FRAME+21*16+1*$SIZE_T`($sp)
+	li		$x30,0x30
+	$PUSH		r28,`$FRAME+21*16+2*$SIZE_T`($sp)
+	li		$x40,0x40
+	$PUSH		r29,`$FRAME+21*16+3*$SIZE_T`($sp)
+	li		$x50,0x50
+	$PUSH		r30,`$FRAME+21*16+4*$SIZE_T`($sp)
+	li		$x60,0x60
+	$PUSH		r31,`$FRAME+21*16+5*$SIZE_T`($sp)
+	li		$x70,0x70
+	mtspr		256,r0
+
+	subi		$rounds,$rounds,3	# -4 in total
+
+	lvx		$rndkey0,$x00,$key1	# load key schedule
+	lvx		v30,$x10,$key1
+	addi		$key1,$key1,0x20
+	lvx		v31,$x00,$key1
+	?vperm		$rndkey0,$rndkey0,v30,$keyperm
+	addi		$key_,$sp,$FRAME+15
+	mtctr		$rounds
+
+Load_xts_enc_key:
+	?vperm		v24,v30,v31,$keyperm
+	lvx		v30,$x10,$key1
+	addi		$key1,$key1,0x20
+	stvx		v24,$x00,$key_		# off-load round[1]
+	?vperm		v25,v31,v30,$keyperm
+	lvx		v31,$x00,$key1
+	stvx		v25,$x10,$key_		# off-load round[2]
+	addi		$key_,$key_,0x20
+	bdnz		Load_xts_enc_key
+
+	lvx		v26,$x10,$key1
+	?vperm		v24,v30,v31,$keyperm
+	lvx		v27,$x20,$key1
+	stvx		v24,$x00,$key_		# off-load round[3]
+	?vperm		v25,v31,v26,$keyperm
+	lvx		v28,$x30,$key1
+	stvx		v25,$x10,$key_		# off-load round[4]
+	addi		$key_,$sp,$FRAME+15	# rewind $key_
+	?vperm		v26,v26,v27,$keyperm
+	lvx		v29,$x40,$key1
+	?vperm		v27,v27,v28,$keyperm
+	lvx		v30,$x50,$key1
+	?vperm		v28,v28,v29,$keyperm
+	lvx		v31,$x60,$key1
+	?vperm		v29,v29,v30,$keyperm
+	lvx		$twk5,$x70,$key1	# borrow $twk5
+	?vperm		v30,v30,v31,$keyperm
+	lvx		v24,$x00,$key_		# pre-load round[1]
+	?vperm		v31,v31,$twk5,$keyperm
+	lvx		v25,$x10,$key_		# pre-load round[2]
+
+	 vperm		$in0,$inout,$inptail,$inpperm
+	 subi		$inp,$inp,31		# undo "caller"
+	vxor		$twk0,$tweak,$rndkey0
+	vsrab		$tmp,$tweak,$seven	# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	vand		$tmp,$tmp,$eighty7
+	 vxor		$out0,$in0,$twk0
+	vxor		$tweak,$tweak,$tmp
+
+	 lvx_u		$in1,$x10,$inp
+	vxor		$twk1,$tweak,$rndkey0
+	vsrab		$tmp,$tweak,$seven	# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	 le?vperm	$in1,$in1,$in1,$leperm
+	vand		$tmp,$tmp,$eighty7
+	 vxor		$out1,$in1,$twk1
+	vxor		$tweak,$tweak,$tmp
+
+	 lvx_u		$in2,$x20,$inp
+	 andi.		$taillen,$len,15
+	vxor		$twk2,$tweak,$rndkey0
+	vsrab		$tmp,$tweak,$seven	# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	 le?vperm	$in2,$in2,$in2,$leperm
+	vand		$tmp,$tmp,$eighty7
+	 vxor		$out2,$in2,$twk2
+	vxor		$tweak,$tweak,$tmp
+
+	 lvx_u		$in3,$x30,$inp
+	 sub		$len,$len,$taillen
+	vxor		$twk3,$tweak,$rndkey0
+	vsrab		$tmp,$tweak,$seven	# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	 le?vperm	$in3,$in3,$in3,$leperm
+	vand		$tmp,$tmp,$eighty7
+	 vxor		$out3,$in3,$twk3
+	vxor		$tweak,$tweak,$tmp
+
+	 lvx_u		$in4,$x40,$inp
+	 subi		$len,$len,0x60
+	vxor		$twk4,$tweak,$rndkey0
+	vsrab		$tmp,$tweak,$seven	# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	 le?vperm	$in4,$in4,$in4,$leperm
+	vand		$tmp,$tmp,$eighty7
+	 vxor		$out4,$in4,$twk4
+	vxor		$tweak,$tweak,$tmp
+
+	 lvx_u		$in5,$x50,$inp
+	 addi		$inp,$inp,0x60
+	vxor		$twk5,$tweak,$rndkey0
+	vsrab		$tmp,$tweak,$seven	# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	 le?vperm	$in5,$in5,$in5,$leperm
+	vand		$tmp,$tmp,$eighty7
+	 vxor		$out5,$in5,$twk5
+	vxor		$tweak,$tweak,$tmp
+
+	vxor		v31,v31,$rndkey0
+	mtctr		$rounds
+	b		Loop_xts_enc6x
+
+.align	5
+Loop_xts_enc6x:
+	vcipher		$out0,$out0,v24
+	vcipher		$out1,$out1,v24
+	vcipher		$out2,$out2,v24
+	vcipher		$out3,$out3,v24
+	vcipher		$out4,$out4,v24
+	vcipher		$out5,$out5,v24
+	lvx		v24,$x20,$key_		# round[3]
+	addi		$key_,$key_,0x20
+
+	vcipher		$out0,$out0,v25
+	vcipher		$out1,$out1,v25
+	vcipher		$out2,$out2,v25
+	vcipher		$out3,$out3,v25
+	vcipher		$out4,$out4,v25
+	vcipher		$out5,$out5,v25
+	lvx		v25,$x10,$key_		# round[4]
+	bdnz		Loop_xts_enc6x
+
+	subic		$len,$len,96		# $len-=96
+	 vxor		$in0,$twk0,v31		# xor with last round key
+	vcipher		$out0,$out0,v24
+	vcipher		$out1,$out1,v24
+	 vsrab		$tmp,$tweak,$seven	# next tweak value
+	 vxor		$twk0,$tweak,$rndkey0
+	 vaddubm	$tweak,$tweak,$tweak
+	vcipher		$out2,$out2,v24
+	vcipher		$out3,$out3,v24
+	 vsldoi		$tmp,$tmp,$tmp,15
+	vcipher		$out4,$out4,v24
+	vcipher		$out5,$out5,v24
+
+	subfe.		r0,r0,r0		# borrow?-1:0
+	 vand		$tmp,$tmp,$eighty7
+	vcipher		$out0,$out0,v25
+	vcipher		$out1,$out1,v25
+	 vxor		$tweak,$tweak,$tmp
+	vcipher		$out2,$out2,v25
+	vcipher		$out3,$out3,v25
+	 vxor		$in1,$twk1,v31
+	 vsrab		$tmp,$tweak,$seven	# next tweak value
+	 vxor		$twk1,$tweak,$rndkey0
+	vcipher		$out4,$out4,v25
+	vcipher		$out5,$out5,v25
+
+	and		r0,r0,$len
+	 vaddubm	$tweak,$tweak,$tweak
+	 vsldoi		$tmp,$tmp,$tmp,15
+	vcipher		$out0,$out0,v26
+	vcipher		$out1,$out1,v26
+	 vand		$tmp,$tmp,$eighty7
+	vcipher		$out2,$out2,v26
+	vcipher		$out3,$out3,v26
+	 vxor		$tweak,$tweak,$tmp
+	vcipher		$out4,$out4,v26
+	vcipher		$out5,$out5,v26
+
+	add		$inp,$inp,r0		# $inp is adjusted in such
+						# way that at exit from the
+						# loop inX-in5 are loaded
+						# with last "words"
+	 vxor		$in2,$twk2,v31
+	 vsrab		$tmp,$tweak,$seven	# next tweak value
+	 vxor		$twk2,$tweak,$rndkey0
+	 vaddubm	$tweak,$tweak,$tweak
+	vcipher		$out0,$out0,v27
+	vcipher		$out1,$out1,v27
+	 vsldoi		$tmp,$tmp,$tmp,15
+	vcipher		$out2,$out2,v27
+	vcipher		$out3,$out3,v27
+	 vand		$tmp,$tmp,$eighty7
+	vcipher		$out4,$out4,v27
+	vcipher		$out5,$out5,v27
+
+	addi		$key_,$sp,$FRAME+15	# rewind $key_
+	 vxor		$tweak,$tweak,$tmp
+	vcipher		$out0,$out0,v28
+	vcipher		$out1,$out1,v28
+	 vxor		$in3,$twk3,v31
+	 vsrab		$tmp,$tweak,$seven	# next tweak value
+	 vxor		$twk3,$tweak,$rndkey0
+	vcipher		$out2,$out2,v28
+	vcipher		$out3,$out3,v28
+	 vaddubm	$tweak,$tweak,$tweak
+	 vsldoi		$tmp,$tmp,$tmp,15
+	vcipher		$out4,$out4,v28
+	vcipher		$out5,$out5,v28
+	lvx		v24,$x00,$key_		# re-pre-load round[1]
+	 vand		$tmp,$tmp,$eighty7
+
+	vcipher		$out0,$out0,v29
+	vcipher		$out1,$out1,v29
+	 vxor		$tweak,$tweak,$tmp
+	vcipher		$out2,$out2,v29
+	vcipher		$out3,$out3,v29
+	 vxor		$in4,$twk4,v31
+	 vsrab		$tmp,$tweak,$seven	# next tweak value
+	 vxor		$twk4,$tweak,$rndkey0
+	vcipher		$out4,$out4,v29
+	vcipher		$out5,$out5,v29
+	lvx		v25,$x10,$key_		# re-pre-load round[2]
+	 vaddubm	$tweak,$tweak,$tweak
+	 vsldoi		$tmp,$tmp,$tmp,15
+
+	vcipher		$out0,$out0,v30
+	vcipher		$out1,$out1,v30
+	 vand		$tmp,$tmp,$eighty7
+	vcipher		$out2,$out2,v30
+	vcipher		$out3,$out3,v30
+	 vxor		$tweak,$tweak,$tmp
+	vcipher		$out4,$out4,v30
+	vcipher		$out5,$out5,v30
+	 vxor		$in5,$twk5,v31
+	 vsrab		$tmp,$tweak,$seven	# next tweak value
+	 vxor		$twk5,$tweak,$rndkey0
+
+	vcipherlast	$out0,$out0,$in0
+	 lvx_u		$in0,$x00,$inp		# load next input block
+	 vaddubm	$tweak,$tweak,$tweak
+	 vsldoi		$tmp,$tmp,$tmp,15
+	vcipherlast	$out1,$out1,$in1
+	 lvx_u		$in1,$x10,$inp
+	vcipherlast	$out2,$out2,$in2
+	 le?vperm	$in0,$in0,$in0,$leperm
+	 lvx_u		$in2,$x20,$inp
+	 vand		$tmp,$tmp,$eighty7
+	vcipherlast	$out3,$out3,$in3
+	 le?vperm	$in1,$in1,$in1,$leperm
+	 lvx_u		$in3,$x30,$inp
+	vcipherlast	$out4,$out4,$in4
+	 le?vperm	$in2,$in2,$in2,$leperm
+	 lvx_u		$in4,$x40,$inp
+	 vxor		$tweak,$tweak,$tmp
+	vcipherlast	$tmp,$out5,$in5		# last block might be needed
+						# in stealing mode
+	 le?vperm	$in3,$in3,$in3,$leperm
+	 lvx_u		$in5,$x50,$inp
+	 addi		$inp,$inp,0x60
+	 le?vperm	$in4,$in4,$in4,$leperm
+	 le?vperm	$in5,$in5,$in5,$leperm
+
+	le?vperm	$out0,$out0,$out0,$leperm
+	le?vperm	$out1,$out1,$out1,$leperm
+	stvx_u		$out0,$x00,$out		# store output
+	 vxor		$out0,$in0,$twk0
+	le?vperm	$out2,$out2,$out2,$leperm
+	stvx_u		$out1,$x10,$out
+	 vxor		$out1,$in1,$twk1
+	le?vperm	$out3,$out3,$out3,$leperm
+	stvx_u		$out2,$x20,$out
+	 vxor		$out2,$in2,$twk2
+	le?vperm	$out4,$out4,$out4,$leperm
+	stvx_u		$out3,$x30,$out
+	 vxor		$out3,$in3,$twk3
+	le?vperm	$out5,$tmp,$tmp,$leperm
+	stvx_u		$out4,$x40,$out
+	 vxor		$out4,$in4,$twk4
+	le?stvx_u	$out5,$x50,$out
+	be?stvx_u	$tmp, $x50,$out
+	 vxor		$out5,$in5,$twk5
+	addi		$out,$out,0x60
+
+	mtctr		$rounds
+	beq		Loop_xts_enc6x		# did $len-=96 borrow?
+
+	addic.		$len,$len,0x60
+	beq		Lxts_enc6x_zero
+	cmpwi		$len,0x20
+	blt		Lxts_enc6x_one
+	nop
+	beq		Lxts_enc6x_two
+	cmpwi		$len,0x40
+	blt		Lxts_enc6x_three
+	nop
+	beq		Lxts_enc6x_four
+
+Lxts_enc6x_five:
+	vxor		$out0,$in1,$twk0
+	vxor		$out1,$in2,$twk1
+	vxor		$out2,$in3,$twk2
+	vxor		$out3,$in4,$twk3
+	vxor		$out4,$in5,$twk4
+
+	bl		_aesp8_xts_enc5x
+
+	le?vperm	$out0,$out0,$out0,$leperm
+	vmr		$twk0,$twk5		# unused tweak
+	le?vperm	$out1,$out1,$out1,$leperm
+	stvx_u		$out0,$x00,$out		# store output
+	le?vperm	$out2,$out2,$out2,$leperm
+	stvx_u		$out1,$x10,$out
+	le?vperm	$out3,$out3,$out3,$leperm
+	stvx_u		$out2,$x20,$out
+	vxor		$tmp,$out4,$twk5	# last block prep for stealing
+	le?vperm	$out4,$out4,$out4,$leperm
+	stvx_u		$out3,$x30,$out
+	stvx_u		$out4,$x40,$out
+	addi		$out,$out,0x50
+	bne		Lxts_enc6x_steal
+	b		Lxts_enc6x_done
+
+.align	4
+Lxts_enc6x_four:
+	vxor		$out0,$in2,$twk0
+	vxor		$out1,$in3,$twk1
+	vxor		$out2,$in4,$twk2
+	vxor		$out3,$in5,$twk3
+	vxor		$out4,$out4,$out4
+
+	bl		_aesp8_xts_enc5x
+
+	le?vperm	$out0,$out0,$out0,$leperm
+	vmr		$twk0,$twk4		# unused tweak
+	le?vperm	$out1,$out1,$out1,$leperm
+	stvx_u		$out0,$x00,$out		# store output
+	le?vperm	$out2,$out2,$out2,$leperm
+	stvx_u		$out1,$x10,$out
+	vxor		$tmp,$out3,$twk4	# last block prep for stealing
+	le?vperm	$out3,$out3,$out3,$leperm
+	stvx_u		$out2,$x20,$out
+	stvx_u		$out3,$x30,$out
+	addi		$out,$out,0x40
+	bne		Lxts_enc6x_steal
+	b		Lxts_enc6x_done
+
+.align	4
+Lxts_enc6x_three:
+	vxor		$out0,$in3,$twk0
+	vxor		$out1,$in4,$twk1
+	vxor		$out2,$in5,$twk2
+	vxor		$out3,$out3,$out3
+	vxor		$out4,$out4,$out4
+
+	bl		_aesp8_xts_enc5x
+
+	le?vperm	$out0,$out0,$out0,$leperm
+	vmr		$twk0,$twk3		# unused tweak
+	le?vperm	$out1,$out1,$out1,$leperm
+	stvx_u		$out0,$x00,$out		# store output
+	vxor		$tmp,$out2,$twk3	# last block prep for stealing
+	le?vperm	$out2,$out2,$out2,$leperm
+	stvx_u		$out1,$x10,$out
+	stvx_u		$out2,$x20,$out
+	addi		$out,$out,0x30
+	bne		Lxts_enc6x_steal
+	b		Lxts_enc6x_done
+
+.align	4
+Lxts_enc6x_two:
+	vxor		$out0,$in4,$twk0
+	vxor		$out1,$in5,$twk1
+	vxor		$out2,$out2,$out2
+	vxor		$out3,$out3,$out3
+	vxor		$out4,$out4,$out4
+
+	bl		_aesp8_xts_enc5x
+
+	le?vperm	$out0,$out0,$out0,$leperm
+	vmr		$twk0,$twk2		# unused tweak
+	vxor		$tmp,$out1,$twk2	# last block prep for stealing
+	le?vperm	$out1,$out1,$out1,$leperm
+	stvx_u		$out0,$x00,$out		# store output
+	stvx_u		$out1,$x10,$out
+	addi		$out,$out,0x20
+	bne		Lxts_enc6x_steal
+	b		Lxts_enc6x_done
+
+.align	4
+Lxts_enc6x_one:
+	vxor		$out0,$in5,$twk0
+	nop
+Loop_xts_enc1x:
+	vcipher		$out0,$out0,v24
+	lvx		v24,$x20,$key_		# round[3]
+	addi		$key_,$key_,0x20
+
+	vcipher		$out0,$out0,v25
+	lvx		v25,$x10,$key_		# round[4]
+	bdnz		Loop_xts_enc1x
+
+	add		$inp,$inp,$taillen
+	cmpwi		$taillen,0
+	vcipher		$out0,$out0,v24
+
+	subi		$inp,$inp,16
+	vcipher		$out0,$out0,v25
+
+	lvsr		$inpperm,0,$taillen
+	vcipher		$out0,$out0,v26
+
+	lvx_u		$in0,0,$inp
+	vcipher		$out0,$out0,v27
+
+	addi		$key_,$sp,$FRAME+15	# rewind $key_
+	vcipher		$out0,$out0,v28
+	lvx		v24,$x00,$key_		# re-pre-load round[1]
+
+	vcipher		$out0,$out0,v29
+	lvx		v25,$x10,$key_		# re-pre-load round[2]
+	 vxor		$twk0,$twk0,v31
+
+	le?vperm	$in0,$in0,$in0,$leperm
+	vcipher		$out0,$out0,v30
+
+	vperm		$in0,$in0,$in0,$inpperm
+	vcipherlast	$out0,$out0,$twk0
+
+	vmr		$twk0,$twk1		# unused tweak
+	vxor		$tmp,$out0,$twk1	# last block prep for stealing
+	le?vperm	$out0,$out0,$out0,$leperm
+	stvx_u		$out0,$x00,$out		# store output
+	addi		$out,$out,0x10
+	bne		Lxts_enc6x_steal
+	b		Lxts_enc6x_done
+
+.align	4
+Lxts_enc6x_zero:
+	cmpwi		$taillen,0
+	beq		Lxts_enc6x_done
+
+	add		$inp,$inp,$taillen
+	subi		$inp,$inp,16
+	lvx_u		$in0,0,$inp
+	lvsr		$inpperm,0,$taillen	# $in5 is no more
+	le?vperm	$in0,$in0,$in0,$leperm
+	vperm		$in0,$in0,$in0,$inpperm
+	vxor		$tmp,$tmp,$twk0
+Lxts_enc6x_steal:
+	vxor		$in0,$in0,$twk0
+	vxor		$out0,$out0,$out0
+	vspltisb	$out1,-1
+	vperm		$out0,$out0,$out1,$inpperm
+	vsel		$out0,$in0,$tmp,$out0	# $tmp is last block, remember?
+
+	subi		r30,$out,17
+	subi		$out,$out,16
+	mtctr		$taillen
+Loop_xts_enc6x_steal:
+	lbzu		r0,1(r30)
+	stb		r0,16(r30)
+	bdnz		Loop_xts_enc6x_steal
+
+	li		$taillen,0
+	mtctr		$rounds
+	b		Loop_xts_enc1x		# one more time...
+
+.align	4
+Lxts_enc6x_done:
+	${UCMP}i	$ivp,0
+	beq		Lxts_enc6x_ret
+
+	vxor		$tweak,$twk0,$rndkey0
+	le?vperm	$tweak,$tweak,$tweak,$leperm
+	stvx_u		$tweak,0,$ivp
+
+Lxts_enc6x_ret:
+	mtlr		r11
+	li		r10,`$FRAME+15`
+	li		r11,`$FRAME+31`
+	stvx		$seven,r10,$sp		# wipe copies of round keys
+	addi		r10,r10,32
+	stvx		$seven,r11,$sp
+	addi		r11,r11,32
+	stvx		$seven,r10,$sp
+	addi		r10,r10,32
+	stvx		$seven,r11,$sp
+	addi		r11,r11,32
+	stvx		$seven,r10,$sp
+	addi		r10,r10,32
+	stvx		$seven,r11,$sp
+	addi		r11,r11,32
+	stvx		$seven,r10,$sp
+	addi		r10,r10,32
+	stvx		$seven,r11,$sp
+	addi		r11,r11,32
+
+	mtspr		256,$vrsave
+	lvx		v20,r10,$sp		# ABI says so
+	addi		r10,r10,32
+	lvx		v21,r11,$sp
+	addi		r11,r11,32
+	lvx		v22,r10,$sp
+	addi		r10,r10,32
+	lvx		v23,r11,$sp
+	addi		r11,r11,32
+	lvx		v24,r10,$sp
+	addi		r10,r10,32
+	lvx		v25,r11,$sp
+	addi		r11,r11,32
+	lvx		v26,r10,$sp
+	addi		r10,r10,32
+	lvx		v27,r11,$sp
+	addi		r11,r11,32
+	lvx		v28,r10,$sp
+	addi		r10,r10,32
+	lvx		v29,r11,$sp
+	addi		r11,r11,32
+	lvx		v30,r10,$sp
+	lvx		v31,r11,$sp
+	$POP		r26,`$FRAME+21*16+0*$SIZE_T`($sp)
+	$POP		r27,`$FRAME+21*16+1*$SIZE_T`($sp)
+	$POP		r28,`$FRAME+21*16+2*$SIZE_T`($sp)
+	$POP		r29,`$FRAME+21*16+3*$SIZE_T`($sp)
+	$POP		r30,`$FRAME+21*16+4*$SIZE_T`($sp)
+	$POP		r31,`$FRAME+21*16+5*$SIZE_T`($sp)
+	addi		$sp,$sp,`$FRAME+21*16+6*$SIZE_T`
+	blr
+	.long		0
+	.byte		0,12,0x04,1,0x80,6,6,0
+	.long		0
+
+.align	5
+_aesp8_xts_enc5x:
+	vcipher		$out0,$out0,v24
+	vcipher		$out1,$out1,v24
+	vcipher		$out2,$out2,v24
+	vcipher		$out3,$out3,v24
+	vcipher		$out4,$out4,v24
+	lvx		v24,$x20,$key_		# round[3]
+	addi		$key_,$key_,0x20
+
+	vcipher		$out0,$out0,v25
+	vcipher		$out1,$out1,v25
+	vcipher		$out2,$out2,v25
+	vcipher		$out3,$out3,v25
+	vcipher		$out4,$out4,v25
+	lvx		v25,$x10,$key_		# round[4]
+	bdnz		_aesp8_xts_enc5x
+
+	add		$inp,$inp,$taillen
+	cmpwi		$taillen,0
+	vcipher		$out0,$out0,v24
+	vcipher		$out1,$out1,v24
+	vcipher		$out2,$out2,v24
+	vcipher		$out3,$out3,v24
+	vcipher		$out4,$out4,v24
+
+	subi		$inp,$inp,16
+	vcipher		$out0,$out0,v25
+	vcipher		$out1,$out1,v25
+	vcipher		$out2,$out2,v25
+	vcipher		$out3,$out3,v25
+	vcipher		$out4,$out4,v25
+	 vxor		$twk0,$twk0,v31
+
+	vcipher		$out0,$out0,v26
+	lvsr		$inpperm,0,$taillen	# $in5 is no more
+	vcipher		$out1,$out1,v26
+	vcipher		$out2,$out2,v26
+	vcipher		$out3,$out3,v26
+	vcipher		$out4,$out4,v26
+	 vxor		$in1,$twk1,v31
+
+	vcipher		$out0,$out0,v27
+	lvx_u		$in0,0,$inp
+	vcipher		$out1,$out1,v27
+	vcipher		$out2,$out2,v27
+	vcipher		$out3,$out3,v27
+	vcipher		$out4,$out4,v27
+	 vxor		$in2,$twk2,v31
+
+	addi		$key_,$sp,$FRAME+15	# rewind $key_
+	vcipher		$out0,$out0,v28
+	vcipher		$out1,$out1,v28
+	vcipher		$out2,$out2,v28
+	vcipher		$out3,$out3,v28
+	vcipher		$out4,$out4,v28
+	lvx		v24,$x00,$key_		# re-pre-load round[1]
+	 vxor		$in3,$twk3,v31
+
+	vcipher		$out0,$out0,v29
+	le?vperm	$in0,$in0,$in0,$leperm
+	vcipher		$out1,$out1,v29
+	vcipher		$out2,$out2,v29
+	vcipher		$out3,$out3,v29
+	vcipher		$out4,$out4,v29
+	lvx		v25,$x10,$key_		# re-pre-load round[2]
+	 vxor		$in4,$twk4,v31
+
+	vcipher		$out0,$out0,v30
+	vperm		$in0,$in0,$in0,$inpperm
+	vcipher		$out1,$out1,v30
+	vcipher		$out2,$out2,v30
+	vcipher		$out3,$out3,v30
+	vcipher		$out4,$out4,v30
+
+	vcipherlast	$out0,$out0,$twk0
+	vcipherlast	$out1,$out1,$in1
+	vcipherlast	$out2,$out2,$in2
+	vcipherlast	$out3,$out3,$in3
+	vcipherlast	$out4,$out4,$in4
+	blr
+        .long   	0
+        .byte   	0,12,0x14,0,0,0,0,0
+
+.align	5
+_aesp8_xts_decrypt6x:
+	$STU		$sp,-`($FRAME+21*16+6*$SIZE_T)`($sp)
+	mflr		r11
+	li		r7,`$FRAME+8*16+15`
+	li		r3,`$FRAME+8*16+31`
+	$PUSH		r11,`$FRAME+21*16+6*$SIZE_T+$LRSAVE`($sp)
+	stvx		v20,r7,$sp		# ABI says so
+	addi		r7,r7,32
+	stvx		v21,r3,$sp
+	addi		r3,r3,32
+	stvx		v22,r7,$sp
+	addi		r7,r7,32
+	stvx		v23,r3,$sp
+	addi		r3,r3,32
+	stvx		v24,r7,$sp
+	addi		r7,r7,32
+	stvx		v25,r3,$sp
+	addi		r3,r3,32
+	stvx		v26,r7,$sp
+	addi		r7,r7,32
+	stvx		v27,r3,$sp
+	addi		r3,r3,32
+	stvx		v28,r7,$sp
+	addi		r7,r7,32
+	stvx		v29,r3,$sp
+	addi		r3,r3,32
+	stvx		v30,r7,$sp
+	stvx		v31,r3,$sp
+	li		r0,-1
+	stw		$vrsave,`$FRAME+21*16-4`($sp)	# save vrsave
+	li		$x10,0x10
+	$PUSH		r26,`$FRAME+21*16+0*$SIZE_T`($sp)
+	li		$x20,0x20
+	$PUSH		r27,`$FRAME+21*16+1*$SIZE_T`($sp)
+	li		$x30,0x30
+	$PUSH		r28,`$FRAME+21*16+2*$SIZE_T`($sp)
+	li		$x40,0x40
+	$PUSH		r29,`$FRAME+21*16+3*$SIZE_T`($sp)
+	li		$x50,0x50
+	$PUSH		r30,`$FRAME+21*16+4*$SIZE_T`($sp)
+	li		$x60,0x60
+	$PUSH		r31,`$FRAME+21*16+5*$SIZE_T`($sp)
+	li		$x70,0x70
+	mtspr		256,r0
+
+	subi		$rounds,$rounds,3	# -4 in total
+
+	lvx		$rndkey0,$x00,$key1	# load key schedule
+	lvx		v30,$x10,$key1
+	addi		$key1,$key1,0x20
+	lvx		v31,$x00,$key1
+	?vperm		$rndkey0,$rndkey0,v30,$keyperm
+	addi		$key_,$sp,$FRAME+15
+	mtctr		$rounds
+
+Load_xts_dec_key:
+	?vperm		v24,v30,v31,$keyperm
+	lvx		v30,$x10,$key1
+	addi		$key1,$key1,0x20
+	stvx		v24,$x00,$key_		# off-load round[1]
+	?vperm		v25,v31,v30,$keyperm
+	lvx		v31,$x00,$key1
+	stvx		v25,$x10,$key_		# off-load round[2]
+	addi		$key_,$key_,0x20
+	bdnz		Load_xts_dec_key
+
+	lvx		v26,$x10,$key1
+	?vperm		v24,v30,v31,$keyperm
+	lvx		v27,$x20,$key1
+	stvx		v24,$x00,$key_		# off-load round[3]
+	?vperm		v25,v31,v26,$keyperm
+	lvx		v28,$x30,$key1
+	stvx		v25,$x10,$key_		# off-load round[4]
+	addi		$key_,$sp,$FRAME+15	# rewind $key_
+	?vperm		v26,v26,v27,$keyperm
+	lvx		v29,$x40,$key1
+	?vperm		v27,v27,v28,$keyperm
+	lvx		v30,$x50,$key1
+	?vperm		v28,v28,v29,$keyperm
+	lvx		v31,$x60,$key1
+	?vperm		v29,v29,v30,$keyperm
+	lvx		$twk5,$x70,$key1	# borrow $twk5
+	?vperm		v30,v30,v31,$keyperm
+	lvx		v24,$x00,$key_		# pre-load round[1]
+	?vperm		v31,v31,$twk5,$keyperm
+	lvx		v25,$x10,$key_		# pre-load round[2]
+
+	 vperm		$in0,$inout,$inptail,$inpperm
+	 subi		$inp,$inp,31		# undo "caller"
+	vxor		$twk0,$tweak,$rndkey0
+	vsrab		$tmp,$tweak,$seven	# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	vand		$tmp,$tmp,$eighty7
+	 vxor		$out0,$in0,$twk0
+	vxor		$tweak,$tweak,$tmp
+
+	 lvx_u		$in1,$x10,$inp
+	vxor		$twk1,$tweak,$rndkey0
+	vsrab		$tmp,$tweak,$seven	# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	 le?vperm	$in1,$in1,$in1,$leperm
+	vand		$tmp,$tmp,$eighty7
+	 vxor		$out1,$in1,$twk1
+	vxor		$tweak,$tweak,$tmp
+
+	 lvx_u		$in2,$x20,$inp
+	 andi.		$taillen,$len,15
+	vxor		$twk2,$tweak,$rndkey0
+	vsrab		$tmp,$tweak,$seven	# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	 le?vperm	$in2,$in2,$in2,$leperm
+	vand		$tmp,$tmp,$eighty7
+	 vxor		$out2,$in2,$twk2
+	vxor		$tweak,$tweak,$tmp
+
+	 lvx_u		$in3,$x30,$inp
+	 sub		$len,$len,$taillen
+	vxor		$twk3,$tweak,$rndkey0
+	vsrab		$tmp,$tweak,$seven	# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	 le?vperm	$in3,$in3,$in3,$leperm
+	vand		$tmp,$tmp,$eighty7
+	 vxor		$out3,$in3,$twk3
+	vxor		$tweak,$tweak,$tmp
+
+	 lvx_u		$in4,$x40,$inp
+	 subi		$len,$len,0x60
+	vxor		$twk4,$tweak,$rndkey0
+	vsrab		$tmp,$tweak,$seven	# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	 le?vperm	$in4,$in4,$in4,$leperm
+	vand		$tmp,$tmp,$eighty7
+	 vxor		$out4,$in4,$twk4
+	vxor		$tweak,$tweak,$tmp
+
+	 lvx_u		$in5,$x50,$inp
+	 addi		$inp,$inp,0x60
+	vxor		$twk5,$tweak,$rndkey0
+	vsrab		$tmp,$tweak,$seven	# next tweak value
+	vaddubm		$tweak,$tweak,$tweak
+	vsldoi		$tmp,$tmp,$tmp,15
+	 le?vperm	$in5,$in5,$in5,$leperm
+	vand		$tmp,$tmp,$eighty7
+	 vxor		$out5,$in5,$twk5
+	vxor		$tweak,$tweak,$tmp
+
+	vxor		v31,v31,$rndkey0
+	mtctr		$rounds
+	b		Loop_xts_dec6x
+
+.align	5
+Loop_xts_dec6x:
+	vncipher	$out0,$out0,v24
+	vncipher	$out1,$out1,v24
+	vncipher	$out2,$out2,v24
+	vncipher	$out3,$out3,v24
+	vncipher	$out4,$out4,v24
+	vncipher	$out5,$out5,v24
+	lvx		v24,$x20,$key_		# round[3]
+	addi		$key_,$key_,0x20
+
+	vncipher	$out0,$out0,v25
+	vncipher	$out1,$out1,v25
+	vncipher	$out2,$out2,v25
+	vncipher	$out3,$out3,v25
+	vncipher	$out4,$out4,v25
+	vncipher	$out5,$out5,v25
+	lvx		v25,$x10,$key_		# round[4]
+	bdnz		Loop_xts_dec6x
+
+	subic		$len,$len,96		# $len-=96
+	 vxor		$in0,$twk0,v31		# xor with last round key
+	vncipher	$out0,$out0,v24
+	vncipher	$out1,$out1,v24
+	 vsrab		$tmp,$tweak,$seven	# next tweak value
+	 vxor		$twk0,$tweak,$rndkey0
+	 vaddubm	$tweak,$tweak,$tweak
+	vncipher	$out2,$out2,v24
+	vncipher	$out3,$out3,v24
+	 vsldoi		$tmp,$tmp,$tmp,15
+	vncipher	$out4,$out4,v24
+	vncipher	$out5,$out5,v24
+
+	subfe.		r0,r0,r0		# borrow?-1:0
+	 vand		$tmp,$tmp,$eighty7
+	vncipher	$out0,$out0,v25
+	vncipher	$out1,$out1,v25
+	 vxor		$tweak,$tweak,$tmp
+	vncipher	$out2,$out2,v25
+	vncipher	$out3,$out3,v25
+	 vxor		$in1,$twk1,v31
+	 vsrab		$tmp,$tweak,$seven	# next tweak value
+	 vxor		$twk1,$tweak,$rndkey0
+	vncipher	$out4,$out4,v25
+	vncipher	$out5,$out5,v25
+
+	and		r0,r0,$len
+	 vaddubm	$tweak,$tweak,$tweak
+	 vsldoi		$tmp,$tmp,$tmp,15
+	vncipher	$out0,$out0,v26
+	vncipher	$out1,$out1,v26
+	 vand		$tmp,$tmp,$eighty7
+	vncipher	$out2,$out2,v26
+	vncipher	$out3,$out3,v26
+	 vxor		$tweak,$tweak,$tmp
+	vncipher	$out4,$out4,v26
+	vncipher	$out5,$out5,v26
+
+	add		$inp,$inp,r0		# $inp is adjusted in such
+						# way that at exit from the
+						# loop inX-in5 are loaded
+						# with last "words"
+	 vxor		$in2,$twk2,v31
+	 vsrab		$tmp,$tweak,$seven	# next tweak value
+	 vxor		$twk2,$tweak,$rndkey0
+	 vaddubm	$tweak,$tweak,$tweak
+	vncipher	$out0,$out0,v27
+	vncipher	$out1,$out1,v27
+	 vsldoi		$tmp,$tmp,$tmp,15
+	vncipher	$out2,$out2,v27
+	vncipher	$out3,$out3,v27
+	 vand		$tmp,$tmp,$eighty7
+	vncipher	$out4,$out4,v27
+	vncipher	$out5,$out5,v27
+
+	addi		$key_,$sp,$FRAME+15	# rewind $key_
+	 vxor		$tweak,$tweak,$tmp
+	vncipher	$out0,$out0,v28
+	vncipher	$out1,$out1,v28
+	 vxor		$in3,$twk3,v31
+	 vsrab		$tmp,$tweak,$seven	# next tweak value
+	 vxor		$twk3,$tweak,$rndkey0
+	vncipher	$out2,$out2,v28
+	vncipher	$out3,$out3,v28
+	 vaddubm	$tweak,$tweak,$tweak
+	 vsldoi		$tmp,$tmp,$tmp,15
+	vncipher	$out4,$out4,v28
+	vncipher	$out5,$out5,v28
+	lvx		v24,$x00,$key_		# re-pre-load round[1]
+	 vand		$tmp,$tmp,$eighty7
+
+	vncipher	$out0,$out0,v29
+	vncipher	$out1,$out1,v29
+	 vxor		$tweak,$tweak,$tmp
+	vncipher	$out2,$out2,v29
+	vncipher	$out3,$out3,v29
+	 vxor		$in4,$twk4,v31
+	 vsrab		$tmp,$tweak,$seven	# next tweak value
+	 vxor		$twk4,$tweak,$rndkey0
+	vncipher	$out4,$out4,v29
+	vncipher	$out5,$out5,v29
+	lvx		v25,$x10,$key_		# re-pre-load round[2]
+	 vaddubm	$tweak,$tweak,$tweak
+	 vsldoi		$tmp,$tmp,$tmp,15
+
+	vncipher	$out0,$out0,v30
+	vncipher	$out1,$out1,v30
+	 vand		$tmp,$tmp,$eighty7
+	vncipher	$out2,$out2,v30
+	vncipher	$out3,$out3,v30
+	 vxor		$tweak,$tweak,$tmp
+	vncipher	$out4,$out4,v30
+	vncipher	$out5,$out5,v30
+	 vxor		$in5,$twk5,v31
+	 vsrab		$tmp,$tweak,$seven	# next tweak value
+	 vxor		$twk5,$tweak,$rndkey0
+
+	vncipherlast	$out0,$out0,$in0
+	 lvx_u		$in0,$x00,$inp		# load next input block
+	 vaddubm	$tweak,$tweak,$tweak
+	 vsldoi		$tmp,$tmp,$tmp,15
+	vncipherlast	$out1,$out1,$in1
+	 lvx_u		$in1,$x10,$inp
+	vncipherlast	$out2,$out2,$in2
+	 le?vperm	$in0,$in0,$in0,$leperm
+	 lvx_u		$in2,$x20,$inp
+	 vand		$tmp,$tmp,$eighty7
+	vncipherlast	$out3,$out3,$in3
+	 le?vperm	$in1,$in1,$in1,$leperm
+	 lvx_u		$in3,$x30,$inp
+	vncipherlast	$out4,$out4,$in4
+	 le?vperm	$in2,$in2,$in2,$leperm
+	 lvx_u		$in4,$x40,$inp
+	 vxor		$tweak,$tweak,$tmp
+	vncipherlast	$out5,$out5,$in5
+	 le?vperm	$in3,$in3,$in3,$leperm
+	 lvx_u		$in5,$x50,$inp
+	 addi		$inp,$inp,0x60
+	 le?vperm	$in4,$in4,$in4,$leperm
+	 le?vperm	$in5,$in5,$in5,$leperm
+
+	le?vperm	$out0,$out0,$out0,$leperm
+	le?vperm	$out1,$out1,$out1,$leperm
+	stvx_u		$out0,$x00,$out		# store output
+	 vxor		$out0,$in0,$twk0
+	le?vperm	$out2,$out2,$out2,$leperm
+	stvx_u		$out1,$x10,$out
+	 vxor		$out1,$in1,$twk1
+	le?vperm	$out3,$out3,$out3,$leperm
+	stvx_u		$out2,$x20,$out
+	 vxor		$out2,$in2,$twk2
+	le?vperm	$out4,$out4,$out4,$leperm
+	stvx_u		$out3,$x30,$out
+	 vxor		$out3,$in3,$twk3
+	le?vperm	$out5,$out5,$out5,$leperm
+	stvx_u		$out4,$x40,$out
+	 vxor		$out4,$in4,$twk4
+	stvx_u		$out5,$x50,$out
+	 vxor		$out5,$in5,$twk5
+	addi		$out,$out,0x60
+
+	mtctr		$rounds
+	beq		Loop_xts_dec6x		# did $len-=96 borrow?
+
+	addic.		$len,$len,0x60
+	beq		Lxts_dec6x_zero
+	cmpwi		$len,0x20
+	blt		Lxts_dec6x_one
+	nop
+	beq		Lxts_dec6x_two
+	cmpwi		$len,0x40
+	blt		Lxts_dec6x_three
+	nop
+	beq		Lxts_dec6x_four
+
+Lxts_dec6x_five:
+	vxor		$out0,$in1,$twk0
+	vxor		$out1,$in2,$twk1
+	vxor		$out2,$in3,$twk2
+	vxor		$out3,$in4,$twk3
+	vxor		$out4,$in5,$twk4
+
+	bl		_aesp8_xts_dec5x
+
+	le?vperm	$out0,$out0,$out0,$leperm
+	vmr		$twk0,$twk5		# unused tweak
+	vxor		$twk1,$tweak,$rndkey0
+	le?vperm	$out1,$out1,$out1,$leperm
+	stvx_u		$out0,$x00,$out		# store output
+	vxor		$out0,$in0,$twk1
+	le?vperm	$out2,$out2,$out2,$leperm
+	stvx_u		$out1,$x10,$out
+	le?vperm	$out3,$out3,$out3,$leperm
+	stvx_u		$out2,$x20,$out
+	le?vperm	$out4,$out4,$out4,$leperm
+	stvx_u		$out3,$x30,$out
+	stvx_u		$out4,$x40,$out
+	addi		$out,$out,0x50
+	bne		Lxts_dec6x_steal
+	b		Lxts_dec6x_done
+
+.align	4
+Lxts_dec6x_four:
+	vxor		$out0,$in2,$twk0
+	vxor		$out1,$in3,$twk1
+	vxor		$out2,$in4,$twk2
+	vxor		$out3,$in5,$twk3
+	vxor		$out4,$out4,$out4
+
+	bl		_aesp8_xts_dec5x
+
+	le?vperm	$out0,$out0,$out0,$leperm
+	vmr		$twk0,$twk4		# unused tweak
+	vmr		$twk1,$twk5
+	le?vperm	$out1,$out1,$out1,$leperm
+	stvx_u		$out0,$x00,$out		# store output
+	vxor		$out0,$in0,$twk5
+	le?vperm	$out2,$out2,$out2,$leperm
+	stvx_u		$out1,$x10,$out
+	le?vperm	$out3,$out3,$out3,$leperm
+	stvx_u		$out2,$x20,$out
+	stvx_u		$out3,$x30,$out
+	addi		$out,$out,0x40
+	bne		Lxts_dec6x_steal
+	b		Lxts_dec6x_done
+
+.align	4
+Lxts_dec6x_three:
+	vxor		$out0,$in3,$twk0
+	vxor		$out1,$in4,$twk1
+	vxor		$out2,$in5,$twk2
+	vxor		$out3,$out3,$out3
+	vxor		$out4,$out4,$out4
+
+	bl		_aesp8_xts_dec5x
+
+	le?vperm	$out0,$out0,$out0,$leperm
+	vmr		$twk0,$twk3		# unused tweak
+	vmr		$twk1,$twk4
+	le?vperm	$out1,$out1,$out1,$leperm
+	stvx_u		$out0,$x00,$out		# store output
+	vxor		$out0,$in0,$twk4
+	le?vperm	$out2,$out2,$out2,$leperm
+	stvx_u		$out1,$x10,$out
+	stvx_u		$out2,$x20,$out
+	addi		$out,$out,0x30
+	bne		Lxts_dec6x_steal
+	b		Lxts_dec6x_done
+
+.align	4
+Lxts_dec6x_two:
+	vxor		$out0,$in4,$twk0
+	vxor		$out1,$in5,$twk1
+	vxor		$out2,$out2,$out2
+	vxor		$out3,$out3,$out3
+	vxor		$out4,$out4,$out4
+
+	bl		_aesp8_xts_dec5x
+
+	le?vperm	$out0,$out0,$out0,$leperm
+	vmr		$twk0,$twk2		# unused tweak
+	vmr		$twk1,$twk3
+	le?vperm	$out1,$out1,$out1,$leperm
+	stvx_u		$out0,$x00,$out		# store output
+	vxor		$out0,$in0,$twk3
+	stvx_u		$out1,$x10,$out
+	addi		$out,$out,0x20
+	bne		Lxts_dec6x_steal
+	b		Lxts_dec6x_done
+
+.align	4
+Lxts_dec6x_one:
+	vxor		$out0,$in5,$twk0
+	nop
+Loop_xts_dec1x:
+	vncipher	$out0,$out0,v24
+	lvx		v24,$x20,$key_		# round[3]
+	addi		$key_,$key_,0x20
+
+	vncipher	$out0,$out0,v25
+	lvx		v25,$x10,$key_		# round[4]
+	bdnz		Loop_xts_dec1x
+
+	subi		r0,$taillen,1
+	vncipher	$out0,$out0,v24
+
+	andi.		r0,r0,16
+	cmpwi		$taillen,0
+	vncipher	$out0,$out0,v25
+
+	sub		$inp,$inp,r0
+	vncipher	$out0,$out0,v26
+
+	lvx_u		$in0,0,$inp
+	vncipher	$out0,$out0,v27
+
+	addi		$key_,$sp,$FRAME+15	# rewind $key_
+	vncipher	$out0,$out0,v28
+	lvx		v24,$x00,$key_		# re-pre-load round[1]
+
+	vncipher	$out0,$out0,v29
+	lvx		v25,$x10,$key_		# re-pre-load round[2]
+	 vxor		$twk0,$twk0,v31
+
+	le?vperm	$in0,$in0,$in0,$leperm
+	vncipher	$out0,$out0,v30
+
+	mtctr		$rounds
+	vncipherlast	$out0,$out0,$twk0
+
+	vmr		$twk0,$twk1		# unused tweak
+	vmr		$twk1,$twk2
+	le?vperm	$out0,$out0,$out0,$leperm
+	stvx_u		$out0,$x00,$out		# store output
+	addi		$out,$out,0x10
+	vxor		$out0,$in0,$twk2
+	bne		Lxts_dec6x_steal
+	b		Lxts_dec6x_done
+
+.align	4
+Lxts_dec6x_zero:
+	cmpwi		$taillen,0
+	beq		Lxts_dec6x_done
+
+	lvx_u		$in0,0,$inp
+	le?vperm	$in0,$in0,$in0,$leperm
+	vxor		$out0,$in0,$twk1
+Lxts_dec6x_steal:
+	vncipher	$out0,$out0,v24
+	lvx		v24,$x20,$key_		# round[3]
+	addi		$key_,$key_,0x20
+
+	vncipher	$out0,$out0,v25
+	lvx		v25,$x10,$key_		# round[4]
+	bdnz		Lxts_dec6x_steal
+
+	add		$inp,$inp,$taillen
+	vncipher	$out0,$out0,v24
+
+	cmpwi		$taillen,0
+	vncipher	$out0,$out0,v25
+
+	lvx_u		$in0,0,$inp
+	vncipher	$out0,$out0,v26
+
+	lvsr		$inpperm,0,$taillen	# $in5 is no more
+	vncipher	$out0,$out0,v27
+
+	addi		$key_,$sp,$FRAME+15	# rewind $key_
+	vncipher	$out0,$out0,v28
+	lvx		v24,$x00,$key_		# re-pre-load round[1]
+
+	vncipher	$out0,$out0,v29
+	lvx		v25,$x10,$key_		# re-pre-load round[2]
+	 vxor		$twk1,$twk1,v31
+
+	le?vperm	$in0,$in0,$in0,$leperm
+	vncipher	$out0,$out0,v30
+
+	vperm		$in0,$in0,$in0,$inpperm
+	vncipherlast	$tmp,$out0,$twk1
+
+	le?vperm	$out0,$tmp,$tmp,$leperm
+	le?stvx_u	$out0,0,$out
+	be?stvx_u	$tmp,0,$out
+
+	vxor		$out0,$out0,$out0
+	vspltisb	$out1,-1
+	vperm		$out0,$out0,$out1,$inpperm
+	vsel		$out0,$in0,$tmp,$out0
+	vxor		$out0,$out0,$twk0
+
+	subi		r30,$out,1
+	mtctr		$taillen
+Loop_xts_dec6x_steal:
+	lbzu		r0,1(r30)
+	stb		r0,16(r30)
+	bdnz		Loop_xts_dec6x_steal
+
+	li		$taillen,0
+	mtctr		$rounds
+	b		Loop_xts_dec1x		# one more time...
+
+.align	4
+Lxts_dec6x_done:
+	${UCMP}i	$ivp,0
+	beq		Lxts_dec6x_ret
+
+	vxor		$tweak,$twk0,$rndkey0
+	le?vperm	$tweak,$tweak,$tweak,$leperm
+	stvx_u		$tweak,0,$ivp
+
+Lxts_dec6x_ret:
+	mtlr		r11
+	li		r10,`$FRAME+15`
+	li		r11,`$FRAME+31`
+	stvx		$seven,r10,$sp		# wipe copies of round keys
+	addi		r10,r10,32
+	stvx		$seven,r11,$sp
+	addi		r11,r11,32
+	stvx		$seven,r10,$sp
+	addi		r10,r10,32
+	stvx		$seven,r11,$sp
+	addi		r11,r11,32
+	stvx		$seven,r10,$sp
+	addi		r10,r10,32
+	stvx		$seven,r11,$sp
+	addi		r11,r11,32
+	stvx		$seven,r10,$sp
+	addi		r10,r10,32
+	stvx		$seven,r11,$sp
+	addi		r11,r11,32
+
+	mtspr		256,$vrsave
+	lvx		v20,r10,$sp		# ABI says so
+	addi		r10,r10,32
+	lvx		v21,r11,$sp
+	addi		r11,r11,32
+	lvx		v22,r10,$sp
+	addi		r10,r10,32
+	lvx		v23,r11,$sp
+	addi		r11,r11,32
+	lvx		v24,r10,$sp
+	addi		r10,r10,32
+	lvx		v25,r11,$sp
+	addi		r11,r11,32
+	lvx		v26,r10,$sp
+	addi		r10,r10,32
+	lvx		v27,r11,$sp
+	addi		r11,r11,32
+	lvx		v28,r10,$sp
+	addi		r10,r10,32
+	lvx		v29,r11,$sp
+	addi		r11,r11,32
+	lvx		v30,r10,$sp
+	lvx		v31,r11,$sp
+	$POP		r26,`$FRAME+21*16+0*$SIZE_T`($sp)
+	$POP		r27,`$FRAME+21*16+1*$SIZE_T`($sp)
+	$POP		r28,`$FRAME+21*16+2*$SIZE_T`($sp)
+	$POP		r29,`$FRAME+21*16+3*$SIZE_T`($sp)
+	$POP		r30,`$FRAME+21*16+4*$SIZE_T`($sp)
+	$POP		r31,`$FRAME+21*16+5*$SIZE_T`($sp)
+	addi		$sp,$sp,`$FRAME+21*16+6*$SIZE_T`
+	blr
+	.long		0
+	.byte		0,12,0x04,1,0x80,6,6,0
+	.long		0
+
+.align	5
+_aesp8_xts_dec5x:
+	vncipher	$out0,$out0,v24
+	vncipher	$out1,$out1,v24
+	vncipher	$out2,$out2,v24
+	vncipher	$out3,$out3,v24
+	vncipher	$out4,$out4,v24
+	lvx		v24,$x20,$key_		# round[3]
+	addi		$key_,$key_,0x20
+
+	vncipher	$out0,$out0,v25
+	vncipher	$out1,$out1,v25
+	vncipher	$out2,$out2,v25
+	vncipher	$out3,$out3,v25
+	vncipher	$out4,$out4,v25
+	lvx		v25,$x10,$key_		# round[4]
+	bdnz		_aesp8_xts_dec5x
+
+	subi		r0,$taillen,1
+	vncipher	$out0,$out0,v24
+	vncipher	$out1,$out1,v24
+	vncipher	$out2,$out2,v24
+	vncipher	$out3,$out3,v24
+	vncipher	$out4,$out4,v24
+
+	andi.		r0,r0,16
+	cmpwi		$taillen,0
+	vncipher	$out0,$out0,v25
+	vncipher	$out1,$out1,v25
+	vncipher	$out2,$out2,v25
+	vncipher	$out3,$out3,v25
+	vncipher	$out4,$out4,v25
+	 vxor		$twk0,$twk0,v31
+
+	sub		$inp,$inp,r0
+	vncipher	$out0,$out0,v26
+	vncipher	$out1,$out1,v26
+	vncipher	$out2,$out2,v26
+	vncipher	$out3,$out3,v26
+	vncipher	$out4,$out4,v26
+	 vxor		$in1,$twk1,v31
+
+	vncipher	$out0,$out0,v27
+	lvx_u		$in0,0,$inp
+	vncipher	$out1,$out1,v27
+	vncipher	$out2,$out2,v27
+	vncipher	$out3,$out3,v27
+	vncipher	$out4,$out4,v27
+	 vxor		$in2,$twk2,v31
+
+	addi		$key_,$sp,$FRAME+15	# rewind $key_
+	vncipher	$out0,$out0,v28
+	vncipher	$out1,$out1,v28
+	vncipher	$out2,$out2,v28
+	vncipher	$out3,$out3,v28
+	vncipher	$out4,$out4,v28
+	lvx		v24,$x00,$key_		# re-pre-load round[1]
+	 vxor		$in3,$twk3,v31
+
+	vncipher	$out0,$out0,v29
+	le?vperm	$in0,$in0,$in0,$leperm
+	vncipher	$out1,$out1,v29
+	vncipher	$out2,$out2,v29
+	vncipher	$out3,$out3,v29
+	vncipher	$out4,$out4,v29
+	lvx		v25,$x10,$key_		# re-pre-load round[2]
+	 vxor		$in4,$twk4,v31
+
+	vncipher	$out0,$out0,v30
+	vncipher	$out1,$out1,v30
+	vncipher	$out2,$out2,v30
+	vncipher	$out3,$out3,v30
+	vncipher	$out4,$out4,v30
+
+	vncipherlast	$out0,$out0,$twk0
+	vncipherlast	$out1,$out1,$in1
+	vncipherlast	$out2,$out2,$in2
+	vncipherlast	$out3,$out3,$in3
+	vncipherlast	$out4,$out4,$in4
+	mtctr		$rounds
+	blr
+        .long   	0
+        .byte   	0,12,0x14,0,0,0,0,0
+___
+}}	}}}
+
+my $consts=1;
+foreach(split("\n",$code)) {
+        s/\`([^\`]*)\`/eval($1)/geo;
+
+	# constants table endian-specific conversion
+	if ($consts && m/\.(long|byte)\s+(.+)\s+(\?[a-z]*)$/o) {
+	    my $conv=$3;
+	    my @bytes=();
+
+	    # convert to endian-agnostic format
+	    if ($1 eq "long") {
+	      foreach (split(/,\s*/,$2)) {
+		my $l = /^0/?oct:int;
+		push @bytes,($l>>24)&0xff,($l>>16)&0xff,($l>>8)&0xff,$l&0xff;
+	      }
+	    } else {
+		@bytes = map(/^0/?oct:int,split(/,\s*/,$2));
+	    }
+
+	    # little-endian conversion
+	    if ($flavour =~ /le$/o) {
+		SWITCH: for($conv)  {
+		    /\?inv/ && do   { @bytes=map($_^0xf,@bytes); last; };
+		    /\?rev/ && do   { @bytes=reverse(@bytes);    last; };
+		}
+	    }
+
+	    #emit
+	    print ".byte\t",join(',',map (sprintf("0x%02x",$_),@bytes)),"\n";
+	    next;
+	}
+	$consts=0 if (m/Lconsts:/o);	# end of table
+
+	# instructions prefixed with '?' are endian-specific and need
+	# to be adjusted accordingly...
+	if ($flavour =~ /le$/o) {	# little-endian
+	    s/le\?//o		or
+	    s/be\?/#be#/o	or
+	    s/\?lvsr/lvsl/o	or
+	    s/\?lvsl/lvsr/o	or
+	    s/\?(vperm\s+v[0-9]+,\s*)(v[0-9]+,\s*)(v[0-9]+,\s*)(v[0-9]+)/$1$3$2$4/o or
+	    s/\?(vsldoi\s+v[0-9]+,\s*)(v[0-9]+,)\s*(v[0-9]+,\s*)([0-9]+)/$1$3$2 16-$4/o or
+	    s/\?(vspltw\s+v[0-9]+,\s*)(v[0-9]+,)\s*([0-9])/$1$2 3-$3/o;
+	} else {			# big-endian
+	    s/le\?/#le#/o	or
+	    s/be\?//o		or
+	    s/\?([a-z]+)/$1/o;
+	}
+
+        print $_,"\n";
+}
+
+close STDOUT;
diff --git a/cbits/asm/generate.sh b/cbits/asm/generate.sh
--- a/cbits/asm/generate.sh
+++ b/cbits/asm/generate.sh
@@ -12,6 +12,8 @@
 #     arm/sha1-armv8.pl		arm/sha512-armv8.pl
 #     arm/keccak1600-armv8.pl	arm/arm-xlate.pl
 #     arm/arm_arch.h
+#     ppc/aesp8-ppc.pl		ppc/ghashp8-ppc.pl
+#     ppc/ppc-xlate.pl
 #
 # The .pl files are the generator, not the product: each one emits
 # assembly for a given "flavour", which is the calling convention and the
@@ -150,4 +152,25 @@
 
 	.section	.note.GNU-stack,"",%progbits
 	NOTE
+done
+
+# POWER8's AES and GHASH, little-endian only.  The generator emits a
+# big-endian flavour from the same source, and crypton does not build for
+# big-endian POWER: nothing here has been able to run that code, and an AES
+# path that has never been executed is not one to check in.  Adding
+# "linux64" to the list below is what it would take.
+#
+# These two read no capability word of their own -- unlike the ARM and x86-64
+# modules above -- so the entry points are all that is renamed.
+for flavour in linux64le; do
+	perl aesp8-ppc.pl $flavour tmp-$flavour.S
+	sed -e 's/\baes_p8_/crypton_aes_p8_/g' \
+	    tmp-$flavour.S > aesp8-ppc-$flavour.S
+
+	perl ghashp8-ppc.pl $flavour tmp-$flavour.S
+	sed -e 's/\bgcm_init_p8/crypton_gcm_init_p8/g' \
+	    -e 's/\bgcm_gmult_p8/crypton_gcm_gmult_p8/g' \
+	    -e 's/\bgcm_ghash_p8/crypton_gcm_ghash_p8/g' \
+	    tmp-$flavour.S > ghashp8-ppc-$flavour.S
+	rm -f tmp-$flavour.S
 done
diff --git a/cbits/asm/ghashp8-ppc-linux64le.S b/cbits/asm/ghashp8-ppc-linux64le.S
new file mode 100644
--- /dev/null
+++ b/cbits/asm/ghashp8-ppc-linux64le.S
@@ -0,0 +1,575 @@
+.machine	"any"
+
+.abiversion	2
+.text
+
+.globl	crypton_gcm_init_p8
+.type	crypton_gcm_init_p8,@function
+.align	5
+crypton_gcm_init_p8:
+.localentry	crypton_gcm_init_p8,0
+
+	li	0,-4096
+	li	8,0x10
+	li	12,-1
+	li	9,0x20
+	or	0,0,0
+	li	10,0x30
+	.long	0x7D202699
+
+	vspltisb	8,-16
+	vspltisb	5,1
+	vaddubm	8,8,8
+	vxor	4,4,4
+	vor	8,8,5
+	vsldoi	8,8,4,15
+	vsldoi	6,4,5,1
+	vaddubm	8,8,8
+	vspltisb	7,7
+	vor	8,8,6
+	vspltb	6,9,0
+	vsl	9,9,5
+	vsrab	6,6,7
+	vand	6,6,8
+	vxor	3,9,6
+
+	vsldoi	9,3,3,8
+	vsldoi	8,4,8,8
+	vsldoi	11,4,9,8
+	vsldoi	10,9,4,8
+
+	.long	0x7D001F99
+	.long	0x7D681F99
+	li	8,0x40
+	.long	0x7D291F99
+	li	9,0x50
+	.long	0x7D4A1F99
+	li	10,0x60
+
+	.long	0x10035CC8
+	.long	0x10234CC8
+	.long	0x104354C8
+
+	.long	0x10E044C8
+
+	vsldoi	5,1,4,8
+	vsldoi	6,4,1,8
+	vxor	0,0,5
+	vxor	2,2,6
+
+	vsldoi	0,0,0,8
+	vxor	0,0,7
+
+	vsldoi	6,0,0,8
+	.long	0x100044C8
+	vxor	6,6,2
+	vxor	16,0,6
+
+	vsldoi	17,16,16,8
+	vsldoi	19,4,17,8
+	vsldoi	18,17,4,8
+
+	.long	0x7E681F99
+	li	8,0x70
+	.long	0x7E291F99
+	li	9,0x80
+	.long	0x7E4A1F99
+	li	10,0x90
+	.long	0x10039CC8
+	.long	0x11B09CC8
+	.long	0x10238CC8
+	.long	0x11D08CC8
+	.long	0x104394C8
+	.long	0x11F094C8
+
+	.long	0x10E044C8
+	.long	0x114D44C8
+
+	vsldoi	5,1,4,8
+	vsldoi	6,4,1,8
+	vsldoi	11,14,4,8
+	vsldoi	9,4,14,8
+	vxor	0,0,5
+	vxor	2,2,6
+	vxor	13,13,11
+	vxor	15,15,9
+
+	vsldoi	0,0,0,8
+	vsldoi	13,13,13,8
+	vxor	0,0,7
+	vxor	13,13,10
+
+	vsldoi	6,0,0,8
+	vsldoi	9,13,13,8
+	.long	0x100044C8
+	.long	0x11AD44C8
+	vxor	6,6,2
+	vxor	9,9,15
+	vxor	0,0,6
+	vxor	13,13,9
+
+	vsldoi	9,0,0,8
+	vsldoi	17,13,13,8
+	vsldoi	11,4,9,8
+	vsldoi	10,9,4,8
+	vsldoi	19,4,17,8
+	vsldoi	18,17,4,8
+
+	.long	0x7D681F99
+	li	8,0xa0
+	.long	0x7D291F99
+	li	9,0xb0
+	.long	0x7D4A1F99
+	li	10,0xc0
+	.long	0x7E681F99
+	.long	0x7E291F99
+	.long	0x7E4A1F99
+
+	or	12,12,12
+	blr	
+.long	0
+.byte	0,12,0x14,0,0,0,2,0
+.long	0
+.size	crypton_gcm_init_p8,.-crypton_gcm_init_p8
+.globl	crypton_gcm_gmult_p8
+.type	crypton_gcm_gmult_p8,@function
+.align	5
+crypton_gcm_gmult_p8:
+.localentry	crypton_gcm_gmult_p8,0
+
+	lis	0,0xfff8
+	li	8,0x10
+	li	12,-1
+	li	9,0x20
+	or	0,0,0
+	li	10,0x30
+	.long	0x7C601E99
+
+	.long	0x7D682699
+	lvsl	12,0,0
+	.long	0x7D292699
+	vspltisb	5,0x07
+	.long	0x7D4A2699
+	vxor	12,12,5
+	.long	0x7D002699
+	vperm	3,3,3,12
+	vxor	4,4,4
+
+	.long	0x10035CC8
+	.long	0x10234CC8
+	.long	0x104354C8
+
+	.long	0x10E044C8
+
+	vsldoi	5,1,4,8
+	vsldoi	6,4,1,8
+	vxor	0,0,5
+	vxor	2,2,6
+
+	vsldoi	0,0,0,8
+	vxor	0,0,7
+
+	vsldoi	6,0,0,8
+	.long	0x100044C8
+	vxor	6,6,2
+	vxor	0,0,6
+
+	vperm	0,0,0,12
+	.long	0x7C001F99
+
+	or	12,12,12
+	blr	
+.long	0
+.byte	0,12,0x14,0,0,0,2,0
+.long	0
+.size	crypton_gcm_gmult_p8,.-crypton_gcm_gmult_p8
+
+.globl	crypton_gcm_ghash_p8
+.type	crypton_gcm_ghash_p8,@function
+.align	5
+crypton_gcm_ghash_p8:
+.localentry	crypton_gcm_ghash_p8,0
+
+	li	0,-4096
+	li	8,0x10
+	li	12,-1
+	li	9,0x20
+	or	0,0,0
+	li	10,0x30
+	.long	0x7C001E99
+
+	.long	0x7D682699
+	li	8,0x40
+	lvsl	12,0,0
+	.long	0x7D292699
+	li	9,0x50
+	vspltisb	5,0x07
+	.long	0x7D4A2699
+	li	10,0x60
+	vxor	12,12,5
+	.long	0x7D002699
+	vperm	0,0,0,12
+	vxor	4,4,4
+
+	cmpldi	6,64
+	bge	.Lgcm_ghash_p8_4x
+
+	.long	0x7C602E99
+	addi	5,5,16
+	subic.	6,6,16
+	vperm	3,3,3,12
+	vxor	3,3,0
+	beq	.Lshort
+
+	.long	0x7E682699
+	li	8,16
+	.long	0x7E292699
+	add	9,5,6
+	.long	0x7E4A2699
+
+
+.align	5
+.Loop_2x:
+	.long	0x7E002E99
+	vperm	16,16,16,12
+
+	subic	6,6,32
+	.long	0x10039CC8
+	.long	0x11B05CC8
+	subfe	0,0,0
+	.long	0x10238CC8
+	.long	0x11D04CC8
+	and	0,0,6
+	.long	0x104394C8
+	.long	0x11F054C8
+	add	5,5,0
+
+	vxor	0,0,13
+	vxor	1,1,14
+
+	.long	0x10E044C8
+
+	vsldoi	5,1,4,8
+	vsldoi	6,4,1,8
+	vxor	2,2,15
+	vxor	0,0,5
+	vxor	2,2,6
+
+	vsldoi	0,0,0,8
+	vxor	0,0,7
+	.long	0x7C682E99
+	addi	5,5,32
+
+	vsldoi	6,0,0,8
+	.long	0x100044C8
+	vperm	3,3,3,12
+	vxor	6,6,2
+	vxor	3,3,6
+	vxor	3,3,0
+	cmpld	9,5
+	bgt	.Loop_2x
+
+	cmplwi	6,0
+	bne	.Leven
+
+.Lshort:
+	.long	0x10035CC8
+	.long	0x10234CC8
+	.long	0x104354C8
+
+	.long	0x10E044C8
+
+	vsldoi	5,1,4,8
+	vsldoi	6,4,1,8
+	vxor	0,0,5
+	vxor	2,2,6
+
+	vsldoi	0,0,0,8
+	vxor	0,0,7
+
+	vsldoi	6,0,0,8
+	.long	0x100044C8
+	vxor	6,6,2
+
+.Leven:
+	vxor	0,0,6
+	vperm	0,0,0,12
+	.long	0x7C001F99
+
+	or	12,12,12
+	blr	
+.long	0
+.byte	0,12,0x14,0,0,0,4,0
+.long	0
+.align	5
+.crypton_gcm_ghash_p8_4x:
+.Lgcm_ghash_p8_4x:
+	stdu	1,-256(1)
+	li	10,63
+	li	11,79
+	stvx	20,10,1
+	addi	10,10,32
+	stvx	21,11,1
+	addi	11,11,32
+	stvx	22,10,1
+	addi	10,10,32
+	stvx	23,11,1
+	addi	11,11,32
+	stvx	24,10,1
+	addi	10,10,32
+	stvx	25,11,1
+	addi	11,11,32
+	stvx	26,10,1
+	addi	10,10,32
+	stvx	27,11,1
+	addi	11,11,32
+	stvx	28,10,1
+	addi	10,10,32
+	stvx	29,11,1
+	addi	11,11,32
+	stvx	30,10,1
+	li	10,0x60
+	stvx	31,11,1
+	li	0,-1
+	stw	12,252(1)
+	or	0,0,0
+
+	lvsl	5,0,8
+
+	li	8,0x70
+	.long	0x7E292699
+	li	9,0x80
+	vspltisb	6,8
+
+	li	10,0x90
+	.long	0x7EE82699
+	li	8,0xa0
+	.long	0x7F092699
+	li	9,0xb0
+	.long	0x7F2A2699
+	li	10,0xc0
+	.long	0x7FA82699
+	li	8,0x10
+	.long	0x7FC92699
+	li	9,0x20
+	.long	0x7FEA2699
+	li	10,0x30
+
+	vsldoi	7,4,6,8
+	vaddubm	18,5,7
+	vaddubm	19,6,18
+
+	srdi	6,6,4
+
+	.long	0x7C602E99
+	.long	0x7E082E99
+	subic.	6,6,8
+	.long	0x7EC92E99
+	.long	0x7F8A2E99
+	addi	5,5,0x40
+	vperm	3,3,3,12
+	vperm	16,16,16,12
+	vperm	22,22,22,12
+	vperm	28,28,28,12
+
+	vxor	2,3,0
+
+	.long	0x11B0BCC8
+	.long	0x11D0C4C8
+	.long	0x11F0CCC8
+
+	vperm	11,17,9,18
+	vperm	5,22,28,19
+	vperm	10,17,9,19
+	vperm	6,22,28,18
+	.long	0x12B68CC8
+	.long	0x12855CC8
+	.long	0x137C4CC8
+	.long	0x134654C8
+
+	vxor	21,21,14
+	vxor	20,20,13
+	vxor	27,27,21
+	vxor	26,26,15
+
+	blt	.Ltail_4x
+
+.Loop_4x:
+	.long	0x7C602E99
+	.long	0x7E082E99
+	subic.	6,6,4
+	.long	0x7EC92E99
+	.long	0x7F8A2E99
+	addi	5,5,0x40
+	vperm	16,16,16,12
+	vperm	22,22,22,12
+	vperm	28,28,28,12
+	vperm	3,3,3,12
+
+	.long	0x1002ECC8
+	.long	0x1022F4C8
+	.long	0x1042FCC8
+	.long	0x11B0BCC8
+	.long	0x11D0C4C8
+	.long	0x11F0CCC8
+
+	vxor	0,0,20
+	vxor	1,1,27
+	vxor	2,2,26
+	vperm	5,22,28,19
+	vperm	6,22,28,18
+
+	.long	0x10E044C8
+	.long	0x12855CC8
+	.long	0x134654C8
+
+	vsldoi	5,1,4,8
+	vsldoi	6,4,1,8
+	vxor	0,0,5
+	vxor	2,2,6
+
+	vsldoi	0,0,0,8
+	vxor	0,0,7
+
+	vsldoi	6,0,0,8
+	.long	0x12B68CC8
+	.long	0x137C4CC8
+	.long	0x100044C8
+
+	vxor	20,20,13
+	vxor	26,26,15
+	vxor	2,2,3
+	vxor	21,21,14
+	vxor	2,2,6
+	vxor	27,27,21
+	vxor	2,2,0
+	bge	.Loop_4x
+
+.Ltail_4x:
+	.long	0x1002ECC8
+	.long	0x1022F4C8
+	.long	0x1042FCC8
+
+	vxor	0,0,20
+	vxor	1,1,27
+
+	.long	0x10E044C8
+
+	vsldoi	5,1,4,8
+	vsldoi	6,4,1,8
+	vxor	2,2,26
+	vxor	0,0,5
+	vxor	2,2,6
+
+	vsldoi	0,0,0,8
+	vxor	0,0,7
+
+	vsldoi	6,0,0,8
+	.long	0x100044C8
+	vxor	6,6,2
+	vxor	0,0,6
+
+	addic.	6,6,4
+	beq	.Ldone_4x
+
+	.long	0x7C602E99
+	cmpldi	6,2
+	li	6,-4
+	blt	.Lone
+	.long	0x7E082E99
+	beq	.Ltwo
+
+.Lthree:
+	.long	0x7EC92E99
+	vperm	3,3,3,12
+	vperm	16,16,16,12
+	vperm	22,22,22,12
+
+	vxor	2,3,0
+	vor	29,23,23
+	vor	30,24,24
+	vor	31,25,25
+
+	vperm	5,16,22,19
+	vperm	6,16,22,18
+	.long	0x12B08CC8
+	.long	0x13764CC8
+	.long	0x12855CC8
+	.long	0x134654C8
+
+	vxor	27,27,21
+	b	.Ltail_4x
+
+.align	4
+.Ltwo:
+	vperm	3,3,3,12
+	vperm	16,16,16,12
+
+	vxor	2,3,0
+	vperm	5,4,16,19
+	vperm	6,4,16,18
+
+	vsldoi	29,4,17,8
+	vor	30,17,17
+	vsldoi	31,17,4,8
+
+	.long	0x12855CC8
+	.long	0x13704CC8
+	.long	0x134654C8
+
+	b	.Ltail_4x
+
+.align	4
+.Lone:
+	vperm	3,3,3,12
+
+	vsldoi	29,4,9,8
+	vor	30,9,9
+	vsldoi	31,9,4,8
+
+	vxor	2,3,0
+	vxor	20,20,20
+	vxor	27,27,27
+	vxor	26,26,26
+
+	b	.Ltail_4x
+
+.Ldone_4x:
+	vperm	0,0,0,12
+	.long	0x7C001F99
+
+	li	10,63
+	li	11,79
+	or	12,12,12
+	lvx	20,10,1
+	addi	10,10,32
+	lvx	21,11,1
+	addi	11,11,32
+	lvx	22,10,1
+	addi	10,10,32
+	lvx	23,11,1
+	addi	11,11,32
+	lvx	24,10,1
+	addi	10,10,32
+	lvx	25,11,1
+	addi	11,11,32
+	lvx	26,10,1
+	addi	10,10,32
+	lvx	27,11,1
+	addi	11,11,32
+	lvx	28,10,1
+	addi	10,10,32
+	lvx	29,11,1
+	addi	11,11,32
+	lvx	30,10,1
+	lvx	31,11,1
+	addi	1,1,256
+	blr	
+.long	0
+.byte	0,12,0x04,0,0x80,0,4,0
+.long	0
+.size	crypton_gcm_ghash_p8,.-crypton_gcm_ghash_p8
+
+.byte	71,72,65,83,72,32,102,111,114,32,80,111,119,101,114,73,83,65,32,50,46,48,55,44,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0
+.align	2
+.align	2
diff --git a/cbits/asm/ghashp8-ppc.pl b/cbits/asm/ghashp8-ppc.pl
new file mode 100644
--- /dev/null
+++ b/cbits/asm/ghashp8-ppc.pl
@@ -0,0 +1,664 @@
+#!/usr/bin/env perl
+#
+# ====================================================================
+# Written by Andy Polyakov, @dot-asm, initially for use in the OpenSSL
+# project. The module is dual licensed under OpenSSL and CRYPTOGAMS
+# licenses depending on where you obtain it. For further details see
+# https://github.com/dot-asm/cryptogams/.
+# ====================================================================
+#
+# GHASH for for PowerISA v2.07.
+#
+# July 2014
+#
+# Accurate performance measurements are problematic, because it's
+# always virtualized setup with possibly throttled processor.
+# Relative comparison is therefore more informative. This initial
+# version is ~2.1x slower than hardware-assisted AES-128-CTR, ~12x
+# faster than "4-bit" integer-only compiler-generated 64-bit code.
+# "Initial version" means that there is room for further improvement.
+
+# May 2016
+#
+# 2x aggregated reduction improves performance by 50% (resulting
+# performance on POWER8 is 1 cycle per processed byte), and 4x
+# aggregated reduction - by 170% or 2.7x (resulting in 0.55 cpb).
+# POWER9 delivers 0.51 cpb.
+
+$flavour=shift;
+$output =shift;
+
+if ($flavour =~ /64/) {
+	$SIZE_T=8;
+	$LRSAVE=2*$SIZE_T;
+	$STU="stdu";
+	$POP="ld";
+	$PUSH="std";
+	$UCMP="cmpld";
+	$SHRI="srdi";
+} elsif ($flavour =~ /32/) {
+	$SIZE_T=4;
+	$LRSAVE=$SIZE_T;
+	$STU="stwu";
+	$POP="lwz";
+	$PUSH="stw";
+	$UCMP="cmplw";
+	$SHRI="srwi";
+} else { die "nonsense $flavour"; }
+
+$sp="r1";
+$FRAME=6*$SIZE_T+13*16;	# 13*16 is for v20-v31 offload
+
+$0 =~ m/(.*[\/\\])[^\/\\]+$/; $dir=$1;
+( $xlate="${dir}ppc-xlate.pl" and -f $xlate ) or
+( $xlate="${dir}../../perlasm/ppc-xlate.pl" and -f $xlate) or
+die "can't locate ppc-xlate.pl";
+
+open STDOUT,"| $^X $xlate $flavour $output" || die "can't call $xlate: $!";
+
+my ($Xip,$Htbl,$inp,$len)=map("r$_",(3..6));	# argument block
+
+my ($Xl,$Xm,$Xh,$IN)=map("v$_",(0..3));
+my ($zero,$t0,$t1,$t2,$xC2,$H,$Hh,$Hl,$lemask)=map("v$_",(4..12));
+my ($Xl1,$Xm1,$Xh1,$IN1,$H2,$H2h,$H2l)=map("v$_",(13..19));
+my $vrsave="r12";
+
+$code=<<___;
+.machine	"any"
+
+.text
+
+.globl	.gcm_init_p8
+.align	5
+.gcm_init_p8:
+	li		r0,-4096
+	li		r8,0x10
+	mfspr		$vrsave,256
+	li		r9,0x20
+	mtspr		256,r0
+	li		r10,0x30
+	lvx_u		$H,0,r4			# load H
+
+	vspltisb	$xC2,-16		# 0xf0
+	vspltisb	$t0,1			# one
+	vaddubm		$xC2,$xC2,$xC2		# 0xe0
+	vxor		$zero,$zero,$zero
+	vor		$xC2,$xC2,$t0		# 0xe1
+	vsldoi		$xC2,$xC2,$zero,15	# 0xe1...
+	vsldoi		$t1,$zero,$t0,1		# ...1
+	vaddubm		$xC2,$xC2,$xC2		# 0xc2...
+	vspltisb	$t2,7
+	vor		$xC2,$xC2,$t1		# 0xc2....01
+	vspltb		$t1,$H,0		# most significant byte
+	vsl		$H,$H,$t0		# H<<=1
+	vsrab		$t1,$t1,$t2		# broadcast carry bit
+	vand		$t1,$t1,$xC2
+	vxor		$IN,$H,$t1		# twisted H
+
+	vsldoi		$H,$IN,$IN,8		# twist even more ...
+	vsldoi		$xC2,$zero,$xC2,8	# 0xc2.0
+	vsldoi		$Hl,$zero,$H,8		# ... and split
+	vsldoi		$Hh,$H,$zero,8
+
+	stvx_u		$xC2,0,r3		# save pre-computed table
+	stvx_u		$Hl,r8,r3
+	li		r8,0x40
+	stvx_u		$H, r9,r3
+	li		r9,0x50
+	stvx_u		$Hh,r10,r3
+	li		r10,0x60
+
+	vpmsumd		$Xl,$IN,$Hl		# H.lo·H.lo
+	vpmsumd		$Xm,$IN,$H		# H.hi·H.lo+H.lo·H.hi
+	vpmsumd		$Xh,$IN,$Hh		# H.hi·H.hi
+
+	vpmsumd		$t2,$Xl,$xC2		# 1st reduction phase
+
+	vsldoi		$t0,$Xm,$zero,8
+	vsldoi		$t1,$zero,$Xm,8
+	vxor		$Xl,$Xl,$t0
+	vxor		$Xh,$Xh,$t1
+
+	vsldoi		$Xl,$Xl,$Xl,8
+	vxor		$Xl,$Xl,$t2
+
+	vsldoi		$t1,$Xl,$Xl,8		# 2nd reduction phase
+	vpmsumd		$Xl,$Xl,$xC2
+	vxor		$t1,$t1,$Xh
+	vxor		$IN1,$Xl,$t1
+
+	vsldoi		$H2,$IN1,$IN1,8
+	vsldoi		$H2l,$zero,$H2,8
+	vsldoi		$H2h,$H2,$zero,8
+
+	stvx_u		$H2l,r8,r3		# save H^2
+	li		r8,0x70
+	stvx_u		$H2,r9,r3
+	li		r9,0x80
+	stvx_u		$H2h,r10,r3
+	li		r10,0x90
+___
+{
+my ($t4,$t5,$t6) = ($Hl,$H,$Hh);
+$code.=<<___;
+	vpmsumd		$Xl,$IN,$H2l		# H.lo·H^2.lo
+	 vpmsumd	$Xl1,$IN1,$H2l		# H^2.lo·H^2.lo
+	vpmsumd		$Xm,$IN,$H2		# H.hi·H^2.lo+H.lo·H^2.hi
+	 vpmsumd	$Xm1,$IN1,$H2		# H^2.hi·H^2.lo+H^2.lo·H^2.hi
+	vpmsumd		$Xh,$IN,$H2h		# H.hi·H^2.hi
+	 vpmsumd	$Xh1,$IN1,$H2h		# H^2.hi·H^2.hi
+
+	vpmsumd		$t2,$Xl,$xC2		# 1st reduction phase
+	 vpmsumd	$t6,$Xl1,$xC2		# 1st reduction phase
+
+	vsldoi		$t0,$Xm,$zero,8
+	vsldoi		$t1,$zero,$Xm,8
+	 vsldoi		$t4,$Xm1,$zero,8
+	 vsldoi		$t5,$zero,$Xm1,8
+	vxor		$Xl,$Xl,$t0
+	vxor		$Xh,$Xh,$t1
+	 vxor		$Xl1,$Xl1,$t4
+	 vxor		$Xh1,$Xh1,$t5
+
+	vsldoi		$Xl,$Xl,$Xl,8
+	 vsldoi		$Xl1,$Xl1,$Xl1,8
+	vxor		$Xl,$Xl,$t2
+	 vxor		$Xl1,$Xl1,$t6
+
+	vsldoi		$t1,$Xl,$Xl,8		# 2nd reduction phase
+	 vsldoi		$t5,$Xl1,$Xl1,8		# 2nd reduction phase
+	vpmsumd		$Xl,$Xl,$xC2
+	 vpmsumd	$Xl1,$Xl1,$xC2
+	vxor		$t1,$t1,$Xh
+	 vxor		$t5,$t5,$Xh1
+	vxor		$Xl,$Xl,$t1
+	 vxor		$Xl1,$Xl1,$t5
+
+	vsldoi		$H,$Xl,$Xl,8
+	 vsldoi		$H2,$Xl1,$Xl1,8
+	vsldoi		$Hl,$zero,$H,8
+	vsldoi		$Hh,$H,$zero,8
+	 vsldoi		$H2l,$zero,$H2,8
+	 vsldoi		$H2h,$H2,$zero,8
+
+	stvx_u		$Hl,r8,r3		# save H^3
+	li		r8,0xa0
+	stvx_u		$H,r9,r3
+	li		r9,0xb0
+	stvx_u		$Hh,r10,r3
+	li		r10,0xc0
+	 stvx_u		$H2l,r8,r3		# save H^4
+	 stvx_u		$H2,r9,r3
+	 stvx_u		$H2h,r10,r3
+
+	mtspr		256,$vrsave
+	blr
+	.long		0
+	.byte		0,12,0x14,0,0,0,2,0
+	.long		0
+.size	.gcm_init_p8,.-.gcm_init_p8
+___
+}
+$code.=<<___;
+.globl	.gcm_gmult_p8
+.align	5
+.gcm_gmult_p8:
+	lis		r0,0xfff8
+	li		r8,0x10
+	mfspr		$vrsave,256
+	li		r9,0x20
+	mtspr		256,r0
+	li		r10,0x30
+	lvx_u		$IN,0,$Xip		# load Xi
+
+	lvx_u		$Hl,r8,$Htbl		# load pre-computed table
+	 le?lvsl	$lemask,r0,r0
+	lvx_u		$H, r9,$Htbl
+	 le?vspltisb	$t0,0x07
+	lvx_u		$Hh,r10,$Htbl
+	 le?vxor	$lemask,$lemask,$t0
+	lvx_u		$xC2,0,$Htbl
+	 le?vperm	$IN,$IN,$IN,$lemask
+	vxor		$zero,$zero,$zero
+
+	vpmsumd		$Xl,$IN,$Hl		# H.lo·Xi.lo
+	vpmsumd		$Xm,$IN,$H		# H.hi·Xi.lo+H.lo·Xi.hi
+	vpmsumd		$Xh,$IN,$Hh		# H.hi·Xi.hi
+
+	vpmsumd		$t2,$Xl,$xC2		# 1st reduction phase
+
+	vsldoi		$t0,$Xm,$zero,8
+	vsldoi		$t1,$zero,$Xm,8
+	vxor		$Xl,$Xl,$t0
+	vxor		$Xh,$Xh,$t1
+
+	vsldoi		$Xl,$Xl,$Xl,8
+	vxor		$Xl,$Xl,$t2
+
+	vsldoi		$t1,$Xl,$Xl,8		# 2nd reduction phase
+	vpmsumd		$Xl,$Xl,$xC2
+	vxor		$t1,$t1,$Xh
+	vxor		$Xl,$Xl,$t1
+
+	le?vperm	$Xl,$Xl,$Xl,$lemask
+	stvx_u		$Xl,0,$Xip		# write out Xi
+
+	mtspr		256,$vrsave
+	blr
+	.long		0
+	.byte		0,12,0x14,0,0,0,2,0
+	.long		0
+.size	.gcm_gmult_p8,.-.gcm_gmult_p8
+
+.globl	.gcm_ghash_p8
+.align	5
+.gcm_ghash_p8:
+	li		r0,-4096
+	li		r8,0x10
+	mfspr		$vrsave,256
+	li		r9,0x20
+	mtspr		256,r0
+	li		r10,0x30
+	lvx_u		$Xl,0,$Xip		# load Xi
+
+	lvx_u		$Hl,r8,$Htbl		# load pre-computed table
+	li		r8,0x40
+	 le?lvsl	$lemask,r0,r0
+	lvx_u		$H, r9,$Htbl
+	li		r9,0x50
+	 le?vspltisb	$t0,0x07
+	lvx_u		$Hh,r10,$Htbl
+	li		r10,0x60
+	 le?vxor	$lemask,$lemask,$t0
+	lvx_u		$xC2,0,$Htbl
+	 le?vperm	$Xl,$Xl,$Xl,$lemask
+	vxor		$zero,$zero,$zero
+
+	${UCMP}i	$len,64
+	bge		Lgcm_ghash_p8_4x
+
+	lvx_u		$IN,0,$inp
+	addi		$inp,$inp,16
+	subic.		$len,$len,16
+	 le?vperm	$IN,$IN,$IN,$lemask
+	vxor		$IN,$IN,$Xl
+	beq		Lshort
+
+	lvx_u		$H2l,r8,$Htbl		# load H^2
+	li		r8,16
+	lvx_u		$H2, r9,$Htbl
+	add		r9,$inp,$len		# end of input
+	lvx_u		$H2h,r10,$Htbl
+	be?b		Loop_2x
+
+.align	5
+Loop_2x:
+	lvx_u		$IN1,0,$inp
+	le?vperm	$IN1,$IN1,$IN1,$lemask
+
+	 subic		$len,$len,32
+	vpmsumd		$Xl,$IN,$H2l		# H^2.lo·Xi.lo
+	 vpmsumd	$Xl1,$IN1,$Hl		# H.lo·Xi+1.lo
+	 subfe		r0,r0,r0		# borrow?-1:0
+	vpmsumd		$Xm,$IN,$H2		# H^2.hi·Xi.lo+H^2.lo·Xi.hi
+	 vpmsumd	$Xm1,$IN1,$H		# H.hi·Xi+1.lo+H.lo·Xi+1.hi
+	 and		r0,r0,$len
+	vpmsumd		$Xh,$IN,$H2h		# H^2.hi·Xi.hi
+	 vpmsumd	$Xh1,$IN1,$Hh		# H.hi·Xi+1.hi
+	 add		$inp,$inp,r0
+
+	vxor		$Xl,$Xl,$Xl1
+	vxor		$Xm,$Xm,$Xm1
+
+	vpmsumd		$t2,$Xl,$xC2		# 1st reduction phase
+
+	vsldoi		$t0,$Xm,$zero,8
+	vsldoi		$t1,$zero,$Xm,8
+	 vxor		$Xh,$Xh,$Xh1
+	vxor		$Xl,$Xl,$t0
+	vxor		$Xh,$Xh,$t1
+
+	vsldoi		$Xl,$Xl,$Xl,8
+	vxor		$Xl,$Xl,$t2
+	 lvx_u		$IN,r8,$inp
+	 addi		$inp,$inp,32
+
+	vsldoi		$t1,$Xl,$Xl,8		# 2nd reduction phase
+	vpmsumd		$Xl,$Xl,$xC2
+	 le?vperm	$IN,$IN,$IN,$lemask
+	vxor		$t1,$t1,$Xh
+	vxor		$IN,$IN,$t1
+	vxor		$IN,$IN,$Xl
+	$UCMP		r9,$inp
+	bgt		Loop_2x			# done yet?
+
+	cmplwi		$len,0
+	bne		Leven
+
+Lshort:
+	vpmsumd		$Xl,$IN,$Hl		# H.lo·Xi.lo
+	vpmsumd		$Xm,$IN,$H		# H.hi·Xi.lo+H.lo·Xi.hi
+	vpmsumd		$Xh,$IN,$Hh		# H.hi·Xi.hi
+
+	vpmsumd		$t2,$Xl,$xC2		# 1st reduction phase
+
+	vsldoi		$t0,$Xm,$zero,8
+	vsldoi		$t1,$zero,$Xm,8
+	vxor		$Xl,$Xl,$t0
+	vxor		$Xh,$Xh,$t1
+
+	vsldoi		$Xl,$Xl,$Xl,8
+	vxor		$Xl,$Xl,$t2
+
+	vsldoi		$t1,$Xl,$Xl,8		# 2nd reduction phase
+	vpmsumd		$Xl,$Xl,$xC2
+	vxor		$t1,$t1,$Xh
+
+Leven:
+	vxor		$Xl,$Xl,$t1
+	le?vperm	$Xl,$Xl,$Xl,$lemask
+	stvx_u		$Xl,0,$Xip		# write out Xi
+
+	mtspr		256,$vrsave
+	blr
+	.long		0
+	.byte		0,12,0x14,0,0,0,4,0
+	.long		0
+___
+{
+my ($Xl3,$Xm2,$IN2,$H3l,$H3,$H3h,
+    $Xh3,$Xm3,$IN3,$H4l,$H4,$H4h) = map("v$_",(20..31));
+my $IN0=$IN;
+my ($H21l,$H21h,$loperm,$hiperm) = ($Hl,$Hh,$H2l,$H2h);
+
+$code.=<<___;
+.align	5
+.gcm_ghash_p8_4x:
+Lgcm_ghash_p8_4x:
+	$STU		$sp,-$FRAME($sp)
+	li		r10,`15+6*$SIZE_T`
+	li		r11,`31+6*$SIZE_T`
+	stvx		v20,r10,$sp
+	addi		r10,r10,32
+	stvx		v21,r11,$sp
+	addi		r11,r11,32
+	stvx		v22,r10,$sp
+	addi		r10,r10,32
+	stvx		v23,r11,$sp
+	addi		r11,r11,32
+	stvx		v24,r10,$sp
+	addi		r10,r10,32
+	stvx		v25,r11,$sp
+	addi		r11,r11,32
+	stvx		v26,r10,$sp
+	addi		r10,r10,32
+	stvx		v27,r11,$sp
+	addi		r11,r11,32
+	stvx		v28,r10,$sp
+	addi		r10,r10,32
+	stvx		v29,r11,$sp
+	addi		r11,r11,32
+	stvx		v30,r10,$sp
+	li		r10,0x60
+	stvx		v31,r11,$sp
+	li		r0,-1
+	stw		$vrsave,`$FRAME-4`($sp)	# save vrsave
+	mtspr		256,r0			# preserve all AltiVec registers
+
+	lvsl		$t0,0,r8		# 0x0001..0e0f
+	#lvx_u		$H2l,r8,$Htbl		# load H^2
+	li		r8,0x70
+	lvx_u		$H2, r9,$Htbl
+	li		r9,0x80
+	vspltisb	$t1,8			# 0x0808..0808
+	#lvx_u		$H2h,r10,$Htbl
+	li		r10,0x90
+	lvx_u		$H3l,r8,$Htbl		# load H^3
+	li		r8,0xa0
+	lvx_u		$H3, r9,$Htbl
+	li		r9,0xb0
+	lvx_u		$H3h,r10,$Htbl
+	li		r10,0xc0
+	lvx_u		$H4l,r8,$Htbl		# load H^4
+	li		r8,0x10
+	lvx_u		$H4, r9,$Htbl
+	li		r9,0x20
+	lvx_u		$H4h,r10,$Htbl
+	li		r10,0x30
+
+	vsldoi		$t2,$zero,$t1,8		# 0x0000..0808
+	vaddubm		$hiperm,$t0,$t2		# 0x0001..1617
+	vaddubm		$loperm,$t1,$hiperm	# 0x0809..1e1f
+
+	$SHRI		$len,$len,4		# this allows to use sign bit
+						# as carry
+	lvx_u		$IN0,0,$inp		# load input
+	lvx_u		$IN1,r8,$inp
+	subic.		$len,$len,8
+	lvx_u		$IN2,r9,$inp
+	lvx_u		$IN3,r10,$inp
+	addi		$inp,$inp,0x40
+	le?vperm	$IN0,$IN0,$IN0,$lemask
+	le?vperm	$IN1,$IN1,$IN1,$lemask
+	le?vperm	$IN2,$IN2,$IN2,$lemask
+	le?vperm	$IN3,$IN3,$IN3,$lemask
+
+	vxor		$Xh,$IN0,$Xl
+
+	 vpmsumd	$Xl1,$IN1,$H3l
+	 vpmsumd	$Xm1,$IN1,$H3
+	 vpmsumd	$Xh1,$IN1,$H3h
+
+	 vperm		$H21l,$H2,$H,$hiperm
+	 vperm		$t0,$IN2,$IN3,$loperm
+	 vperm		$H21h,$H2,$H,$loperm
+	 vperm		$t1,$IN2,$IN3,$hiperm
+	 vpmsumd	$Xm2,$IN2,$H2		# H^2.lo·Xi+2.hi+H^2.hi·Xi+2.lo
+	 vpmsumd	$Xl3,$t0,$H21l		# H^2.lo·Xi+2.lo+H.lo·Xi+3.lo
+	 vpmsumd	$Xm3,$IN3,$H		# H.hi·Xi+3.lo  +H.lo·Xi+3.hi
+	 vpmsumd	$Xh3,$t1,$H21h		# H^2.hi·Xi+2.hi+H.hi·Xi+3.hi
+
+	 vxor		$Xm2,$Xm2,$Xm1
+	 vxor		$Xl3,$Xl3,$Xl1
+	 vxor		$Xm3,$Xm3,$Xm2
+	 vxor		$Xh3,$Xh3,$Xh1
+
+	blt		Ltail_4x
+
+Loop_4x:
+	lvx_u		$IN0,0,$inp
+	lvx_u		$IN1,r8,$inp
+	subic.		$len,$len,4
+	lvx_u		$IN2,r9,$inp
+	lvx_u		$IN3,r10,$inp
+	addi		$inp,$inp,0x40
+	le?vperm	$IN1,$IN1,$IN1,$lemask
+	le?vperm	$IN2,$IN2,$IN2,$lemask
+	le?vperm	$IN3,$IN3,$IN3,$lemask
+	le?vperm	$IN0,$IN0,$IN0,$lemask
+
+	vpmsumd		$Xl,$Xh,$H4l		# H^4.lo·Xi.lo
+	vpmsumd		$Xm,$Xh,$H4		# H^4.hi·Xi.lo+H^4.lo·Xi.hi
+	vpmsumd		$Xh,$Xh,$H4h		# H^4.hi·Xi.hi
+	 vpmsumd	$Xl1,$IN1,$H3l
+	 vpmsumd	$Xm1,$IN1,$H3
+	 vpmsumd	$Xh1,$IN1,$H3h
+
+	vxor		$Xl,$Xl,$Xl3
+	vxor		$Xm,$Xm,$Xm3
+	vxor		$Xh,$Xh,$Xh3
+	 vperm		$t0,$IN2,$IN3,$loperm
+	 vperm		$t1,$IN2,$IN3,$hiperm
+
+	vpmsumd		$t2,$Xl,$xC2		# 1st reduction phase
+	 vpmsumd	$Xl3,$t0,$H21l		# H.lo·Xi+3.lo  +H^2.lo·Xi+2.lo
+	 vpmsumd	$Xh3,$t1,$H21h		# H.hi·Xi+3.hi  +H^2.hi·Xi+2.hi
+
+	vsldoi		$t0,$Xm,$zero,8
+	vsldoi		$t1,$zero,$Xm,8
+	vxor		$Xl,$Xl,$t0
+	vxor		$Xh,$Xh,$t1
+
+	vsldoi		$Xl,$Xl,$Xl,8
+	vxor		$Xl,$Xl,$t2
+
+	vsldoi		$t1,$Xl,$Xl,8		# 2nd reduction phase
+	 vpmsumd	$Xm2,$IN2,$H2		# H^2.hi·Xi+2.lo+H^2.lo·Xi+2.hi
+	 vpmsumd	$Xm3,$IN3,$H		# H.hi·Xi+3.lo  +H.lo·Xi+3.hi
+	vpmsumd		$Xl,$Xl,$xC2
+
+	 vxor		$Xl3,$Xl3,$Xl1
+	 vxor		$Xh3,$Xh3,$Xh1
+	vxor		$Xh,$Xh,$IN0
+	 vxor		$Xm2,$Xm2,$Xm1
+	vxor		$Xh,$Xh,$t1
+	 vxor		$Xm3,$Xm3,$Xm2
+	vxor		$Xh,$Xh,$Xl
+	bge		Loop_4x
+
+Ltail_4x:
+	vpmsumd		$Xl,$Xh,$H4l		# H^4.lo·Xi.lo
+	vpmsumd		$Xm,$Xh,$H4		# H^4.hi·Xi.lo+H^4.lo·Xi.hi
+	vpmsumd		$Xh,$Xh,$H4h		# H^4.hi·Xi.hi
+
+	vxor		$Xl,$Xl,$Xl3
+	vxor		$Xm,$Xm,$Xm3
+
+	vpmsumd		$t2,$Xl,$xC2		# 1st reduction phase
+
+	vsldoi		$t0,$Xm,$zero,8
+	vsldoi		$t1,$zero,$Xm,8
+	 vxor		$Xh,$Xh,$Xh3
+	vxor		$Xl,$Xl,$t0
+	vxor		$Xh,$Xh,$t1
+
+	vsldoi		$Xl,$Xl,$Xl,8
+	vxor		$Xl,$Xl,$t2
+
+	vsldoi		$t1,$Xl,$Xl,8		# 2nd reduction phase
+	vpmsumd		$Xl,$Xl,$xC2
+	vxor		$t1,$t1,$Xh
+	vxor		$Xl,$Xl,$t1
+
+	addic.		$len,$len,4
+	beq		Ldone_4x
+
+	lvx_u		$IN0,0,$inp
+	${UCMP}i	$len,2
+	li		$len,-4
+	blt		Lone
+	lvx_u		$IN1,r8,$inp
+	beq		Ltwo
+
+Lthree:
+	lvx_u		$IN2,r9,$inp
+	le?vperm	$IN0,$IN0,$IN0,$lemask
+	le?vperm	$IN1,$IN1,$IN1,$lemask
+	le?vperm	$IN2,$IN2,$IN2,$lemask
+
+	vxor		$Xh,$IN0,$Xl
+	vmr		$H4l,$H3l
+	vmr		$H4, $H3
+	vmr		$H4h,$H3h
+
+	vperm		$t0,$IN1,$IN2,$loperm
+	vperm		$t1,$IN1,$IN2,$hiperm
+	vpmsumd		$Xm2,$IN1,$H2		# H^2.lo·Xi+1.hi+H^2.hi·Xi+1.lo
+	vpmsumd		$Xm3,$IN2,$H		# H.hi·Xi+2.lo  +H.lo·Xi+2.hi
+	vpmsumd		$Xl3,$t0,$H21l		# H^2.lo·Xi+1.lo+H.lo·Xi+2.lo
+	vpmsumd		$Xh3,$t1,$H21h		# H^2.hi·Xi+1.hi+H.hi·Xi+2.hi
+
+	vxor		$Xm3,$Xm3,$Xm2
+	b		Ltail_4x
+
+.align	4
+Ltwo:
+	le?vperm	$IN0,$IN0,$IN0,$lemask
+	le?vperm	$IN1,$IN1,$IN1,$lemask
+
+	vxor		$Xh,$IN0,$Xl
+	vperm		$t0,$zero,$IN1,$loperm
+	vperm		$t1,$zero,$IN1,$hiperm
+
+	vsldoi		$H4l,$zero,$H2,8
+	vmr		$H4, $H2
+	vsldoi		$H4h,$H2,$zero,8
+
+	vpmsumd		$Xl3,$t0, $H21l		# H.lo·Xi+1.lo
+	vpmsumd		$Xm3,$IN1,$H		# H.hi·Xi+1.lo+H.lo·Xi+2.hi
+	vpmsumd		$Xh3,$t1, $H21h		# H.hi·Xi+1.hi
+
+	b		Ltail_4x
+
+.align	4
+Lone:
+	le?vperm	$IN0,$IN0,$IN0,$lemask
+
+	vsldoi		$H4l,$zero,$H,8
+	vmr		$H4, $H
+	vsldoi		$H4h,$H,$zero,8
+
+	vxor		$Xh,$IN0,$Xl
+	vxor		$Xl3,$Xl3,$Xl3
+	vxor		$Xm3,$Xm3,$Xm3
+	vxor		$Xh3,$Xh3,$Xh3
+
+	b		Ltail_4x
+
+Ldone_4x:
+	le?vperm	$Xl,$Xl,$Xl,$lemask
+	stvx_u		$Xl,0,$Xip		# write out Xi
+
+	li		r10,`15+6*$SIZE_T`
+	li		r11,`31+6*$SIZE_T`
+	mtspr		256,$vrsave
+	lvx		v20,r10,$sp
+	addi		r10,r10,32
+	lvx		v21,r11,$sp
+	addi		r11,r11,32
+	lvx		v22,r10,$sp
+	addi		r10,r10,32
+	lvx		v23,r11,$sp
+	addi		r11,r11,32
+	lvx		v24,r10,$sp
+	addi		r10,r10,32
+	lvx		v25,r11,$sp
+	addi		r11,r11,32
+	lvx		v26,r10,$sp
+	addi		r10,r10,32
+	lvx		v27,r11,$sp
+	addi		r11,r11,32
+	lvx		v28,r10,$sp
+	addi		r10,r10,32
+	lvx		v29,r11,$sp
+	addi		r11,r11,32
+	lvx		v30,r10,$sp
+	lvx		v31,r11,$sp
+	addi		$sp,$sp,$FRAME
+	blr
+	.long		0
+	.byte		0,12,0x04,0,0x80,0,4,0
+	.long		0
+___
+}
+$code.=<<___;
+.size	.gcm_ghash_p8,.-.gcm_ghash_p8
+
+.asciz  "GHASH for PowerISA 2.07, CRYPTOGAMS by <appro\@openssl.org>"
+.align  2
+___
+
+foreach (split("\n",$code)) {
+	s/\`([^\`]*)\`/eval $1/geo;
+
+	if ($flavour =~ /le$/o) {	# little-endian
+	    s/le\?//o		or
+	    s/be\?/#be#/o;
+	} else {
+	    s/le\?/#le#/o	or
+	    s/be\?//o;
+	}
+	print $_,"\n";
+}
+
+close STDOUT; # enforce flush
diff --git a/cbits/asm/ppc-xlate.pl b/cbits/asm/ppc-xlate.pl
new file mode 100644
--- /dev/null
+++ b/cbits/asm/ppc-xlate.pl
@@ -0,0 +1,352 @@
+#!/usr/bin/env perl
+
+# PowerPC assembler distiller by \@dot-asm.
+
+################################################################
+# Recognized "flavour"-s are:
+#
+# linux{32|64}[le]  GNU assembler and ELF symbol decorations,
+#                   with little-endian option
+# linux64v2         GNU asssembler and big-endian instantiation
+#                   of latest ELF specification
+# aix{32|64}        AIX assembler and symbol decorations
+# osx{32|64}        Mac OS X assembler and symbol decoratons
+
+my $flavour = shift;
+my $output = shift;
+open STDOUT,">$output" || die "can't open $output: $!";
+
+my %GLOBALS;
+my %TYPES;
+my $dotinlocallabels=($flavour=~/linux/)?1:0;
+
+################################################################
+# directives which need special treatment on different platforms
+################################################################
+my $type = sub {
+    my ($dir,$name,$type) = @_;
+
+    $TYPES{$name} = $type;
+    if ($flavour =~ /linux/) {
+	$name =~ s|^\.||;
+	".type	$name,$type";
+    } else {
+	"";
+    }
+};
+my $globl = sub {
+    my $junk = shift;
+    my $name = shift;
+    my $global = \$GLOBALS{$name};
+    my $type = \$TYPES{$name};
+    my $ret;
+
+    $name =~ s|^\.||;
+
+    SWITCH: for ($flavour) {
+	/aix/		&& do { if (!$$type) {
+				    $$type = "\@function";
+				}
+				if ($$type =~ /function/) {
+				    $name = ".$name";
+				}
+				last;
+			      };
+	/osx/		&& do { $name = "_$name";
+				last;
+			      };
+	/linux.*(32|64(le|v2))/
+			&& do {	$ret .= ".globl	$name";
+				if (!$$type) {
+				    $ret .= "\n.type	$name,\@function";
+				    $$type = "\@function";
+				}
+				last;
+			      };
+	/linux.*64/	&& do {	$ret .= ".globl	$name";
+				if (!$$type) {
+				    $ret .= "\n.type	$name,\@function";
+				    $$type = "\@function";
+				}
+				if ($$type =~ /function/) {
+				    $ret .= "\n.section	\".opd\",\"aw\"";
+				    $ret .= "\n.align	3";
+				    $ret .= "\n$name:";
+				    $ret .= "\n.quad	.$name,.TOC.\@tocbase,0";
+				    $ret .= "\n.previous";
+				    $name = ".$name";
+				}
+				last;
+			      };
+    }
+
+    $ret = ".globl	$name" if (!$ret);
+    $$global = $name;
+    $ret;
+};
+my $text = sub {
+    my $ret = ($flavour =~ /aix/) ? ".csect\t.text[PR],7" : ".text";
+    $ret = ".abiversion	2\n".$ret	if ($flavour =~ /linux.*64(le|v2)/);
+    $ret;
+};
+my $machine = sub {
+    my $junk = shift;
+    my $arch = shift;
+    if ($flavour =~ /osx/)
+    {	$arch =~ s/\"//g;
+	$arch = ($flavour=~/64/) ? "ppc970-64" : "ppc970" if ($arch eq "any");
+    }
+    ".machine	$arch";
+};
+my $size = sub {
+    if ($flavour =~ /linux/)
+    {	shift;
+	my $name = shift;
+	my $real = $GLOBALS{$name} ? \$GLOBALS{$name} : \$name;
+	my $ret  = ".size	$$real,.-$$real";
+	$name =~ s|^\.||;
+	if ($$real ne $name) {
+	    $ret .= "\n.size	$name,.-$$real";
+	}
+	$ret;
+    }
+    else
+    {	"";	}
+};
+my $asciz = sub {
+    shift;
+    my $line = join(",",@_);
+    if ($line =~ /^"(.*)"$/)
+    {	".byte	" . join(",",unpack("C*",$1),0) . "\n.align	2";	}
+    else
+    {	"";	}
+};
+my $quad = sub {
+    shift;
+    my @ret;
+    my ($hi,$lo);
+    for (@_) {
+	if (/^0x([0-9a-f]*?)([0-9a-f]{1,8})$/io)
+	{  $hi=$1?"0x$1":"0"; $lo="0x$2";  }
+	elsif (/^([0-9]+)$/o)
+	{  $hi=$1>>32; $lo=$1&0xffffffff;  } # error-prone with 32-bit perl
+	else
+	{  $hi=undef; $lo=$_; }
+
+	if (defined($hi))
+	{  push(@ret,$flavour=~/le$/o?".long\t$lo,$hi":".long\t$hi,$lo");  }
+	else
+	{  push(@ret,".quad	$lo");  }
+    }
+    join("\n",@ret);
+};
+
+################################################################
+# simplified mnemonics not handled by at least one assembler
+################################################################
+my $cmplw = sub {
+    my $f = shift;
+    my $cr = 0; $cr = shift if ($#_>1);
+    # Some out-of-date 32-bit GNU assembler just can't handle cmplw...
+    ($flavour =~ /linux.*32/) ?
+	"	.long	".sprintf "0x%x",31<<26|$cr<<23|$_[0]<<16|$_[1]<<11|64 :
+	"	cmplw	".join(',',$cr,@_);
+};
+my $bdnz = sub {
+    my $f = shift;
+    my $bo = $f=~/[\+\-]/ ? 16+9 : 16;	# optional "to be taken" hint
+    "	bc	$bo,0,".shift;
+} if ($flavour!~/linux/);
+my $bltlr = sub {
+    my $f = shift;
+    my $bo = $f=~/\-/ ? 12+2 : 12;	# optional "not to be taken" hint
+    ($flavour =~ /linux/) ?		# GNU as doesn't allow most recent hints
+	"	.long	".sprintf "0x%x",19<<26|$bo<<21|16<<1 :
+	"	bclr	$bo,0";
+};
+my $bnelr = sub {
+    my $f = shift;
+    my $bo = $f=~/\-/ ? 4+2 : 4;	# optional "not to be taken" hint
+    ($flavour =~ /linux/) ?		# GNU as doesn't allow most recent hints
+	"	.long	".sprintf "0x%x",19<<26|$bo<<21|2<<16|16<<1 :
+	"	bclr	$bo,2";
+};
+my $beqlr = sub {
+    my $f = shift;
+    my $bo = $f=~/-/ ? 12+2 : 12;	# optional "not to be taken" hint
+    ($flavour =~ /linux/) ?		# GNU as doesn't allow most recent hints
+	"	.long	".sprintf "0x%X",19<<26|$bo<<21|2<<16|16<<1 :
+	"	bclr	$bo,2";
+};
+# GNU assembler can't handle extrdi rA,rS,16,48, or when sum of last two
+# arguments is 64, with "operand out of range" error.
+my $extrdi = sub {
+    my ($f,$ra,$rs,$n,$b) = @_;
+    $b = ($b+$n)&63; $n = 64-$n;
+    "	rldicl	$ra,$rs,$b,$n";
+};
+my $vmr = sub {
+    my ($f,$vx,$vy) = @_;
+    "	vor	$vx,$vy,$vy";
+};
+
+# Some ABIs specify vrsave, special-purpose register #256, as reserved
+# for system use.
+my $no_vrsave = ($flavour =~ /aix|linux64(le|v2)/);
+my $mtspr = sub {
+    my ($f,$idx,$ra) = @_;
+    if ($idx == 256 && $no_vrsave) {
+	"	or	$ra,$ra,$ra";
+    } else {
+	"	mtspr	$idx,$ra";
+    }
+};
+my $mfspr = sub {
+    my ($f,$rd,$idx) = @_;
+    if ($idx == 256 && $no_vrsave) {
+	"	li	$rd,-1";
+    } else {
+	"	mfspr	$rd,$idx";
+    }
+};
+
+# PowerISA 2.06 stuff
+sub vsxmem_op {
+    my ($f, $vrt, $ra, $rb, $op) = @_;
+    "	.long	".sprintf "0x%X",(31<<26)|($vrt<<21)|($ra<<16)|($rb<<11)|($op*2+1);
+}
+# made-up unaligned memory reference AltiVec/VMX instructions
+my $lvx_u	= sub {	vsxmem_op(@_, 844); };	# lxvd2x
+my $stvx_u	= sub {	vsxmem_op(@_, 972); };	# stxvd2x
+my $lvdx_u	= sub {	vsxmem_op(@_, 588); };	# lxsdx
+my $stvdx_u	= sub {	vsxmem_op(@_, 716); };	# stxsdx
+my $lvx_4w	= sub { vsxmem_op(@_, 780); };	# lxvw4x
+my $stvx_4w	= sub { vsxmem_op(@_, 908); };	# stxvw4x
+my $lvx_splt	= sub { vsxmem_op(@_, 332); };	# lxvdsx
+# VSX instruction[s] masqueraded as made-up AltiVec/VMX
+my $vpermdi	= sub {				# xxpermdi
+    my ($f, $vrt, $vra, $vrb, $dm) = @_;
+    $dm = oct($dm) if ($dm =~ /^0/);
+    "	.long	".sprintf "0x%X",(60<<26)|($vrt<<21)|($vra<<16)|($vrb<<11)|($dm<<8)|(10<<3)|7;
+};
+
+# PowerISA 2.07 stuff
+sub vcrypto_op {
+    my ($f, $vrt, $vra, $vrb, $op) = @_;
+    "	.long	".sprintf "0x%X",(4<<26)|($vrt<<21)|($vra<<16)|($vrb<<11)|$op;
+}
+sub vfour {
+    my ($f, $vrt, $vra, $vrb, $vrc, $op) = @_;
+    "	.long	".sprintf "0x%X",(4<<26)|($vrt<<21)|($vra<<16)|($vrb<<11)|($vrc<<6)|$op;
+};
+my $vcipher	= sub { vcrypto_op(@_, 1288); };
+my $vcipherlast	= sub { vcrypto_op(@_, 1289); };
+my $vncipher	= sub { vcrypto_op(@_, 1352); };
+my $vncipherlast= sub { vcrypto_op(@_, 1353); };
+my $vsbox	= sub { vcrypto_op(@_, 0, 1480); };
+my $vshasigmad	= sub { my ($st,$six)=splice(@_,-2); vcrypto_op(@_, $st<<4|$six, 1730); };
+my $vshasigmaw	= sub { my ($st,$six)=splice(@_,-2); vcrypto_op(@_, $st<<4|$six, 1666); };
+my $vpmsumb	= sub { vcrypto_op(@_, 1032); };
+my $vpmsumd	= sub { vcrypto_op(@_, 1224); };
+my $vpmsubh	= sub { vcrypto_op(@_, 1096); };
+my $vpmsumw	= sub { vcrypto_op(@_, 1160); };
+# These are not really crypto, but vcrypto_op template works
+my $vaddudm	= sub { vcrypto_op(@_, 192);  };
+my $vadduqm	= sub { vcrypto_op(@_, 256);  };
+my $vmuleuw	= sub { vcrypto_op(@_, 648);  };
+my $vmulouw	= sub { vcrypto_op(@_, 136);  };
+my $vrld	= sub { vcrypto_op(@_, 196);  };
+my $vsld	= sub { vcrypto_op(@_, 1476); };
+my $vsrd	= sub { vcrypto_op(@_, 1732); };
+my $vsubudm	= sub { vcrypto_op(@_, 1216); };
+my $vaddcuq	= sub { vcrypto_op(@_, 320);  };
+my $vaddeuqm	= sub { vfour(@_,60); };
+my $vaddecuq	= sub { vfour(@_,61); };
+my $vmrgew	= sub { vfour(@_,0,1932); };
+my $vmrgow	= sub { vfour(@_,0,1676); };
+
+my $mtsle	= sub {
+    my ($f, $arg) = @_;
+    "	.long	".sprintf "0x%X",(31<<26)|($arg<<21)|(147*2);
+};
+
+# VSX instructions masqueraded as AltiVec/VMX
+my $mtvrd	= sub {
+    my ($f, $vrt, $ra) = @_;
+    "	.long	".sprintf "0x%X",(31<<26)|($vrt<<21)|($ra<<16)|(179<<1)|1;
+};
+my $mtvrwz	= sub {
+    my ($f, $vrt, $ra) = @_;
+    "	.long	".sprintf "0x%X",(31<<26)|($vrt<<21)|($ra<<16)|(243<<1)|1;
+};
+my $lvwzx_u	= sub { vsxmem_op(@_, 12); };	# lxsiwzx
+my $stvwx_u	= sub { vsxmem_op(@_, 140); };	# stxsiwx
+
+# PowerISA 3.0 stuff
+my $maddhdu	= sub { vfour(@_,49); };
+my $maddld	= sub { vfour(@_,51); };
+my $darn = sub {
+    my ($f, $rt, $l) = @_;
+    "	.long	".sprintf "0x%X",(31<<26)|($rt<<21)|($l<<16)|(755<<1);
+};
+my $iseleq = sub {
+    my ($f, $rt, $ra, $rb) = @_;
+    "	.long	".sprintf "0x%X",(31<<26)|($rt<<21)|($ra<<16)|($rb<<11)|(2<<6)|30;
+};
+# VSX instruction[s] masqueraded as made-up AltiVec/VMX
+my $vspltib	= sub {				# xxspltib
+    my ($f, $vrt, $imm8) = @_;
+    $imm8 = oct($imm8) if ($imm8 =~ /^0/);
+    $imm8 &= 0xff;
+    "	.long	".sprintf "0x%X",(60<<26)|($vrt<<21)|($imm8<<11)|(360<<1)|1;
+};
+
+# PowerISA 3.0B stuff
+my $addex = sub {
+    my ($f, $rt, $ra, $rb, $cy) = @_;	# only cy==0 is specified in 3.0B
+    "	.long	".sprintf "0x%X",(31<<26)|($rt<<21)|($ra<<16)|($rb<<11)|($cy<<9)|(170<<1);
+};
+my $vmsumudm	= sub { vfour(@_,35); };
+
+while($line=<>) {
+
+    $line =~ s|[#!;].*$||;	# get rid of asm-style comments...
+    $line =~ s|/\*.*\*/||;	# ... and C-style comments...
+    $line =~ s|^\s+||;		# ... and skip white spaces in beginning...
+    $line =~ s|\s+$||;		# ... and at the end
+
+    {
+	$line =~ s|\.L(\w+)|L$1|g;	# common denominator for Locallabel
+	$line =~ s|\bL(\w+)|\.L$1|g	if ($dotinlocallabels);
+    }
+
+    {
+	$line =~ s|(^[\.\w]+)\:\s*||;
+	my $label = $1;
+	if ($label) {
+	    my $xlated = ($GLOBALS{$label} or $label);
+	    print "$xlated:";
+	    if ($flavour =~ /linux.*64(le|v2)/) {
+		if ($TYPES{$label} =~ /function/) {
+		    printf "\n.localentry	%s,0\n",$xlated;
+		}
+	    }
+	}
+    }
+
+    {
+	$line =~ s|^\s*(\.?)(\w+)([\.\+\-]?)\s*||;
+	my $c = $1; $c = "\t" if ($c eq "");
+	my $mnemonic = $2;
+	my $f = $3;
+	my $opcode = eval("\$$mnemonic");
+	$line =~ s/\b(c?[rf]|v|vs)([0-9]+)\b/$2/g if ($c ne "." and $flavour !~ /osx/);
+	if (ref($opcode) eq 'CODE') { $line = &$opcode($f,split(/,\s*/,$line)); }
+	elsif ($mnemonic)           { $line = $c.$mnemonic.$f."\t".$line; }
+    }
+
+    print $line if ($line);
+    print "\n";
+}
+
+close STDOUT;
diff --git a/cbits/bearssl/LICENSE b/cbits/bearssl/LICENSE
new file mode 100644
--- /dev/null
+++ b/cbits/bearssl/LICENSE
@@ -0,0 +1,21 @@
+Copyright (c) 2016 Thomas Pornin <pornin@bolet.org>
+
+Permission is hereby granted, free of charge, to any person obtaining 
+a copy of this software and associated documentation files (the
+"Software"), to deal in the Software without restriction, including
+without limitation the rights to use, copy, modify, merge, publish,
+distribute, sublicense, and/or sell copies of the Software, and to
+permit persons to whom the Software is furnished to do so, subject to
+the following conditions:
+
+The above copyright notice and this permission notice shall be 
+included in all copies or substantial portions of the Software.
+
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, 
+EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
+MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND 
+NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
+BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
+ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
+CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+SOFTWARE.
diff --git a/cbits/bearssl/README.md b/cbits/bearssl/README.md
new file mode 100644
--- /dev/null
+++ b/cbits/bearssl/README.md
@@ -0,0 +1,57 @@
+# BearSSL
+
+Constant-time AES and GHASH from [BearSSL](https://bearssl.org/), vendored
+here.  `VERSION` holds the upstream release these files came from, and
+`import.sh` fetches them again.
+
+## Why
+
+crypton's portable AES and GHASH are table-driven, and both index their
+tables with a secret: `cbits/aes/generic.c` with a byte of the state, in
+every round and in the key expansion, and `cbits/aes/gf.c` with a nibble of
+the GHASH accumulator.  Both are variable-time by construction, and the
+cache-timing attacks on that shape of code are old and well documented.
+
+The processors that have AES instructions do not run any of it -- crypton
+asks them first.  What runs it is everything else: ppc64le and s390x, where
+the instructions exist but crypton has no path to them; 32-bit ARM, where
+the same is true; riscv64 and loongarch64; and the boards that genuinely
+have no AES instructions at all, of which the Raspberry Pi 3 and 4 are by
+some distance the largest population.
+
+BearSSL's answer is bitslicing.  `aes_ct64` holds four blocks interleaved
+across eight 64-bit words and computes the S-box as boolean algebra, so
+there is no table and no address derived from a secret; `ghash_ctmul64`
+builds the GF(2^128) multiply out of shifts, masks and integer multiplies
+rather than a table of H.  Both are plain C99 and assume nothing beyond
+`uint64_t`.
+
+## What is here
+
+| file | from |
+| --- | --- |
+| `aes_ct64.c` | `src/symcipher/aes_ct64.c` |
+| `aes_ct64_enc.c` | `src/symcipher/aes_ct64_enc.c` |
+| `aes_ct64_dec.c` | `src/symcipher/aes_ct64_dec.c` |
+| `ghash_ctmul64.c` | `src/hash/ghash_ctmul64.c` |
+| `dec32le.c` | `src/codec/dec32le.c`, for `br_range_dec32le` |
+| `LICENSE` | `LICENSE.txt` -- MIT, (c) 2016 Thomas Pornin |
+
+`inner.h` is **crypton's, not upstream's**.  Each of those five opens with
+`#include "inner.h"`, and upstream's is some two thousand lines declaring
+the whole library; these five want six things from it.  So this one gives
+those six and nothing else, which is what lets the five stay byte for byte
+what upstream ships.  `import.sh` does not overwrite it.
+
+The byte-order helpers in it are written rather than copied, in their plain
+portable form without upstream's unaligned-access fast paths, so that no
+platform configuration comes with them.
+
+## Keeping it honest
+
+`cbits/tests/bearssl_diff.c` checks this against the implementation it
+replaces: the FIPS-197 vectors first, so that agreement means AES and not
+merely that both sides compute the same wrong thing, then a run of random
+keys and blocks, then GHASH against the 4-bit table.  Given any argument it
+corrupts three results on purpose and the comparison has to notice -- a
+differential test that cannot fail has said nothing.
diff --git a/cbits/bearssl/VERSION b/cbits/bearssl/VERSION
new file mode 100644
--- /dev/null
+++ b/cbits/bearssl/VERSION
@@ -0,0 +1,1 @@
+0.6
diff --git a/cbits/bearssl/aes_ct64.c b/cbits/bearssl/aes_ct64.c
new file mode 100644
--- /dev/null
+++ b/cbits/bearssl/aes_ct64.c
@@ -0,0 +1,398 @@
+/*
+ * Copyright (c) 2016 Thomas Pornin <pornin@bolet.org>
+ *
+ * Permission is hereby granted, free of charge, to any person obtaining 
+ * a copy of this software and associated documentation files (the
+ * "Software"), to deal in the Software without restriction, including
+ * without limitation the rights to use, copy, modify, merge, publish,
+ * distribute, sublicense, and/or sell copies of the Software, and to
+ * permit persons to whom the Software is furnished to do so, subject to
+ * the following conditions:
+ *
+ * The above copyright notice and this permission notice shall be 
+ * included in all copies or substantial portions of the Software.
+ *
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, 
+ * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
+ * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND 
+ * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
+ * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
+ * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
+ * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+ * SOFTWARE.
+ */
+
+#include "inner.h"
+
+/* see inner.h */
+void
+br_aes_ct64_bitslice_Sbox(uint64_t *q)
+{
+	/*
+	 * This S-box implementation is a straightforward translation of
+	 * the circuit described by Boyar and Peralta in "A new
+	 * combinational logic minimization technique with applications
+	 * to cryptology" (https://eprint.iacr.org/2009/191.pdf).
+	 *
+	 * Note that variables x* (input) and s* (output) are numbered
+	 * in "reverse" order (x0 is the high bit, x7 is the low bit).
+	 */
+
+	uint64_t x0, x1, x2, x3, x4, x5, x6, x7;
+	uint64_t y1, y2, y3, y4, y5, y6, y7, y8, y9;
+	uint64_t y10, y11, y12, y13, y14, y15, y16, y17, y18, y19;
+	uint64_t y20, y21;
+	uint64_t z0, z1, z2, z3, z4, z5, z6, z7, z8, z9;
+	uint64_t z10, z11, z12, z13, z14, z15, z16, z17;
+	uint64_t t0, t1, t2, t3, t4, t5, t6, t7, t8, t9;
+	uint64_t t10, t11, t12, t13, t14, t15, t16, t17, t18, t19;
+	uint64_t t20, t21, t22, t23, t24, t25, t26, t27, t28, t29;
+	uint64_t t30, t31, t32, t33, t34, t35, t36, t37, t38, t39;
+	uint64_t t40, t41, t42, t43, t44, t45, t46, t47, t48, t49;
+	uint64_t t50, t51, t52, t53, t54, t55, t56, t57, t58, t59;
+	uint64_t t60, t61, t62, t63, t64, t65, t66, t67;
+	uint64_t s0, s1, s2, s3, s4, s5, s6, s7;
+
+	x0 = q[7];
+	x1 = q[6];
+	x2 = q[5];
+	x3 = q[4];
+	x4 = q[3];
+	x5 = q[2];
+	x6 = q[1];
+	x7 = q[0];
+
+	/*
+	 * Top linear transformation.
+	 */
+	y14 = x3 ^ x5;
+	y13 = x0 ^ x6;
+	y9 = x0 ^ x3;
+	y8 = x0 ^ x5;
+	t0 = x1 ^ x2;
+	y1 = t0 ^ x7;
+	y4 = y1 ^ x3;
+	y12 = y13 ^ y14;
+	y2 = y1 ^ x0;
+	y5 = y1 ^ x6;
+	y3 = y5 ^ y8;
+	t1 = x4 ^ y12;
+	y15 = t1 ^ x5;
+	y20 = t1 ^ x1;
+	y6 = y15 ^ x7;
+	y10 = y15 ^ t0;
+	y11 = y20 ^ y9;
+	y7 = x7 ^ y11;
+	y17 = y10 ^ y11;
+	y19 = y10 ^ y8;
+	y16 = t0 ^ y11;
+	y21 = y13 ^ y16;
+	y18 = x0 ^ y16;
+
+	/*
+	 * Non-linear section.
+	 */
+	t2 = y12 & y15;
+	t3 = y3 & y6;
+	t4 = t3 ^ t2;
+	t5 = y4 & x7;
+	t6 = t5 ^ t2;
+	t7 = y13 & y16;
+	t8 = y5 & y1;
+	t9 = t8 ^ t7;
+	t10 = y2 & y7;
+	t11 = t10 ^ t7;
+	t12 = y9 & y11;
+	t13 = y14 & y17;
+	t14 = t13 ^ t12;
+	t15 = y8 & y10;
+	t16 = t15 ^ t12;
+	t17 = t4 ^ t14;
+	t18 = t6 ^ t16;
+	t19 = t9 ^ t14;
+	t20 = t11 ^ t16;
+	t21 = t17 ^ y20;
+	t22 = t18 ^ y19;
+	t23 = t19 ^ y21;
+	t24 = t20 ^ y18;
+
+	t25 = t21 ^ t22;
+	t26 = t21 & t23;
+	t27 = t24 ^ t26;
+	t28 = t25 & t27;
+	t29 = t28 ^ t22;
+	t30 = t23 ^ t24;
+	t31 = t22 ^ t26;
+	t32 = t31 & t30;
+	t33 = t32 ^ t24;
+	t34 = t23 ^ t33;
+	t35 = t27 ^ t33;
+	t36 = t24 & t35;
+	t37 = t36 ^ t34;
+	t38 = t27 ^ t36;
+	t39 = t29 & t38;
+	t40 = t25 ^ t39;
+
+	t41 = t40 ^ t37;
+	t42 = t29 ^ t33;
+	t43 = t29 ^ t40;
+	t44 = t33 ^ t37;
+	t45 = t42 ^ t41;
+	z0 = t44 & y15;
+	z1 = t37 & y6;
+	z2 = t33 & x7;
+	z3 = t43 & y16;
+	z4 = t40 & y1;
+	z5 = t29 & y7;
+	z6 = t42 & y11;
+	z7 = t45 & y17;
+	z8 = t41 & y10;
+	z9 = t44 & y12;
+	z10 = t37 & y3;
+	z11 = t33 & y4;
+	z12 = t43 & y13;
+	z13 = t40 & y5;
+	z14 = t29 & y2;
+	z15 = t42 & y9;
+	z16 = t45 & y14;
+	z17 = t41 & y8;
+
+	/*
+	 * Bottom linear transformation.
+	 */
+	t46 = z15 ^ z16;
+	t47 = z10 ^ z11;
+	t48 = z5 ^ z13;
+	t49 = z9 ^ z10;
+	t50 = z2 ^ z12;
+	t51 = z2 ^ z5;
+	t52 = z7 ^ z8;
+	t53 = z0 ^ z3;
+	t54 = z6 ^ z7;
+	t55 = z16 ^ z17;
+	t56 = z12 ^ t48;
+	t57 = t50 ^ t53;
+	t58 = z4 ^ t46;
+	t59 = z3 ^ t54;
+	t60 = t46 ^ t57;
+	t61 = z14 ^ t57;
+	t62 = t52 ^ t58;
+	t63 = t49 ^ t58;
+	t64 = z4 ^ t59;
+	t65 = t61 ^ t62;
+	t66 = z1 ^ t63;
+	s0 = t59 ^ t63;
+	s6 = t56 ^ ~t62;
+	s7 = t48 ^ ~t60;
+	t67 = t64 ^ t65;
+	s3 = t53 ^ t66;
+	s4 = t51 ^ t66;
+	s5 = t47 ^ t65;
+	s1 = t64 ^ ~s3;
+	s2 = t55 ^ ~t67;
+
+	q[7] = s0;
+	q[6] = s1;
+	q[5] = s2;
+	q[4] = s3;
+	q[3] = s4;
+	q[2] = s5;
+	q[1] = s6;
+	q[0] = s7;
+}
+
+/* see inner.h */
+void
+br_aes_ct64_ortho(uint64_t *q)
+{
+#define SWAPN(cl, ch, s, x, y)   do { \
+		uint64_t a, b; \
+		a = (x); \
+		b = (y); \
+		(x) = (a & (uint64_t)cl) | ((b & (uint64_t)cl) << (s)); \
+		(y) = ((a & (uint64_t)ch) >> (s)) | (b & (uint64_t)ch); \
+	} while (0)
+
+#define SWAP2(x, y)    SWAPN(0x5555555555555555, 0xAAAAAAAAAAAAAAAA,  1, x, y)
+#define SWAP4(x, y)    SWAPN(0x3333333333333333, 0xCCCCCCCCCCCCCCCC,  2, x, y)
+#define SWAP8(x, y)    SWAPN(0x0F0F0F0F0F0F0F0F, 0xF0F0F0F0F0F0F0F0,  4, x, y)
+
+	SWAP2(q[0], q[1]);
+	SWAP2(q[2], q[3]);
+	SWAP2(q[4], q[5]);
+	SWAP2(q[6], q[7]);
+
+	SWAP4(q[0], q[2]);
+	SWAP4(q[1], q[3]);
+	SWAP4(q[4], q[6]);
+	SWAP4(q[5], q[7]);
+
+	SWAP8(q[0], q[4]);
+	SWAP8(q[1], q[5]);
+	SWAP8(q[2], q[6]);
+	SWAP8(q[3], q[7]);
+}
+
+/* see inner.h */
+void
+br_aes_ct64_interleave_in(uint64_t *q0, uint64_t *q1, const uint32_t *w)
+{
+	uint64_t x0, x1, x2, x3;
+
+	x0 = w[0];
+	x1 = w[1];
+	x2 = w[2];
+	x3 = w[3];
+	x0 |= (x0 << 16);
+	x1 |= (x1 << 16);
+	x2 |= (x2 << 16);
+	x3 |= (x3 << 16);
+	x0 &= (uint64_t)0x0000FFFF0000FFFF;
+	x1 &= (uint64_t)0x0000FFFF0000FFFF;
+	x2 &= (uint64_t)0x0000FFFF0000FFFF;
+	x3 &= (uint64_t)0x0000FFFF0000FFFF;
+	x0 |= (x0 << 8);
+	x1 |= (x1 << 8);
+	x2 |= (x2 << 8);
+	x3 |= (x3 << 8);
+	x0 &= (uint64_t)0x00FF00FF00FF00FF;
+	x1 &= (uint64_t)0x00FF00FF00FF00FF;
+	x2 &= (uint64_t)0x00FF00FF00FF00FF;
+	x3 &= (uint64_t)0x00FF00FF00FF00FF;
+	*q0 = x0 | (x2 << 8);
+	*q1 = x1 | (x3 << 8);
+}
+
+/* see inner.h */
+void
+br_aes_ct64_interleave_out(uint32_t *w, uint64_t q0, uint64_t q1)
+{
+	uint64_t x0, x1, x2, x3;
+
+	x0 = q0 & (uint64_t)0x00FF00FF00FF00FF;
+	x1 = q1 & (uint64_t)0x00FF00FF00FF00FF;
+	x2 = (q0 >> 8) & (uint64_t)0x00FF00FF00FF00FF;
+	x3 = (q1 >> 8) & (uint64_t)0x00FF00FF00FF00FF;
+	x0 |= (x0 >> 8);
+	x1 |= (x1 >> 8);
+	x2 |= (x2 >> 8);
+	x3 |= (x3 >> 8);
+	x0 &= (uint64_t)0x0000FFFF0000FFFF;
+	x1 &= (uint64_t)0x0000FFFF0000FFFF;
+	x2 &= (uint64_t)0x0000FFFF0000FFFF;
+	x3 &= (uint64_t)0x0000FFFF0000FFFF;
+	w[0] = (uint32_t)x0 | (uint32_t)(x0 >> 16);
+	w[1] = (uint32_t)x1 | (uint32_t)(x1 >> 16);
+	w[2] = (uint32_t)x2 | (uint32_t)(x2 >> 16);
+	w[3] = (uint32_t)x3 | (uint32_t)(x3 >> 16);
+}
+
+static const unsigned char Rcon[] = {
+	0x01, 0x02, 0x04, 0x08, 0x10, 0x20, 0x40, 0x80, 0x1B, 0x36
+};
+
+static uint32_t
+sub_word(uint32_t x)
+{
+	uint64_t q[8];
+
+	memset(q, 0, sizeof q);
+	q[0] = x;
+	br_aes_ct64_ortho(q);
+	br_aes_ct64_bitslice_Sbox(q);
+	br_aes_ct64_ortho(q);
+	return (uint32_t)q[0];
+}
+
+/* see inner.h */
+unsigned
+br_aes_ct64_keysched(uint64_t *comp_skey, const void *key, size_t key_len)
+{
+	unsigned num_rounds;
+	int i, j, k, nk, nkf;
+	uint32_t tmp;
+	uint32_t skey[60];
+
+	switch (key_len) {
+	case 16:
+		num_rounds = 10;
+		break;
+	case 24:
+		num_rounds = 12;
+		break;
+	case 32:
+		num_rounds = 14;
+		break;
+	default:
+		/* abort(); */
+		return 0;
+	}
+	nk = (int)(key_len >> 2);
+	nkf = (int)((num_rounds + 1) << 2);
+	br_range_dec32le(skey, (key_len >> 2), key);
+	tmp = skey[(key_len >> 2) - 1];
+	for (i = nk, j = 0, k = 0; i < nkf; i ++) {
+		if (j == 0) {
+			tmp = (tmp << 24) | (tmp >> 8);
+			tmp = sub_word(tmp) ^ Rcon[k];
+		} else if (nk > 6 && j == 4) {
+			tmp = sub_word(tmp);
+		}
+		tmp ^= skey[i - nk];
+		skey[i] = tmp;
+		if (++ j == nk) {
+			j = 0;
+			k ++;
+		}
+	}
+
+	for (i = 0, j = 0; i < nkf; i += 4, j += 2) {
+		uint64_t q[8];
+
+		br_aes_ct64_interleave_in(&q[0], &q[4], skey + i);
+		q[1] = q[0];
+		q[2] = q[0];
+		q[3] = q[0];
+		q[5] = q[4];
+		q[6] = q[4];
+		q[7] = q[4];
+		br_aes_ct64_ortho(q);
+		comp_skey[j + 0] =
+			  (q[0] & (uint64_t)0x1111111111111111)
+			| (q[1] & (uint64_t)0x2222222222222222)
+			| (q[2] & (uint64_t)0x4444444444444444)
+			| (q[3] & (uint64_t)0x8888888888888888);
+		comp_skey[j + 1] =
+			  (q[4] & (uint64_t)0x1111111111111111)
+			| (q[5] & (uint64_t)0x2222222222222222)
+			| (q[6] & (uint64_t)0x4444444444444444)
+			| (q[7] & (uint64_t)0x8888888888888888);
+	}
+	return num_rounds;
+}
+
+/* see inner.h */
+void
+br_aes_ct64_skey_expand(uint64_t *skey,
+	unsigned num_rounds, const uint64_t *comp_skey)
+{
+	unsigned u, v, n;
+
+	n = (num_rounds + 1) << 1;
+	for (u = 0, v = 0; u < n; u ++, v += 4) {
+		uint64_t x0, x1, x2, x3;
+
+		x0 = x1 = x2 = x3 = comp_skey[u];
+		x0 &= (uint64_t)0x1111111111111111;
+		x1 &= (uint64_t)0x2222222222222222;
+		x2 &= (uint64_t)0x4444444444444444;
+		x3 &= (uint64_t)0x8888888888888888;
+		x1 >>= 1;
+		x2 >>= 2;
+		x3 >>= 3;
+		skey[v + 0] = (x0 << 4) - x0;
+		skey[v + 1] = (x1 << 4) - x1;
+		skey[v + 2] = (x2 << 4) - x2;
+		skey[v + 3] = (x3 << 4) - x3;
+	}
+}
diff --git a/cbits/bearssl/aes_ct64_dec.c b/cbits/bearssl/aes_ct64_dec.c
new file mode 100644
--- /dev/null
+++ b/cbits/bearssl/aes_ct64_dec.c
@@ -0,0 +1,159 @@
+/*
+ * Copyright (c) 2016 Thomas Pornin <pornin@bolet.org>
+ *
+ * Permission is hereby granted, free of charge, to any person obtaining 
+ * a copy of this software and associated documentation files (the
+ * "Software"), to deal in the Software without restriction, including
+ * without limitation the rights to use, copy, modify, merge, publish,
+ * distribute, sublicense, and/or sell copies of the Software, and to
+ * permit persons to whom the Software is furnished to do so, subject to
+ * the following conditions:
+ *
+ * The above copyright notice and this permission notice shall be 
+ * included in all copies or substantial portions of the Software.
+ *
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, 
+ * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
+ * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND 
+ * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
+ * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
+ * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
+ * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+ * SOFTWARE.
+ */
+
+#include "inner.h"
+
+/* see inner.h */
+void
+br_aes_ct64_bitslice_invSbox(uint64_t *q)
+{
+	/*
+	 * See br_aes_ct_bitslice_invSbox(). This is the natural extension
+	 * to 64-bit registers.
+	 */
+	uint64_t q0, q1, q2, q3, q4, q5, q6, q7;
+
+	q0 = ~q[0];
+	q1 = ~q[1];
+	q2 = q[2];
+	q3 = q[3];
+	q4 = q[4];
+	q5 = ~q[5];
+	q6 = ~q[6];
+	q7 = q[7];
+	q[7] = q1 ^ q4 ^ q6;
+	q[6] = q0 ^ q3 ^ q5;
+	q[5] = q7 ^ q2 ^ q4;
+	q[4] = q6 ^ q1 ^ q3;
+	q[3] = q5 ^ q0 ^ q2;
+	q[2] = q4 ^ q7 ^ q1;
+	q[1] = q3 ^ q6 ^ q0;
+	q[0] = q2 ^ q5 ^ q7;
+
+	br_aes_ct64_bitslice_Sbox(q);
+
+	q0 = ~q[0];
+	q1 = ~q[1];
+	q2 = q[2];
+	q3 = q[3];
+	q4 = q[4];
+	q5 = ~q[5];
+	q6 = ~q[6];
+	q7 = q[7];
+	q[7] = q1 ^ q4 ^ q6;
+	q[6] = q0 ^ q3 ^ q5;
+	q[5] = q7 ^ q2 ^ q4;
+	q[4] = q6 ^ q1 ^ q3;
+	q[3] = q5 ^ q0 ^ q2;
+	q[2] = q4 ^ q7 ^ q1;
+	q[1] = q3 ^ q6 ^ q0;
+	q[0] = q2 ^ q5 ^ q7;
+}
+
+static void
+add_round_key(uint64_t *q, const uint64_t *sk)
+{
+	int i;
+
+	for (i = 0; i < 8; i ++) {
+		q[i] ^= sk[i];
+	}
+}
+
+static void
+inv_shift_rows(uint64_t *q)
+{
+	int i;
+
+	for (i = 0; i < 8; i ++) {
+		uint64_t x;
+
+		x = q[i];
+		q[i] = (x & (uint64_t)0x000000000000FFFF)
+			| ((x & (uint64_t)0x000000000FFF0000) << 4)
+			| ((x & (uint64_t)0x00000000F0000000) >> 12)
+			| ((x & (uint64_t)0x000000FF00000000) << 8)
+			| ((x & (uint64_t)0x0000FF0000000000) >> 8)
+			| ((x & (uint64_t)0x000F000000000000) << 12)
+			| ((x & (uint64_t)0xFFF0000000000000) >> 4);
+	}
+}
+
+static inline uint64_t
+rotr32(uint64_t x)
+{
+	return (x << 32) | (x >> 32);
+}
+
+static void
+inv_mix_columns(uint64_t *q)
+{
+	uint64_t q0, q1, q2, q3, q4, q5, q6, q7;
+	uint64_t r0, r1, r2, r3, r4, r5, r6, r7;
+
+	q0 = q[0];
+	q1 = q[1];
+	q2 = q[2];
+	q3 = q[3];
+	q4 = q[4];
+	q5 = q[5];
+	q6 = q[6];
+	q7 = q[7];
+	r0 = (q0 >> 16) | (q0 << 48);
+	r1 = (q1 >> 16) | (q1 << 48);
+	r2 = (q2 >> 16) | (q2 << 48);
+	r3 = (q3 >> 16) | (q3 << 48);
+	r4 = (q4 >> 16) | (q4 << 48);
+	r5 = (q5 >> 16) | (q5 << 48);
+	r6 = (q6 >> 16) | (q6 << 48);
+	r7 = (q7 >> 16) | (q7 << 48);
+
+	q[0] = q5 ^ q6 ^ q7 ^ r0 ^ r5 ^ r7 ^ rotr32(q0 ^ q5 ^ q6 ^ r0 ^ r5);
+	q[1] = q0 ^ q5 ^ r0 ^ r1 ^ r5 ^ r6 ^ r7 ^ rotr32(q1 ^ q5 ^ q7 ^ r1 ^ r5 ^ r6);
+	q[2] = q0 ^ q1 ^ q6 ^ r1 ^ r2 ^ r6 ^ r7 ^ rotr32(q0 ^ q2 ^ q6 ^ r2 ^ r6 ^ r7);
+	q[3] = q0 ^ q1 ^ q2 ^ q5 ^ q6 ^ r0 ^ r2 ^ r3 ^ r5 ^ rotr32(q0 ^ q1 ^ q3 ^ q5 ^ q6 ^ q7 ^ r0 ^ r3 ^ r5 ^ r7);
+	q[4] = q1 ^ q2 ^ q3 ^ q5 ^ r1 ^ r3 ^ r4 ^ r5 ^ r6 ^ r7 ^ rotr32(q1 ^ q2 ^ q4 ^ q5 ^ q7 ^ r1 ^ r4 ^ r5 ^ r6);
+	q[5] = q2 ^ q3 ^ q4 ^ q6 ^ r2 ^ r4 ^ r5 ^ r6 ^ r7 ^ rotr32(q2 ^ q3 ^ q5 ^ q6 ^ r2 ^ r5 ^ r6 ^ r7);
+	q[6] = q3 ^ q4 ^ q5 ^ q7 ^ r3 ^ r5 ^ r6 ^ r7 ^ rotr32(q3 ^ q4 ^ q6 ^ q7 ^ r3 ^ r6 ^ r7);
+	q[7] = q4 ^ q5 ^ q6 ^ r4 ^ r6 ^ r7 ^ rotr32(q4 ^ q5 ^ q7 ^ r4 ^ r7);
+}
+
+/* see inner.h */
+void
+br_aes_ct64_bitslice_decrypt(unsigned num_rounds,
+	const uint64_t *skey, uint64_t *q)
+{
+	unsigned u;
+
+	add_round_key(q, skey + (num_rounds << 3));
+	for (u = num_rounds - 1; u > 0; u --) {
+		inv_shift_rows(q);
+		br_aes_ct64_bitslice_invSbox(q);
+		add_round_key(q, skey + (u << 3));
+		inv_mix_columns(q);
+	}
+	inv_shift_rows(q);
+	br_aes_ct64_bitslice_invSbox(q);
+	add_round_key(q, skey);
+}
diff --git a/cbits/bearssl/aes_ct64_enc.c b/cbits/bearssl/aes_ct64_enc.c
new file mode 100644
--- /dev/null
+++ b/cbits/bearssl/aes_ct64_enc.c
@@ -0,0 +1,115 @@
+/*
+ * Copyright (c) 2016 Thomas Pornin <pornin@bolet.org>
+ *
+ * Permission is hereby granted, free of charge, to any person obtaining 
+ * a copy of this software and associated documentation files (the
+ * "Software"), to deal in the Software without restriction, including
+ * without limitation the rights to use, copy, modify, merge, publish,
+ * distribute, sublicense, and/or sell copies of the Software, and to
+ * permit persons to whom the Software is furnished to do so, subject to
+ * the following conditions:
+ *
+ * The above copyright notice and this permission notice shall be 
+ * included in all copies or substantial portions of the Software.
+ *
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, 
+ * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
+ * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND 
+ * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
+ * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
+ * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
+ * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+ * SOFTWARE.
+ */
+
+#include "inner.h"
+
+static inline void
+add_round_key(uint64_t *q, const uint64_t *sk)
+{
+	q[0] ^= sk[0];
+	q[1] ^= sk[1];
+	q[2] ^= sk[2];
+	q[3] ^= sk[3];
+	q[4] ^= sk[4];
+	q[5] ^= sk[5];
+	q[6] ^= sk[6];
+	q[7] ^= sk[7];
+}
+
+static inline void
+shift_rows(uint64_t *q)
+{
+	int i;
+
+	for (i = 0; i < 8; i ++) {
+		uint64_t x;
+
+		x = q[i];
+		q[i] = (x & (uint64_t)0x000000000000FFFF)
+			| ((x & (uint64_t)0x00000000FFF00000) >> 4)
+			| ((x & (uint64_t)0x00000000000F0000) << 12)
+			| ((x & (uint64_t)0x0000FF0000000000) >> 8)
+			| ((x & (uint64_t)0x000000FF00000000) << 8)
+			| ((x & (uint64_t)0xF000000000000000) >> 12)
+			| ((x & (uint64_t)0x0FFF000000000000) << 4);
+	}
+}
+
+static inline uint64_t
+rotr32(uint64_t x)
+{
+	return (x << 32) | (x >> 32);
+}
+
+static inline void
+mix_columns(uint64_t *q)
+{
+	uint64_t q0, q1, q2, q3, q4, q5, q6, q7;
+	uint64_t r0, r1, r2, r3, r4, r5, r6, r7;
+
+	q0 = q[0];
+	q1 = q[1];
+	q2 = q[2];
+	q3 = q[3];
+	q4 = q[4];
+	q5 = q[5];
+	q6 = q[6];
+	q7 = q[7];
+	r0 = (q0 >> 16) | (q0 << 48);
+	r1 = (q1 >> 16) | (q1 << 48);
+	r2 = (q2 >> 16) | (q2 << 48);
+	r3 = (q3 >> 16) | (q3 << 48);
+	r4 = (q4 >> 16) | (q4 << 48);
+	r5 = (q5 >> 16) | (q5 << 48);
+	r6 = (q6 >> 16) | (q6 << 48);
+	r7 = (q7 >> 16) | (q7 << 48);
+
+	q[0] = q7 ^ r7 ^ r0 ^ rotr32(q0 ^ r0);
+	q[1] = q0 ^ r0 ^ q7 ^ r7 ^ r1 ^ rotr32(q1 ^ r1);
+	q[2] = q1 ^ r1 ^ r2 ^ rotr32(q2 ^ r2);
+	q[3] = q2 ^ r2 ^ q7 ^ r7 ^ r3 ^ rotr32(q3 ^ r3);
+	q[4] = q3 ^ r3 ^ q7 ^ r7 ^ r4 ^ rotr32(q4 ^ r4);
+	q[5] = q4 ^ r4 ^ r5 ^ rotr32(q5 ^ r5);
+	q[6] = q5 ^ r5 ^ r6 ^ rotr32(q6 ^ r6);
+	q[7] = q6 ^ r6 ^ r7 ^ rotr32(q7 ^ r7);
+}
+
+/* see inner.h */
+void
+br_aes_ct64_bitslice_encrypt(unsigned num_rounds,
+	const uint64_t *skey, uint64_t *q)
+{
+	unsigned u;
+
+	add_round_key(q, skey);
+	for (u = 1; u < num_rounds; u ++) {
+		br_aes_ct64_bitslice_Sbox(q);
+		shift_rows(q);
+		mix_columns(q);
+		add_round_key(q, skey + (u << 3));
+	}
+	br_aes_ct64_bitslice_Sbox(q);
+	shift_rows(q);
+	add_round_key(q, skey + (num_rounds << 3));
+}
diff --git a/cbits/bearssl/dec32le.c b/cbits/bearssl/dec32le.c
new file mode 100644
--- /dev/null
+++ b/cbits/bearssl/dec32le.c
@@ -0,0 +1,38 @@
+/*
+ * Copyright (c) 2016 Thomas Pornin <pornin@bolet.org>
+ *
+ * Permission is hereby granted, free of charge, to any person obtaining 
+ * a copy of this software and associated documentation files (the
+ * "Software"), to deal in the Software without restriction, including
+ * without limitation the rights to use, copy, modify, merge, publish,
+ * distribute, sublicense, and/or sell copies of the Software, and to
+ * permit persons to whom the Software is furnished to do so, subject to
+ * the following conditions:
+ *
+ * The above copyright notice and this permission notice shall be 
+ * included in all copies or substantial portions of the Software.
+ *
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, 
+ * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
+ * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND 
+ * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
+ * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
+ * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
+ * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+ * SOFTWARE.
+ */
+
+#include "inner.h"
+
+/* see inner.h */
+void
+br_range_dec32le(uint32_t *v, size_t num, const void *src)
+{
+	const unsigned char *buf;
+
+	buf = src;
+	while (num -- > 0) {
+		*v ++ = br_dec32le(buf);
+		buf += 4;
+	}
+}
diff --git a/cbits/bearssl/ghash_ctmul64.c b/cbits/bearssl/ghash_ctmul64.c
new file mode 100644
--- /dev/null
+++ b/cbits/bearssl/ghash_ctmul64.c
@@ -0,0 +1,154 @@
+/*
+ * Copyright (c) 2016 Thomas Pornin <pornin@bolet.org>
+ *
+ * Permission is hereby granted, free of charge, to any person obtaining 
+ * a copy of this software and associated documentation files (the
+ * "Software"), to deal in the Software without restriction, including
+ * without limitation the rights to use, copy, modify, merge, publish,
+ * distribute, sublicense, and/or sell copies of the Software, and to
+ * permit persons to whom the Software is furnished to do so, subject to
+ * the following conditions:
+ *
+ * The above copyright notice and this permission notice shall be 
+ * included in all copies or substantial portions of the Software.
+ *
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, 
+ * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
+ * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND 
+ * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
+ * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
+ * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
+ * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+ * SOFTWARE.
+ */
+
+#include "inner.h"
+
+/*
+ * This is the 64-bit variant of br_ghash_ctmul32(), with 64-bit operands
+ * and bit reversal of 64-bit words.
+ */
+
+static inline uint64_t
+bmul64(uint64_t x, uint64_t y)
+{
+	uint64_t x0, x1, x2, x3;
+	uint64_t y0, y1, y2, y3;
+	uint64_t z0, z1, z2, z3;
+
+	x0 = x & (uint64_t)0x1111111111111111;
+	x1 = x & (uint64_t)0x2222222222222222;
+	x2 = x & (uint64_t)0x4444444444444444;
+	x3 = x & (uint64_t)0x8888888888888888;
+	y0 = y & (uint64_t)0x1111111111111111;
+	y1 = y & (uint64_t)0x2222222222222222;
+	y2 = y & (uint64_t)0x4444444444444444;
+	y3 = y & (uint64_t)0x8888888888888888;
+	z0 = (x0 * y0) ^ (x1 * y3) ^ (x2 * y2) ^ (x3 * y1);
+	z1 = (x0 * y1) ^ (x1 * y0) ^ (x2 * y3) ^ (x3 * y2);
+	z2 = (x0 * y2) ^ (x1 * y1) ^ (x2 * y0) ^ (x3 * y3);
+	z3 = (x0 * y3) ^ (x1 * y2) ^ (x2 * y1) ^ (x3 * y0);
+	z0 &= (uint64_t)0x1111111111111111;
+	z1 &= (uint64_t)0x2222222222222222;
+	z2 &= (uint64_t)0x4444444444444444;
+	z3 &= (uint64_t)0x8888888888888888;
+	return z0 | z1 | z2 | z3;
+}
+
+static uint64_t
+rev64(uint64_t x)
+{
+#define RMS(m, s)   do { \
+		x = ((x & (uint64_t)(m)) << (s)) \
+			| ((x >> (s)) & (uint64_t)(m)); \
+	} while (0)
+
+	RMS(0x5555555555555555,  1);
+	RMS(0x3333333333333333,  2);
+	RMS(0x0F0F0F0F0F0F0F0F,  4);
+	RMS(0x00FF00FF00FF00FF,  8);
+	RMS(0x0000FFFF0000FFFF, 16);
+	return (x << 32) | (x >> 32);
+
+#undef RMS
+}
+
+/* see bearssl_ghash.h */
+void
+br_ghash_ctmul64(void *y, const void *h, const void *data, size_t len)
+{
+	const unsigned char *buf, *hb;
+	unsigned char *yb;
+	uint64_t y0, y1;
+	uint64_t h0, h1, h2, h0r, h1r, h2r;
+
+	buf = data;
+	yb = y;
+	hb = h;
+	y1 = br_dec64be(yb);
+	y0 = br_dec64be(yb + 8);
+	h1 = br_dec64be(hb);
+	h0 = br_dec64be(hb + 8);
+	h0r = rev64(h0);
+	h1r = rev64(h1);
+	h2 = h0 ^ h1;
+	h2r = h0r ^ h1r;
+	while (len > 0) {
+		const unsigned char *src;
+		unsigned char tmp[16];
+		uint64_t y0r, y1r, y2, y2r;
+		uint64_t z0, z1, z2, z0h, z1h, z2h;
+		uint64_t v0, v1, v2, v3;
+
+		if (len >= 16) {
+			src = buf;
+			buf += 16;
+			len -= 16;
+		} else {
+			memcpy(tmp, buf, len);
+			memset(tmp + len, 0, (sizeof tmp) - len);
+			src = tmp;
+			len = 0;
+		}
+		y1 ^= br_dec64be(src);
+		y0 ^= br_dec64be(src + 8);
+
+		y0r = rev64(y0);
+		y1r = rev64(y1);
+		y2 = y0 ^ y1;
+		y2r = y0r ^ y1r;
+
+		z0 = bmul64(y0, h0);
+		z1 = bmul64(y1, h1);
+		z2 = bmul64(y2, h2);
+		z0h = bmul64(y0r, h0r);
+		z1h = bmul64(y1r, h1r);
+		z2h = bmul64(y2r, h2r);
+		z2 ^= z0 ^ z1;
+		z2h ^= z0h ^ z1h;
+		z0h = rev64(z0h) >> 1;
+		z1h = rev64(z1h) >> 1;
+		z2h = rev64(z2h) >> 1;
+
+		v0 = z0;
+		v1 = z0h ^ z2;
+		v2 = z1 ^ z2h;
+		v3 = z1h;
+
+		v3 = (v3 << 1) | (v2 >> 63);
+		v2 = (v2 << 1) | (v1 >> 63);
+		v1 = (v1 << 1) | (v0 >> 63);
+		v0 = (v0 << 1);
+
+		v2 ^= v0 ^ (v0 >> 1) ^ (v0 >> 2) ^ (v0 >> 7);
+		v1 ^= (v0 << 63) ^ (v0 << 62) ^ (v0 << 57);
+		v3 ^= v1 ^ (v1 >> 1) ^ (v1 >> 2) ^ (v1 >> 7);
+		v2 ^= (v1 << 63) ^ (v1 << 62) ^ (v1 << 57);
+
+		y0 = v2;
+		y1 = v3;
+	}
+
+	br_enc64be(yb, y1);
+	br_enc64be(yb + 8, y0);
+}
diff --git a/cbits/bearssl/import.sh b/cbits/bearssl/import.sh
new file mode 100644
--- /dev/null
+++ b/cbits/bearssl/import.sh
@@ -0,0 +1,34 @@
+#!/bin/sh
+# Re-import the vendored parts of BearSSL.
+#
+# Only the five files crypton calls are kept, and they are kept unmodified,
+# so that `diff` against a new release is readable.  What they want from
+# upstream's two-thousand-line src/inner.h is supplied by the inner.h here,
+# which is crypton's and is NOT overwritten by this script.  Run this from
+# cbits/bearssl:
+#
+#     ./import.sh [version]
+#
+# and commit the result together with the VERSION line it writes, so that
+# the tree always says which upstream release it holds.
+set -eu
+
+VER=${1:-0.6}
+HERE=$(cd "$(dirname "$0")" && pwd)
+TMP=$(mktemp -d)
+trap 'rm -rf "$TMP"' EXIT
+
+curl -sSL "https://bearssl.org/bearssl-$VER.tar.gz" -o "$TMP/b.tgz"
+tar xzf "$TMP/b.tgz" -C "$TMP"
+SRC="$TMP/bearssl-$VER"
+
+cp "$SRC/LICENSE.txt"                    "$HERE/LICENSE"
+cp "$SRC/src/symcipher/aes_ct64.c"       "$HERE/"
+cp "$SRC/src/symcipher/aes_ct64_enc.c"   "$HERE/"
+cp "$SRC/src/symcipher/aes_ct64_dec.c"   "$HERE/"
+cp "$SRC/src/hash/ghash_ctmul64.c"       "$HERE/"
+cp "$SRC/src/codec/dec32le.c"            "$HERE/"
+echo "$VER" > "$HERE/VERSION"
+
+echo "imported BearSSL $VER"
+echo "now run the differential test in cbits/tests/bearssl_diff.c"
diff --git a/cbits/bearssl/inner.h b/cbits/bearssl/inner.h
new file mode 100644
--- /dev/null
+++ b/cbits/bearssl/inner.h
@@ -0,0 +1,92 @@
+/*
+ * Stands in for BearSSL's own src/inner.h.
+ *
+ * The five .c files beside this one are upstream's, byte for byte, and each
+ * of them opens with #include "inner.h".  Upstream's is some two thousand
+ * lines and declares the whole library; these five want six things from it.
+ * So this header gives those six and nothing else, and the upstream files
+ * stay unmodified -- which is what makes `diff` against a new BearSSL
+ * release readable.  See README.md.
+ *
+ * The byte-order helpers are written here rather than copied, in their plain
+ * portable form without upstream's unaligned-access fast paths, so that they
+ * carry no platform configuration with them.  cbits/tests/bearssl_diff.c
+ * checks the whole thing against crypton's existing implementation.
+ */
+#ifndef CRYPTON_BEARSSL_INNER_H
+#define CRYPTON_BEARSSL_INNER_H
+
+#include <stddef.h>
+#include <stdint.h>
+#include <string.h>
+
+static inline uint32_t
+br_dec32le(const void *src)
+{
+	const unsigned char *b = src;
+
+	return (uint32_t)b[0]
+	    | ((uint32_t)b[1] << 8)
+	    | ((uint32_t)b[2] << 16)
+	    | ((uint32_t)b[3] << 24);
+}
+
+static inline void
+br_enc32le(void *dst, uint32_t x)
+{
+	unsigned char *b = dst;
+
+	b[0] = (unsigned char)x;
+	b[1] = (unsigned char)(x >> 8);
+	b[2] = (unsigned char)(x >> 16);
+	b[3] = (unsigned char)(x >> 24);
+}
+
+static inline uint64_t
+br_dec64be(const void *src)
+{
+	const unsigned char *b = src;
+	uint64_t x = 0;
+	int i;
+
+	for (i = 0; i < 8; i++)
+		x = (x << 8) | (uint64_t)b[i];
+	return x;
+}
+
+static inline void
+br_enc64be(void *dst, uint64_t x)
+{
+	unsigned char *b = dst;
+	int i;
+
+	for (i = 7; i >= 0; i--) {
+		b[i] = (unsigned char)(x & 0xff);
+		x >>= 8;
+	}
+}
+
+/* dec32le.c */
+void br_range_dec32le(uint32_t *v, size_t num, const void *src);
+
+/* aes_ct64.c */
+void br_aes_ct64_bitslice_Sbox(uint64_t *q);
+void br_aes_ct64_ortho(uint64_t *q);
+void br_aes_ct64_interleave_in(uint64_t *q0, uint64_t *q1, const uint32_t *w);
+void br_aes_ct64_interleave_out(uint32_t *w, uint64_t q0, uint64_t q1);
+unsigned br_aes_ct64_keysched(uint64_t *comp_skey, const void *key,
+	size_t key_len);
+void br_aes_ct64_skey_expand(uint64_t *skey, unsigned num_rounds,
+	const uint64_t *comp_skey);
+
+/* aes_ct64_enc.c, aes_ct64_dec.c */
+void br_aes_ct64_bitslice_encrypt(unsigned num_rounds, const uint64_t *skey,
+	uint64_t *q);
+void br_aes_ct64_bitslice_invSbox(uint64_t *q);
+void br_aes_ct64_bitslice_decrypt(unsigned num_rounds, const uint64_t *skey,
+	uint64_t *q);
+
+/* ghash_ctmul64.c */
+void br_ghash_ctmul64(void *y, const void *h, const void *data, size_t len);
+
+#endif
diff --git a/cbits/crypton_aes.c b/cbits/crypton_aes.c
--- a/cbits/crypton_aes.c
+++ b/cbits/crypton_aes.c
@@ -38,6 +38,9 @@
 #include <aes/generic.h>
 #include <aes/gf.h>
 #include <aes/x86ni.h>
+#ifdef WITH_PPC8_CRYPTO
+#include <aes/ppc8.h>
+#endif
 #ifdef WITH_GCM_FUSED
 #include <aes/gcm_fused_x86.h>
 #endif
@@ -54,6 +57,8 @@
                              uint32_t spoint, aes_block *input, uint32_t nb_blocks);
 void crypton_aes_generic_gcm_encrypt(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length);
 void crypton_aes_generic_gcm_decrypt(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length);
+void crypton_aes_bitsliced_gcm_encrypt(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length);
+void crypton_aes_bitsliced_gcm_decrypt(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length);
 void crypton_aes_generic_ocb_encrypt(uint8_t *output, aes_ocb *ocb, aes_key *key, uint8_t *input, uint32_t length);
 void crypton_aes_generic_ocb_decrypt(uint8_t *output, aes_ocb *ocb, aes_key *key, uint8_t *input, uint32_t length);
 void crypton_aes_generic_ccm_encrypt(uint8_t *output, aes_ccm *ccm, aes_key *key, uint8_t *input, uint32_t length);
@@ -208,7 +213,7 @@
 typedef void (*gf_mul_f)(block128 *a, const table_4bit htable);
 typedef void (*gf_mul4_f)(block128 *a, const block128 *blocks, const table_4bit htable);
 
-#if defined(WITH_AESNI) || defined(WITH_ARMV8_CRYPTO)
+#if defined(WITH_AESNI) || defined(WITH_ARMV8_CRYPTO) || defined(WITH_PPC8_CRYPTO)
 #define GET_INIT(strength) \
 	((init_f) (crypton_aes_branch_table[INIT_128 + strength]))
 #define GET_ECB_ENCRYPT(strength) \
@@ -255,12 +260,16 @@
 #define GET_ECB_DECRYPT(strength) crypton_aes_generic_decrypt_ecb
 #define GET_CBC_ENCRYPT(strength) crypton_aes_generic_encrypt_cbc
 #define GET_CBC_DECRYPT(strength) crypton_aes_generic_decrypt_cbc
-#define GET_CTR_ENCRYPT(strength) crypton_aes_generic_encrypt_ctr
-#define GET_C32_ENCRYPT(strength) crypton_aes_generic_encrypt_c32
+/* No accelerator is compiled in, so the portable schedule is the only one
+ * a key can hold and the four-block CTR is named here rather than installed
+ * at run time.  This is the build ppc64le, s390x, riscv64, 32-bit ARM and
+ * -f-support_aesni all get, and nothing in it reads the branch table. */
+#define GET_CTR_ENCRYPT(strength) crypton_aes_bitsliced_encrypt_ctr
+#define GET_C32_ENCRYPT(strength) crypton_aes_bitsliced_encrypt_c32
 #define GET_XTS_ENCRYPT(strength) crypton_aes_generic_encrypt_xts
 #define GET_XTS_DECRYPT(strength) crypton_aes_generic_decrypt_xts
-#define GET_GCM_ENCRYPT(strength) crypton_aes_generic_gcm_encrypt
-#define GET_GCM_DECRYPT(strength) crypton_aes_generic_gcm_decrypt
+#define GET_GCM_ENCRYPT(strength) crypton_aes_bitsliced_gcm_encrypt
+#define GET_GCM_DECRYPT(strength) crypton_aes_bitsliced_gcm_decrypt
 #define GET_OCB_ENCRYPT(strength) crypton_aes_generic_ocb_encrypt
 #define GET_OCB_DECRYPT(strength) crypton_aes_generic_ocb_decrypt
 #define GET_CCM_ENCRYPT(strength) crypton_aes_generic_ccm_encrypt
@@ -272,10 +281,6 @@
 #define crypton_gf_mul4(a,b,t) crypton_aes_generic_gf_mul4(a,b,t)
 #endif
 
-#define CPU_AESNI        0
-#define CPU_PCLMUL       1
-#define CPU_OPTION_COUNT 2
-
 static uint8_t crypton_aes_cpu_options[CPU_OPTION_COUNT] = {};
 
 #if defined(ARCH_X86) && defined(WITH_AESNI)
@@ -440,6 +445,74 @@
  * A constructor runs while there is one thread, which is the cheapest way to
  * have no race at all: no flag to test, no lock to take, and one less thing
  * for crypton_aes_initkey to do per key. */
+/*
+ * The two CTR entries, where the portable implementation is the one that
+ * will run.  This is for the build that has an accelerator compiled in and
+ * did not find it on the processor; a build with no accelerator at all
+ * names them directly in the GET_ macros above and never reads this table.
+ *
+ * They cannot be the defaults.  crypton_aes_generic_encrypt_c32 is not
+ * overridden on AArch64 -- the ARMv8 table has no C32 of its own -- and
+ * neither CCM nor OCB is overridden anywhere, so those reach the block
+ * function through the branch table and work whatever is installed.  A CTR
+ * that read the portable schedule directly would be wrong on exactly those
+ * machines.  So these go in only once nothing has claimed AES.
+ */
+static void initialize_table_bitsliced(void)
+{
+	crypton_aes_branch_table[ENCRYPT_CTR_128] = crypton_aes_bitsliced_encrypt_ctr;
+	crypton_aes_branch_table[ENCRYPT_CTR_192] = crypton_aes_bitsliced_encrypt_ctr;
+	crypton_aes_branch_table[ENCRYPT_CTR_256] = crypton_aes_bitsliced_encrypt_ctr;
+	crypton_aes_branch_table[ENCRYPT_C32_128] = crypton_aes_bitsliced_encrypt_c32;
+	crypton_aes_branch_table[ENCRYPT_C32_192] = crypton_aes_bitsliced_encrypt_c32;
+	crypton_aes_branch_table[ENCRYPT_C32_256] = crypton_aes_bitsliced_encrypt_c32;
+	crypton_aes_branch_table[ENCRYPT_GCM_128] = crypton_aes_bitsliced_gcm_encrypt;
+	crypton_aes_branch_table[ENCRYPT_GCM_192] = crypton_aes_bitsliced_gcm_encrypt;
+	crypton_aes_branch_table[ENCRYPT_GCM_256] = crypton_aes_bitsliced_gcm_encrypt;
+	crypton_aes_branch_table[DECRYPT_GCM_128] = crypton_aes_bitsliced_gcm_decrypt;
+	crypton_aes_branch_table[DECRYPT_GCM_192] = crypton_aes_bitsliced_gcm_decrypt;
+	crypton_aes_branch_table[DECRYPT_GCM_256] = crypton_aes_bitsliced_gcm_decrypt;
+}
+
+#ifdef WITH_PPC8_CRYPTO
+/*
+ * POWER8.  One function per operation rather than three: the assembly reads
+ * the round count out of the key, so the same entry serves every key size.
+ *
+ * GCM and the 32-bit counter are left where they are.  Neither has an entry
+ * in the assembly, and the generic loops reach the block function and the
+ * GHASH through this table, so both get the instructions anyway.
+ */
+static void initialize_table_ppc8(void)
+{
+	int sz;
+
+	if (!crypton_aes_ppc8_available())
+		return;
+	/* what stops initialize_table_bitsliced below from putting its own CTR
+	 * in, which would read this key as a bitsliced schedule */
+	crypton_aes_cpu_options[CPU_AESNI] = 1;
+	crypton_aes_cpu_options[CPU_PCLMUL] = 1;
+
+	for (sz = 0; sz < 3; sz++) {
+		crypton_aes_branch_table[INIT_128 + sz] = crypton_aes_ppc8_init;
+		crypton_aes_branch_table[ENCRYPT_BLOCK_128 + sz] = crypton_aes_ppc8_encrypt_block;
+		crypton_aes_branch_table[DECRYPT_BLOCK_128 + sz] = crypton_aes_ppc8_decrypt_block;
+		crypton_aes_branch_table[ENCRYPT_ECB_128 + sz] = crypton_aes_ppc8_encrypt_ecb;
+		crypton_aes_branch_table[DECRYPT_ECB_128 + sz] = crypton_aes_ppc8_decrypt_ecb;
+		crypton_aes_branch_table[ENCRYPT_CBC_128 + sz] = crypton_aes_ppc8_encrypt_cbc;
+		crypton_aes_branch_table[DECRYPT_CBC_128 + sz] = crypton_aes_ppc8_decrypt_cbc;
+		crypton_aes_branch_table[ENCRYPT_CTR_128 + sz] = crypton_aes_ppc8_encrypt_ctr;
+		crypton_aes_branch_table[ENCRYPT_XTS_128 + sz] = crypton_aes_ppc8_encrypt_xts;
+		crypton_aes_branch_table[DECRYPT_XTS_128 + sz] = crypton_aes_ppc8_decrypt_xts;
+	}
+
+	crypton_aes_branch_table[GHASH_HINIT]   = crypton_aes_ppc8_hinit;
+	crypton_aes_branch_table[GHASH_GF_MUL]  = crypton_aes_ppc8_gf_mul;
+	crypton_aes_branch_table[GHASH_GF_MUL4] = crypton_aes_ppc8_gf_mul4;
+}
+#endif
+
 static void crypton_aes_cpu_setup(void)
 {
 #if defined(ARCH_X86) && defined(WITH_AESNI)
@@ -448,6 +521,11 @@
 #ifdef WITH_ARMV8_CRYPTO
 	initialize_table_armv8();
 #endif
+#ifdef WITH_PPC8_CRYPTO
+	initialize_table_ppc8();
+#endif
+	if (crypton_aes_cpu_options[CPU_AESNI] == 0)
+		initialize_table_bitsliced();
 }
 
 __attribute__((constructor))
@@ -1136,18 +1214,30 @@
 	block128_xor((block128 *) tag, &ocb->sum_aad);
 }
 
+/*
+ * These three reach cbits/aes/generic.c directly rather than through the
+ * branch table, so they run only where the portable implementation was the
+ * one installed -- which is what lets them read its schedule and hand its
+ * core four blocks at a time.  The CTR entries below cannot do the same:
+ * they go through the branch table for the block, and so run on accelerated
+ * machines too.
+ */
 void crypton_aes_generic_encrypt_ecb(aes_block *output, aes_key *key, aes_block *input, uint32_t nb_blocks)
 {
-	for ( ; nb_blocks-- > 0; input++, output++) {
-		crypton_aes_generic_encrypt_block(output, key, input);
-	}
+	aes_sched sched;
+
+	crypton_aes_generic_schedule(&sched, key);
+	crypton_aes_generic_blocks((uint8_t *) output, (const uint8_t *) input,
+	                           nb_blocks, &sched, 0);
 }
 
 void crypton_aes_generic_decrypt_ecb(aes_block *output, aes_key *key, aes_block *input, uint32_t nb_blocks)
 {
-	for ( ; nb_blocks-- > 0; input++, output++) {
-		crypton_aes_generic_decrypt_block(output, key, input);
-	}
+	aes_sched sched;
+
+	crypton_aes_generic_schedule(&sched, key);
+	crypton_aes_generic_blocks((uint8_t *) output, (const uint8_t *) input,
+	                           nb_blocks, &sched, 1);
 }
 
 void crypton_aes_generic_encrypt_cbc(aes_block *output, aes_key *key, aes_block *iv, aes_block *input, uint32_t nb_blocks)
@@ -1163,18 +1253,35 @@
 	}
 }
 
+/*
+ * Decryption, unlike encryption, does not wait for the block before it:
+ * every ciphertext block is ready at once and the chaining is a XOR
+ * afterwards.  So four at a pass, with the ciphertext copied aside first,
+ * since the caller is allowed to decrypt in place and the next group's IV
+ * is the last ciphertext block of this one.
+ */
 void crypton_aes_generic_decrypt_cbc(aes_block *output, aes_key *key, aes_block *ivini, aes_block *input, uint32_t nb_blocks)
 {
-	aes_block block, blocko;
+	aes_sched sched;
 	aes_block iv;
+	uint8_t ct[64], plain[64];
 
-	/* preload IV in block */
+	crypton_aes_generic_schedule(&sched, key);
 	block128_copy(&iv, ivini);
-	for ( ; nb_blocks-- > 0; input++, output++) {
-		block128_copy(&block, (block128 *) input);
-		crypton_aes_generic_decrypt_block(&blocko, key, &block);
-		block128_vxor((block128 *) output, &blocko, &iv);
-		block128_copy(&iv, &block);
+	while (nb_blocks > 0) {
+		uint32_t n = nb_blocks < 4 ? nb_blocks : 4;
+		uint32_t i;
+
+		memcpy(ct, input, n * 16);
+		crypton_aes_generic_blocks(plain, ct, n, &sched, 1);
+		for (i = 0; i < n; i++) {
+			block128_vxor((block128 *) (output + i),
+			              (block128 *) (plain + 16 * i), &iv);
+			block128_copy(&iv, (block128 *) (ct + 16 * i));
+		}
+		input += n;
+		output += n;
+		nb_blocks -= n;
 	}
 }
 
@@ -1229,7 +1336,9 @@
 void crypton_aes_generic_encrypt_xts(aes_block *output, aes_key *k1, aes_key *k2, aes_block *dataunit,
                              uint32_t spoint, aes_block *input, uint32_t nb_blocks)
 {
-	aes_block block, tweak;
+	aes_sched sched;
+	aes_block tweak;
+	uint8_t buf[64], tw[64];
 
 	/* load IV and encrypt it using k2 as the tweak */
 	block128_copy(&tweak, dataunit);
@@ -1239,17 +1348,34 @@
 	while (spoint-- > 0)
 		crypton_aes_generic_gf_mulx(&tweak);
 
-	for ( ; nb_blocks-- > 0; input++, output++, crypton_aes_generic_gf_mulx(&tweak)) {
-		block128_vxor(&block, input, &tweak);
-		crypton_aes_encrypt_block(&block, k1, &block);
-		block128_vxor(output, &block, &tweak);
+	crypton_aes_generic_schedule(&sched, k1);
+	while (nb_blocks > 0) {
+		uint32_t n = nb_blocks < 4 ? nb_blocks : 4;
+		uint32_t i;
+
+		/* the tweaks for a group depend on nothing but each other, so
+		 * all four are known before any block is enciphered */
+		for (i = 0; i < n; i++) {
+			block128_copy((block128 *) (tw + 16 * i), &tweak);
+			block128_vxor((block128 *) (buf + 16 * i), input + i, &tweak);
+			crypton_aes_generic_gf_mulx(&tweak);
+		}
+		crypton_aes_generic_blocks(buf, buf, n, &sched, 0);
+		for (i = 0; i < n; i++)
+			block128_vxor(output + i, (block128 *) (buf + 16 * i),
+			              (block128 *) (tw + 16 * i));
+		input += n;
+		output += n;
+		nb_blocks -= n;
 	}
 }
 
 void crypton_aes_generic_decrypt_xts(aes_block *output, aes_key *k1, aes_key *k2, aes_block *dataunit,
                              uint32_t spoint, aes_block *input, uint32_t nb_blocks)
 {
-	aes_block block, tweak;
+	aes_sched sched;
+	aes_block tweak;
+	uint8_t buf[64], tw[64];
 
 	/* load IV and encrypt it using k2 as the tweak */
 	block128_copy(&tweak, dataunit);
@@ -1259,10 +1385,25 @@
 	while (spoint-- > 0)
 		crypton_aes_generic_gf_mulx(&tweak);
 
-	for ( ; nb_blocks-- > 0; input++, output++, crypton_aes_generic_gf_mulx(&tweak)) {
-		block128_vxor(&block, input, &tweak);
-		crypton_aes_decrypt_block(&block, k1, &block);
-		block128_vxor(output, &block, &tweak);
+	crypton_aes_generic_schedule(&sched, k1);
+	while (nb_blocks > 0) {
+		uint32_t n = nb_blocks < 4 ? nb_blocks : 4;
+		uint32_t i;
+
+		/* the tweaks for a group depend on nothing but each other, so
+		 * all four are known before any block is enciphered */
+		for (i = 0; i < n; i++) {
+			block128_copy((block128 *) (tw + 16 * i), &tweak);
+			block128_vxor((block128 *) (buf + 16 * i), input + i, &tweak);
+			crypton_aes_generic_gf_mulx(&tweak);
+		}
+		crypton_aes_generic_blocks(buf, buf, n, &sched, 1);
+		for (i = 0; i < n; i++)
+			block128_vxor(output + i, (block128 *) (buf + 16 * i),
+			              (block128 *) (tw + 16 * i));
+		input += n;
+		output += n;
+		nb_blocks -= n;
 	}
 }
 
@@ -1491,6 +1632,116 @@
 		block128_zero(&tmp);
 		block128_copy_bytes(&tmp, input, length);
 		ccm_cbcmac_add(ccm, key, &tmp);
+	}
+}
+
+/*
+ * GCM where the portable implementation is the one that will run.
+ *
+ * The same shape as the two above, with two differences: the four counter
+ * blocks of a group are encrypted together, which is what the bitsliced
+ * core is for, and the schedule is expanded once for the message rather
+ * than once per block.
+ *
+ * These cannot replace the generic pair.  That pair also runs on an x86
+ * machine that has AES-NI and no carry-less multiply, where crypton_aes.c
+ * installs the AES-NI block function and leaves GCM alone -- and there the
+ * key holds the AES-NI schedule, not this one.  So the choice is made where
+ * the rest of the portable entries are chosen.
+ */
+void crypton_aes_bitsliced_gcm_encrypt(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length)
+{
+	aes_sched sched;
+	aes_block out;
+
+	crypton_aes_generic_schedule(&sched, key);
+	gcm->length_input += length;
+	for (; length >= 64; input += 64, output += 64, length -= 64) {
+		aes_block buf[4];
+		int i;
+
+		for (i = 0; i < 4; i++) {
+			block128_inc32_be(&gcm->civ);
+			block128_copy(&buf[i], &gcm->civ);
+		}
+		crypton_aes_generic_blocks((uint8_t *) buf, (const uint8_t *) buf,
+		                           4, &sched, 0);
+		for (i = 0; i < 4; i++)
+			block128_xor(&buf[i], (block128 *) (input + 16 * i));
+		gcm_ghash_add4(gcm, buf);
+		for (i = 0; i < 4; i++)
+			block128_copy((block128 *) (output + 16 * i), &buf[i]);
+	}
+	for (; length >= 16; input += 16, output += 16, length -= 16) {
+		block128_inc32_be(&gcm->civ);
+		crypton_aes_generic_blocks((uint8_t *) &out,
+		                           (const uint8_t *) &gcm->civ, 1, &sched, 0);
+		block128_xor(&out, (block128 *) input);
+		gcm_ghash_add(gcm, &out);
+		block128_copy((block128 *) output, &out);
+	}
+	if (length > 0) {
+		aes_block tmp;
+		uint32_t i;
+
+		block128_inc32_be(&gcm->civ);
+		crypton_aes_generic_blocks((uint8_t *) &out,
+		                           (const uint8_t *) &gcm->civ, 1, &sched, 0);
+		block128_zero(&tmp);
+		block128_copy_bytes(&tmp, input, length);
+		block128_xor_bytes(&tmp, out.b, length);
+		gcm_ghash_add(gcm, &tmp);
+		for (i = 0; i < length; i++)
+			output[i] = tmp.b[i];
+	}
+}
+
+void crypton_aes_bitsliced_gcm_decrypt(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length)
+{
+	aes_sched sched;
+	aes_block out;
+
+	crypton_aes_generic_schedule(&sched, key);
+	gcm->length_input += length;
+	/* GHASH all four ciphertext blocks before writing any plaintext, since
+	 * output may be input */
+	for (; length >= 64; input += 64, output += 64, length -= 64) {
+		aes_block buf[4];
+		int i;
+
+		gcm_ghash_add4(gcm, (const block128 *) input);
+		for (i = 0; i < 4; i++) {
+			block128_inc32_be(&gcm->civ);
+			block128_copy(&buf[i], &gcm->civ);
+		}
+		crypton_aes_generic_blocks((uint8_t *) buf, (const uint8_t *) buf,
+		                           4, &sched, 0);
+		for (i = 0; i < 4; i++) {
+			block128_xor(&buf[i], (block128 *) (input + 16 * i));
+			block128_copy((block128 *) (output + 16 * i), &buf[i]);
+		}
+	}
+	for (; length >= 16; input += 16, output += 16, length -= 16) {
+		block128_inc32_be(&gcm->civ);
+		crypton_aes_generic_blocks((uint8_t *) &out,
+		                           (const uint8_t *) &gcm->civ, 1, &sched, 0);
+		gcm_ghash_add(gcm, (block128 *) input);
+		block128_xor(&out, (block128 *) input);
+		block128_copy((block128 *) output, &out);
+	}
+	if (length > 0) {
+		aes_block tmp;
+		uint32_t i;
+
+		block128_inc32_be(&gcm->civ);
+		crypton_aes_generic_blocks((uint8_t *) &out,
+		                           (const uint8_t *) &gcm->civ, 1, &sched, 0);
+		block128_zero(&tmp);
+		block128_copy_bytes(&tmp, input, length);
+		gcm_ghash_add(gcm, &tmp);
+		block128_xor_bytes(&tmp, out.b, length);
+		for (i = 0; i < length; i++)
+			output[i] = tmp.b[i];
 	}
 }
 
diff --git a/cbits/crypton_armv8_target.h b/cbits/crypton_armv8_target.h
--- a/cbits/crypton_armv8_target.h
+++ b/cbits/crypton_armv8_target.h
@@ -23,7 +23,19 @@
 #define CRYPTON_ARMV8_TARGET_H
 
 #ifdef WITH_TARGET_ATTRIBUTES
-#if defined(__clang__)
+#if defined(__arm__)
+/*
+ * AArch32 asks for an FPU rather than an architecture extension, and the
+ * spelling differs between the compilers in ways that cannot be checked from
+ * here.  Rather than guess, the cabal file passes -mfpu=crypto-neon-fp-armv8
+ * for the whole component on this architecture whatever this flag says, and
+ * these expand to nothing.  A global target option is safe because a
+ * compiler emits these instructions where an intrinsic asks for them and
+ * nowhere else.
+ */
+#define CRYPTON_TARGET_ARMV8_CRYPTO
+#define CRYPTON_TARGET_ARMV8_SHA3
+#elif defined(__clang__)
 #define CRYPTON_TARGET_ARMV8_CRYPTO __attribute__((target("+crypto")))
 #define CRYPTON_TARGET_ARMV8_SHA3 __attribute__((target("+sha3")))
 #else
diff --git a/cbits/crypton_cpu.c b/cbits/crypton_cpu.c
--- a/cbits/crypton_cpu.c
+++ b/cbits/crypton_cpu.c
@@ -60,7 +60,7 @@
     CRYPTON_ARMCAP_NEON;
 #endif
 
-#if defined(__aarch64__)
+#if defined(__aarch64__) || defined(__arm__)
 #if defined(__APPLE__)
 #include <sys/sysctl.h>
 #elif defined(__linux__)
@@ -71,14 +71,19 @@
 #endif
 
 /*
- * The HWCAP bit positions are the ARM ELF ABI's, so every system that
- * reports through the auxiliary vector agrees on them.  What differs is
- * AT_HWCAP itself -- 16 on Linux, 25 on FreeBSD -- and that comes from each
- * system's own header, which is why the tag is never written out here.
- * Defined only where the system's headers did not define them, the way
- * compiler-rt does it, so that a system which reports through the auxiliary
- * vector without shipping the ARM names still compiles.
+ * The bit positions are the ARM ELF ABI's, so every system that reports
+ * through the auxiliary vector agrees on them.  What differs is the tag --
+ * AT_HWCAP is 16 on Linux and 25 on FreeBSD -- and that comes from each
+ * system's own header, which is why no tag is written out here.  Defined
+ * only where the system's headers did not define them, the way compiler-rt
+ * does it, so that a system which reports through the auxiliary vector
+ * without shipping the ARM names still compiles.
+ *
+ * The two execution states do not share a word or an order.  AArch64 puts
+ * these in AT_HWCAP; AArch32 has filled that word with older features and
+ * puts the cryptographic ones in AT_HWCAP2, starting again from bit zero.
  */
+#if defined(__aarch64__)
 #ifndef HWCAP_AES
 #define HWCAP_AES    (1 << 3)
 #endif
@@ -97,6 +102,20 @@
 #ifndef HWCAP_SHA512
 #define HWCAP_SHA512 (1 << 21)
 #endif
+#else
+#ifndef HWCAP2_AES
+#define HWCAP2_AES   (1 << 0)
+#endif
+#ifndef HWCAP2_PMULL
+#define HWCAP2_PMULL (1 << 1)
+#endif
+#ifndef HWCAP2_SHA1
+#define HWCAP2_SHA1  (1 << 2)
+#endif
+#ifndef HWCAP2_SHA2
+#define HWCAP2_SHA2  (1 << 3)
+#endif
+#endif
 
 #if defined(__APPLE__)
 static int apple_has(const char *name)
@@ -145,7 +164,7 @@
 			f |= CRYPTON_ARM_SHA512;
 		if (apple_has("hw.optional.arm.FEAT_SHA3"))
 			f |= CRYPTON_ARM_SHA3;
-#else
+#elif defined(__aarch64__)
 		unsigned long cap = 0;
 
 #if defined(__linux__)
@@ -160,14 +179,67 @@
 		if (cap & HWCAP_SHA2)   f |= CRYPTON_ARM_SHA2;
 		if (cap & HWCAP_SHA512) f |= CRYPTON_ARM_SHA512;
 		if (cap & HWCAP_SHA3)   f |= CRYPTON_ARM_SHA3;
+#else
+		/* AArch32: the second word, and nothing later than SHA-256 to
+		 * ask about */
+		unsigned long cap = 0;
+
+#if defined(__linux__)
+		cap = getauxval(AT_HWCAP2);
+#elif defined(__FreeBSD__)
+		if (elf_aux_info(AT_HWCAP2, &cap, sizeof(cap)) != 0)
+			cap = 0;
 #endif
+		if (cap & HWCAP2_AES)   f |= CRYPTON_ARM_AES;
+		if (cap & HWCAP2_PMULL) f |= CRYPTON_ARM_PMULL;
+		if (cap & HWCAP2_SHA1)  f |= CRYPTON_ARM_SHA1;
+		if (cap & HWCAP2_SHA2)  f |= CRYPTON_ARM_SHA2;
+#endif
 		features = f;
 		resolved = 1;
 	}
 	return features;
 }
-#endif /* __aarch64__ */
+#endif /* __aarch64__ || __arm__ */
 
+#if defined(__powerpc64__) || defined(__PPC64__)
+#if defined(__linux__)
+#include <sys/auxv.h>
+#elif defined(__FreeBSD__)
+#include <sys/auxv.h>
+#endif
+
+/*
+ * PPC_FEATURE2_VEC_CRYPTO, from the Power ELF ABI.  Defined here only where
+ * the system's headers did not, so that a system reporting through the
+ * auxiliary vector without shipping the name still compiles.
+ */
+#ifndef PPC_FEATURE2_VEC_CRYPTO
+#define PPC_FEATURE2_VEC_CRYPTO 0x02000000
+#endif
+
+unsigned int crypton_ppc_features(void)
+{
+	static unsigned int features;
+	static int resolved;
+
+	if (!resolved) {
+		unsigned long cap = 0;
+
+#if defined(__linux__)
+		cap = getauxval(AT_HWCAP2);
+#elif defined(__FreeBSD__)
+		if (elf_aux_info(AT_HWCAP2, &cap, sizeof(cap)) != 0)
+			cap = 0;
+#endif
+		features = (cap & PPC_FEATURE2_VEC_CRYPTO)
+		         ? CRYPTON_PPC_VCRYPTO : 0;
+		resolved = 1;
+	}
+	return features;
+}
+#endif /* __powerpc64__ */
+
 #ifdef ARCH_X86
 static void cpuid(uint32_t info, uint32_t *eax, uint32_t *ebx, uint32_t *ecx, uint32_t *edx)
 {
@@ -421,3 +493,108 @@
 #endif
 
 #endif
+
+/*
+ * What Crypto.System.CPU reports.
+ *
+ * The numbers are that module's, and this answers for them.  The old shape
+ * was the other way round -- Haskell read crypton_aes.c's two-entry array by
+ * the Enum index of its own constructors -- which tied the list of names to
+ * the indices of an array that had no reason to grow, and so it never did.
+ *
+ * Nothing here is cached: every answer is either a static array filled by a
+ * constructor, a value resolved once and remembered by the function that
+ * owns it, or a runtime check that does its own remembering.  Haskell calls
+ * this once, for a CAF.
+ */
+uint8_t *crypton_aes_cpu_init(void);
+
+#ifdef WITH_ARMV8_CRYPTO
+int crypton_aes_armv8_available(void);
+int crypton_aes_armv8_pmull_available(void);
+#endif
+#ifdef WITH_ARMV8_SHA1
+extern int crypton_sha1_armv8_available(void);
+#endif
+#ifdef WITH_ARMV8_SHA2
+extern int crypton_sha256_armv8_available(void);
+#endif
+#ifdef WITH_ARMV8_SHA512
+extern int crypton_sha512_armv8_available(void);
+#endif
+
+/* Keep in step with Crypto/System/CPU.hs, which is where these are named.
+ * 2 is missing: it was RDRAND, which crypton no longer dispatches on.  The
+ * number is left out rather than reused, so that the rest keep the values
+ * they had. */
+#define OPT_AESNI      0
+#define OPT_PCLMUL     1
+#define OPT_SSSE3      3
+#define OPT_AVX        4
+#define OPT_AVX2       5
+#define OPT_SHANI      6
+#define OPT_MOVBE      7
+#define OPT_ADX        8
+#define OPT_VAES       9
+#define OPT_VAES512   10
+#define OPT_NEON      11
+#define OPT_ARMAES    12
+#define OPT_ARMPMULL  13
+#define OPT_ARMSHA1   14
+#define OPT_ARMSHA2   15
+#define OPT_ARMSHA512 16
+#define OPT_PPCAES    17
+#define OPT_PPCVPMSUM 18
+
+int crypton_cpu_option(unsigned int option)
+{
+#ifdef ARCH_X86
+	uint32_t f = crypton_x86_simd_features();
+
+	switch (option) {
+	case OPT_AESNI:    return crypton_aes_cpu_init()[CPU_AESNI] != 0;
+	case OPT_PCLMUL:   return crypton_aes_cpu_init()[CPU_PCLMUL] != 0;
+	case OPT_SSSE3:    return (f & CRYPTON_X86_SSSE3)   != 0;
+	case OPT_AVX:      return (f & CRYPTON_X86_AVX)     != 0;
+	case OPT_AVX2:     return (f & CRYPTON_X86_AVX2)    != 0;
+	case OPT_SHANI:    return (f & CRYPTON_X86_SHA_NI)  != 0;
+	case OPT_MOVBE:    return (f & CRYPTON_X86_MOVBE)   != 0;
+	case OPT_ADX:      return (f & CRYPTON_X86_ADX)     != 0;
+	case OPT_VAES:     return (f & CRYPTON_X86_VAES)    != 0;
+	case OPT_VAES512:  return (f & CRYPTON_X86_VAES512) != 0;
+	default:           return 0;
+	}
+#else
+	switch (option) {
+	/* NEON is not optional on AArch64, and crypton_armcap_P is set from
+	 * the start for it.  Where there is no ARM assembly at all there is
+	 * nothing to say. */
+#ifdef CRYPTON_ARM_ASM
+	case OPT_NEON:     return (crypton_armcap_P & CRYPTON_ARMCAP_NEON) != 0;
+#endif
+#ifdef WITH_ARMV8_CRYPTO
+	/* The array rather than the check: it is what the branch table was
+	 * actually built from, so it says what will run, not merely what the
+	 * processor has. */
+	case OPT_ARMAES:   return crypton_aes_cpu_init()[CPU_AESNI]  != 0;
+	case OPT_ARMPMULL: return crypton_aes_cpu_init()[CPU_PCLMUL] != 0;
+#endif
+#ifdef WITH_ARMV8_SHA1
+	case OPT_ARMSHA1:  return crypton_sha1_armv8_available()   != 0;
+#endif
+#ifdef WITH_ARMV8_SHA2
+	case OPT_ARMSHA2:  return crypton_sha256_armv8_available() != 0;
+#endif
+#ifdef WITH_ARMV8_SHA512
+	case OPT_ARMSHA512: return crypton_sha512_armv8_available() != 0;
+#endif
+#ifdef WITH_PPC8_CRYPTO
+	/* The array rather than the check, for the same reason as the ARM pair
+	 * above: it says what the branch table was built from. */
+	case OPT_PPCAES:    return crypton_aes_cpu_init()[CPU_AESNI]  != 0;
+	case OPT_PPCVPMSUM: return crypton_aes_cpu_init()[CPU_PCLMUL] != 0;
+#endif
+	default:           return 0;
+	}
+#endif
+}
diff --git a/cbits/crypton_cpu.h b/cbits/crypton_cpu.h
--- a/cbits/crypton_cpu.h
+++ b/cbits/crypton_cpu.h
@@ -107,8 +107,13 @@
  * which systems someone had thought of.  It is what left FreeBSD running
  * the table-driven AES and the C SHA on hardware that has the
  * instructions.  In one place the next system is added once.
+ *
+ * AArch32 answers here too.  It has the same AES, PMULL, SHA-1 and SHA-256
+ * instructions, reports them in a different word of the auxiliary vector,
+ * and has no SHA-512 or SHA-3 to report at all -- those two are AArch64's
+ * alone, so the bits exist and are never set.
  */
-#if defined(__aarch64__)
+#if defined(__aarch64__) || defined(__arm__)
 #define CRYPTON_ARM_AES    (1u << 0)
 #define CRYPTON_ARM_PMULL  (1u << 1)
 #define CRYPTON_ARM_SHA1   (1u << 2)
@@ -117,6 +122,40 @@
 #define CRYPTON_ARM_SHA3   (1u << 5)
 unsigned int crypton_arm_features(void);
 #endif
+
+/*
+ * Which of the optional PowerISA instruction sets this processor has.
+ *
+ * One bit so far: the vector AES and the vector carry-less multiply that
+ * PowerISA 2.07 brought and POWER8 was the first to implement.  The system
+ * reports it in the second capability word, as AArch32 does, and with its
+ * own numbering again.
+ */
+#if defined(__powerpc64__) || defined(__PPC64__)
+#define CRYPTON_PPC_VCRYPTO (1u << 0)
+unsigned int crypton_ppc_features(void);
+#endif
+
+/*
+ * The two slots of the array crypton_aes.c fills and crypton_aes_cpu_init
+ * hands out.  Here rather than in that file because crypton_cpu.c reads
+ * them too, to answer for Crypto.System.CPU.
+ */
+#define CPU_AESNI        0
+#define CPU_PCLMUL       1
+#define CPU_OPTION_COUNT 2
+
+/*
+ * One question at a time, numbered the way Crypto.System.CPU numbers its
+ * ProcessorOption rather than the way any array here is indexed.  The
+ * numbering is the module's to choose and this answers for it; reading an
+ * array by the Haskell constructor's Enum index is what kept that type at
+ * three names.  Returns 1 if the processor has it and crypton was built to
+ * use it, 0 otherwise -- including for anything this does not know, so an
+ * older library answers a newer caller rather than failing to link.
+ */
+int crypton_cpu_option(unsigned int option);
+
 #ifdef USE_AESNI
 void crypton_aesni_initialize_hw(void (*init_table)(int, int));
 #else
diff --git a/cbits/crypton_rdrand.c b/cbits/crypton_rdrand.c
deleted file mode 100644
--- a/cbits/crypton_rdrand.c
+++ /dev/null
@@ -1,126 +0,0 @@
-/*
- * Copyright (C) Thomas DuBuisson
- * Copyright (C) 2013 Vincent Hanquez <tab@snarc.org>
- *
- * All rights reserved.
- * 
- * Redistribution and use in source and binary forms, with or without
- * modification, are permitted provided that the following conditions
- * are met:
- * 1. Redistributions of source code must retain the above copyright
- *    notice, this list of conditions and the following disclaimer.
- * 2. Redistributions in binary form must reproduce the above copyright
- *    notice, this list of conditions and the following disclaimer in the
- *    documentation and/or other materials provided with the distribution.
- * 3. Neither the name of the author nor the names of his contributors
- *    may be used to endorse or promote products derived from this software
- *    without specific prior written permission.
- * 
- * THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
- * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
- * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
- * ARE DISCLAIMED.  IN NO EVENT SHALL THE AUTHORS OR CONTRIBUTORS BE LIABLE
- * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
- * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
- * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
- * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
- * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
- * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
- * SUCH DAMAGE.
- */
-
-#include <stdint.h>
-#include <stdlib.h>
-#include <stdio.h>
-#include <string.h>
-
-int crypton_cpu_has_rdrand()
-{
-	uint32_t ax,bx,cx,dx,func=1;
-#if defined(__PIC__) && defined(__i386__)
-	__asm__ volatile ("mov %%ebx, %%edi;" "cpuid;" "xchgl %%ebx, %%edi;"
-		: "=a" (ax), "=D" (bx), "=c" (cx), "=d" (dx) : "a" (func));
-#else
-	__asm__ volatile ("cpuid": "=a" (ax), "=b" (bx), "=c" (cx), "=d" (dx) : "a" (func));
-#endif
-	return (cx & 0x40000000);
-}
-
-/* inline encoding of 'rdrand %rax' to cover old binutils
- * - no inputs
- * - 'cc' to the clobber list as we modify condition code.
- * - output of rdrand in rax and have a 8 bit error condition
- */
-#define inline_rdrand_rax(val, err) \
-	asm(".byte 0x48,0x0f,0xc7,0xf0; setc %1" \
-	   : "=a" (val), "=q" (err) \
-	   : \
-	   : "cc")
-
-/* inline encoding of 'rdrand %eax' to cover old binutils
- * - no inputs
- * - 'cc' to the clobber list as we modify condition code.
- * - output of rdrand in eax and have a 8 bit error condition
- */
-#define inline_rdrand_eax(val, err) \
-	asm(".byte 0x0f,0xc7,0xf0; setc %1" \
-	   : "=a" (val), "=q" (err) \
-	   : \
-	   : "cc")
-
-#ifdef __x86_64__
-# define RDRAND_SZ 8
-# define RDRAND_T  uint64_t
-#define inline_rdrand(val, err) err = crypton_rdrand_step(&val)
-#else
-# define RDRAND_SZ 4
-# define RDRAND_T  uint32_t
-#define inline_rdrand(val, err) err = crypton_rdrand_step(&val)
-#endif
-
-/* sadly many people are still using an old binutils,
- * leading to report that instruction is not recognized.
- */
-#if 1
-/* Returns 1 on success */
-static inline int crypton_rdrand_step(RDRAND_T *buffer)
-{
-	unsigned char err;
-	asm volatile ("rdrand %0; setc %1" : "=r" (*buffer), "=qm" (err));
-	return (int) err;
-}
-#endif
-
-/* Returns the number of bytes successfully generated */
-int crypton_get_rand_bytes(uint8_t *buffer, size_t len)
-{
-	RDRAND_T tmp;
-	int aligned = (intptr_t) buffer % RDRAND_SZ;
-	int orig_len = len;
-	int to_alignment = RDRAND_SZ - aligned;
-	uint8_t ok;
-
-	if (aligned != 0) {
-		inline_rdrand(tmp, ok);
-		if (!ok)
-			return 0;
-		memcpy(buffer, (uint8_t *) &tmp, to_alignment);
-		buffer += to_alignment;
-		len -= to_alignment;
-	}
-
-	for (; len >= RDRAND_SZ; buffer += RDRAND_SZ, len -= RDRAND_SZ) {
-		inline_rdrand(tmp, ok);
-		if (!ok)
-			return (orig_len - len);
-		*((RDRAND_T *) buffer) = tmp;
-	}
-
-	if (len > 0) {
-		inline_rdrand(tmp, ok);
-		if (!ok)
-			return (orig_len - len);
-		memcpy(buffer, (uint8_t *) &tmp, len);
-	}
-	return orig_len;
-}
diff --git a/cbits/crypton_sysdrg.c b/cbits/crypton_sysdrg.c
new file mode 100644
--- /dev/null
+++ b/cbits/crypton_sysdrg.c
@@ -0,0 +1,388 @@
+/*
+ * The generator behind MonadRandom IO.
+ *
+ * A ChaCha20 DRBG per operating system thread, seeded from a process-wide
+ * DRBG, which is itself seeded from the system entropy pool.  This is the
+ * shape RFC 9180's neighbours and the other libraries have settled on, and
+ * it was asked for in #298.
+ *
+ * Per operating system thread and not per Haskell thread: a forkIO thread
+ * moves between capabilities, so state kept against it would be shared by
+ * threads running at the same time.  That is why the state is here and
+ * reached through pthread_getspecific rather than held in Haskell.
+ *
+ * Three things force a reseed:
+ *
+ *   - a thread has produced CRYPTON_THREAD_RESEED bytes,
+ *   - the process DRBG has issued CRYPTON_GLOBAL_RESEED bytes of seed,
+ *   - the process has forked.
+ *
+ * The last one is the one that bites.  A child inherits its parent's state
+ * and would otherwise produce the same stream; the generation counter below
+ * is bumped in the child by a pthread_atfork handler, and every generator
+ * compares against it before it answers.
+ */
+
+#include <stdint.h>
+#include <string.h>
+#include <stdlib.h>
+
+#include "crypton_chacha.h"
+#include "crypton_sha512.h"
+
+#ifdef _WIN32
+#include <windows.h>
+#else
+#include <pthread.h>
+#include <unistd.h>
+#endif
+
+/* from crypton_sysrandom.c */
+int crypton_sysrandom_available(void);
+int crypton_sysrandom_bytes(uint8_t *buf, int len);
+
+#define CHACHA_ROUNDS      20
+#define SEED_KEY           32
+#define SEED_IV             8
+#define SEED_LEN           (SEED_KEY + SEED_IV)
+
+#define CRYPTON_THREAD_RESEED  (1u << 20)
+#define CRYPTON_GLOBAL_RESEED  (1u << 20)
+
+typedef struct {
+	crypton_chacha_context ctx;
+	uint64_t used;
+	uint32_t generation;
+	int seeded;
+} drg_t;
+
+static drg_t global_drg;
+static volatile uint32_t fork_generation = 0;
+
+#ifdef _WIN32
+static CRITICAL_SECTION global_lock;
+/* Fls and not Tls: TlsAlloc has no destructor, so the state of every thread
+ * that ever drew a byte would be left allocated and unscrubbed when the
+ * thread ended.  FlsAlloc takes the callback that pthread_key_create does. */
+static DWORD thread_slot;
+static INIT_ONCE init_once = INIT_ONCE_STATIC_INIT;
+#else
+static pthread_mutex_t global_lock = PTHREAD_MUTEX_INITIALIZER;
+static pthread_key_t thread_slot;
+static pthread_once_t init_once = PTHREAD_ONCE_INIT;
+#endif
+
+static void scrub(void *p, size_t n)
+{
+	volatile uint8_t *q = (volatile uint8_t *) p;
+	while (n--) *q++ = 0;
+}
+
+/*
+ * Seed material for the process DRBG.
+ *
+ * What the system gives goes through SHA-512 rather than into the key
+ * directly, so that the key is a fixed size whatever the call returns, and
+ * so that a second source could be added without any of this changing shape.
+ * There is one source today and the note below says why.
+ */
+static int seed_from_system(uint8_t out[SEED_LEN])
+{
+	struct sha512_ctx h;
+	uint8_t buf[64];
+	uint8_t digest[64];
+	int got;
+
+	crypton_sha512_init(&h);
+
+	/*
+	 * The system call only.  Where there is none -- an old kernel, a BSD
+	 * this does not know -- seeding fails and the caller keeps the path
+	 * it has, rather than this file growing a second copy of the device
+	 * reading that Crypto.Random.Entropy.Unix already does.
+	 *
+	 * And RDRAND is not mixed in beside it, which it was until
+	 * @vdukhovni pointed out that it cannot add anything: every system
+	 * this code runs on already feeds RDRAND into the pool the call below
+	 * draws from.  Linux does that whatever random.trust_cpu says -- that
+	 * setting decides whether the contribution is *credited*, not whether
+	 * it is mixed -- so the instruction's output is in this buffer
+	 * already.
+	 *
+	 * The argument for keeping it was defence in depth against that pool
+	 * having gone wrong.  It does not survive the above: a pool that has
+	 * gone wrong has gone wrong with RDRAND already in it, so asking the
+	 * instruction a second time covers only the case where the kernel's
+	 * mixing is broken and the instruction is not.  That is not worth a
+	 * second code path, a flag, and the standing invitation to read this
+	 * as a second source when it is the same one twice.
+	 */
+	got = crypton_sysrandom_available() ? crypton_sysrandom_bytes(buf, 64) : 0;
+	if (got <= 0) {
+		/* Nothing else here is a seed on its own, so there is nothing to
+		 * go on with: return before drawing anything that would only be
+		 * thrown away. */
+		scrub(&h, sizeof h);
+		scrub(buf, sizeof buf);
+		return 0;
+	}
+	crypton_sha512_update(&h, buf, (uint32_t) got);
+
+
+	crypton_sha512_finalize(&h, digest);
+	memcpy(out, digest, SEED_LEN);
+
+	scrub(&h, sizeof h);
+	scrub(buf, sizeof buf);
+	scrub(digest, sizeof digest);
+	return 1;
+}
+
+static void drg_seed(drg_t *d, const uint8_t seed[SEED_LEN])
+{
+	crypton_chacha_init(&d->ctx, CHACHA_ROUNDS, SEED_KEY, seed,
+	                    SEED_IV, seed + SEED_KEY);
+	d->used = 0;
+	d->generation = fork_generation;
+	d->seeded = 1;
+}
+
+/*
+ * Forget the key that produced the bytes just handed out.
+ *
+ * Without this the key stands until the next reseed, and anyone who reads a
+ * generator's state can wind the counter back and reproduce everything it
+ * has issued since -- up to CRYPTON_THREAD_RESEED bytes that were meant to
+ * be secret.  Taking the next forty bytes of keystream as the new key and
+ * nonce, and dropping the old ones, puts that out of reach one step after
+ * it is issued: ChaCha20 does not run backwards, and the key that would
+ * have been needed is gone.  It is what arc4random does.
+ *
+ * crypton_chacha_init memsets the whole context, so the counter returns to
+ * zero and the tail of a part-used block goes with the old key rather than
+ * being handed out under the new one.
+ *
+ * d->used is not touched.  It counts what callers were given, which is what
+ * the reseed interval is written in terms of; these forty bytes are the
+ * cost of the rekey and not an answer to anybody.
+ */
+static void drg_rekey(drg_t *d)
+{
+	uint8_t next[SEED_LEN];
+
+	crypton_chacha_generate(next, &d->ctx, SEED_LEN);
+	crypton_chacha_init(&d->ctx, CHACHA_ROUNDS, SEED_KEY, next,
+	                    SEED_IV, next + SEED_KEY);
+	scrub(next, sizeof next);
+}
+
+static int drg_stale(const drg_t *d, uint64_t limit)
+{
+	return !d->seeded || d->used >= limit || d->generation != fork_generation;
+}
+
+/* Bytes from the process DRBG, which is only ever asked for seed material. */
+static int global_bytes(uint8_t *out, uint32_t len)
+{
+	int ok = 1;
+
+#ifdef _WIN32
+	EnterCriticalSection(&global_lock);
+#else
+	pthread_mutex_lock(&global_lock);
+#endif
+	if (drg_stale(&global_drg, CRYPTON_GLOBAL_RESEED)) {
+		uint8_t seed[SEED_LEN];
+		ok = seed_from_system(seed);
+		if (ok)
+			drg_seed(&global_drg, seed);
+		scrub(seed, sizeof seed);
+	}
+	if (ok) {
+		crypton_chacha_generate(out, &global_drg.ctx, len);
+		global_drg.used += len;
+		drg_rekey(&global_drg);
+	}
+#ifdef _WIN32
+	LeaveCriticalSection(&global_lock);
+#else
+	pthread_mutex_unlock(&global_lock);
+#endif
+	return ok;
+}
+
+#ifdef _WIN32
+static void WINAPI thread_free(void *p)
+#else
+static void thread_free(void *p)
+#endif
+{
+	if (p) {
+		scrub(p, sizeof(drg_t));
+		free(p);
+	}
+}
+
+#ifndef _WIN32
+/* All three handlers, not just the child's.  The child's first draw has to
+ * reseed -- the generation has changed -- and reseeding takes global_lock.
+ * A fork made while another thread held it would hand the child a mutex
+ * locked by a thread that did not come across, and the child would wait on
+ * it for ever.  So the lock is taken before the fork and released on both
+ * sides of it, which is the state the child needs it in. */
+static void before_fork(void)
+{
+	pthread_mutex_lock(&global_lock);
+}
+
+static void after_fork_in_parent(void)
+{
+	pthread_mutex_unlock(&global_lock);
+}
+
+static void after_fork_in_child(void)
+{
+	pthread_mutex_unlock(&global_lock);
+	fork_generation++;
+}
+#endif
+
+#ifdef CRYPTON_SYSDRG_TESTING
+/* For cbits/tests/sysdrg and nothing else: hold and release the process
+ * generator's lock, so that a fork can be made to happen while another
+ * thread holds it.  Nothing outside that test declares these, and the
+ * library is never built with this defined. */
+void crypton_sysdrg_test_lock(void);
+void crypton_sysdrg_test_unlock(void);
+
+static drg_t *this_thread(void);
+
+/* The calling thread's ChaCha key, which is d[4..11] of the state.  A test
+ * uses it to ask whether the key that produced a draw is still there
+ * afterwards; see "a draw replaces the key that made it" in
+ * cbits/tests/sysdrg. */
+void crypton_sysdrg_test_key(uint8_t out[32]);
+
+void crypton_sysdrg_test_key(uint8_t out[32])
+{
+	drg_t *d = this_thread();
+	int i;
+
+	if (!d)
+		return;
+	for (i = 0; i < 8; i++) {
+		uint32_t w = d->ctx.st.d[4 + i];
+		out[i * 4 + 0] = (uint8_t) (w);
+		out[i * 4 + 1] = (uint8_t) (w >> 8);
+		out[i * 4 + 2] = (uint8_t) (w >> 16);
+		out[i * 4 + 3] = (uint8_t) (w >> 24);
+	}
+}
+
+void crypton_sysdrg_test_lock(void)
+{
+#ifndef _WIN32
+	pthread_mutex_lock(&global_lock);
+#endif
+}
+
+void crypton_sysdrg_test_unlock(void)
+{
+#ifndef _WIN32
+	pthread_mutex_unlock(&global_lock);
+#endif
+}
+#endif
+
+#ifdef _WIN32
+static BOOL CALLBACK init_slot(PINIT_ONCE o, PVOID p, PVOID *c)
+{
+	(void) o; (void) p; (void) c;
+	InitializeCriticalSection(&global_lock);
+	thread_slot = FlsAlloc(thread_free);
+	return TRUE;
+}
+#else
+static void init_slot(void)
+{
+	pthread_key_create(&thread_slot, thread_free);
+	pthread_atfork(before_fork, after_fork_in_parent, after_fork_in_child);
+}
+#endif
+
+static drg_t *this_thread(void)
+{
+	drg_t *d;
+
+#ifdef _WIN32
+	InitOnceExecuteOnce(&init_once, init_slot, NULL, NULL);
+	if (thread_slot == FLS_OUT_OF_INDEXES)
+		return NULL;
+	d = (drg_t *) FlsGetValue(thread_slot);
+#else
+	pthread_once(&init_once, init_slot);
+	d = (drg_t *) pthread_getspecific(thread_slot);
+#endif
+	if (!d) {
+		d = (drg_t *) calloc(1, sizeof(drg_t));
+		if (!d)
+			return NULL;
+#ifdef _WIN32
+		if (!FlsSetValue(thread_slot, d)) {
+			free(d);
+			return NULL;
+		}
+#else
+		if (pthread_setspecific(thread_slot, d) != 0) {
+			free(d);
+			return NULL;
+		}
+#endif
+	}
+	return d;
+}
+
+/* Returns the number of bytes written, which is len unless there was no
+ * seed to be had -- and then it is 0, so that the caller can say so rather
+ * than hand back a buffer it cannot vouch for. */
+int crypton_sysdrg_bytes(uint8_t *out, int len)
+{
+	drg_t *d;
+
+	if (len < 0)
+		return 0;
+	d = this_thread();
+	if (!d)
+		return 0;
+
+	if (drg_stale(d, CRYPTON_THREAD_RESEED)) {
+		uint8_t seed[SEED_LEN];
+		int ok = global_bytes(seed, SEED_LEN);
+		if (ok)
+			drg_seed(d, seed);
+		scrub(seed, sizeof seed);
+		if (!ok)
+			return 0;
+	}
+
+	crypton_chacha_generate(out, &d->ctx, (uint32_t) len);
+	d->used += (uint64_t) len;
+	drg_rekey(d);
+	return len;
+}
+
+/* For the tests: how many times the process has been seen to fork. */
+uint32_t crypton_sysdrg_generation(void)
+{
+	return fork_generation;
+}
+
+/* For the tests: bytes this thread's generator has produced since it was
+ * last seeded.  A generator shared between threads would carry the first
+ * thread's count into the second; a per-thread one starts again at zero,
+ * and that is the only way from outside to tell the two apart. */
+uint64_t crypton_sysdrg_thread_used(void)
+{
+	drg_t *d = this_thread();
+	return d ? d->used : 0;
+}
diff --git a/cbits/crypton_sysrandom.c b/cbits/crypton_sysrandom.c
new file mode 100644
--- /dev/null
+++ b/cbits/crypton_sysrandom.c
@@ -0,0 +1,119 @@
+/*
+ * The kernel's own random number generator, reached without a file
+ * descriptor: getrandom(2) on Linux and FreeBSD, getentropy(3) on the
+ * systems that have that instead.
+ *
+ * This is preferred over reading /dev/urandom because it needs no path and
+ * no descriptor, so it still works where /dev is not mounted or not
+ * populated -- a minimal container, a chroot, a sandbox -- and it cannot be
+ * defeated by exhausting the descriptor table.
+ *
+ * Availability is decided at run time as well as at compile time: a binary
+ * built against headers that declare getrandom can still run on a kernel
+ * that does not implement it, and says so with ENOSYS.
+ */
+
+#include <stddef.h>
+#include <stdint.h>
+#include <errno.h>
+
+#if defined(__linux__)
+#include <unistd.h>
+#include <sys/syscall.h>
+#ifdef SYS_getrandom
+#define CRYPTON_SYSRANDOM_GETRANDOM 1
+#endif
+#elif defined(__FreeBSD__)
+#include <sys/param.h>
+#if __FreeBSD_version >= 1200000
+#include <sys/random.h>
+#define CRYPTON_SYSRANDOM_GETRANDOM 1
+#endif
+#elif defined(__APPLE__)
+/* getentropy(3) is declared in sys/random.h on Darwin, since 10.12 -- but
+ * macOS only.  Crypto.Random says iOS does not allow it, which is why that
+ * platform builds with INSECURE_ENTROPY; until someone can try it there,
+ * iOS keeps the path it has rather than gaining an untested one. */
+#include <TargetConditionals.h>
+#if defined(TARGET_OS_OSX) && TARGET_OS_OSX
+#include <sys/random.h>
+#define CRYPTON_SYSRANDOM_GETENTROPY 1
+#endif
+#elif defined(__OpenBSD__)
+#include <unistd.h>
+#define CRYPTON_SYSRANDOM_GETENTROPY 1
+#elif defined(__NetBSD__)
+#include <sys/param.h>
+#if __NetBSD_Version__ >= 1000000000
+#include <sys/random.h>
+#define CRYPTON_SYSRANDOM_GETENTROPY 1
+#endif
+#endif
+
+/* getentropy(3) refuses more than 256 bytes in one call. */
+#define CRYPTON_GETENTROPY_MAX 256
+
+#if defined(CRYPTON_SYSRANDOM_GETRANDOM)
+
+static int sysrandom_once(uint8_t *buf, size_t n)
+{
+#if defined(__linux__)
+	return (int) syscall(SYS_getrandom, buf, n, 0);
+#else
+	return (int) getrandom(buf, n, 0);
+#endif
+}
+
+#elif defined(CRYPTON_SYSRANDOM_GETENTROPY)
+
+static int sysrandom_once(uint8_t *buf, size_t n)
+{
+	if (n > CRYPTON_GETENTROPY_MAX)
+		n = CRYPTON_GETENTROPY_MAX;
+	if (getentropy(buf, n) != 0)
+		return -1;
+	return (int) n;
+}
+
+#else
+
+static int sysrandom_once(uint8_t *buf, size_t n)
+{
+	(void) buf; (void) n;
+	errno = ENOSYS;
+	return -1;
+}
+
+#endif
+
+/* Is the call there, on this kernel, right now?  A zero-length request
+ * answers that without consuming anything. */
+int crypton_sysrandom_available(void)
+{
+	uint8_t b;
+	int r = sysrandom_once(&b, 0);
+	return r < 0 ? 0 : 1;
+}
+
+/* Fill the buffer.  Returns the number of bytes written, which is n unless
+ * the call failed for a reason other than being interrupted. */
+int crypton_sysrandom_bytes(uint8_t *buf, int len)
+{
+	size_t n = (size_t) len;
+	size_t done = 0;
+
+	if (len < 0)
+		return 0;
+	while (done < n) {
+		int r = sysrandom_once(buf + done, n - done);
+		if (r < 0) {
+			if (errno == EINTR)
+				continue;
+			break;
+		}
+		if (r == 0)
+			break;
+		done += (size_t) r;
+	}
+	return (int) done;
+}
diff --git a/cbits/tests/bearssl_diff.c b/cbits/tests/bearssl_diff.c
new file mode 100644
--- /dev/null
+++ b/cbits/tests/bearssl_diff.c
@@ -0,0 +1,432 @@
+/*
+ * crypton's portable AES and GHASH against the BearSSL they are written on.
+ *
+ * This began as the migration check: for one commit, cbits/aes/generic.c and
+ * cbits/aes/gf.c were still the table-driven implementations, and this said
+ * the vendored code computed what they computed before they were replaced.
+ *
+ * Since the replacement the two sides are no longer independent, and what is
+ * left is still worth checking: everything in generic.c and gf.c is now
+ * crypton's own glue -- a schedule kept compressed and carried through
+ * memcpy, the interleave-and-ortho idiom around a single block, three of four
+ * lanes left idle, and a GHASH entry that reaches the same multiply by handing
+ * it a block of zeros.  Each of those is somewhere a mistake would live, and
+ * each is compared here against calling BearSSL directly.
+ *
+ * FIPS-197's own vectors come first either way, so that agreement means AES
+ * rather than two callers agreeing on something that is not.
+ *
+ * Run with an argument to corrupt results on purpose and see the comparison
+ * notice: a differential test that cannot fail has said nothing.
+ */
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <stdint.h>
+
+#include "bearssl/inner.h"
+#include "crypton_aes.h"
+#include "aes/generic.h"
+#include "aes/gf.h"
+#include "aes/block128.h"
+
+/*
+ * crypton_aes.c's table, which is a global, and the three key sizes of the
+ * two CTR entries this looks for in it.  Searching rather than indexing
+ * because the index is an enum private to that file.
+ */
+#define BRANCH_TABLE_SEARCH 64
+extern void *crypton_aes_branch_table[];
+
+/* the two GCM implementations, which crypton_aes.c declares but no header
+ * does: in this build the generic one reaches the portable block function
+ * too, so the two have to agree exactly, tag and all */
+void crypton_aes_generic_gcm_encrypt(uint8_t *, aes_gcm *, aes_key *, uint8_t *, uint32_t);
+void crypton_aes_generic_gcm_decrypt(uint8_t *, aes_gcm *, aes_key *, uint8_t *, uint32_t);
+void crypton_aes_bitsliced_gcm_encrypt(uint8_t *, aes_gcm *, aes_key *, uint8_t *, uint32_t);
+void crypton_aes_bitsliced_gcm_decrypt(uint8_t *, aes_gcm *, aes_key *, uint8_t *, uint32_t);
+
+/* ---- the vendored code, one block at a time, as aes_ct64_cbcenc.c does ---- */
+
+static void bear_key(uint64_t *comp_skey, unsigned *nr,
+                     const uint8_t *key, size_t len)
+{
+	*nr = br_aes_ct64_keysched(comp_skey, key, len);
+}
+
+static void bear_block(uint8_t *out, const uint64_t *comp_skey, unsigned nr,
+                       const uint8_t *in, int decrypt)
+{
+	uint64_t sk_exp[120];
+	uint32_t w[4];
+	uint64_t q[8];
+
+	br_aes_ct64_skey_expand(sk_exp, nr, comp_skey);
+	w[0] = br_dec32le(in);
+	w[1] = br_dec32le(in + 4);
+	w[2] = br_dec32le(in + 8);
+	w[3] = br_dec32le(in + 12);
+	memset(q, 0, sizeof q);
+	br_aes_ct64_interleave_in(&q[0], &q[4], w);
+	br_aes_ct64_ortho(q);
+	if (decrypt)
+		br_aes_ct64_bitslice_decrypt(nr, sk_exp, q);
+	else
+		br_aes_ct64_bitslice_encrypt(nr, sk_exp, q);
+	br_aes_ct64_ortho(q);
+	br_aes_ct64_interleave_out(w, q[0], q[4]);
+	br_enc32le(out, w[0]);
+	br_enc32le(out + 4, w[1]);
+	br_enc32le(out + 8, w[2]);
+	br_enc32le(out + 12, w[3]);
+}
+
+/* ---- crypton's GHASH, driven the way crypton_aes.c drives it ---- */
+
+static void crypton_ghash(uint8_t *y, const uint8_t *h,
+                          const uint8_t *data, size_t len)
+{
+	table_4bit ht;
+	block128 acc;
+	size_t i;
+
+	crypton_aes_generic_hinit(ht, (const block128 *) h);
+	block128_zero(&acc);
+	for (i = 0; i < len; i += 16) {
+		block128_xor_bytes(&acc, data + i, 16);
+		crypton_aes_generic_gf_mul(&acc, ht);
+	}
+	memcpy(y, &acc, 16);
+}
+
+/* ---- the comparison ---- */
+
+static int failures;
+static int sabotage;
+
+static void same(const char *what, const uint8_t *a, const uint8_t *b, size_t n)
+{
+	if (memcmp(a, b, n) != 0) {
+		size_t i;
+
+		printf("  MISMATCH %s\n    bearssl ", what);
+		for (i = 0; i < n; i++) printf("%02x", a[i]);
+		printf("\n    crypton ");
+		for (i = 0; i < n; i++) printf("%02x", b[i]);
+		printf("\n");
+		failures++;
+	}
+}
+
+static uint32_t rnd_state = 1;
+static uint8_t rnd(void)
+{
+	rnd_state = rnd_state * 1103515245u + 12345u;
+	return (uint8_t)(rnd_state >> 16);
+}
+static void rnd_fill(uint8_t *p, size_t n)
+{
+	while (n--) *p++ = rnd();
+}
+
+int main(int argc, char **argv)
+{
+	/* FIPS-197 C.1, C.2, C.3 */
+	static const uint8_t pt[16] = {
+		0x00,0x11,0x22,0x33,0x44,0x55,0x66,0x77,
+		0x88,0x99,0xaa,0xbb,0xcc,0xdd,0xee,0xff };
+	static const uint8_t k128[16] = {
+		0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15 };
+	static const uint8_t k192[24] = {
+		0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23 };
+	static const uint8_t k256[32] = {
+		0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,
+		16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31 };
+	static const uint8_t c128[16] = {
+		0x69,0xc4,0xe0,0xd8,0x6a,0x7b,0x04,0x30,
+		0xd8,0xcd,0xb7,0x80,0x70,0xb4,0xc5,0x5a };
+	static const uint8_t c192[16] = {
+		0xdd,0xa9,0x7c,0xa4,0x86,0x4c,0xdf,0xe0,
+		0x6e,0xaf,0x70,0xa0,0xec,0x0d,0x71,0x91 };
+	static const uint8_t c256[16] = {
+		0x8e,0xa2,0xb7,0xca,0x51,0x67,0x45,0xbf,
+		0xea,0xfc,0x49,0x90,0x4b,0x49,0x60,0x89 };
+	const uint8_t *keys[3] = { k128, k192, k256 };
+	const uint8_t *cts[3]  = { c128, c192, c256 };
+	size_t klens[3] = { 16, 24, 32 };
+	uint64_t comp[30];
+	unsigned nr;
+	uint8_t a[16], b[16];
+	int i, round;
+
+	sabotage = argc > 1;
+	printf("== FIPS-197, so that agreement means AES ==\n");
+	for (i = 0; i < 3; i++) {
+		bear_key(comp, &nr, keys[i], klens[i]);
+		bear_block(a, comp, nr, pt, 0);
+		if (sabotage && i == 1) a[0] ^= 1;
+		same("bearssl vs FIPS-197", a, cts[i], 16);
+		bear_block(b, comp, nr, cts[i], 1);
+		same("bearssl decrypt vs plaintext", b, pt, 16);
+	}
+
+	printf("== crypton's glue vs BearSSL direct, 2000 random keys and blocks ==\n");
+	for (round = 0; round < 2000; round++) {
+		uint8_t key[32], in[16], e1[16], e2[16], d1[16], d2[16];
+		size_t kl = klens[round % 3];
+		aes_key ck;
+
+		rnd_fill(key, kl);
+		rnd_fill(in, 16);
+
+		bear_key(comp, &nr, key, kl);
+		bear_block(e1, comp, nr, in, 0);
+		bear_block(d1, comp, nr, in, 1);
+
+		crypton_aes_generic_init(&ck, key, (uint8_t) kl);
+		crypton_aes_generic_encrypt_block((aes_block *) e2, &ck,
+		                                  (aes_block *) in);
+		crypton_aes_generic_decrypt_block((aes_block *) d2, &ck,
+		                                  (aes_block *) in);
+		if (sabotage && round == 7) e1[3] ^= 0x10;
+		same("encrypt", e1, e2, 16);
+		same("decrypt", d1, d2, 16);
+	}
+
+	printf("== GHASH: crypton's entries vs br_ghash_ctmul64, 500 messages ==\n");
+	for (round = 0; round < 500; round++) {
+		uint8_t h[16], data[256], y1[16], y2[16];
+		size_t len = 16u * (size_t)(1 + (round % 16));
+
+		rnd_fill(h, 16);
+		rnd_fill(data, len);
+
+		memset(y1, 0, 16);
+		br_ghash_ctmul64(y1, h, data, len);
+		crypton_ghash(y2, h, data, len);
+		if (sabotage && round == 3) y1[15] ^= 0x80;
+		same("ghash", y1, y2, 16);
+	}
+
+	printf("== gf_mul4: the four-block entry against the same four blocks ==\n");
+	for (round = 0; round < 500; round++) {
+		uint8_t h[16], data[64], y1[16];
+		table_4bit ht;
+		block128 acc;
+
+		rnd_fill(h, 16);
+		rnd_fill(data, sizeof data);
+
+		memset(y1, 0, 16);
+		br_ghash_ctmul64(y1, h, data, sizeof data);
+
+		crypton_aes_generic_hinit(ht, (const block128 *) h);
+		block128_zero(&acc);
+		crypton_aes_generic_gf_mul4(&acc, (const block128 *) data, ht);
+		if (sabotage && round == 11) y1[0] ^= 0x40;
+		same("gf_mul4", y1, (const uint8_t *) &acc, 16);
+	}
+
+	/*
+	 * And that the wide entries are the ones installed.  Nothing else
+	 * here would notice if they were not: the entries they replace are
+	 * correct too, just a block at a time, so every answer above would
+	 * be the same and only the speed would be gone.
+	 */
+	/*
+	 * The Haskell suite reaches the four-block pass, but thinly: its
+	 * vectors are mostly a block or three long, and breaking that pass
+	 * alone fails sixteen of its examples where breaking every pass
+	 * fails seven hundred.  So the lane logic and the counter are
+	 * covered here instead, where the lengths can be chosen.
+	 */
+	printf("== many blocks at a pass against one at a time ==\n");
+	for (round = 0; round < 200; round++) {
+		uint8_t key[32], in[16 * 9], wide[16 * 9], single[16 * 9];
+		size_t kl = klens[round % 3];
+		uint32_t nb = 1 + (round % 9);
+		aes_sched sched;
+		aes_key ck;
+		uint32_t i;
+		int dec = round & 1;
+
+		rnd_fill(key, kl);
+		rnd_fill(in, nb * 16);
+		crypton_aes_generic_init(&ck, key, (uint8_t) kl);
+
+		crypton_aes_generic_schedule(&sched, &ck);
+		crypton_aes_generic_blocks(wide, in, nb, &sched, dec);
+
+		for (i = 0; i < nb; i++) {
+			if (dec)
+				crypton_aes_generic_decrypt_block(
+				    (aes_block *) (single + 16 * i), &ck,
+				    (aes_block *) (in + 16 * i));
+			else
+				crypton_aes_generic_encrypt_block(
+				    (aes_block *) (single + 16 * i), &ck,
+				    (aes_block *) (in + 16 * i));
+		}
+		if (sabotage && round == 5) wide[16] ^= 2;
+		same("wide pass", wide, single, nb * 16);
+	}
+
+	printf("== the four-block CTR against one block at a time ==\n");
+	for (round = 0; round < 200; round++) {
+		uint8_t key[32], iv[16], in[200], got[200], want[200];
+		size_t kl = klens[round % 3];
+		/* lengths that land on, before and after a group of four */
+		uint32_t len = 1 + (round % 200);
+		aes_key ck;
+		aes_block counter, ks;
+		uint32_t done, n, i;
+		int c32 = round & 1;
+
+		rnd_fill(key, kl);
+		rnd_fill(iv, 16);
+		rnd_fill(in, len);
+		crypton_aes_generic_init(&ck, key, (uint8_t) kl);
+
+		if (c32)
+			crypton_aes_bitsliced_encrypt_c32(got, &ck,
+			    (aes_block *) iv, in, len);
+		else
+			crypton_aes_bitsliced_encrypt_ctr(got, &ck,
+			    (aes_block *) iv, in, len);
+
+		block128_copy(&counter, (block128 *) iv);
+		for (done = 0; done < len; done += 16) {
+			crypton_aes_generic_encrypt_block(&ks, &ck, &counter);
+			n = len - done < 16 ? len - done : 16;
+			for (i = 0; i < n; i++)
+				want[done + i] = ((uint8_t *) &ks)[i]
+				               ^ in[done + i];
+			if (c32)
+				block128_inc32_le(&counter);
+			else
+				block128_inc_be(&counter);
+		}
+		if (sabotage && round == 9) got[len - 1] ^= 4;
+		same("ctr", got, want, len);
+	}
+
+	printf("== XTS four at a pass against one at a time ==\n");
+	for (round = 0; round < 200; round++) {
+		uint8_t key[32], key2[32], du[16], in[16 * 11];
+		uint8_t got[16 * 11], want[16 * 11];
+		size_t kl = klens[round % 3];
+		uint32_t nb = 1 + (round % 11);
+		uint32_t spoint = round % 3;
+		aes_key k1, k2;
+		aes_block tweak;
+		uint32_t i, j;
+		int dec = round & 1;
+
+		rnd_fill(key, kl);
+		rnd_fill(key2, kl);
+		rnd_fill(du, 16);
+		rnd_fill(in, nb * 16);
+		crypton_aes_initkey(&k1, key, (uint8_t) kl);
+		crypton_aes_initkey(&k2, key2, (uint8_t) kl);
+
+		{
+			aes_block d;
+			memcpy(&d, du, 16);
+			if (dec)
+				crypton_aes_decrypt_xts((aes_block *) got, &k1, &k2,
+				    &d, spoint, (aes_block *) in, nb);
+			else
+				crypton_aes_encrypt_xts((aes_block *) got, &k1, &k2,
+				    &d, spoint, (aes_block *) in, nb);
+		}
+
+		/* the same thing a block at a time */
+		memcpy(&tweak, du, 16);
+		crypton_aes_generic_encrypt_block(&tweak, &k2, &tweak);
+		for (j = 0; j < spoint; j++)
+			crypton_aes_generic_gf_mulx((block128 *) &tweak);
+		for (i = 0; i < nb; i++) {
+			aes_block t;
+
+			block128_vxor(&t, (block128 *) (in + 16 * i), &tweak);
+			if (dec)
+				crypton_aes_generic_decrypt_block(&t, &k1, &t);
+			else
+				crypton_aes_generic_encrypt_block(&t, &k1, &t);
+			block128_vxor((block128 *) (want + 16 * i), &t, &tweak);
+			crypton_aes_generic_gf_mulx((block128 *) &tweak);
+		}
+		if (sabotage && round == 17) got[16] ^= 8;
+		same("xts", got, want, nb * 16);
+	}
+
+	printf("== the four-block GCM against the one-block GCM ==\n");
+	for (round = 0; round < 200; round++) {
+		uint8_t key[32], iv[12], in[300], ga[300], gb[300];
+		size_t kl = klens[round % 3];
+		uint32_t len = 1 + (round % 300);
+		aes_key ck;
+		aes_gcm g1, g2;
+		int dec = round & 1;
+
+		rnd_fill(key, kl);
+		rnd_fill(iv, sizeof iv);
+		rnd_fill(in, len);
+		crypton_aes_initkey(&ck, key, (uint8_t) kl);
+		crypton_aes_gcm_init(&g1, &ck, iv, sizeof iv);
+		memcpy(&g2, &g1, sizeof g1);
+
+		if (dec) {
+			crypton_aes_generic_gcm_decrypt(ga, &g1, &ck, in, len);
+			crypton_aes_bitsliced_gcm_decrypt(gb, &g2, &ck, in, len);
+		} else {
+			crypton_aes_generic_gcm_encrypt(ga, &g1, &ck, in, len);
+			crypton_aes_bitsliced_gcm_encrypt(gb, &g2, &ck, in, len);
+		}
+		if (sabotage && round == 13) gb[0] ^= 1;
+		same("gcm text", ga, gb, len);
+		/* the running GHASH and counter, which the tag is made from */
+		same("gcm state", (const uint8_t *) &g1, (const uint8_t *) &g2,
+		     sizeof g1);
+	}
+
+	printf("== which implementation the build chose ==\n");
+#if !defined(WITH_AESNI) && !defined(WITH_ARMV8_CRYPTO)
+	/*
+	 * With no accelerator compiled in, crypton_aes.c does not read the
+	 * branch table at all -- its GET_ macros name the portable entries
+	 * directly -- so there is nothing here to look at, and which
+	 * implementation runs is settled by the preprocessor.  Scanning the
+	 * table in this build was a check that passed while telling nothing,
+	 * which is how the four-block CTR came to be written, installed, and
+	 * never called.
+	 */
+	printf("  named at compile time; the table is not read in this build\n");
+	(void) crypton_aes_branch_table;
+	if (sabotage) { /* nothing to corrupt here */ }
+#else
+	{
+		int i, ctr = 0, c32 = 0;
+
+		for (i = 0; i < BRANCH_TABLE_SEARCH; i++) {
+			if (crypton_aes_branch_table[i] ==
+			    (void *) crypton_aes_bitsliced_encrypt_ctr)
+				ctr++;
+			if (crypton_aes_branch_table[i] ==
+			    (void *) crypton_aes_bitsliced_encrypt_c32)
+				c32++;
+		}
+		printf("  CTR entries %d, C32 entries %d\n", ctr, c32);
+		if (sabotage) { ctr = 0; }
+		if (ctr != 3 || c32 != 3) {
+			printf("  MISMATCH the portable CTR entries were not installed\n");
+			failures++;
+		}
+	}
+#endif
+
+	printf("%s: %d mismatch(es)%s\n",
+	       failures ? "FAIL" : "ok", failures,
+	       sabotage ? "  (sabotage was asked for)" : "");
+	return failures != 0;
+}
diff --git a/cbits/tests/ct/README b/cbits/tests/ct/README
--- a/cbits/tests/ct/README
+++ b/cbits/tests/ct/README
@@ -26,29 +26,33 @@
   chapoly  the ChaCha20 key, the plaintext, and the Poly1305 key
   aes      the AES key and the plaintext, through ECB and GCM
   aes_armv8  the same driver again, on AArch64, against the instructions
+  aes_x86ni  the same driver again, on x86-64, against AES-NI
 
-The aes driver is built against cbits/aes/generic.c and cbits/aes/gf.c on
-purpose, rather than whatever the machine offers.  AES-NI and the ARMv8
-instructions do not look anything up and would report nothing, which would say
-nothing about the table-driven code every other machine runs.  That code is
-variable-time by construction -- a table index is a byte of the state -- and
-so is the table-driven GHASH beside it.  A report from the aes driver is
-therefore expected and is a property of those implementations, not a defect
-found in them; it is here so that the size of it is written down rather than
-assumed.  Everything else is expected to be silent.
+One driver, built three times, against the portable C and against each of
+the two instruction sets.  All three must report nothing at all -- not
+"nothing outside known.txt", nothing.  The portable AES and GHASH are
+bitsliced and table-free; AESE, AESMC, PMULL, AESENC and PCLMULQDQ look
+nothing up.  None of the three has anywhere for a secret to decide a branch
+or an address.
 
-On AArch64 the same driver is then built a second time, as aes_armv8, against
-cbits/aes/armv8.c with the crypto extension turned on.  AESE, AESMC and PMULL
-look nothing up and branch on nothing, so that run must report nothing at all
--- not "nothing outside known.txt", nothing; a table site appearing there
-would mean the dispatch had not picked the instructions.  The two runs keep
-each other honest: the table-driven one has to report and the instruction one
-has to be silent, and either going the wrong way says the run is not
-measuring what it claims to.
+This entry used to read the other way round.  The portable implementation
+was table-driven -- an S-box indexed by a byte of the state, and Shoup's
+4-bit table for the GHASH -- so the aes driver was *required* to report, and
+known.txt carried four entries saying so.  That requirement was also what
+kept the pair honest: if the portable build had quietly taken an accelerated
+path it would have fallen silent, and the silence would have failed it.
 
-Until that was added the harness ran only on x86-64, so crypton's AArch64 AES
-and GHASH had never been put to it -- which was noticed when the GHASH was
-rewritten.
+With the tables gone that check goes too, so the question is now asked
+outright rather than read off a leak.  Each build of the driver prints which
+implementation its dispatch took, and run.sh requires the answer it expects
+before it looks at anything else.  A build that takes the wrong path fails
+on that line rather than passing by saying nothing.
+
+The x86-64 build is new with the rewrite.  Before it, AES-NI and PCLMULQDQ
+had never been put to this harness at all: their being constant time was an
+inference from the instruction specifications rather than a measurement of
+crypton's code.  The AArch64 build came earlier, when the GHASH was
+rewritten and it was noticed that the harness had only ever run on x86-64.
 
 What round eight found
 ----------------------
diff --git a/cbits/tests/ct/ct_aes.c b/cbits/tests/ct/ct_aes.c
--- a/cbits/tests/ct/ct_aes.c
+++ b/cbits/tests/ct/ct_aes.c
@@ -5,13 +5,24 @@
  * is table-driven and is variable-time by construction, which is a property
  * of that code rather than a defect in it.  See cbits/tests/ct/README. */
 #include "tests/ct/ct.h"
+#include <stdio.h>
 #include <string.h>
 #include "crypton_aes.h"
+#include "crypton_cpu.h"
 
+/* crypton_aes.c fills this; run.sh reads the line below to tell the three
+ * builds of this driver apart.  They are all silent now, so which one took
+ * which implementation cannot be read off a leak any more. */
+uint8_t *crypton_aes_cpu_init(void);
+
 int main(void) {
     aes_key k;
     aes_gcm_key gk;
     uint8_t key[32], pt[256], ct[256 + 16], iv[12];
+
+    printf("dispatch aes=%d pclmul=%d\n",
+           crypton_aes_cpu_init()[CPU_AESNI] != 0,
+           crypton_aes_cpu_init()[CPU_PCLMUL] != 0);
 
     ct_fill(key, sizeof key);
     ct_fill(pt, sizeof pt);
diff --git a/cbits/tests/ct/known.txt b/cbits/tests/ct/known.txt
--- a/cbits/tests/ct/known.txt
+++ b/cbits/tests/ct/known.txt
@@ -22,15 +22,6 @@
 # arithmetic, and it too goes the same way on every valid input.
 decaf.c:136       assert(ret) in crypton_gf_invert
 
-# The table-driven AES and the table-driven GHASH index with a byte of the
-# state, which is what makes them fast and what makes them variable-time.
-# That is a property of those implementations rather than a defect in them;
-# a machine with AES-NI or the ARMv8 instructions runs neither.
-generic.c         the AES tables, in key expansion and in the rounds
-gf.c              the GHASH table
-crypton_aes.c     the same tables, attributed to the code that inlines them
-block128.h        likewise
-
 # AArch64 only.  gcc keeps a carry in the flags and takes it out with `cset`
 # or `cinc`, where on x86-64 it uses `adc` and the carry never leaves the
 # data path.  memcheck calls `cset` a conditional move and reports it, and
diff --git a/cbits/tests/ct/run.sh b/cbits/tests/ct/run.sh
--- a/cbits/tests/ct/run.sh
+++ b/cbits/tests/ct/run.sh
@@ -27,17 +27,25 @@
 decaf_inc="-DCRYPTON_DECAF_WORD_BITS=64 -I$D/include -I$D/p448
            -I$D/include/arch_ref64 -I$D/p448/arch_ref64"
 
-# The generic C, not whatever the machine happens to offer.  A build that
-# takes AES-NI reports nothing from the AES driver and says nothing about the
-# table-driven code every other machine runs.
-aes_src="cbits/crypton_aes.c cbits/aes/generic.c cbits/aes/gf.c"
+# The portable C, not whatever the machine happens to offer, and the BearSSL
+# it is written on.  All three builds below have to be silent, so what tells
+# them apart is which implementation each one's dispatch chose -- which the
+# driver prints and run_one checks, since otherwise a build that quietly took
+# the wrong path would pass by saying nothing.
+aes_src="cbits/crypton_aes.c cbits/aes/generic.c cbits/aes/gf.c
+         cbits/bearssl/aes_ct64.c cbits/bearssl/aes_ct64_enc.c
+         cbits/bearssl/aes_ct64_dec.c cbits/bearssl/ghash_ctmul64.c
+         cbits/bearssl/dec32le.c"
 
-# And, on AArch64, the same driver again against the instructions.  That one
-# has to be silent; this one has to report.  Either going the wrong way says
-# the run is not measuring what it claims to.
+# The same driver again against the instructions, on each architecture that
+# has them.
 armv8_src="$aes_src cbits/aes/armv8.c cbits/crypton_cpu.c"
 armv8_inc="-DWITH_ARMV8_CRYPTO -march=armv8-a+crypto -Icbits/aes"
 
+x86ni_src="$aes_src cbits/aes/x86ni.c cbits/crypton_cpu.c
+           cbits/aes/gcm_vaes_x86.c cbits/aes/gcm_vaes512_x86.c"
+x86ni_inc="-DWITH_AESNI -DWITH_PCLMUL -maes -mpclmul -mssse3 -msse4.1 -Icbits/aes"
+
 status=0
 have_valgrind=no
 ct_define=
@@ -67,6 +75,32 @@
 mlkem_native_inc="$mlkem_inc -DCRYPTON_MLKEM_NATIVE_BACKEND"
 
 
+# Does this processor have AES instructions at all?
+#
+# Asked of the system rather than of crypton, because the question being
+# settled is whether crypton found what is there -- and a witness that comes
+# from the code under test cannot answer that.  "Cannot tell" counts as
+# having them, so an unfamiliar system turns into a loud failure rather than
+# a silent skip.
+#
+# CRYPTON_CT_CPUINFO is for testing this function itself; nothing else sets
+# it.
+machine_has_aes() {
+	cpuinfo=${CRYPTON_CT_CPUINFO:-/proc/cpuinfo}
+	if [ -r "$cpuinfo" ]; then
+		# "aes" in Features on ARM, in flags on x86; -w so that a model
+		# name containing the letters does not answer for the flags
+		grep -qw aes "$cpuinfo"
+		return
+	fi
+	if [ "$(uname -s)" = Darwin ]; then
+		[ "$(sysctl -n hw.optional.arm.FEAT_AES 2>/dev/null)" = 1 ] && return 0
+		[ "$(sysctl -n hw.optional.aes 2>/dev/null)" = 1 ] && return 0
+		return 1
+	fi
+	return 0
+}
+
 # run_one <name> <sources> <includes> [driver]
 #
 # The driver defaults to ct_<name>.c.  It is given separately where one
@@ -79,6 +113,36 @@
 		-o "$out/$name" "cbits/tests/ct/ct_$drv.c" $srcs 2> "$out/$name.cc" || {
 		echo "FAIL $name did not build"; sed -n '1,12p' "$out/$name.cc"; status=1; return
 	}
+	# Which implementation did this build's dispatch actually take?  The
+	# driver says, and the three AES builds want different answers.  This
+	# used to be inferred: the portable build was required to report,
+	# because the tables leaked, and silence meant it had taken an
+	# accelerated path instead.  The tables are gone and all three are
+	# silent now, so the question is asked outright rather than read off a
+	# leak that no longer happens.
+	case $name in
+	aes | aes_armv8 | aes_x86ni)
+		want=0
+		[ "$name" = aes ] || want=1
+		got=$("$out/$name" 2>/dev/null | sed -n 's/^dispatch aes=\([01]\).*/\1/p')
+		if [ "$got" != "$want" ]; then
+			# A processor with no AES instructions is not a failure of
+			# this driver; there is simply nothing here to measure, and
+			# the portable one above has already measured what does run.
+			# Boards in this position are common -- the Raspberry Pi 2,
+			# 3 and 4 among them, on either word size.
+			if [ "$want" = 1 ] && [ "$got" = 0 ] && ! machine_has_aes; then
+				echo "skip $name: this processor has no AES instructions,"
+				echo "     so there is nothing here to measure"
+				return
+			fi
+			echo "FAIL $name: dispatch took aes=$got, wanted aes=$want --"
+			echo "     this build is not running the implementation it is named for"
+			status=1
+			return
+		fi
+		;;
+	esac
 	if [ "$have_valgrind" = no ]; then
 		"$out/$name" > /dev/null 2>&1 && echo "built $name (no valgrind here; nothing checked)" \
 			|| { echo "FAIL $name did not run"; status=1; }
@@ -123,27 +187,17 @@
 			echo "ok   canary: reported $n, so the marking works"
 		fi
 		;;
-	aes)
-		# Silence would mean the build took an accelerated path and so
-		# measured nothing; the tables reporting is the point.
-		if [ "$n" -eq 0 ]; then
-			echo "FAIL aes: reported nothing, so this build did not take the"
-			echo "     table-driven code the driver exists to measure"
-			status=1
-		else
-			echo "note aes: $n report(s), from $(echo "$sites" | tr '\n' ' ')"
-		fi
-		;;
-	aes_armv8)
-		# The opposite demand, and known.txt does not apply: the entries in
-		# it are for the tables, and this build is not supposed to reach
-		# them.  Anything at all here is a finding, including a table site,
-		# which would mean the dispatch did not pick the instructions.
+	aes | aes_armv8 | aes_x86ni)
+		# Nothing here looks anything up: the portable AES and GHASH are
+		# bitsliced, and the two instruction sets branch on nothing.  So
+		# anything at all is a finding and known.txt does not apply --
+		# until the tables went this entry read the other way round, with
+		# the portable build required to report.
 		if [ "$n" -eq 0 ]; then
-			echo "ok   aes_armv8: the instructions decided nothing"
+			echo "ok   $name: the secret decided nothing"
 		else
-			echo "FAIL aes_armv8: $n report(s) from the AArch64 AES or GHASH,"
-			echo "     which look nothing up and should branch on nothing:"
+			echo "FAIL $name: $n report(s) from AES or GHASH,"
+			echo "     which should branch on nothing:"
 			for site in $sites; do echo "         $site"; done
 			sed -n '/Conditional jump\|Use of uninitialised/,/^==[0-9]*== $/p' \
 				"$out/$name.log" | head -30 | sed 's/^/    /'
@@ -183,7 +237,10 @@
 # and the build would not even compile.
 case $(uname -m) in
 aarch64 | arm64)
-	run_one aes_armv8 "$armv8_src" "$armv8_inc"
+	run_one aes_armv8 "$armv8_src" "$armv8_inc" aes
+	;;
+x86_64 | amd64)
+	run_one aes_x86ni "$x86ni_src" "$x86ni_inc" aes
 	;;
 esac
 
diff --git a/cbits/tests/endian/endian.c b/cbits/tests/endian/endian.c
--- a/cbits/tests/endian/endian.c
+++ b/cbits/tests/endian/endian.c
@@ -25,6 +25,9 @@
 #include "crypton_chacha.h"
 #include "crypton_salsa.h"
 #include "crypton_poly1305.h"
+#include "crypton_aes.h"
+#include "aes/gf.h"
+#include "aes/block128.h"
 
 /* The skein headers spell the prefix "cryponite", which nothing defines. */
 void crypton_skein256_init(struct skein256_ctx *ctx, uint32_t hashlen);
@@ -92,6 +95,131 @@
         }                                                                   \
     } while (0)
 
+/*
+ * AES, which reaches further into the byte order than the hashes above do.
+ *
+ * The portable implementation keeps its schedule as 64-bit words and reads
+ * its input through br_dec32le; the GHASH beside it reads H and the
+ * accumulator as big-endian words; the counter modes carry a counter that is
+ * incremented big-endian and stored little-endian in one case and the other
+ * way round in another; and XTS doubles its tweak in GF(2^128) through
+ * cpu_to_le64.  Every one of those is a place where a big-endian machine can
+ * differ, and none of them was asked about here until now.
+ *
+ * The entries called are the public ones, so this is whichever
+ * implementation the build installed -- which, for the build this harness
+ * makes, is the portable one.  That is the one a big-endian machine runs:
+ * crypton has no AES instructions for s390x, and the POWER8 ones are
+ * little-endian only.
+ */
+static void aes_answers(void) {
+    static const uint8_t keylens[] = {16, 24, 32};
+    /* multiples of the block, for the modes that take whole blocks */
+    static const uint32_t blocks[] = {1, 2, 4, 7};
+    /* and byte counts, including a partial block, for the ones that do not */
+    static const uint32_t bytes[] = {0, 1, 15, 16, 17, 64, 100};
+    uint8_t key[32], key2[32], iv[16], out[128], tmp[128];
+    char nm[128];
+    size_t ki, li;
+    uint32_t i;
+
+    for (i = 0; i < 32; i++) { key[i] = (uint8_t)(i * 3 + 1);
+                               key2[i] = (uint8_t)(i * 5 + 2); }
+    for (i = 0; i < 16; i++) iv[i] = (uint8_t)(i * 11 + 7);
+
+    for (ki = 0; ki < sizeof keylens / sizeof *keylens; ki++) {
+        uint8_t kl = keylens[ki];
+        aes_key k, k2;
+        aes_gcm_key gk;
+
+        crypton_aes_initkey(&k, key, kl);
+        crypton_aes_initkey(&k2, key2, kl);
+        crypton_aes_gcm_key_init(&gk, &k);
+
+        for (li = 0; li < sizeof blocks / sizeof *blocks; li++) {
+            uint32_t nb = blocks[li];
+            aes_block ivb;
+
+            crypton_aes_encrypt_ecb((aes_block *)out, &k, (aes_block *)buf, nb);
+            snprintf(nm, sizeof nm, "aes%u-ecb/%u", kl * 8, nb);
+            answer(nm, out, nb * 16);
+
+            crypton_aes_decrypt_ecb((aes_block *)out, &k, (aes_block *)buf, nb);
+            snprintf(nm, sizeof nm, "aes%u-ecbd/%u", kl * 8, nb);
+            answer(nm, out, nb * 16);
+
+            memcpy(&ivb, iv, 16);
+            crypton_aes_encrypt_cbc((aes_block *)out, &k, &ivb, (aes_block *)buf, nb);
+            snprintf(nm, sizeof nm, "aes%u-cbc/%u", kl * 8, nb);
+            answer(nm, out, nb * 16);
+
+            memcpy(&ivb, iv, 16);
+            crypton_aes_decrypt_cbc((aes_block *)out, &k, &ivb, (aes_block *)buf, nb);
+            snprintf(nm, sizeof nm, "aes%u-cbcd/%u", kl * 8, nb);
+            answer(nm, out, nb * 16);
+
+            memcpy(&ivb, iv, 16);
+            crypton_aes_encrypt_xts((aes_block *)out, &k, &k2, &ivb, 0,
+                                    (aes_block *)buf, nb);
+            snprintf(nm, sizeof nm, "aes%u-xts/%u", kl * 8, nb);
+            answer(nm, out, nb * 16);
+
+            memcpy(&ivb, iv, 16);
+            crypton_aes_decrypt_xts((aes_block *)out, &k, &k2, &ivb, 0,
+                                    (aes_block *)buf, nb);
+            snprintf(nm, sizeof nm, "aes%u-xtsd/%u", kl * 8, nb);
+            answer(nm, out, nb * 16);
+
+            /* a starting point too, since that is extra tweak doubling */
+            memcpy(&ivb, iv, 16);
+            crypton_aes_encrypt_xts((aes_block *)out, &k, &k2, &ivb, 3,
+                                    (aes_block *)buf, nb);
+            snprintf(nm, sizeof nm, "aes%u-xts-sp3/%u", kl * 8, nb);
+            answer(nm, out, nb * 16);
+        }
+
+        for (li = 0; li < sizeof bytes / sizeof *bytes; li++) {
+            uint32_t n = bytes[li];
+            aes_block ivb;
+
+            memcpy(&ivb, iv, 16);
+            crypton_aes_encrypt_ctr(out, &k, &ivb, buf, n);
+            snprintf(nm, sizeof nm, "aes%u-ctr/%u", kl * 8, n);
+            answer(nm, out, n);
+
+            /* the tag goes after the ciphertext, so this answers for both */
+            crypton_aes_gcm_full_encrypt(out, &gk, &k, iv, 12, buf, 13,
+                                         buf, n, 16);
+            snprintf(nm, sizeof nm, "aes%u-gcm/%u", kl * 8, n);
+            answer(nm, out, n + 16);
+        }
+    }
+
+    /* GHASH and POLYVAL on their own, which the modes above reach only
+     * through whatever length they were given */
+    {
+        table_4bit ht;
+        block128 acc;
+        aes_polyval pv;
+
+        crypton_aes_generic_hinit(ht, (const block128 *)buf);
+        memcpy(&acc, buf + 16, 16);
+        crypton_aes_generic_gf_mul(&acc, ht);
+        answer("ghash-mul", (const uint8_t *)&acc, 16);
+
+        crypton_aes_generic_hinit(ht, (const block128 *)buf);
+        memcpy(&acc, buf + 16, 16);
+        crypton_aes_generic_gf_mul4(&acc, (const block128 *)(buf + 32), ht);
+        answer("ghash-mul4", (const uint8_t *)&acc, 16);
+
+        memcpy(tmp, buf, 16);
+        crypton_aes_polyval_init(&pv, (const aes_block *)tmp);
+        crypton_aes_polyval_update(&pv, buf + 16, 64);
+        crypton_aes_polyval_finalize(&pv, (aes_block *)out);
+        answer("polyval", out, 16);
+    }
+}
+
 int main(int argc, char **argv) {
     generating = (argc > 1 && strcmp(argv[1], "generate") == 0);
     vf = fopen(argc > 2 ? argv[2] : "cbits/tests/endian/vectors.txt",
@@ -182,6 +310,8 @@
             answer(nmbuf, mac, sizeof mac);
         }
     }
+
+    aes_answers();
 
     if (!generating && failures == 0)
         printf("ok   %d answers match the little-endian ones\n", checked);
diff --git a/cbits/tests/endian/run.sh b/cbits/tests/endian/run.sh
--- a/cbits/tests/endian/run.sh
+++ b/cbits/tests/endian/run.sh
@@ -6,6 +6,14 @@
 # loaders in crypton_align.h -- rewritten from word-typed casts to memcpy in
 # #256 -- and eight more decide something from the byte order themselves.
 #
+# AES is here as well, and it reaches further into the byte order than the
+# hashes do: a schedule kept as 64-bit words, a GHASH that reads H and its
+# accumulator as big-endian words, counters incremented one way and stored
+# the other, and an XTS tweak doubled through cpu_to_le64.  It is the
+# portable implementation that answers, which is what a big-endian machine
+# runs: crypton has no AES instructions for s390x, and the POWER8 ones are
+# little-endian only.
+#
 # The two sides cannot be compared in one run the way the 32-bit harness
 # compares two builds, because the machine doing the comparing has only one
 # byte order.  So the answers are frozen: vectors.txt is what this code gives
@@ -26,7 +34,11 @@
       cbits/crypton_sha256.c cbits/crypton_sha512.c cbits/crypton_sha3.c
       cbits/crypton_ripemd.c cbits/crypton_skein256.c cbits/crypton_skein512.c
       cbits/crypton_tiger.c cbits/crypton_whirlpool.c
-      cbits/crypton_chacha.c cbits/crypton_salsa.c cbits/crypton_poly1305.c"
+      cbits/crypton_chacha.c cbits/crypton_salsa.c cbits/crypton_poly1305.c
+      cbits/crypton_aes.c cbits/aes/generic.c cbits/aes/gf.c
+      cbits/bearssl/aes_ct64.c cbits/bearssl/aes_ct64_enc.c
+      cbits/bearssl/aes_ct64_dec.c cbits/bearssl/ghash_ctmul64.c
+      cbits/bearssl/dec32le.c"
 
 # Generating is done under the sanitizers, since a driver that writes out of
 # bounds would otherwise freeze whatever it happened to leave behind.  That
@@ -38,7 +50,7 @@
 fi
 
 # shellcheck disable=SC2086
-$cc -O2 -g $san -Icbits -Icbits/include64 -o "$out/endian" \
+$cc -O2 -g $san -Icbits -Icbits/aes -Icbits/include64 -o "$out/endian" \
 	cbits/tests/endian/endian.c $srcs
 
 if [ "$mode" = generate ]; then
diff --git a/cbits/tests/endian/vectors.txt b/cbits/tests/endian/vectors.txt
--- a/cbits/tests/endian/vectors.txt
+++ b/cbits/tests/endian/vectors.txt
@@ -250,3 +250,132 @@
 chacha20/1000 064429ccbd6ef85054af07f8dee3a283e75e844df9a59b1b76addd7d41fb554946f5dd12b49b97f4a72f05b0b41c557e8c69da034ae9dddc8a54dbf245d2bcff
 salsa20/1000 9ee7f7b5774db507d8e78fe7398594de9b40444a5301c14dfbc675e6310f9aa659e9cca9892563df9386a60607a2e18781d35d4897ea2b029db7b8819cafae17
 poly1305/1000 c403a63071ea2f2e732472db8e7b79c0
+aes128-ecb/1 51d0767be0142b54f38422f098a979c1
+aes128-ecbd/1 0772f11dd1e53677251c6d2b42c12a84
+aes128-cbc/1 53423fee1b0b9dda9006158bb0e3c6e6
+aes128-cbcd/1 0060ec35e2db7f237a7618abc9578b28
+aes128-xts/1 2634e65251d6fed878a1b3546061cd8f
+aes128-xtsd/1 9d22db2b665e917344f60d1561481854
+aes128-xts-sp3/1 1d40e186532c4338b08ebb40039c49b6
+aes128-ecb/2 51d0767be0142b54f38422f098a979c1f5e0ee3fe17beef5b33b72d8f60ec17c
+aes128-ecbd/2 0772f11dd1e53677251c6d2b42c12a843d8abf116e0bd219a05d2c1d89e8faee
+aes128-cbc/2 53423fee1b0b9dda9006158bb0e3c6e6b1acc41a9ff16d75c2170e3242346760
+aes128-cbcd/2 0060ec35e2db7f237a7618abc9578b283d8db1047228f82898626a50ddb39887
+aes128-xts/2 2634e65251d6fed878a1b3546061cd8ff2e82679d8d7ade78233b89a76958855
+aes128-xtsd/2 9d22db2b665e917344f60d1561481854c8d8c3fe1890b52db7ca720cd253192d
+aes128-xts-sp3/2 1d40e186532c4338b08ebb40039c49b6cc2b2b42b37e9412005737634804ad97
+aes128-ecb/4 51d0767be0142b54f38422f098a979c1f5e0ee3fe17beef5b33b72d8f60ec17c527ee4edb0bd6e9346d1148c3763613b5112f9a67b30c06fd511e725279957b9
+aes128-ecbd/4 0772f11dd1e53677251c6d2b42c12a843d8abf116e0bd219a05d2c1d89e8faeed21af82debb90e26d30fd338766a0ff47234606001901e7888e61a64468495be
+aes128-cbc/4 53423fee1b0b9dda9006158bb0e3c6e6b1acc41a9ff16d75c2170e324234676075d60f392793fd0623859fffb40b23585955ff09a2c20c02b9c2baf084d6943b
+aes128-cbcd/4 0060ec35e2db7f237a7618abc9578b283d8db1047228f82898626a50ddb39887a26d86a8672a94877ba06585b2a1dd2d8d326d741ab23748bfd85f2815def4d6
+aes128-xts/4 2634e65251d6fed878a1b3546061cd8ff2e82679d8d7ade78233b89a76958855da809daab8adf228232779804848093c5aa478a2d7e0c11b58dfbfd19b94ea36
+aes128-xtsd/4 9d22db2b665e917344f60d1561481854c8d8c3fe1890b52db7ca720cd253192dedd68b0854e3340afb2d8a34942bcd5479bed9f04ee52769479c2192544eaa2f
+aes128-xts-sp3/4 1d40e186532c4338b08ebb40039c49b6cc2b2b42b37e9412005737634804ad970343c410aa37b63b6595f3e6e8bbcb8bc8581ec02491cb1e3df920165a8d05c7
+aes128-ecb/7 51d0767be0142b54f38422f098a979c1f5e0ee3fe17beef5b33b72d8f60ec17c527ee4edb0bd6e9346d1148c3763613b5112f9a67b30c06fd511e725279957b977a15ad7bbc6a54c141a2d07c4b233000adedca5ef085da03709e978ed1aca3500e2bd1f0ad7a29ade429446d2ddfb93
+aes128-ecbd/7 0772f11dd1e53677251c6d2b42c12a843d8abf116e0bd219a05d2c1d89e8faeed21af82debb90e26d30fd338766a0ff47234606001901e7888e61a64468495be2de8cb16ce03066f65758235de9386de764ddfa85c70fae9b4c80eed9b53a216d5c34dcea96fa27e88fe002306a3548d
+aes128-cbc/7 53423fee1b0b9dda9006158bb0e3c6e6b1acc41a9ff16d75c2170e324234676075d60f392793fd0623859fffb40b23585955ff09a2c20c02b9c2baf084d6943b84e8cef5ab838dc0864175cab14a1026f6230f8df62be8d6794dfb9aba0622e62e6a358aab88832f2b18d4819eca4f91
+aes128-cbcd/7 0060ec35e2db7f237a7618abc9578b283d8db1047228f82898626a50ddb39887a26d86a8672a94877ba06585b2a1dd2d8d326d741ab23748bfd85f2815def4d6429eb69245919fcfc2db37891d5957068848d3bb4651d2c682f54aa6c90ac271bbb6314d23fe3ae12e53b498c46a845a
+aes128-xts/7 2634e65251d6fed878a1b3546061cd8ff2e82679d8d7ade78233b89a76958855da809daab8adf228232779804848093c5aa478a2d7e0c11b58dfbfd19b94ea3651d66c7adda6172d3e8672cc65cce7dbd8d9d6f32bf8f087867527f0a8128fdd80ed3a97398f66577f2041342c585e72
+aes128-xtsd/7 9d22db2b665e917344f60d1561481854c8d8c3fe1890b52db7ca720cd253192dedd68b0854e3340afb2d8a34942bcd5479bed9f04ee52769479c2192544eaa2fa426c177c9b397f0bb380e3991c69479759294d63723eeb04aae18e48f84fa213dcf254eb1d532252c6328a8fdd8976f
+aes128-xts-sp3/7 1d40e186532c4338b08ebb40039c49b6cc2b2b42b37e9412005737634804ad970343c410aa37b63b6595f3e6e8bbcb8bc8581ec02491cb1e3df920165a8d05c7fa1c26910201c2ac91dab7ffcb1995c57cfc19a1a8ffbb3e11debe1e9de63147bd99f1f0f9af66d827b18a427ed02d32
+aes128-ctr/0 -
+aes128-gcm/0 91527d0ffa250047dc10c74aaf7dc630
+aes128-ctr/1 f3
+aes128-gcm/1 110be12ced502543dc2b098ce214c704a9
+aes128-ctr/15 f3e218ff08f7128f6c9e088df658bf
+aes128-gcm/15 11e696e7a098883e54b19200262e669d12ac829a6a3eed6933b267f13690e7
+aes128-ctr/16 f3e218ff08f7128f6c9e088df658bfcc
+aes128-gcm/16 11e696e7a098883e54b19200262e6622968a6e12ea8c1d162967acbccc339980
+aes128-ctr/17 f3e218ff08f7128f6c9e088df658bfccea
+aes128-gcm/17 11e696e7a098883e54b19200262e6622dd540e6e9683dd7026b3b24579826c0593
+aes128-ctr/64 f3e218ff08f7128f6c9e088df658bfccea245fbfae1cb12ed69d5f01a5a6d7f161f8ae682ba4555a088d89fe91fdfb2890935fe9d0ae19f0ac4f73e19d26d1cc
+aes128-gcm/64 11e696e7a098883e54b19200262e6622dd7f1ac8145981a493fdaea4641bf49da46e0d311c3daf1665fd5fdf025965d9312f324d4c646d3f3f85ab13d236716fdb10935bd1c334f13b56a98aa09ef476
+aes128-ctr/100 f3e218ff08f7128f6c9e088df658bfccea245fbfae1cb12ed69d5f01a5a6d7f161f8ae682ba4555a088d89fe91fdfb2890935fe9d0ae19f0ac4f73e19d26d1cc8c30a3d0749027d7cff23bb8dbccb1a8540feb89f5ca9447a5cdef585d59125c58b73b22
+aes128-gcm/100 11e696e7a098883e54b19200262e6622dd7f1ac8145981a493fdaea4641bf49da46e0d311c3daf1665fd5fdf025965d9312f324d4c646d3f3f85ab13d236716fc40f23dfc154e04b18dfc89ae50a386551eab4dd02434c526f9fc45a278b21a485347381adbe43ff0386975d584a98adacbde34a
+aes192-ecb/1 b7bd8a4e647f413adc38c4464e4434b6
+aes192-ecbd/1 4308adb49827c2741aabb7c40c5b63a5
+aes192-cbc/1 9fdd2c700fd7e916fd89805f18439a27
+aes192-cbcd/1 441ab09cab198b2045c1c24487cdc209
+aes192-xts/1 c2eea1b9f5f0fd170168a3e2af1d059a
+aes192-xtsd/1 5a3a19588f9700d5482eae9dd316051d
+aes192-xts-sp3/1 f1dba9fc3e3d43d3a46608690e98b0e8
+aes192-ecb/2 b7bd8a4e647f413adc38c4464e4434b67d307330822566b908545f9e8de18c1a
+aes192-ecbd/2 4308adb49827c2741aabb7c40c5b63a5e6015605e3e6c8df51f2de769db588e2
+aes192-cbc/2 9fdd2c700fd7e916fd89805f18439a2754e55a17706f184fe6b2185ef87eb075
+aes192-cbcd/2 441ab09cab198b2045c1c24487cdc209e6065810ffc5e2ee69cd983bc9eeea8b
+aes192-xts/2 c2eea1b9f5f0fd170168a3e2af1d059af4c8308582684da17d5b6d631459d6d5
+aes192-xtsd/2 5a3a19588f9700d5482eae9dd316051dab8c3b6155002903c90664413bd96653
+aes192-xts-sp3/2 f1dba9fc3e3d43d3a46608690e98b0e823173a551464b1a3ef6095560902a48c
+aes192-ecb/4 b7bd8a4e647f413adc38c4464e4434b67d307330822566b908545f9e8de18c1ac8d6dd6f7bee6e99e4682aef343f7cd0ea985b24e2368b8d1acd3b0bf164a30b
+aes192-ecbd/4 4308adb49827c2741aabb7c40c5b63a5e6015605e3e6c8df51f2de769db588e2f4a1ce5088ed9eee6f721f73986dcf23ffc534ac5d16cb7010015dc1b50330ce
+aes192-cbc/4 9fdd2c700fd7e916fd89805f18439a2754e55a17706f184fe6b2185ef87eb075061361ddd895a18991e3068ef563dcea087fc1517810a4ac222a391215026095
+aes192-cbcd/4 441ab09cab198b2045c1c24487cdc209e6065810ffc5e2ee69cd983bc9eeea8b84d6b0d5047e044fc7dda9ce5ca61dfa00c339b84634e240273f188de65951a6
+aes192-xts/4 c2eea1b9f5f0fd170168a3e2af1d059af4c8308582684da17d5b6d631459d6d52e43efd39bfc3ebbbbf189a4bfb1071bbc3851a67841dc4206c4f0f86fca0fed
+aes192-xtsd/4 5a3a19588f9700d5482eae9dd316051dab8c3b6155002903c90664413bd9665351b41f0a6b8b2afd38a19274800e86739bb50da065cc97f9818895e062c6be6c
+aes192-xts-sp3/4 f1dba9fc3e3d43d3a46608690e98b0e823173a551464b1a3ef6095560902a48cdc267dd87440e9a910bedc98758b512799ad24644966748404717b701ad70804
+aes192-ecb/7 b7bd8a4e647f413adc38c4464e4434b67d307330822566b908545f9e8de18c1ac8d6dd6f7bee6e99e4682aef343f7cd0ea985b24e2368b8d1acd3b0bf164a30b225dbf0a85eb782bbaf30c02286d91d271bbfb08e60fc12977e7eea142abc72bed4ade399f048bfa5d5a48f02ac47e22
+aes192-ecbd/7 4308adb49827c2741aabb7c40c5b63a5e6015605e3e6c8df51f2de769db588e2f4a1ce5088ed9eee6f721f73986dcf23ffc534ac5d16cb7010015dc1b50330ce1b0b2582b28d8f6174ee21d4cefdc31632f8ca413704961b7da50fdc2de61e730ab37b8b3148660f36335f6d4a01d85a
+aes192-cbc/7 9fdd2c700fd7e916fd89805f18439a2754e55a17706f184fe6b2185ef87eb075061361ddd895a18991e3068ef563dcea087fc1517810a4ac222a391215026095582b2b9c34d2b9b40ef88f9ec4459fc671f5ecb0d23defc6ea99f6075174381b861dec529ac5a49c9c55efae85aca707
+aes192-cbcd/7 441ab09cab198b2045c1c24487cdc209e6065810ffc5e2ee69cd983bc9eeea8b84d6b0d5047e044fc7dda9ce5ca61dfa00c339b84634e240273f188de65951a6747d5806391f16c1d34094680d3712ceccfdc6522d25be344b984b977fbf7e1464c60708bbd9fe90909eebd688c8088d
+aes192-xts/7 c2eea1b9f5f0fd170168a3e2af1d059af4c8308582684da17d5b6d631459d6d52e43efd39bfc3ebbbbf189a4bfb1071bbc3851a67841dc4206c4f0f86fca0fedf90c2617695920f2f5611aab587198b0263b555f0a93d1580c22822b6fe3451e31e5f67c9526f5128d87bc586e61a255
+aes192-xtsd/7 5a3a19588f9700d5482eae9dd316051dab8c3b6155002903c90664413bd9665351b41f0a6b8b2afd38a19274800e86739bb50da065cc97f9818895e062c6be6cfa00838607d2c4a55642a4973aac6fbdd5e9c514e69e8569d1559acde57fc207a5709b8f0eff9a93ad7dc6c170fda2c0
+aes192-xts-sp3/7 f1dba9fc3e3d43d3a46608690e98b0e823173a551464b1a3ef6095560902a48cdc267dd87440e9a910bedc98758b512799ad24644966748404717b701ad70804506ba3f939ed54c43dc721d3532f83cf267aa4c32d9543edb71de9548dacd5171e76bef732fdad2c49ebb781d639a445
+aes192-ctr/0 -
+aes192-gcm/0 b03daf3f9346095b3083d09391d47cee
+aes192-ctr/1 72
+aes192-gcm/1 bdc01f6ed1a9e3dc9be2c4d9e8d8d3b584
+aes192-ctr/15 72138e46596ca4b6de00c2aeadc2c1
+aes192-gcm/15 bd19761039873f6946e29ba643ee41c6ca3791dc7497012fce9f48631f599c
+aes192-ctr/16 72138e46596ca4b6de00c2aeadc2c17c
+aes192-gcm/16 bd19761039873f6946e29ba643ee4175c85f71850653f05f2410354e2e93cdc9
+aes192-ctr/17 72138e46596ca4b6de00c2aeadc2c17c02
+aes192-gcm/17 bd19761039873f6946e29ba643ee4175b3e87ccbca9af385c12d31382f6e721151
+aes192-ctr/64 72138e46596ca4b6de00c2aeadc2c17c02df852e3c7abed9e3cd5a94d24c71074945c7835771862ec94dbc9c8ba47ef305131b7f3933a150025d8a796c5889b2
+aes192-gcm/64 bd19761039873f6946e29ba643ee4175b3c8f8e942762e364f67333b85f6c220999554b2d6e78a47fc789581a39e89a4b6a5539491894c85fbba17f03e183e3b27736f65aaba8f86ba884618e2c75b83
+aes192-ctr/100 72138e46596ca4b6de00c2aeadc2c17c02df852e3c7abed9e3cd5a94d24c71074945c7835771862ec94dbc9c8ba47ef305131b7f3933a150025d8a796c5889b2aea7d364add8934a993f8532bae1b59ec0c4d75bf59dca8e265de412ed73052188706192
+aes192-gcm/100 bd19761039873f6946e29ba643ee4175b3c8f8e942762e364f67333b85f6c220999554b2d6e78a47fc789581a39e89a4b6a5539491894c85fbba17f03e183e3bb42b4d0d362fc35b7719b76c299e857e82b8b127e9cbc732559c26b6338ef92f153271feb0462a5c14552476a3a41aae5c275596
+aes256-ecb/1 791925c38818304f047d756d78ef8f13
+aes256-ecbd/1 831e330b93ed108f70c15c463224b5c6
+aes256-cbc/1 6b1dfef4ab6d3fd0a32a3c3a7c9603d5
+aes256-cbcd/1 840c2e23a0d359db2fab29c6b9b2146a
+aes256-xts/1 b90a69094799c4e679fc829084d41055
+aes256-xtsd/1 18db4b08618f8573974e5c85d0954302
+aes256-xts-sp3/1 d5f9ef046dd7e688938aab18d3d68ce5
+aes256-ecb/2 791925c38818304f047d756d78ef8f139da36087c3b26165970cf0987eee7d70
+aes256-ecbd/2 831e330b93ed108f70c15c463224b5c6f2f7d848bb39a82dd65b4db54a6f54ff
+aes256-cbc/2 6b1dfef4ab6d3fd0a32a3c3a7c9603d5090fa2b707cd96fea1132c143d5683fa
+aes256-cbcd/2 840c2e23a0d359db2fab29c6b9b2146af2f0d65da71a821cee640bf81e343696
+aes256-xts/2 b90a69094799c4e679fc829084d4105597001399d1f01a458a64913345271c27
+aes256-xtsd/2 18db4b08618f8573974e5c85d0954302511a3cdde7128329882a30bd28169980
+aes256-xts-sp3/2 d5f9ef046dd7e688938aab18d3d68ce5724313716ca706a5b7a914024b26cfc1
+aes256-ecb/4 791925c38818304f047d756d78ef8f139da36087c3b26165970cf0987eee7d70aff88ab2d21b91927e27057d2f91f923cd1e35c75f2b3111adc0fd520ae368b4
+aes256-ecbd/4 831e330b93ed108f70c15c463224b5c6f2f7d848bb39a82dd65b4db54a6f54ff6c4c59c5bd048efc1c7cf8353e29eae59fd397d041edd80e0c4a41824a3b7bc1
+aes256-cbc/4 6b1dfef4ab6d3fd0a32a3c3a7c9603d5090fa2b707cd96fea1132c143d5683fae5e8441b18737ef3581adef0e182aea3b94a6e5e4925464a95ddb6b409f19e88
+aes256-cbcd/4 840c2e23a0d359db2fab29c6b9b2146af2f0d65da71a821cee640bf81e3436961c3b27403197145db4d34e88fae2383c60d59ac45acff13e3b7404ce19611aa9
+aes256-xts/4 b90a69094799c4e679fc829084d4105597001399d1f01a458a64913345271c27261700fa455603c6d5370d3867bf3213717c7241850cc16b414cf4598472f9e7
+aes256-xtsd/4 18db4b08618f8573974e5c85d0954302511a3cdde7128329882a30bd28169980f52cc7d5d434263c1c22188e735a49d6130e0683a55e636739ba7abb743e7b09
+aes256-xts-sp3/4 d5f9ef046dd7e688938aab18d3d68ce5724313716ca706a5b7a914024b26cfc1b47c048f8548cc7a80a22a9dcaf25065efd935f4730020a085cfd489b8a98cab
+aes256-ecb/7 791925c38818304f047d756d78ef8f139da36087c3b26165970cf0987eee7d70aff88ab2d21b91927e27057d2f91f923cd1e35c75f2b3111adc0fd520ae368b4b6d87226c642479229919a9c47bfca56ca5607a049cd7f423fb0556511f5db4bb8f7165904e83e4da028f316252e9f5d
+aes256-ecbd/7 831e330b93ed108f70c15c463224b5c6f2f7d848bb39a82dd65b4db54a6f54ff6c4c59c5bd048efc1c7cf8353e29eae59fd397d041edd80e0c4a41824a3b7bc1eaa8daa1a6f40e115cfd93684f1b7ad8b66b09a303b2488f93eaac36762e571d4f00c320962c9dc1607e8d884579f306
+aes256-cbc/7 6b1dfef4ab6d3fd0a32a3c3a7c9603d5090fa2b707cd96fea1132c143d5683fae5e8441b18737ef3581adef0e182aea3b94a6e5e4925464a95ddb6b409f19e88f104a783b152e599e90b85fc60f57aef7ee95c1608290490adf2688b4683b2d70a6b16291cb5a49008dd74dbf8c16191
+aes256-cbcd/7 840c2e23a0d359db2fab29c6b9b2146af2f0d65da71a821cee640bf81e3436961c3b27403197145db4d34e88fae2383c60d59ac45acff13e3b7404ce19611aa985dea7252d6697b1fb5326d48cd1ab00486e05b0199360a0a5d7e87d2477377a2175bfa31cbd055ec6d3393387b023d1
+aes256-xts/7 b90a69094799c4e679fc829084d4105597001399d1f01a458a64913345271c27261700fa455603c6d5370d3867bf3213717c7241850cc16b414cf4598472f9e77778636d8d94ec90c8bc220a7075ab56b4758635686c3985c243224397844894defd980efc5ce05ed451cec1af9583e9
+aes256-xtsd/7 18db4b08618f8573974e5c85d0954302511a3cdde7128329882a30bd28169980f52cc7d5d434263c1c22188e735a49d6130e0683a55e636739ba7abb743e7b09cdc794b975bdb545b6a5a69fde322ceb0f3da5d8a1dd3103dee190f9fb65d8a0e86ddffed9358c0872862ae921e7ba75
+aes256-xts-sp3/7 d5f9ef046dd7e688938aab18d3d68ce5724313716ca706a5b7a914024b26cfc1b47c048f8548cc7a80a22a9dcaf25065efd935f4730020a085cfd489b8a98cab5e7eb266eb52f17b6f78bcb0996b96fc7ca3a671856b32fe2e8c9341c50a70c2d40906b8d450fed98e6a5f6a4d5f84ab
+aes256-ctr/0 -
+aes256-gcm/0 8316aabedf3935666f9d8b2dd6b5b356
+aes256-ctr/1 21
+aes256-gcm/1 7b8fcd0cf59d4bccd66dd123de4c1c7aa4
+aes256-ctr/15 219430a80fe1a08ed512ff1c9cee61
+aes256-gcm/15 7bfc609c1f245b464a8e0763c5e8b7a1dba82dac1ce33ecb0350b7d9e582df
+aes256-ctr/16 219430a80fe1a08ed512ff1c9cee61dc
+aes256-gcm/16 7bfc609c1f245b464a8e0763c5e8b7f9ae001617c759469bd602cd92dd4bf7e9
+aes256-ctr/17 219430a80fe1a08ed512ff1c9cee61dc52
+aes256-gcm/17 7bfc609c1f245b464a8e0763c5e8b7f96b70cf1b59a4a56949e6437f0b9b7b257b
+aes256-ctr/64 219430a80fe1a08ed512ff1c9cee61dc52867258efbdf3943f48d30ef7da12c7bfb56c293beeddc92b1c270b7110c0e2db805ddad464f0eac9d46e7109001374
+aes256-gcm/64 7bfc609c1f245b464a8e0763c5e8b7f96bc3d807f39fd255cbf14006baf6bbccb5d2509a2b3cba5a4c7ee680ee09979e810f94882ff5c0b6e572d397aef389ec72e6a3f5eb4be76004c10d8c5881492a
+aes256-ctr/100 219430a80fe1a08ed512ff1c9cee61dc52867258efbdf3943f48d30ef7da12c7bfb56c293beeddc92b1c270b7110c0e2db805ddad464f0eac9d46e71090013745c52cea313ba7efea2eac9da963dcf1bc484a1c1924fd8f91fd8bf774dc77b472356c522
+aes256-gcm/100 7bfc609c1f245b464a8e0763c5e8b7f96bc3d807f39fd255cbf14006baf6bbccb5d2509a2b3cba5a4c7ee680ee09979e810f94882ff5c0b6e572d397aef389ecb7bd0a0a1f4d638d3e2557ca92beb2bf4228499ac6cc38d20e0539cd4658ed2a7d24f4ab32d4b6ea69814b4c2718fd1dcdb7e16a
+ghash-mul c65b128f165974f311df03df3486ce71
+ghash-mul4 6bfd46cd3d314cac2d75ead0e96092c7
+polyval bbc31d0761373f0b2737904b9a76fb27
diff --git a/cbits/tests/fuzz/run.sh b/cbits/tests/fuzz/run.sh
--- a/cbits/tests/fuzz/run.sh
+++ b/cbits/tests/fuzz/run.sh
@@ -31,7 +31,10 @@
 	canary)  echo "" ;;
 	ed25519) echo "cbits/ed25519/ed25519.c cbits/crypton_sha512.c" ;;
 	p256)    echo "cbits/p256/p256.c cbits/p256/p256_ec.c" ;;
-	aead)    echo "cbits/crypton_aes.c cbits/aes/generic.c cbits/aes/gf.c" ;;
+	aead)    echo "cbits/crypton_aes.c cbits/aes/generic.c cbits/aes/gf.c
+	               cbits/bearssl/aes_ct64.c cbits/bearssl/aes_ct64_enc.c
+	               cbits/bearssl/aes_ct64_dec.c cbits/bearssl/ghash_ctmul64.c
+	               cbits/bearssl/dec32le.c" ;;
 	decaf)   echo "$D/ed448goldilocks/decaf_all.c $D/ed448goldilocks/eddsa.c
 	               $D/ed448goldilocks/scalar.c $D/p448/f_arithmetic.c
 	               $D/p448/f_generic.c $D/utils.c
diff --git a/cbits/tests/ppc8_diff.c b/cbits/tests/ppc8_diff.c
new file mode 100644
--- /dev/null
+++ b/cbits/tests/ppc8_diff.c
@@ -0,0 +1,254 @@
+/*
+ * The POWER8 path against the portable one.
+ *
+ * crypton's own test suite exercises whichever implementation the machine
+ * installed, so on a POWER8 it never runs the portable code and nothing
+ * compares the two.  CI has no POWER8 either.  This does the comparison in
+ * one process, with both compiled in, and runs under qemu-ppc64le.
+ *
+ *     powerpc64le-linux-gnu-gcc -O2 -static -Icbits -Icbits/aes \
+ *         -DWITH_PPC8_CRYPTO -o ppc8_diff \
+ *         cbits/tests/ppc8_diff.c cbits/crypton_aes.c cbits/aes/generic.c \
+ *         cbits/aes/gf.c cbits/aes/ppc8.c cbits/crypton_cpu.c \
+ *         cbits/asm/aesp8-ppc-linux64le.S cbits/asm/ghashp8-ppc-linux64le.S \
+ *         cbits/bearssl/*.c
+ *     qemu-ppc64le-static ./ppc8_diff
+ *
+ * Given any argument it corrupts a result in each section, so that the
+ * comparison can be seen to notice: one that cannot fail has said nothing.
+ *
+ * One trap in writing this, worth naming because it looks like a crash in
+ * the code under test.  Several of the generic mode loops reach the block
+ * function through the branch table, which in this process holds the POWER8
+ * entries -- so calling them as the reference hands a portable key to the
+ * POWER8 block function.  The references below are therefore the entries
+ * that reach cbits/aes/generic.c directly, and XTS, whose generic loop takes
+ * its tweak through the table, is written out here instead.
+ */
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <stdint.h>
+
+#include "crypton_aes.h"
+#include "aes/generic.h"
+#include "aes/gf.h"
+#include "aes/block128.h"
+#include "aes/ppc8.h"
+
+/* the portable CTR, which reaches generic.c rather than the branch table */
+void crypton_aes_bitsliced_encrypt_ctr(uint8_t *output, aes_key *key,
+                                       aes_block *iv, uint8_t *input,
+                                       uint32_t len);
+
+static int failures;
+static int sabotage;
+
+static void same(const char *what, const uint8_t *a, const uint8_t *b, size_t n)
+{
+	size_t i;
+
+	if (memcmp(a, b, n) == 0)
+		return;
+	printf("  MISMATCH %s\n    ppc8     ", what);
+	for (i = 0; i < n && i < 32; i++) printf("%02x", a[i]);
+	printf("\n    portable ");
+	for (i = 0; i < n && i < 32; i++) printf("%02x", b[i]);
+	printf("\n");
+	failures++;
+}
+
+static uint32_t rnd_state = 1;
+static uint8_t rnd(void)
+{
+	rnd_state = rnd_state * 1103515245u + 12345u;
+	return (uint8_t) (rnd_state >> 16);
+}
+static void rnd_fill(uint8_t *p, size_t n)
+{
+	while (n--) *p++ = rnd();
+}
+
+/*
+ * XTS with the portable block function, written out because the generic
+ * loop takes its tweak through the branch table.
+ */
+static void xts_reference(uint8_t *out, aes_key *k1, aes_key *k2,
+                          const uint8_t *dataunit, uint32_t spoint,
+                          const uint8_t *in, uint32_t nb, int decrypt)
+{
+	block128 tweak, t;
+	uint32_t i;
+
+	memcpy(&tweak, dataunit, 16);
+	crypton_aes_generic_encrypt_block(&tweak, k2, &tweak);
+	while (spoint-- > 0)
+		crypton_aes_generic_gf_mulx(&tweak);
+
+	for (i = 0; i < nb; i++) {
+		block128_vxor(&t, (const block128 *) (in + 16 * i), &tweak);
+		if (decrypt)
+			crypton_aes_generic_decrypt_block(&t, k1, &t);
+		else
+			crypton_aes_generic_encrypt_block(&t, k1, &t);
+		block128_vxor((block128 *) (out + 16 * i), &t, &tweak);
+		crypton_aes_generic_gf_mulx(&tweak);
+	}
+}
+
+/* the two inits write different things into the same aes_key, so each side
+ * gets its own */
+static void keys(aes_key *p8, aes_key *gen, const uint8_t *k, uint8_t len)
+{
+	memset(p8, 0, sizeof *p8);
+	memset(gen, 0, sizeof *gen);
+	p8->nbr = gen->nbr = len == 16 ? 10 : len == 24 ? 12 : 14;
+	crypton_aes_ppc8_init(p8, (uint8_t *) k, len);
+	crypton_aes_generic_init(gen, (uint8_t *) k, len);
+}
+
+int main(int argc, char **argv)
+{
+	static const uint8_t klens[3] = { 16, 24, 32 };
+	int round;
+
+	setvbuf(stdout, NULL, _IONBF, 0);   /* so a crash does not eat the section it was in */
+	sabotage = argc > 1;
+
+	printf("== the dispatch would take: aes=%d ==\n",
+	       crypton_aes_ppc8_available());
+
+	printf("== ECB and CBC, both directions ==\n");
+	for (round = 0; round < 300; round++) {
+		uint8_t key[32], in[16 * 9], a[16 * 9], b[16 * 9], iv[16];
+		uint8_t len = klens[round % 3];
+		uint32_t nb = 1 + (round % 9);
+		aes_key kp, kg;
+
+		rnd_fill(key, len);
+		rnd_fill(in, nb * 16);
+		rnd_fill(iv, 16);
+		keys(&kp, &kg, key, len);
+
+		crypton_aes_ppc8_encrypt_ecb((aes_block *) a, &kp, (aes_block *) in, nb);
+		crypton_aes_generic_encrypt_ecb((aes_block *) b, &kg, (aes_block *) in, nb);
+		if (sabotage && round == 3) a[0] ^= 1;
+		same("ecb encrypt", a, b, nb * 16);
+
+		crypton_aes_ppc8_decrypt_ecb((aes_block *) a, &kp, (aes_block *) in, nb);
+		crypton_aes_generic_decrypt_ecb((aes_block *) b, &kg, (aes_block *) in, nb);
+		same("ecb decrypt", a, b, nb * 16);
+
+		{
+			aes_block iv1, iv2;
+			memcpy(&iv1, iv, 16); memcpy(&iv2, iv, 16);
+			crypton_aes_ppc8_encrypt_cbc((aes_block *) a, &kp, &iv1, (aes_block *) in, nb);
+			crypton_aes_generic_encrypt_cbc((aes_block *) b, &kg, &iv2, (aes_block *) in, nb);
+			same("cbc encrypt", a, b, nb * 16);
+
+			memcpy(&iv1, iv, 16); memcpy(&iv2, iv, 16);
+			crypton_aes_ppc8_decrypt_cbc((aes_block *) a, &kp, &iv1, (aes_block *) in, nb);
+			crypton_aes_generic_decrypt_cbc((aes_block *) b, &kg, &iv2, (aes_block *) in, nb);
+			same("cbc decrypt", a, b, nb * 16);
+		}
+	}
+
+	/*
+	 * CTR, and the one place the two counters have to be reconciled by
+	 * hand.  crypton counts over all 128 bits and the assembly over the
+	 * low 32, so the work is handed over in runs that stop where that word
+	 * wraps.  A buffer that never reaches the wrap exercises none of that,
+	 * so the IV here is placed a few blocks before it on purpose.
+	 */
+	printf("== CTR, including the 32-bit counter wrapping ==\n");
+	for (round = 0; round < 200; round++) {
+		uint8_t key[32], in[600], a[600], b[600], iv[16];
+		uint8_t len = klens[round % 3];
+		uint32_t n = 1 + (round % 600);
+		aes_key kp, kg;
+		aes_block iv1, iv2;
+		uint32_t before = round % 5;   /* blocks left before the wrap */
+
+		rnd_fill(key, len);
+		rnd_fill(in, n);
+		rnd_fill(iv, 16);
+		/* put the low word within a few blocks of wrapping, and for a
+		 * third of the rounds exactly on it */
+		iv[12] = 0xff; iv[13] = 0xff; iv[14] = 0xff;
+		iv[15] = (uint8_t) (0x100 - before - 1);
+		keys(&kp, &kg, key, len);
+
+		memcpy(&iv1, iv, 16); memcpy(&iv2, iv, 16);
+		crypton_aes_ppc8_encrypt_ctr(a, &kp, &iv1, in, n);
+		crypton_aes_bitsliced_encrypt_ctr(b, &kg, &iv2, in, n);
+		if (sabotage && round == 7) a[n - 1] ^= 2;
+		same("ctr across the wrap", a, b, n);
+	}
+
+	printf("== XTS, both directions and a starting point ==\n");
+	for (round = 0; round < 200; round++) {
+		uint8_t k1[32], k2[32], in[16 * 11], a[16 * 11], b[16 * 11], du[16];
+		uint8_t len = klens[round % 3];
+		uint32_t nb = 1 + (round % 11);
+		uint32_t spoint = round % 4;
+		aes_key p1, g1, p2, g2;
+		aes_block d1;
+
+		rnd_fill(k1, len);
+		rnd_fill(k2, len);
+		rnd_fill(in, nb * 16);
+		rnd_fill(du, 16);
+		keys(&p1, &g1, k1, len);
+		keys(&p2, &g2, k2, len);
+
+		memcpy(&d1, du, 16);
+		crypton_aes_ppc8_encrypt_xts((aes_block *) a, &p1, &p2, &d1, spoint,
+		                             (aes_block *) in, nb);
+		xts_reference(b, &g1, &g2, du, spoint, in, nb, 0);
+		/* a[0], not a further block: at this round nb is 1 and anything
+		 * past the first block is outside what same() is given */
+		if (sabotage && round == 11) a[0] ^= 4;
+		same("xts encrypt", a, b, nb * 16);
+
+		memcpy(&d1, du, 16);
+		crypton_aes_ppc8_decrypt_xts((aes_block *) a, &p1, &p2, &d1, spoint,
+		                             (aes_block *) in, nb);
+		xts_reference(b, &g1, &g2, du, spoint, in, nb, 1);
+		same("xts decrypt", a, b, nb * 16);
+	}
+
+	/*
+	 * GHASH, where the two arguments do not take the same convention and
+	 * ppc8.c has to swap one of them.  Two different H and a non-zero
+	 * accumulator, because a symmetric H or a zero start can hide a wrong
+	 * convention.
+	 */
+	printf("== GHASH ==\n");
+	for (round = 0; round < 300; round++) {
+		uint8_t h[16], data[64], start[16];
+		table_4bit tp, tg;
+		block128 ap, ag;
+
+		rnd_fill(h, 16);
+		rnd_fill(data, sizeof data);
+		rnd_fill(start, 16);
+
+		crypton_aes_ppc8_hinit(tp, (const block128 *) h);
+		crypton_aes_generic_hinit(tg, (const block128 *) h);
+
+		memcpy(&ap, start, 16); memcpy(&ag, start, 16);
+		crypton_aes_ppc8_gf_mul4(&ap, (const block128 *) data, tp);
+		crypton_aes_generic_gf_mul4(&ag, (const block128 *) data, tg);
+		if (sabotage && round == 5) ((uint8_t *) &ap)[0] ^= 8;
+		same("gf_mul4", (const uint8_t *) &ap, (const uint8_t *) &ag, 16);
+
+		memcpy(&ap, start, 16); memcpy(&ag, start, 16);
+		crypton_aes_ppc8_gf_mul(&ap, tp);
+		crypton_aes_generic_gf_mul(&ag, tg);
+		same("gf_mul", (const uint8_t *) &ap, (const uint8_t *) &ag, 16);
+	}
+
+	printf("%s: %d mismatch(es)%s\n", failures ? "FAIL" : "ok", failures,
+	       sabotage ? "  (sabotage was asked for)" : "");
+	return failures != 0;
+}
diff --git a/cbits/tests/sysdrg/run.sh b/cbits/tests/sysdrg/run.sh
new file mode 100644
--- /dev/null
+++ b/cbits/tests/sysdrg/run.sh
@@ -0,0 +1,22 @@
+#!/bin/sh
+# Does the generator behind MonadRandom IO survive the things that break a
+# generator: a second thread, a reseed, and a fork?
+#
+# None of this can be asked from Haskell alone.  A forkIO thread is not an
+# operating system thread, and the Haskell test suite cannot fork the way a
+# child process must for the question to mean anything.
+#
+# Usage: cbits/tests/sysdrg/run.sh [build-dir]
+set -eu
+
+cbits=$(cd "$(dirname "$0")/../.." && pwd)
+out=${1:-$(mktemp -d)}
+
+${CC:-cc} -O2 -Wall -Wextra -DCRYPTON_SYSDRG_TESTING -I"$cbits" -o "$out/sysdrg" \
+    "$cbits/tests/sysdrg/sysdrg.c" \
+    "$cbits/crypton_sysdrg.c" \
+    "$cbits/crypton_sysrandom.c" \
+    "$cbits/crypton_chacha.c" \
+    "$cbits/crypton_sha512.c"
+
+"$out/sysdrg"
diff --git a/cbits/tests/sysdrg/sysdrg.c b/cbits/tests/sysdrg/sysdrg.c
new file mode 100644
--- /dev/null
+++ b/cbits/tests/sysdrg/sysdrg.c
@@ -0,0 +1,140 @@
+#include <stdio.h>
+#include <stdint.h>
+#include <string.h>
+#include <stdlib.h>
+#include <unistd.h>
+#include <pthread.h>
+#include <sys/wait.h>
+
+void crypton_sysdrg_test_key(uint8_t out[32]);
+void crypton_sysdrg_test_lock(void);
+void crypton_sysdrg_test_unlock(void);
+int crypton_sysdrg_bytes(uint8_t *out, int len);
+uint32_t crypton_sysdrg_generation(void);
+uint64_t crypton_sysdrg_thread_used(void);
+static uint64_t used_in_thread;
+
+static int fail = 0;
+static void check(const char *what, int ok) {
+    printf("%-52s %s\n", what, ok ? "ok" : "FAIL");
+    if (!ok) fail = 1;
+}
+
+/* Holds the process generator's lock for a moment, so that a fork can be
+ * made to happen while it is held.  Releasing after a delay rather than on
+ * demand is deliberate: a prepare handler has to be able to take the lock,
+ * and a holder that waited to be told would stop the fork instead of the
+ * child. */
+static pthread_mutex_t gate = PTHREAD_MUTEX_INITIALIZER;
+static pthread_cond_t gate_cv = PTHREAD_COND_INITIALIZER;
+static int holding = 0;
+
+static void *lock_holder(void *arg) {
+    (void) arg;
+    crypton_sysdrg_test_lock();
+    pthread_mutex_lock(&gate);
+    holding = 1;
+    pthread_cond_broadcast(&gate_cv);
+    pthread_mutex_unlock(&gate);
+    usleep(200000);
+    crypton_sysdrg_test_unlock();
+    return NULL;
+}
+
+static void *thread_body(void *arg) {
+    uint8_t *out = (uint8_t *) arg;
+    check("a second OS thread gets bytes", crypton_sysdrg_bytes(out, 32) == 32);
+    used_in_thread = crypton_sysdrg_thread_used();
+    return NULL;
+}
+
+int main(void) {
+    uint8_t a[32], b[32], big[4096];
+    memset(a, 0, 32); memset(b, 0, 32);
+
+    check("asking for 32 bytes gives 32", crypton_sysdrg_bytes(a, 32) == 32);
+    check("asking again gives something else",
+          crypton_sysdrg_bytes(b, 32) == 32 && memcmp(a, b, 32) != 0);
+    check("a 4096-byte request is filled", crypton_sysdrg_bytes(big, 4096) == 4096);
+    check("zero length is fine", crypton_sysdrg_bytes(a, 0) == 0);
+
+    /* two OS threads must not share a stream.  The main thread has already
+     * produced over 4 KiB by here, so a shared generator would show it. */
+    uint8_t t1[32], t2[32];
+    pthread_t p1, p2;
+    pthread_create(&p1, NULL, thread_body, t1);
+    pthread_join(p1, NULL);
+    pthread_create(&p2, NULL, thread_body, t2);
+    pthread_join(p2, NULL);
+    check("two OS threads do not produce the same bytes", memcmp(t1, t2, 32) != 0);
+    check("a new OS thread starts its own stream, not the main one's",
+          used_in_thread == 32);
+
+    /* crossing the per-thread reseed limit (1 MiB) must not repeat */
+    uint8_t before[32], after[32];
+    crypton_sysdrg_bytes(before, 32);
+    for (int i = 0; i < 1100; i++) { uint8_t junk[1024]; crypton_sysdrg_bytes(junk, 1024); }
+    crypton_sysdrg_bytes(after, 32);
+    check("output after a reseed differs from before", memcmp(before, after, 32) != 0);
+
+    /* fork: parent and child must diverge */
+    int fds[2];
+    if (pipe(fds) != 0) { perror("pipe"); return 2; }
+    uint8_t parent[32], child[32];
+    pid_t pid = fork();
+    if (pid == 0) {
+        close(fds[0]);
+        crypton_sysdrg_bytes(child, 32);
+        ssize_t w = write(fds[1], child, 32);
+        _exit(w == 32 ? 0 : 1);
+    }
+    close(fds[1]);
+    crypton_sysdrg_bytes(parent, 32);
+    ssize_t r = read(fds[0], child, 32);
+    int status = 0; waitpid(pid, &status, 0);
+    check("the child was able to produce bytes", r == 32 && status == 0);
+    check("parent and child do not produce the same bytes", memcmp(parent, child, 32) != 0);
+    check("the fork was noticed", crypton_sysdrg_generation() == 0);
+
+    /* Backtracking resistance.  The key that produced a draw is replaced
+     * once the bytes are out, so a state read afterwards is not the state
+     * that made them and cannot be wound back to remake them.  Comparing
+     * the key before and against the key after is the whole of it: without
+     * the rekey it is the same key and the counter alone says where to
+     * start. */
+    uint8_t key_before[32], key_after[32], drawn[32];
+    memset(key_before, 0, 32); memset(key_after, 0, 32);
+    crypton_sysdrg_test_key(key_before);
+    crypton_sysdrg_bytes(drawn, 32);
+    crypton_sysdrg_test_key(key_after);
+    check("a draw replaces the key that made it",
+          memcmp(key_before, key_after, 32) != 0);
+
+    /* A fork while another thread holds the process generator's lock.  The
+     * child's first draw has to reseed, the generation having changed, and
+     * reseeding takes that lock.  With no prepare and parent handlers the
+     * child inherits it locked and waits on it for ever -- the thread that
+     * would release it did not come across the fork.  The alarm is what
+     * turns that into a failure rather than a hung test run. */
+    pthread_t holder;
+    if (pthread_create(&holder, NULL, lock_holder, NULL) != 0) {
+        perror("pthread_create"); return 2;
+    }
+    pthread_mutex_lock(&gate);
+    while (!holding) pthread_cond_wait(&gate_cv, &gate);
+    pthread_mutex_unlock(&gate);
+
+    pid_t locked_pid = fork();
+    if (locked_pid == 0) {
+        uint8_t c[32];
+        alarm(5);
+        _exit(crypton_sysdrg_bytes(c, 32) == 32 ? 0 : 1);
+    }
+    pthread_join(holder, NULL);
+    int locked_status = 0;
+    waitpid(locked_pid, &locked_status, 0);
+    check("a child forked while the lock was held can draw",
+          WIFEXITED(locked_status) && WEXITSTATUS(locked_status) == 0);
+
+    return fail;
+}
diff --git a/crypton.cabal b/crypton.cabal
--- a/crypton.cabal
+++ b/crypton.cabal
@@ -1,10 +1,11 @@
 cabal-version:      3.0
 name:               crypton
-version:            2.1.10
+version:            2.2.0
 -- crypton's own code is BSD-3-Clause.  The parts of
 -- cbits/aes/gcm_fused_x86.c that follow picotls's fusion are MIT, and the
--- vendored s2n-bignum assembly in cbits/s2n is taken under ISC; each has
--- its licence beside it, and they are listed below.  The CRYPTOGAMS
+-- vendored s2n-bignum assembly in cbits/s2n is taken under ISC, and the
+-- constant-time AES and GHASH in cbits/bearssl are BearSSL's under MIT;
+-- each has its licence beside it, and they are listed below.  The CRYPTOGAMS
 -- assembly in cbits/asm is taken under its BSD-3-Clause option, and the
 -- AArch64 multiply-accumulate loop in cbits/crypton_bignum.h follows Go's,
 -- which is BSD-3-Clause too; the first term already covers both.
@@ -15,6 +16,7 @@
     cbits/aes/LICENSE.fusion
     cbits/asm/LICENSE.cryptogams
     cbits/s2n/LICENSE
+    cbits/bearssl/LICENSE
     cbits/mlkem/LICENSE
     cbits/mldsa/LICENSE
 copyright:
@@ -46,6 +48,9 @@
     cbits/asm/README.md
     cbits/asm/aesni-gcm-x86_64.pl
     cbits/asm/arm-xlate.pl
+    cbits/asm/aesp8-ppc.pl
+    cbits/asm/ghashp8-ppc.pl
+    cbits/asm/ppc-xlate.pl
     cbits/asm/arm_arch.h
     cbits/asm/chacha-armv8.pl
     cbits/asm/chacha-x86_64.pl
@@ -146,6 +151,11 @@
     cbits/s2n/import.sh
     cbits/s2n/include/*.h
     cbits/s2n/x86_att/*.S
+    cbits/bearssl/README.md
+    cbits/bearssl/VERSION
+    cbits/bearssl/import.sh
+    cbits/bearssl/inner.h
+    cbits/tests/*.c
     cbits/tests/ct/*.c
     cbits/tests/ct/*.h
     cbits/tests/ct/known.txt
@@ -167,9 +177,13 @@
     cbits/tests/scrub/README
     cbits/tests/scrub/known.txt
     cbits/tests/scrub/run.sh
+    cbits/tests/sysdrg/*.c
+    cbits/tests/sysdrg/run.sh
     cbits/tests/width/*.c
     cbits/tests/width/run.sh
     tests/*.hs
+    tests/tutorial/extract.awk
+    tests/tutorial/run.sh
 
 extra-doc-files:
     CHANGELOG.md
@@ -185,12 +199,6 @@
 
     manual:      True
 
-flag support_rdrand
-    description:
-        allow compilation with RDRAND on system and architecture that supports it
-
-    manual:      True
-
 flag support_pclmuldq
     description:
         Allow compilation with pclmuldq on architecture that supports it
@@ -355,6 +363,17 @@
         cbits/crypton_chacha.c
         cbits/crypton_chachapoly.c
         cbits/crypton_cpu.c
+        cbits/crypton_sysdrg.c
+        cbits/crypton_sysrandom.c
+        -- BearSSL's constant-time AES and GHASH, which cbits/aes/generic.c
+        -- and cbits/aes/gf.c are written on top of.  Unconditional because
+        -- those two are: every one of the three AES branches below names
+        -- them, accelerated or not.
+        cbits/bearssl/aes_ct64.c
+        cbits/bearssl/aes_ct64_dec.c
+        cbits/bearssl/aes_ct64_enc.c
+        cbits/bearssl/dec32le.c
+        cbits/bearssl/ghash_ctmul64.c
         cbits/crypton_des.c
         cbits/crypton_ecc.c
         cbits/crypton_f2m.c
@@ -441,6 +460,7 @@
         Crypto.Random.Entropy.Source
         Crypto.Random.HmacDRG
         Crypto.Random.Probabilistic
+        Crypto.Random.SysDRG
         Crypto.Random.SystemDRG
 
     default-language: Haskell2010
@@ -723,11 +743,6 @@
             cc-options:  -mavx2 -mbmi2
             asm-options: -mavx2 -mbmi2
 
-    if ((flag(support_rdrand) && (arch(i386) || arch(x86_64))) && !os(windows))
-        cpp-options:   -DSUPPORT_RDRAND
-        c-sources:     cbits/crypton_rdrand.c
-        other-modules: Crypto.Random.Entropy.RDRand
-
     if (flag(support_aesni) && arch(aarch64))
         cc-options: -DWITH_ARMV8_CRYPTO
         c-sources:
@@ -739,6 +754,52 @@
         if !flag(use_target_attributes)
             cc-options: -march=armv8-a+crypto
 
+    -- The same instructions, reached from AArch32.  cbits/aes/armv8.c needs
+    -- nothing changed to compile here: every intrinsic it uses exists in both
+    -- execution states, including the PMULL2 it takes through
+    -- vmull_high_p64, which the compiler lowers to a pair of VMULL.P64.  What
+    -- differs is how the operating system reports the instructions --
+    -- AArch32 has filled AT_HWCAP with older features and puts these in
+    -- AT_HWCAP2 -- which cbits/crypton_cpu.c now asks for.
+    --
+    -- -mfpu is passed whatever use_target_attributes says, unlike the branch
+    -- above.  AArch32 names an FPU rather than an architecture extension and
+    -- the two compilers spell the function attribute for it differently; a
+    -- global option is safe here, since a compiler emits these instructions
+    -- where an intrinsic asks for them and nowhere else.
+    --
+    -- A soft-float ARM that cannot take -mfpu fails loudly at the first C
+    -- file rather than quietly, and -f-support_aesni is the way out.
+    if (flag(support_aesni) && arch(arm))
+        cc-options: -DWITH_ARMV8_CRYPTO -mfpu=crypto-neon-fp-armv8
+        c-sources:
+            cbits/aes/generic.c
+            cbits/aes/gf.c
+            cbits/aes/armv8.c
+            cbits/crypton_aes.c
+
+    -- POWER8's vector AES and vector carry-less multiply, through the
+    -- CRYPTOGAMS assembly in cbits/asm.  No cc-options beyond the define:
+    -- cbits/aes/ppc8.c calls that assembly rather than using intrinsics of
+    -- its own, and the assembly says .machine "any", so the assembler takes
+    -- the instructions without being told about the processor.
+    --
+    -- Little-endian only.  The generator emits a big-endian flavour from the
+    -- same source, and it is not checked in: nothing here has been able to
+    -- run it, and an AES path that has never been executed is not one to
+    -- ship.  cbits/asm/generate.sh says what adding it would take.
+    if (flag(support_aesni) && arch(ppc64le))
+        cc-options: -DWITH_PPC8_CRYPTO
+        c-sources:
+            cbits/aes/generic.c
+            cbits/aes/gf.c
+            cbits/aes/ppc8.c
+            cbits/crypton_aes.c
+
+        asm-sources:
+            cbits/asm/aesp8-ppc-linux64le.S
+            cbits/asm/ghashp8-ppc-linux64le.S
+
     if arch(aarch64)
         cc-options:
             -DWITH_ARMV8_SHA1 -DWITH_ARMV8_SHA2 -DWITH_ARMV8_SHA3
@@ -864,7 +925,7 @@
     -- branch had already named every file it names.  Cabal drops the repeats,
     -- so nothing was built twice, but the line read as the fallback for a
     -- platform with no AES instructions and was not one.
-    if !((flag(support_aesni) && arch(aarch64)) || (flag(support_aesni) && (arch(i386) || arch(x86_64))))
+    if !((flag(support_aesni) && (arch(aarch64) || arch(arm))) || (flag(support_aesni) && (arch(i386) || arch(x86_64))) || (flag(support_aesni) && arch(ppc64le)))
         c-sources:
             cbits/aes/generic.c
             cbits/aes/gf.c
@@ -901,7 +962,9 @@
         build-depends:   Win32 <2.15
 
     else
-        other-modules: Crypto.Random.Entropy.Unix
+        other-modules:
+            Crypto.Random.Entropy.SysRandom
+            Crypto.Random.Entropy.Unix
 
     if (impl(ghc >=0) && flag(integer-gmp))
         build-depends: integer-gmp <1.2
@@ -986,7 +1049,9 @@
         PubKey.RabinSpec
         PubKey.SecrecySpec
         PubKey.RSASpec
+        EntropySpec
         RuntimeSpec
+        SysDRGSpec
         StreamCipher.ChaChaPoly1305Spec
         StreamCipher.ChaChaSpec
         StreamCipher.RC4Spec
@@ -1007,6 +1072,45 @@
     default-language: Haskell2010
     ghc-options:
         -Wall -fno-warn-orphans -fno-warn-missing-signatures -rtsopts
+        -threaded "-with-rtsopts=-N2"
+
+-- The generator behind MonadRandom and a fork made from Haskell.  Its own
+-- executables because the suite above runs with -N2, where GHC does not
+-- support forkProcess, and because the answer may differ between the
+-- threaded runtime and the one without it -- so it is built for both.
+-- Not on Windows, which has no fork and no unix package.
+test-suite test-forkprocess-threaded
+    type:             exitcode-stdio-1.0
+    main-is:          ForkProcess.hs
+    hs-source-dirs:   tests/forkprocess
+    default-language: Haskell2010
+    ghc-options:
+        -Wall -rtsopts -threaded "-with-rtsopts=-N1"
+
+    if os(windows)
+        buildable: False
+    else
+        build-depends:
+            base >=4.13 && <5,
+            bytestring,
+            crypton,
+            unix
+
+test-suite test-forkprocess-unthreaded
+    type:             exitcode-stdio-1.0
+    main-is:          ForkProcess.hs
+    hs-source-dirs:   tests/forkprocess
+    default-language: Haskell2010
+    ghc-options:      -Wall -rtsopts
+
+    if os(windows)
+        buildable: False
+    else
+        build-depends:
+            base >=4.13 && <5,
+            bytestring,
+            crypton,
+            unix
 
 benchmark bench-crypton
     type:             exitcode-stdio-1.0
diff --git a/tests/EntropySpec.hs b/tests/EntropySpec.hs
new file mode 100644
--- /dev/null
+++ b/tests/EntropySpec.hs
@@ -0,0 +1,58 @@
+-- | The system source of entropy.
+--
+-- There is nothing here to check the bytes against -- they are supposed to
+-- be unpredictable, and a test that said otherwise would be a test of the
+-- kernel.  What can be checked is that the right number of them comes back,
+-- which is where an implementation of the call with a limit per request
+-- goes wrong: @getentropy(3)@ refuses more than 256 bytes at a time, so a
+-- backend that forgets to loop answers a short buffer for anything larger.
+module EntropySpec (spec) where
+
+import Control.Exception (try)
+import qualified Data.ByteString as BS
+import Data.Maybe (catMaybes)
+import Foreign.Marshal.Alloc (allocaBytes)
+import Test.Hspec
+
+import Crypto.Random.Entropy (EntropyError (..), getEntropy)
+import Crypto.Random.Entropy.Unsafe (gatherBackend, replenish, supportedBackends)
+
+spec :: Spec
+spec = describe "the system entropy source" $ do
+    mapM_ lengthCase [0, 1, 31, 32, 255, 256, 257, 512, 1000]
+
+    it "does not answer the same thing twice" $ do
+        a <- getEntropy 64 :: IO BS.ByteString
+        b <- getEntropy 64 :: IO BS.ByteString
+        a `shouldNotBe` b
+
+    it "is not answering a constant" $ do
+        b <- getEntropy 1024 :: IO BS.ByteString
+        BS.length (BS.filter (== BS.head b) b) `shouldSatisfy` (< 64)
+
+    -- getEntropy would not notice: replenish tops a short answer up from
+    -- the next backend, so the buffer comes back full either way and the
+    -- system call is quietly replaced by the device file.  The backend has
+    -- to be asked on its own.
+    it "the best backend fills a buffer larger than one request on its own" $ do
+        bs <- catMaybes `fmap` sequence supportedBackends
+        case bs of
+            [] -> expectationFailure "no source of entropy on this system"
+            (b : _) -> do
+                n <- allocaBytes 1000 $ \ptr -> gatherBackend b ptr 1000
+                n `shouldBe` 1000
+
+    -- A system with no source of entropy is the one failure this library
+    -- cannot answer, and it used to be an `error`, which a caller could not
+    -- tell from a bug.  There is no way to take the sources away from a
+    -- running machine, but replenish can be handed the empty list, which is
+    -- the same question.
+    it "says so with an exception when there is no source at all" $ do
+        r <- allocaBytes 16 $ \ptr -> try (replenish 16 [] ptr)
+        r `shouldBe` Left NoEntropySource
+
+lengthCase :: Int -> Spec
+lengthCase n =
+    it ("gives back the " ++ show n ++ " bytes asked for") $ do
+        b <- getEntropy n :: IO BS.ByteString
+        BS.length b `shouldBe` n
diff --git a/tests/PubKey/RSASpec.hs b/tests/PubKey/RSASpec.hs
--- a/tests/PubKey/RSASpec.hs
+++ b/tests/PubKey/RSASpec.hs
@@ -1,3 +1,7 @@
+-- The properties below hold the new entry points against sign, signSafer
+-- and verify, which are deprecated as of this release.  Comparing against
+-- them is the point, so the warning is off here and nowhere else.
+{-# OPTIONS_GHC -Wno-deprecations #-}
 {-# LANGUAGE ExistentialQuantification #-}
 {-# LANGUAGE OverloadedStrings #-}
 
diff --git a/tests/RuntimeSpec.hs b/tests/RuntimeSpec.hs
--- a/tests/RuntimeSpec.hs
+++ b/tests/RuntimeSpec.hs
@@ -1,7 +1,72 @@
+-- | What 'processorOptions' says about the machine it is running on.
+--
+-- A test cannot know what processor it is on, so it cannot check that the
+-- answer is right.  What it can check is that the answer is well formed:
+-- that every name is distinct, that nothing is reported under a name from
+-- another architecture, and that the two questions which do not name one
+-- agree with the list.  The list is printed as well, because a CI log that
+-- says what each runner reported is the only way the detection itself gets
+-- looked at.
 module RuntimeSpec (spec) where
 
-import Crypto.System.CPU
+import Data.List (nub, sort)
 import Test.Hspec
 
+import Crypto.System.CPU (
+    ProcessorOption (..),
+    hasAESAcceleration,
+    hasGHASHAcceleration,
+    processorOptions,
+ )
+
+-- | Every name this module gives, which is also the set 'Show' has to
+-- cover.
+x86Options :: [ProcessorOption]
+x86Options =
+    [AESNI, PCLMUL, SSSE3, AVX, AVX2, SHANI, MOVBE, ADX, VAES, VAES512]
+
+armOptions :: [ProcessorOption]
+armOptions = [NEON, ARMAES, ARMPMULL, ARMSHA1, ARMSHA2, ARMSHA512]
+
+ppcOptions :: [ProcessorOption]
+ppcOptions = [PPCAES, PPCVPMSUM]
+
 spec :: Spec
-spec = it "CPU" $ putStrLn (show processorOptions)
+spec = describe "processorOptions" $ do
+    it "CPU" $ putStrLn (show processorOptions)
+
+    it "gives every name a number of its own" $
+        -- two patterns sharing a number would make one of them unreachable
+        -- and the other print under the wrong name
+        length (nub (x86Options ++ armOptions ++ ppcOptions))
+            `shouldBe` length (x86Options ++ armOptions ++ ppcOptions)
+
+    it "has a name for every option it names" $
+        -- Show falls back to "ProcessorOption n" for what it does not know,
+        -- which is for values from a newer release, not for these
+        filter (startsWith "ProcessorOption " . show) (x86Options ++ armOptions ++ ppcOptions)
+            `shouldBe` []
+
+    it "reports each option at most once, in order" $ do
+        processorOptions `shouldBe` sort processorOptions
+        nub processorOptions `shouldBe` processorOptions
+
+    it "does not mix one architecture's names with another's" $ do
+        let reported g = any (`elem` g) processorOptions
+        -- a machine reporting two of these groups is the
+        -- AArch64-says-AESNI fault this replaced
+        length (filter reported [x86Options, armOptions, ppcOptions])
+            `shouldSatisfy` (<= 1)
+
+    -- Half a check, and which half depends on the machine: where the
+    -- processor has AES this fails if the answer is broken to False and
+    -- passes if it is broken to True, and the other way round on a machine
+    -- without it.  Both were tried.
+    it "answers the architecture-free questions from the same list" $ do
+        hasAESAcceleration
+            `shouldBe` any (`elem` processorOptions) [AESNI, ARMAES, PPCAES]
+        hasGHASHAcceleration
+            `shouldBe` any (`elem` processorOptions) [PCLMUL, ARMPMULL, PPCVPMSUM]
+
+startsWith :: String -> String -> Bool
+startsWith p s = take (length p) s == p
diff --git a/tests/SysDRGSpec.hs b/tests/SysDRGSpec.hs
new file mode 100644
--- /dev/null
+++ b/tests/SysDRGSpec.hs
@@ -0,0 +1,100 @@
+-- | The generator behind 'MonadRandom' for 'IO'.
+--
+-- What can be asked from Haskell is narrow.  A @forkIO@ thread is not an
+-- operating system thread, and this process cannot fork, so the questions
+-- that matter most -- does a new operating system thread start its own
+-- stream, does a child after @fork@ diverge from its parent -- are in
+-- @cbits\/tests\/sysdrg@ instead.
+--
+-- What is left is still worth asking: the generator keeps a lock and a
+-- thread-local slot, and it is reached through a @safe@ foreign call, so
+-- many Haskell threads drawing at once is exactly the shape that deadlocks
+-- or hands two of them the same bytes.  That needs @-threaded@ and more
+-- than one capability, which is why the suite has them.
+module SysDRGSpec (spec) where
+
+import Control.Concurrent (
+    forkIO,
+    getNumCapabilities,
+    setNumCapabilities,
+    yield,
+ )
+import Control.Concurrent.MVar (newEmptyMVar, putMVar, takeMVar)
+import Control.Exception (finally)
+import Control.Monad (forM, forM_)
+import qualified Data.ByteString as BS
+import Data.List (group, nub, sort)
+import Test.Hspec
+
+import Crypto.Random (getRandomBytes)
+
+spec :: Spec
+spec = describe "the generator behind MonadRandom IO" $ do
+    it "runs on more than one capability, or the rest proves little" $ do
+        n <- getNumCapabilities
+        n `shouldSatisfy` (> 1)
+
+    mapM_ lengthCase [0, 1, 32, 1000, 4096]
+
+    it "gives every thread something different" $ do
+        -- plainly, rather than through async, which is not a dependency here
+        boxes <- forM [1 .. 256 :: Int] $ \_ -> do
+            box <- newEmptyMVar
+            _ <- forkIO $ do
+                b <- getRandomBytes 32 :: IO BS.ByteString
+                putMVar box b
+            return box
+        bss <- mapM takeMVar boxes
+        length (nub bss) `shouldBe` 256
+
+    it "keeps the streams apart while setNumCapabilities changes underneath" $ do
+        -- The state is held against the operating system thread rather than
+        -- against the capability, which is what makes this safe: a forkIO
+        -- thread moves between capabilities, and a capability is served by
+        -- different worker threads over its life, so state held against one
+        -- would be shared by threads running at the same time.  Raising the
+        -- count makes the runtime create worker threads, each of which has
+        -- to start its own stream; lowering it disables capabilities under
+        -- threads that are drawing.
+        n0 <- getNumCapabilities
+        let flips = concat (replicate 3 [1, 2, min 8 (n0 * 2), n0])
+            threads = 64 :: Int
+            draws = 16 :: Int
+        bss <-
+            ( do
+                flipped <- newEmptyMVar
+                _ <- forkIO $ do
+                    forM_ flips $ \n -> setNumCapabilities n >> yield
+                    putMVar flipped ()
+                boxes <- forM [1 .. threads] $ \_ -> do
+                    box <- newEmptyMVar
+                    _ <- forkIO $ do
+                        bs <- forM [1 .. draws] $ \_ -> do
+                            b <- getRandomBytes 32 :: IO BS.ByteString
+                            yield
+                            return b
+                        putMVar box bs
+                    return box
+                bss <- concat <$> mapM takeMVar boxes
+                takeMVar flipped
+                return bss
+            )
+                `finally` setNumCapabilities n0
+        -- sorted and grouped rather than nub, which is quadratic and this
+        -- list is long enough for that to show
+        length bss `shouldBe` threads * draws
+        length (group (sort bss)) `shouldBe` threads * draws
+
+    it "does not repeat across a reseed" $ do
+        -- the per-thread generator reseeds after a mebibyte
+        first <- getRandomBytes 32 :: IO BS.ByteString
+        forM_ [1 .. 300 :: Int] $ \_ ->
+            (getRandomBytes 4096 :: IO BS.ByteString) >>= \b -> b `seq` return ()
+        second <- getRandomBytes 32 :: IO BS.ByteString
+        second `shouldNotBe` first
+
+lengthCase :: Int -> Spec
+lengthCase n =
+    it ("gives back the " ++ show n ++ " bytes asked for") $ do
+        b <- getRandomBytes n :: IO BS.ByteString
+        BS.length b `shouldBe` n
diff --git a/tests/forkprocess/ForkProcess.hs b/tests/forkprocess/ForkProcess.hs
new file mode 100644
--- /dev/null
+++ b/tests/forkprocess/ForkProcess.hs
@@ -0,0 +1,81 @@
+-- | Does the generator behind 'MonadRandom' notice a fork made from
+-- Haskell?
+--
+-- @cbits\/tests\/sysdrg@ asks the same question of @fork(2)@ called from C,
+-- in a process with no runtime system in it at all.  That leaves the case
+-- anyone actually meets untested: 'forkProcess', with the runtime's own
+-- threads about.  The handler is registered with @pthread_atfork@, which
+-- libc runs for every @fork(2)@ whoever calls it, so it should reach this
+-- too -- should, which is why this is a test and not a comment.
+--
+-- It cannot live in the main suite.  That one runs with @-N2@, where GHC
+-- says 'forkProcess' is not supported; this needs its own runtime options,
+-- and is built twice, once threaded with one capability and once not
+-- threaded at all.  Both are configurations GHC supports 'forkProcess' in.
+--
+-- The check is that parent and child disagree.  Without fork detection the
+-- child carries on the parent's stream, so the next block each of them
+-- draws is the same block -- they would agree exactly, which is the fault.
+module Main (main) where
+
+import Control.Concurrent (getNumCapabilities, rtsSupportsBoundThreads)
+import Control.Monad (when)
+import qualified Data.ByteString as B
+import System.Exit (exitFailure)
+import System.IO (hClose, hFlush, hPutStrLn, stderr, stdout)
+import System.Posix.IO (closeFd, createPipe, fdToHandle)
+import System.Posix.Process (ProcessStatus (..), forkProcess, getProcessStatus)
+
+import Crypto.Random (getRandomBytes)
+
+draw :: IO B.ByteString
+draw = getRandomBytes 32
+
+main :: IO ()
+main = do
+    caps <- getNumCapabilities
+    putStrLn $
+        "threaded: "
+            ++ show rtsSupportsBoundThreads
+            ++ ", capabilities: "
+            ++ show caps
+    -- GHC supports forkProcess with -threaded only while one capability is
+    -- in use.  Saying so here means a change to the runtime options shows
+    -- up as a failure rather than as a test that quietly means nothing.
+    when (rtsSupportsBoundThreads && caps /= 1) $
+        die "this test needs one capability when threaded"
+
+    -- Before the fork, or the child inherits whatever is still in the
+    -- buffer and writes it out again when it exits.
+    hFlush stdout
+
+    -- Draw once first, so that both sides inherit a generator that has been
+    -- seeded.  A child of an unseeded one would seed itself for the first
+    -- time and differ for that reason instead of this one.
+    _ <- draw
+
+    (readEnd, writeEnd) <- createPipe
+    pid <- forkProcess $ do
+        closeFd readEnd
+        b <- draw
+        h <- fdToHandle writeEnd
+        B.hPut h b
+        hClose h
+    closeFd writeEnd
+    hr <- fdToHandle readEnd
+    fromChild <- B.hGet hr 32
+    hClose hr
+    status <- getProcessStatus True False pid
+
+    fromParent <- draw
+
+    case status of
+        Just (Exited _) -> return ()
+        other -> die ("the child did not exit cleanly: " ++ show other)
+    when (B.length fromChild /= 32) $
+        die ("the child sent " ++ show (B.length fromChild) ++ " bytes, not 32")
+    when (fromChild == fromParent) $
+        die "parent and child drew the same bytes: the fork went unnoticed"
+    putStrLn "parent and child drew different bytes"
+  where
+    die msg = hPutStrLn stderr ("FAIL: " ++ msg) >> exitFailure
diff --git a/tests/tutorial/extract.awk b/tests/tutorial/extract.awk
new file mode 100644
--- /dev/null
+++ b/tests/tutorial/extract.awk
@@ -0,0 +1,53 @@
+# Pull every code block out of a Haddock module and write each one as a
+# compilable Haskell module.
+#
+# A block is a maximal run of lines beginning "-- >", which is what Haddock
+# renders as a code sample.  The run is ended by any line that is not one,
+# so two samples separated by a line of prose are two modules and cannot
+# refer to each other -- which is the point: each sample has to stand on its
+# own, because that is how a reader will copy it.
+#
+# The name comes from the "-- $section" marker the block sits under, so a
+# compiler message names the section of the tutorial that is wrong rather
+# than a number.  The module header goes after any LANGUAGE pragmas, since
+# those have to come first.
+#
+# Writes one file per block into the directory given as -v out=, and prints
+# each module name on stdout.
+
+/^-- \$[A-Za-z_][A-Za-z0-9_]*$/ {
+    section = substr($0, 5)
+    nth = 0
+    next
+}
+
+/^-- >/ {
+    if (!inblock) {
+        inblock = 1
+        nlines = 0
+        nth++
+        mod = "Example_" section "_" nth
+        file = out "/" mod ".hs"
+    }
+    # "-- > foo" carries a space that is not part of the code; "-- >" alone
+    # is a blank line inside the block.
+    line = (length($0) > 5) ? substr($0, 6) : ""
+    lines[++nlines] = line
+    next
+}
+
+{ if (inblock) flush() }
+
+END { if (inblock) flush() }
+
+function flush(   i, k) {
+    # Pragmas, then the header, then the rest.
+    k = 1
+    while (k <= nlines && (lines[k] ~ /^\{-#/ || lines[k] == "")) k++
+    for (i = 1; i < k; i++) print lines[i] > file
+    print "module " mod " where" > file
+    for (i = k; i <= nlines; i++) print lines[i] > file
+    close(file)
+    print mod
+    inblock = 0
+}
diff --git a/tests/tutorial/run.sh b/tests/tutorial/run.sh
new file mode 100644
--- /dev/null
+++ b/tests/tutorial/run.sh
@@ -0,0 +1,126 @@
+#!/bin/sh
+# Does the tutorial still compile?
+#
+# Crypto/Tutorial.hs is the one place in the package whose code the compiler
+# never sees: every example in it is a Haddock block, so an API change
+# silently leaves it wrong and the next reader copies something that does
+# not build.  That has happened -- taking a checked key in Poly1305 made the
+# tutorial's crypto_box stop type-checking, and it was noticed by reading
+# the module rather than by anything here.
+#
+# So the blocks are pulled out and type-checked.  Each one becomes a module
+# of its own, because each one is a thing a reader copies whole; a block
+# that needs a definition from the block above it will fail here, which is
+# the right answer.
+#
+# Checked against this tree rather than against whatever crypton is
+# installed: cabal repl has the project's library as its home package, so a
+# tutorial written for an API this working tree has not got cannot pass by
+# finding the API in a released version on the machine.  -fno-code because
+# nothing is run -- the question is only whether it compiles.
+#
+# -Wall as well, and a warning in a tutorial counts as a failure: an unused
+# import is three words a reader will paste into their own file.  Only the
+# extracted modules are held to it, not the library loaded beside them.
+#
+# $GHC names a compiler other than the one on the path, and $BUILDDIR a
+# build directory to keep that compiler's products out of the usual one.
+# CI has one compiler per job and sets neither; locally they are how an
+# example is asked whether it also builds on the oldest GHC the package
+# supports, which is the reader this is most likely to have failed.
+#
+# The first module is wrong on purpose and has to be reported, or the run
+# proves nothing: a repl that failed to start, a ghci whose message format
+# changed, or an extractor that produced no modules would otherwise all look
+# like a tutorial that compiles.
+set -eu
+
+cd "$(dirname "$0")/../.."
+tutorial=Crypto/Tutorial.hs
+
+work=$(mktemp -d)
+trap 'rm -rf "$work"' EXIT INT TERM
+
+modules=$(awk -v out="$work" -f tests/tutorial/extract.awk "$tutorial")
+count=$(printf '%s\n' "$modules" | grep -c . || true)
+if [ "$count" -eq 0 ]; then
+    echo "FAIL no code blocks found in $tutorial"
+    exit 1
+fi
+echo "$count code blocks in $tutorial"
+
+# The deliberate one.  hashWith returns a Digest, and has since the module
+# was written, so this is the shape of every break this harness is for: the
+# tutorial says something the library no longer agrees with.
+cat > "$work/Calibration.hs" <<'HS'
+module Calibration where
+import Crypto.Hash (SHA1 (..), hashWith)
+thisIsNotADigest :: Int
+thisIsNotADigest = hashWith SHA1 "the harness has to report this"
+HS
+
+{
+    echo ':!echo @@ Calibration'
+    echo ":load $work/Calibration.hs"
+    for m in $modules; do
+        echo ":!echo @@ $m"
+        echo ":load $work/$m.hs"
+    done
+    echo ':quit'
+} > "$work/script"
+
+# Not cabal's -v0: that reaches ghci too and takes the "Ok, N modules
+# loaded." line with it, which is what is read below.
+cabal repl crypton ${GHC:+-w "$GHC"} ${BUILDDIR:+--builddir="$BUILDDIR"} \
+    --repl-options=-i"$work" --repl-options=-fno-code \
+    --repl-options=-Wall --repl-options=-Wno-missing-home-modules \
+    < "$work/script" > "$work/log" 2>&1 || true
+
+awk -v expected="$count" '
+    # ghci writes its prompt before the echo, so the marker is at the end
+    # of the line rather than the start of it.
+    /@@ [A-Za-z_]/ { mod = $NF; verdict[mod] = "no answer"; order[++n] = mod; next }
+    mod == "" { next }
+    /^Ok, [0-9]+ modules? loaded\.$/     { verdict[mod] = "ok";     next }
+    /^Failed, [0-9]+ modules? loaded\.$/ { verdict[mod] = "failed"; next }
+    # Only the extracted modules are held to -Wall; the library is loaded
+    # beside them and is not what this is asking about.
+    /Example_[A-Za-z0-9_]*\.hs:[0-9]+:[0-9]+: warning:/ { warned[mod] = 1 }
+    # Everything ghci said about a module before it gave its verdict, minus
+    # the progress lines, which are the bulk of it and say nothing.
+    /^(ghci> )?\[ *[0-9]+ of [0-9]+\] Compiling/ { next }
+    { if (verdict[mod] == "no answer" && $0 != "") detail[mod] = detail[mod] $0 "\n" }
+    END {
+        bad = 0
+        for (i = 1; i <= n; i++) {
+            m = order[i]
+            want = (m == "Calibration") ? "failed" : "ok"
+            got = verdict[m]
+            if (got == want && want == "ok" && warned[m]) got = "warned"
+            if (got == want) {
+                printf "ok   %s%s\n", m, (m == "Calibration" ? "   (reported, as it must be)" : "")
+            } else {
+                bad++
+                printf "FAIL %s: expected %s, got %s\n", m, want, got
+                printf "%s", detail[m]
+            }
+        }
+        # Nothing answered, or not everything did: the run itself did not
+        # happen the way this script assumes, so the log is worth seeing.
+        if (n != expected + 1) {
+            printf "FAIL %d modules answered, %d expected\n", n - 1, expected
+            exit 2
+        }
+        exit (bad > 0)
+    }
+' "$work/log" || {
+    status=$?
+    if [ "$status" -eq 2 ]; then
+        echo
+        echo "--- the last of the log ---"
+        tail -60 "$work/log"
+    fi
+    exit 1
+}
+
+echo "every example in $tutorial compiles"
