diff --git a/CHANGELOG b/CHANGELOG
--- a/CHANGELOG
+++ b/CHANGELOG
@@ -1,5 +1,17 @@
 # Changelog
 
+- 0.2.5 (2026-10-10)
+  * Further defends MAC comparison timing against "helpful" LLVM
+    optimizations.
+  * Detects support for the ARM SHA512 extensions at runtime, rather
+    than assuming it on all aarch64 builds, which crashed on aarch64
+    CPUs lacking them (e.g. Graviton2, Raspberry Pi).
+  * The portable implementation now lives in an internal
+    'Crypto.Hash.SHA512.Pure' module, and
+    'Crypto.Hash.SHA512.Arm' is no longer exposed. Both are hidden
+    internal modules; the public API is unchanged.
+  * Improves 'hash_lazy' and 'hmac_lazy' performance.
+
 - 0.2.4 (2026-07-06)
   * Reverts the unrolled MAC comparison introduced in 0.2.3, which was
     found to introduce timing variation on both aarch64 and x86-64 when
diff --git a/cbits/sha512_arm.c b/cbits/sha512_arm.c
--- a/cbits/sha512_arm.c
+++ b/cbits/sha512_arm.c
@@ -1,10 +1,19 @@
 #include <stdint.h>
 #include <string.h>
 
-#if defined(__aarch64__) && defined(__ARM_FEATURE_SHA512)
+#if defined(__aarch64__)
 
 #include <arm_neon.h>
 
+#if defined(__APPLE__)
+#include <sys/sysctl.h>
+#elif defined(__linux__)
+#include <sys/auxv.h>
+#ifndef HWCAP_SHA512
+#define HWCAP_SHA512 (1UL << 21)
+#endif
+#endif
+
 static const uint64_t K[80] = {
     0x428a2f98d728ae22ULL, 0x7137449123ef65cdULL,
     0xb5c0fbcfec4d3b2fULL, 0xe9b5dba58189dbbcULL,
@@ -55,7 +64,11 @@
  * block: pointer to 16 uint64_t words (already native endian)
  *
  * The state is updated in place.
+ *
+ * Compiled for the SHA512 extension only; callers must first check
+ * 'sha512_arm_available'.
  */
+__attribute__((target("+sha3")))
 void sha512_block_arm(uint64_t *state, const uint64_t *block) {
     /* Load current hash state */
     uint64x2_t ab = vld1q_u64(&state[0]);
@@ -446,14 +459,34 @@
     vst1q_u64(&state[6], gh);
 }
 
-/* Return 1 if ARM SHA512 is available, 0 otherwise */
+#if defined(__APPLE__)
+static int sysctl_flag(const char *name) {
+    int val = 0;
+    size_t len = sizeof(val);
+    if (sysctlbyname(name, &val, &len, NULL, 0) != 0) return 0;
+    return val != 0;
+}
+#endif
+
+/*
+ * Return 1 if the CPU implements the ARM SHA512 extension, 0
+ * otherwise.  We only know how to ask Darwin and Linux, and report 0
+ * (selecting the pure Haskell fallback) on any other OS.
+ */
 int sha512_arm_available(void) {
-    return 1;
+#if defined(__APPLE__)
+    return sysctl_flag("hw.optional.arm.FEAT_SHA512")
+        || sysctl_flag("hw.optional.armv8_2_sha512");
+#elif defined(__linux__)
+    return (getauxval(AT_HWCAP) & HWCAP_SHA512) != 0;
+#else
+    return 0;
+#endif
 }
 
 #else
 
-/* Stub implementations when ARM SHA512 is not available */
+/* Stub implementations for non-aarch64 builds */
 void sha512_block_arm(uint64_t *state, const uint64_t *block) {
     (void)state;
     (void)block;
diff --git a/lib/Crypto/Hash/SHA512.hs b/lib/Crypto/Hash/SHA512.hs
--- a/lib/Crypto/Hash/SHA512.hs
+++ b/lib/Crypto/Hash/SHA512.hs
@@ -1,9 +1,4 @@
 {-# OPTIONS_HADDOCK prune #-}
-{-# LANGUAGE BangPatterns #-}
-{-# LANGUAGE MagicHash #-}
-{-# LANGUAGE PatternSynonyms #-}
-{-# LANGUAGE UnboxedTuples #-}
-{-# LANGUAGE UnliftedNewtypes #-}
 
 -- |
 -- Module: Crypto.Hash.SHA512
@@ -36,20 +31,12 @@
   ) where
 
 import qualified Data.ByteString as BS
-import qualified Data.ByteString.Internal as BI
-import qualified Data.ByteString.Unsafe as BU
 import Data.Word (Word8, Word64)
 import Foreign.Ptr (Ptr)
-import qualified GHC.Exts as Exts
 import qualified Crypto.Hash.SHA512.Arm as Arm
-import Crypto.Hash.SHA512.Internal
+import Crypto.Hash.SHA512.Internal (MAC(..), Registers)
 import qualified Crypto.Hash.SHA512.Lazy as Lazy
-
--- utilities ------------------------------------------------------------------
-
-fi :: (Integral a, Num b) => a -> b
-fi = fromIntegral
-{-# INLINE fi #-}
+import qualified Crypto.Hash.SHA512.Pure as Pure
 
 -- hash -----------------------------------------------------------------------
 
@@ -63,39 +50,9 @@
 hash :: BS.ByteString -> BS.ByteString
 hash m
   | Arm.sha512_arm_available = Arm.hash m
-  | otherwise = cat (_hash 0 (iv ()) m)
+  | otherwise = Pure.hash m
 {-# INLINABLE hash #-}
 
-_hash
-  :: Word64        -- ^ extra prefix length for padding calculations
-  -> Registers     -- ^ register state
-  -> BS.ByteString -- ^ input
-  -> Registers
-_hash el rs m@(BI.PS _ _ l) = do
-  let !state = _hash_blocks rs m
-      !fin@(BI.PS _ _ ll) = BU.unsafeDrop (l - l `rem` 128) m
-      !total = el + fi l
-  if   ll < 112
-  then
-    let !ult = parse_pad1 fin total
-    in  update state ult
-  else
-    let !(# pen, ult #) = parse_pad2 fin total
-    in  update (update state pen) ult
-{-# INLINABLE _hash #-}
-
-_hash_blocks
-  :: Registers     -- ^ state
-  -> BS.ByteString -- ^ input
-  -> Registers
-_hash_blocks rs m@(BI.PS _ _ l) = loop rs 0 where
-  loop !acc !j
-    | j + 128 > l = acc
-    | otherwise   =
-        let !nacc = update acc (parse m j)
-        in  loop nacc (j + 128)
-{-# INLINABLE _hash_blocks #-}
-
 -- hmac ----------------------------------------------------------------------
 
 -- | Produce a message authentication code for a strict bytestring,
@@ -111,26 +68,9 @@
 hmac :: BS.ByteString -> BS.ByteString -> MAC
 hmac k m
   | Arm.sha512_arm_available = MAC (Arm.hmac k m)
-  | otherwise = MAC (cat (_hmac (prep_key k) m))
+  | otherwise = MAC (Pure.hmac k m)
 {-# INLINABLE hmac #-}
 
-prep_key :: BS.ByteString -> Block
-prep_key k@(BI.PS _ _ l)
-    | l > 128   = parse_key (hash k)
-    | otherwise = parse_key k
-{-# INLINABLE prep_key #-}
-
-_hmac
-  :: Block          -- ^ padded key
-  -> BS.ByteString  -- ^ message
-  -> Registers
-_hmac k m =
-  let !rs0   = update (iv ()) (xor k (Exts.wordToWord64# 0x3636363636363636##))
-      !block = pad_registers_with_length (_hash 128 rs0 m)
-      !rs1   = update (iv ()) (xor k (Exts.wordToWord64# 0x5C5C5C5C5C5C5C5C##))
-  in  update rs1 block
-{-# INLINABLE _hmac #-}
-
 -- the following functions are useful when we want to avoid allocating certain
 -- components of the HMAC key and message on the heap.
 
@@ -145,25 +85,9 @@
   -> IO ()
 _hmac_rr rp bp k m
   | Arm.sha512_arm_available = Arm._hmac_rr rp bp k m
-  | otherwise = do
-      let !key   = pad_registers k
-          !block = pad_registers_with_length m
-          !rs    = _hmac_bb key block
-      poke_registers rp rs
+  | otherwise = Pure._hmac_rr rp bp k m
 {-# INLINABLE _hmac_rr #-}
 
-_hmac_bb
-  :: Block     -- ^ key
-  -> Block     -- ^ message
-  -> Registers
-_hmac_bb k m =
-  let !rs0   = update (iv ()) (xor k (Exts.wordToWord64# 0x3636363636363636##))
-      !rs1   = update rs0 m
-      !inner = pad_registers_with_length rs1
-      !rs2   = update (iv ()) (xor k (Exts.wordToWord64# 0x5C5C5C5C5C5C5C5C##))
-  in  update rs2 inner
-{-# INLINABLE _hmac_bb #-}
-
 -- Calculate hmac(k, m) where m is the concatenation of v (registers), a
 -- separator byte, and a ByteString. This avoids allocating 'v' on the
 -- heap.
@@ -179,46 +103,5 @@
   -> IO ()
 _hmac_rsb rp bp k v sep dat
   | Arm.sha512_arm_available = Arm._hmac_rsb rp bp k v sep dat
-  | otherwise = do
-      let !key   = pad_registers k
-          !rs0   = update (iv ()) (xor key (Exts.wordToWord64# 0x3636363636363636##))
-          !inner = _hash_vsb 128 rs0 v sep dat
-          !block = pad_registers_with_length inner
-          !rs1   = update (iv ()) (xor key (Exts.wordToWord64# 0x5C5C5C5C5C5C5C5C##))
-          !rs    = update rs1 block
-      poke_registers rp rs
+  | otherwise = Pure._hmac_rsb rp bp k v sep dat
 {-# INLINABLE _hmac_rsb #-}
-
--- hash(v || sep || dat) with a custom initial state and extra
--- prefix length. used for producing a more specialized hmac.
-_hash_vsb
-  :: Word64        -- ^ extra prefix length
-  -> Registers     -- ^ initial state
-  -> Registers     -- ^ v
-  -> Word8         -- ^ sep
-  -> BS.ByteString -- ^ dat
-  -> Registers
-_hash_vsb el rs0 v sep dat@(BI.PS _ _ l)
-  | l >= 63 =
-      -- first block is complete
-      let !b0    = parse_vsb v sep dat
-          !rs1   = update rs0 b0
-          !rest  = BU.unsafeDrop 63 dat
-          !rlen  = l - 63
-          !rs2   = _hash_blocks rs1 rest
-          !flen  = rlen `rem` 128
-          !fin   = BU.unsafeDrop (rlen - flen) rest
-          !total = el + 65 + fi l
-      in  if   flen < 112
-          then update rs2 (parse_pad1 fin total)
-          else let !(# pen, ult #) = parse_pad2 fin total
-               in  update (update rs2 pen) ult
-  | otherwise =
-      -- message < 128 bytes, goes straight to padding
-      let !total = el + 65 + fi l
-      in  if   65 + l < 112
-          then update rs0 (parse_pad1_vsb v sep dat total)
-          else let !(# pen, ult #) = parse_pad2_vsb v sep dat total
-               in  update (update rs0 pen) ult
-{-# INLINABLE _hash_vsb #-}
-
diff --git a/lib/Crypto/Hash/SHA512/Arm.hs b/lib/Crypto/Hash/SHA512/Arm.hs
--- a/lib/Crypto/Hash/SHA512/Arm.hs
+++ b/lib/Crypto/Hash/SHA512/Arm.hs
@@ -24,6 +24,7 @@
 import qualified Data.ByteString.Internal as BI
 import qualified Data.ByteString.Unsafe as BU
 import Data.Word (Word8, Word64)
+import Foreign.C.Types (CInt(..))
 import Foreign.Marshal.Alloc (allocaBytes)
 import Foreign.Ptr (Ptr)
 import qualified GHC.Exts as Exts
@@ -38,7 +39,7 @@
   c_sha512_block :: Ptr Word64 -> Ptr Word64 -> IO ()
 
 foreign import ccall unsafe "sha512_arm_available"
-  c_sha512_arm_available :: IO Int
+  c_sha512_arm_available :: IO CInt
 
 -- utilities ------------------------------------------------------------------
 
diff --git a/lib/Crypto/Hash/SHA512/Internal.hs b/lib/Crypto/Hash/SHA512/Internal.hs
--- a/lib/Crypto/Hash/SHA512/Internal.hs
+++ b/lib/Crypto/Hash/SHA512/Internal.hs
@@ -50,6 +50,7 @@
   , poke_registers
   ) where
 
+import Data.Barrier (barrier)
 import qualified Data.Bits as B
 import qualified Data.ByteString as BS
 import qualified Data.ByteString.Internal as BI
@@ -89,10 +90,13 @@
       -- fused fold: OR the bytewise XORs into an accumulator
       -- directly, rather than via packZipWith, so no intermediate
       -- ByteString holding the (secret-derived) difference bytes
-      -- is ever materialised on the heap.
+      -- is ever materialised on the heap. The accumulator is routed
+      -- through 'barrier' before the zero-test so the LLVM backend
+      -- cannot recover the array-equality idiom and short-circuit on
+      -- the first mismatch (see "Data.Barrier").
       go :: Word8 -> Int -> Bool
       go !acc !i
-        | i == la   = acc == 0
+        | i == la   = barrier acc == 0
         | otherwise =
             let !x = BU.unsafeIndex a i
                 !y = BU.unsafeIndex b i
diff --git a/lib/Crypto/Hash/SHA512/Lazy.hs b/lib/Crypto/Hash/SHA512/Lazy.hs
--- a/lib/Crypto/Hash/SHA512/Lazy.hs
+++ b/lib/Crypto/Hash/SHA512/Lazy.hs
@@ -22,10 +22,10 @@
   ) where
 
 import Crypto.Hash.SHA512.Internal
+import qualified Crypto.Hash.SHA512.Pure as Pure
 import qualified Data.Bits as B
 import qualified Data.ByteString as BS
 import qualified Data.ByteString.Builder as BSB
-import qualified Data.ByteString.Builder.Extra as BE
 import qualified Data.ByteString.Internal as BI
 import qualified Data.ByteString.Lazy as BL
 import qualified Data.ByteString.Lazy.Internal as BLI
@@ -77,26 +77,20 @@
 
 -- k such that (l + 1 + k) mod 128 = 112
 sol :: Word64 -> Word64
-sol l =
-  let r = 112 - fi l `rem` 128 - 1 :: Integer -- fi prevents underflow
-  in  fi (if r < 0 then r + 128 else r)
+sol l = (111 - l) B..&. 127 -- wraps mod 2^64, a multiple of 128
 
 -- RFC 6234 4.1 (lazy)
 pad_lazy :: BL.ByteString -> BL.ByteString
 pad_lazy (BL.toChunks -> m) = BL.fromChunks (walk 0 m) where
   walk !l bs = case bs of
     (c:cs) -> c : walk (l + fi (BS.length c)) cs
-    [] -> padding l (sol l) (BSB.word8 0x80)
+    [] -> [padding l]
 
-  padding l k bs
-    | k == 0 =
-          pure
-        . to_strict
-          -- more efficient for small builder
-        $ bs <> BSB.word64BE 0x00 <> BSB.word64BE (l * 8)
-    | otherwise =
-        let nacc = bs <> BSB.word8 0x00
-        in  padding l (pred k) nacc
+  padding l = to_strict $
+       BSB.word8 0x80
+    <> BSB.byteString (BS.replicate (fi (sol l)) 0x00)
+    <> BSB.word64BE 0x00
+    <> BSB.word64BE (l * 8)
 
 -- | Compute a condensed representation of a lazy bytestring via
 --   SHA-512.
@@ -137,37 +131,6 @@
         step4 = hash_lazy step3
         step5 = BS.map (B.xor 0x5C) step1
         step6 = step5 <> step4
-    in  MAC (hash step6)
+    in  MAC (Pure.hash step6)
   where
-    hash bs = cat (go (iv ()) (pad bs)) where
-      go :: Registers -> BS.ByteString -> Registers
-      go !acc b
-        | BS.null b = acc
-        | otherwise = case unsafe_splitAt 128 b of
-            SSPair c r -> go (update acc (parse c 0)) r
-
-      pad m@(BI.PS _ _ (fi -> len))
-          | len < 256 = to_strict_small padded
-          | otherwise = to_strict padded
-        where
-          padded = BSB.byteString m
-                <> fill (sol len) (BSB.word8 0x80)
-                <> BSB.word64BE 0x00
-                <> BSB.word64BE (len * 8)
-
-          to_strict_small = BL.toStrict . BE.toLazyByteStringWith
-            (BE.safeStrategy 256 BE.smallChunkSize) mempty
-
-          fill j !acc
-            | j `rem` 8 == 0 = loop64 j acc
-            | otherwise = loop8 j acc
-
-          loop64 j !acc
-            | j == 0 = acc
-            | otherwise = loop64 (j - 8) (acc <> BSB.word64BE 0x00)
-
-          loop8 j !acc
-            | j == 0 = acc
-            | otherwise = loop8 (pred j) (acc <> BSB.word8 0x00)
-
-    !(k, lk) = if l > 128 then (hash mk, 64) else (mk, l)
+    !(k, lk) = if l > 128 then (Pure.hash mk, 64) else (mk, l)
diff --git a/lib/Crypto/Hash/SHA512/Pure.hs b/lib/Crypto/Hash/SHA512/Pure.hs
new file mode 100644
--- /dev/null
+++ b/lib/Crypto/Hash/SHA512/Pure.hs
@@ -0,0 +1,189 @@
+{-# OPTIONS_HADDOCK hide #-}
+{-# LANGUAGE BangPatterns #-}
+{-# LANGUAGE MagicHash #-}
+{-# LANGUAGE UnboxedTuples #-}
+
+-- |
+-- Module: Crypto.Hash.SHA512.Pure
+-- Copyright: (c) 2024 Jared Tobin
+-- License: MIT
+-- Maintainer: Jared Tobin <jared@ppad.tech>
+--
+-- Pure Haskell SHA-512 and HMAC-SHA512 implementations for strict
+-- ByteStrings, used when the ARM cryptographic extensions are
+-- unavailable.
+
+module Crypto.Hash.SHA512.Pure (
+    hash
+  , hmac
+  , _hmac_rr
+  , _hmac_rsb
+  ) where
+
+import qualified Data.ByteString as BS
+import qualified Data.ByteString.Internal as BI
+import qualified Data.ByteString.Unsafe as BU
+import Data.Word (Word8, Word64)
+import Foreign.Ptr (Ptr)
+import qualified GHC.Exts as Exts
+import Crypto.Hash.SHA512.Internal
+
+-- utilities ------------------------------------------------------------------
+
+fi :: (Integral a, Num b) => a -> b
+fi = fromIntegral
+{-# INLINE fi #-}
+
+-- hash -----------------------------------------------------------------------
+
+-- | Compute a condensed representation of a strict bytestring via
+--   SHA-512, using the pure Haskell implementation.
+hash :: BS.ByteString -> BS.ByteString
+hash m = cat (_hash 0 (iv ()) m)
+{-# INLINABLE hash #-}
+
+_hash
+  :: Word64        -- ^ extra prefix length for padding calculations
+  -> Registers     -- ^ register state
+  -> BS.ByteString -- ^ input
+  -> Registers
+_hash el rs m@(BI.PS _ _ l) = do
+  let !state = _hash_blocks rs m
+      !fin@(BI.PS _ _ ll) = BU.unsafeDrop (l - l `rem` 128) m
+      !total = el + fi l
+  if   ll < 112
+  then
+    let !ult = parse_pad1 fin total
+    in  update state ult
+  else
+    let !(# pen, ult #) = parse_pad2 fin total
+    in  update (update state pen) ult
+{-# INLINABLE _hash #-}
+
+_hash_blocks
+  :: Registers     -- ^ state
+  -> BS.ByteString -- ^ input
+  -> Registers
+_hash_blocks rs m@(BI.PS _ _ l) = loop rs 0 where
+  loop !acc !j
+    | j + 128 > l = acc
+    | otherwise   =
+        let !nacc = update acc (parse m j)
+        in  loop nacc (j + 128)
+{-# INLINABLE _hash_blocks #-}
+
+-- hmac ----------------------------------------------------------------------
+
+-- | Produce a message authentication code for a strict bytestring,
+--   based on the provided (strict, bytestring) key, via HMAC-SHA512,
+--   using the pure Haskell implementation.
+hmac :: BS.ByteString -> BS.ByteString -> BS.ByteString
+hmac k m = cat (_hmac (prep_key k) m)
+{-# INLINABLE hmac #-}
+
+prep_key :: BS.ByteString -> Block
+prep_key k@(BI.PS _ _ l)
+    | l > 128   = parse_key (hash k)
+    | otherwise = parse_key k
+{-# INLINABLE prep_key #-}
+
+_hmac
+  :: Block          -- ^ padded key
+  -> BS.ByteString  -- ^ message
+  -> Registers
+_hmac k m =
+  let !rs0   = update (iv ()) (xor k (Exts.wordToWord64# 0x3636363636363636##))
+      !block = pad_registers_with_length (_hash 128 rs0 m)
+      !rs1   = update (iv ()) (xor k (Exts.wordToWord64# 0x5C5C5C5C5C5C5C5C##))
+  in  update rs1 block
+{-# INLINABLE _hmac #-}
+
+-- the following functions are useful when we want to avoid allocating certain
+-- components of the HMAC key and message on the heap.
+
+-- Computes hmac(k, v) when k and v are Registers.
+--
+-- The 64-byte result is written to the destination pointer.
+_hmac_rr
+  :: Ptr Word64    -- ^ destination (8 Word64s)
+  -> Ptr Word64    -- ^ scratch block buffer (unused)
+  -> Registers     -- ^ key
+  -> Registers     -- ^ message
+  -> IO ()
+_hmac_rr rp _ k m = do
+  let !key   = pad_registers k
+      !block = pad_registers_with_length m
+      !rs    = _hmac_bb key block
+  poke_registers rp rs
+{-# INLINABLE _hmac_rr #-}
+
+_hmac_bb
+  :: Block     -- ^ key
+  -> Block     -- ^ message
+  -> Registers
+_hmac_bb k m =
+  let !rs0   = update (iv ()) (xor k (Exts.wordToWord64# 0x3636363636363636##))
+      !rs1   = update rs0 m
+      !inner = pad_registers_with_length rs1
+      !rs2   = update (iv ()) (xor k (Exts.wordToWord64# 0x5C5C5C5C5C5C5C5C##))
+  in  update rs2 inner
+{-# INLINABLE _hmac_bb #-}
+
+-- Calculate hmac(k, m) where m is the concatenation of v (registers), a
+-- separator byte, and a ByteString. This avoids allocating 'v' on the
+-- heap.
+--
+-- The 64-byte result is written to the destination pointer.
+_hmac_rsb
+  :: Ptr Word64    -- ^ destination pointer (8 x Word64)
+  -> Ptr Word64    -- ^ scratch block pointer (unused)
+  -> Registers     -- ^ k
+  -> Registers     -- ^ v
+  -> Word8         -- ^ separator byte
+  -> BS.ByteString -- ^ data
+  -> IO ()
+_hmac_rsb rp _ k v sep dat = do
+  let !key   = pad_registers k
+      !rs0   = update (iv ())
+                 (xor key (Exts.wordToWord64# 0x3636363636363636##))
+      !inner = _hash_vsb 128 rs0 v sep dat
+      !block = pad_registers_with_length inner
+      !rs1   = update (iv ())
+                 (xor key (Exts.wordToWord64# 0x5C5C5C5C5C5C5C5C##))
+      !rs    = update rs1 block
+  poke_registers rp rs
+{-# INLINABLE _hmac_rsb #-}
+
+-- hash(v || sep || dat) with a custom initial state and extra
+-- prefix length. used for producing a more specialized hmac.
+_hash_vsb
+  :: Word64        -- ^ extra prefix length
+  -> Registers     -- ^ initial state
+  -> Registers     -- ^ v
+  -> Word8         -- ^ sep
+  -> BS.ByteString -- ^ dat
+  -> Registers
+_hash_vsb el rs0 v sep dat@(BI.PS _ _ l)
+  | l >= 63 =
+      -- first block is complete
+      let !b0    = parse_vsb v sep dat
+          !rs1   = update rs0 b0
+          !rest  = BU.unsafeDrop 63 dat
+          !rlen  = l - 63
+          !rs2   = _hash_blocks rs1 rest
+          !flen  = rlen `rem` 128
+          !fin   = BU.unsafeDrop (rlen - flen) rest
+          !total = el + 65 + fi l
+      in  if   flen < 112
+          then update rs2 (parse_pad1 fin total)
+          else let !(# pen, ult #) = parse_pad2 fin total
+               in  update (update rs2 pen) ult
+  | otherwise =
+      -- message < 128 bytes, goes straight to padding
+      let !total = el + 65 + fi l
+      in  if   65 + l < 112
+          then update rs0 (parse_pad1_vsb v sep dat total)
+          else let !(# pen, ult #) = parse_pad2_vsb v sep dat total
+               in  update (update rs0 pen) ult
+{-# INLINABLE _hash_vsb #-}
+
diff --git a/lib/Data/Barrier.hs b/lib/Data/Barrier.hs
new file mode 100644
--- /dev/null
+++ b/lib/Data/Barrier.hs
@@ -0,0 +1,34 @@
+{-# OPTIONS_HADDOCK hide #-}
+
+-- |
+-- Module: Data.Barrier
+-- Copyright: (c) 2026 Jared Tobin
+-- License: MIT
+-- Maintainer: Jared Tobin <jared@ppad.tech>
+--
+-- An optimisation barrier for constant-time code.
+
+module Data.Barrier (
+    barrier
+  ) where
+
+import Data.Word (Word8)
+
+-- | Identity on 'Word8', but opaque to the optimiser. A constant-time
+--   accumulate-then-compare (e.g. an OR-fold of bytewise XORs, tested
+--   against zero) routes its accumulator through this before the
+--   zero-test, so the compiler cannot recognise it as an array-equality
+--   test and lower it to a short-circuiting byte comparison (which would
+--   leak the mismatch position).
+--
+--   Both properties are required and must not be \"tidied\" away:
+--
+--     * @NOINLINE@ -- if GHC inlines it, the LLVM backend regains the
+--       accumulator's definition and short-circuits again.
+--     * a /separate/ module -- a caller then compiles @barrier@ to an
+--       external call it cannot see through. Inline it into the caller
+--       and the barrier is gone.
+--
+barrier :: Word8 -> Word8
+barrier x = x
+{-# NOINLINE barrier #-}
diff --git a/ppad-sha512.cabal b/ppad-sha512.cabal
--- a/ppad-sha512.cabal
+++ b/ppad-sha512.cabal
@@ -1,6 +1,6 @@
 cabal-version:      3.0
 name:               ppad-sha512
-version:            0.2.4
+version:            0.2.5
 synopsis:           The SHA-512 and HMAC-SHA512 algorithms
 license:            MIT
 license-file:       LICENSE
@@ -37,16 +37,17 @@
     ghc-options: -fllvm -O2
   exposed-modules:
       Crypto.Hash.SHA512
-      Crypto.Hash.SHA512.Arm
       Crypto.Hash.SHA512.Internal
       Crypto.Hash.SHA512.Lazy
+      Crypto.Hash.SHA512.Pure
+  other-modules:
+      Crypto.Hash.SHA512.Arm
+      Data.Barrier
   build-depends:
       base >= 4.9 && < 5
     , bytestring >= 0.9 && < 0.13
   c-sources:
       cbits/sha512_arm.c
-  if arch(aarch64)
-    cc-options: -march=armv8.2-a+sha3
   if flag(sanitize)
     cc-options: -fsanitize=address,undefined -fno-omit-frame-pointer
     ghc-options: -optl=-fsanitize=address,undefined
diff --git a/test/Property.hs b/test/Property.hs
--- a/test/Property.hs
+++ b/test/Property.hs
@@ -1,13 +1,26 @@
 {-# OPTIONS_GHC -fno-warn-orphans #-}
+{-# LANGUAGE MagicHash #-}
 
 module Property (
     properties
   ) where
 
+import Data.Bits ((.|.))
+import qualified Data.Bits as B
 import qualified Data.ByteString as BS
+import qualified Data.ByteString.Builder as BSB
 import qualified Data.ByteString.Lazy as BL
+import Data.Word (Word8, Word64)
+import Foreign.Marshal.Alloc (allocaBytes)
+import Foreign.Marshal.Array (peekArray)
+import Foreign.Ptr (Ptr)
+import qualified GHC.Exts as Exts
+import qualified GHC.Word
 import Crypto.Hash.SHA512
+import Crypto.Hash.SHA512.Internal (Registers(..))
+import qualified Crypto.Hash.SHA512.Pure as Pure
 import Test.Tasty
+import qualified Test.Tasty.HUnit as H
 import qualified Test.Tasty.QuickCheck as Q
 import Test.QuickCheck.Instances.ByteString ()
 
@@ -41,6 +54,102 @@
       lazy_chunked = BL.fromChunks (chunk_by (map Q.getPositive chunks) m)
   in  hmac_lazy k lazy_single == hmac_lazy k lazy_chunked
 
+-- dispatching vs. pure implementations ------------------------------------
+--
+-- on hosts with the ARM SHA512 extension, the public functions use it, so
+-- these compare it against the pure Haskell implementation.
+
+-- messages spanning several blocks
+newtype Msg = Msg BS.ByteString
+  deriving Show
+
+instance Q.Arbitrary Msg where
+  arbitrary = do
+    l <- Q.chooseInt (0, 8 * 128)
+    Msg . BS.pack <$> Q.vectorOf l Q.arbitrary
+
+-- register-sized strings
+newtype Bytes64 = Bytes64 BS.ByteString
+  deriving Show
+
+instance Q.Arbitrary Bytes64 where
+  arbitrary = Bytes64 . BS.pack <$> Q.vectorOf 64 Q.arbitrary
+
+-- big-endian words of a 64-byte string, as registers
+to_registers :: BS.ByteString -> Registers
+to_registers bs = R (w 0) (w 1) (w 2) (w 3) (w 4) (w 5) (w 6) (w 7) where
+  w :: Int -> Exts.Word64#
+  w i = case word64 (BS.take 8 (BS.drop (8 * i) bs)) of
+    GHC.Word.W64# x -> x
+  word64 :: BS.ByteString -> Word64
+  word64 = BS.foldl' (\acc b -> acc `B.shiftL` 8 .|. fromIntegral b) 0
+
+-- run a destination-pointer HMAC, returning the big-endian digest
+run_ptr :: (Ptr Word64 -> Ptr Word64 -> IO ()) -> IO BS.ByteString
+run_ptr act = allocaBytes 64 $ \rp -> allocaBytes 128 $ \bp -> do
+  act rp bp
+  ws <- peekArray 8 rp
+  pure (BL.toStrict (BSB.toLazyByteString (foldMap BSB.word64BE ws)))
+
+unmac :: MAC -> BS.ByteString
+unmac (MAC m) = m
+
+-- deterministic input of the given length
+msg :: Int -> BS.ByteString
+msg l = BS.pack (fmap fromIntegral (take l [(7 :: Int), 20 ..]))
+
+hash_matches_pure :: Msg -> Bool
+hash_matches_pure (Msg m) = hash m == Pure.hash m
+
+hmac_matches_pure :: Msg -> Msg -> Bool
+hmac_matches_pure (Msg k) (Msg m) = unmac (hmac k m) == Pure.hmac k m
+
+hmac_rr_matches_hmac :: Bytes64 -> Bytes64 -> Q.Property
+hmac_rr_matches_hmac (Bytes64 k) (Bytes64 m) = Q.ioProperty $ do
+  let expected = unmac (hmac k m)
+  a <- run_ptr (\rp bp -> _hmac_rr rp bp (to_registers k) (to_registers m))
+  b <- run_ptr (\rp bp ->
+         Pure._hmac_rr rp bp (to_registers k) (to_registers m))
+  pure (a == expected && b == expected)
+
+hmac_rsb_matches_hmac :: Bytes64 -> Bytes64 -> Word8 -> Msg -> Q.Property
+hmac_rsb_matches_hmac (Bytes64 k) (Bytes64 v) sep (Msg dat) =
+  Q.ioProperty (rsb_matches k v sep dat)
+
+rsb_matches
+  :: BS.ByteString -> BS.ByteString -> Word8 -> BS.ByteString -> IO Bool
+rsb_matches k v sep dat = do
+  let expected = unmac (hmac k (v <> BS.singleton sep <> dat))
+  a <- run_ptr (\rp bp ->
+         _hmac_rsb rp bp (to_registers k) (to_registers v) sep dat)
+  b <- run_ptr (\rp bp ->
+         Pure._hmac_rsb rp bp (to_registers k) (to_registers v) sep dat)
+  pure (a == expected && b == expected)
+
+-- every input length through three blocks, hitting each padding case
+hash_all_lengths :: TestTree
+hash_all_lengths = H.testCase "hash ~ Pure.hash (all lengths)" $
+  H.assertBool mempty $
+    all (\l -> hash (msg l) == Pure.hash (msg l)) [0 .. 3 * 128]
+
+hmac_all_lengths :: TestTree
+hmac_all_lengths = H.testCase "hmac ~ Pure.hmac (all lengths)" $
+  H.assertBool mempty $
+    all (\l -> unmac (hmac (msg 64) (msg l)) == Pure.hmac (msg 64) (msg l))
+      [0 .. 3 * 128]
+
+hmac_all_key_lengths :: TestTree
+hmac_all_key_lengths = H.testCase "hmac ~ Pure.hmac (all key lengths)" $
+  H.assertBool mempty $
+    all (\l -> unmac (hmac (msg l) (msg 3)) == Pure.hmac (msg l) (msg 3))
+      [0 .. 2 * 128 + 1]
+
+hmac_rsb_all_lengths :: TestTree
+hmac_rsb_all_lengths = H.testCase "_hmac_rsb ~ hmac (all lengths)" $ do
+  oks <- traverse (rsb_matches (msg 64) (BS.reverse (msg 64)) 0x01 . msg)
+           [0 .. 3 * 128]
+  H.assertBool mempty (and oks)
+
 chunk_by :: [Int] -> BS.ByteString -> [BS.ByteString]
 chunk_by _ bs | BS.null bs = []
 chunk_by [] bs = [bs]
@@ -62,4 +171,18 @@
       Q.withMaxSuccess 1000 hash_chunking
   , Q.testProperty "hmac_lazy chunking-invariant" $
       Q.withMaxSuccess 1000 hmac_chunking
+  , testGroup "dispatch ~ pure" [
+      Q.testProperty "hash ~ Pure.hash" $
+        Q.withMaxSuccess 1000 hash_matches_pure
+    , Q.testProperty "hmac ~ Pure.hmac" $
+        Q.withMaxSuccess 1000 hmac_matches_pure
+    , Q.testProperty "_hmac_rr k m ~ hmac (cat k) (cat m)" $
+        Q.withMaxSuccess 1000 hmac_rr_matches_hmac
+    , Q.testProperty "_hmac_rsb k v s d ~ hmac (cat k) (cat v <> s <> d)" $
+        Q.withMaxSuccess 1000 hmac_rsb_matches_hmac
+    , hash_all_lengths
+    , hmac_all_lengths
+    , hmac_all_key_lengths
+    , hmac_rsb_all_lengths
+    ]
   ]
