bytestring-encodings 0.1.0.1 → 0.2.0.1
raw patch · 6 files changed
+255/−237 lines, 6 filesdep ~basePVP ok
version bump matches the API change (PVP)
Dependency ranges changed: base
API changes (from Hackage documentation)
- Data.ByteString.Ascii: isAscii :: ByteString -> Bool
- Data.ByteString.Utf8: instance GHC.Classes.Eq Data.ByteString.Utf8.Utf8
- Data.ByteString.Utf8: isUtf8 :: ByteString -> Bool
- Data.ByteString.Utf8: isUtf8' :: ByteString -> Bool
+ Data.ByteString.Encodings: isAscii :: ByteString -> Bool
+ Data.ByteString.Encodings: isUtf8 :: ByteString -> Bool
+ Data.ByteString.Encodings: isUtf8' :: ByteString -> Bool
+ Data.ByteString.Internal.Ascii: isAscii :: ByteString -> Bool
+ Data.ByteString.Internal.Utf8: instance GHC.Classes.Eq Data.ByteString.Internal.Utf8.Utf8
+ Data.ByteString.Internal.Utf8: isUtf8 :: ByteString -> Bool
+ Data.ByteString.Internal.Utf8: isUtf8' :: ByteString -> Bool
Files
- Data/ByteString/Ascii.hs +0/−98
- Data/ByteString/Encodings.hs +10/−0
- Data/ByteString/Internal/Ascii.hs +106/−0
- Data/ByteString/Internal/Utf8.hs +131/−0
- Data/ByteString/Utf8.hs +0/−125
- bytestring-encodings.cabal +8/−14
− Data/ByteString/Ascii.hs
@@ -1,98 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE MagicHash #-}-{-# LANGUAGE NoImplicitPrelude #-}--{-# OPTIONS_GHC -O2 #-}--module Data.ByteString.Ascii- ( isAscii- ) where--import Data.Bits ( (.&.) )-import Data.ByteString.Internal (ByteString(..), accursedUnutterablePerformIO)-import Data.Word (Word8, Word64)-import GHC.Base-import GHC.Ptr-import Foreign.ForeignPtr (withForeignPtr)-import Foreign.Storable (peek)--import qualified Data.ByteString as B---- | Binary: 1000000010000000100000001000000010000000100000001000000010000000--- Decimal: 9259542123273814144-m64 :: Word64-m64 = 0x8080808080808080---- | Binary: 10000000--- Decimal: 128-m8 :: Word8-m8 = 0x80--isAsciiPtrW64 :: Ptr Word64 -> Ptr Word64 -> IO Bool-isAsciiPtrW64 !p !q- | p == q = pure True- | otherwise = do- c <- peek p- if isAsciiW64 c- then isAsciiPtrW64 (p `plusPtr` 8) q- else pure False--isAsciiPtrW8 :: Ptr Word8 -> Ptr Word8 -> IO Bool-isAsciiPtrW8 !p !q- | p == q = pure True- | otherwise = do- c <- peek p- if isAsciiW8 c- then isAsciiPtrW8 (p `plusPtr` 1) q- else pure False--isAsciiW64 :: Word64 -> Bool-isAsciiW64 !w = w .&. m64 == 0--isAsciiW8 :: Word8 -> Bool-isAsciiW8 !w = w .&. m8 == 0--isAsciiSmall :: ByteString -> Bool-isAsciiSmall !b = B.all (< m8) b--alignPtrPos :: Ptr a -> Ptr a-alignPtrPos addr@(Ptr a) - = case remAddr# a 8# of- 0# -> addr- n -> Ptr (plusAddr# a (8# -# n))--alignPtrNeg :: Ptr a -> Ptr a-alignPtrNeg addr@(Ptr a)- = case remAddr# a 8# of- 0# -> addr- n -> Ptr (plusAddr# a (negateInt# n))---- | 'isAscii' can tell if a given 'ByteString' is ASCII-encoded.-isAscii :: ByteString -> Bool-isAscii (PS _ _ 0) = True-isAscii b@(PS fp (I# o#) len@(I# l#)) =- accursedUnutterablePerformIO- $ withForeignPtr fp- $ \(Ptr addr) ->- if len < 8- then pure (isAsciiSmall b)- else do- let- startPre, endPre, startPost, endPost :: Ptr Word8- startMid, endMid :: Ptr Word64- - startPre = Ptr (plusAddr# addr o#)- endPre = alignPtrPos startPre- startMid = castPtr endPre- endMid = castPtr startPost- startPost = alignPtrNeg endPost- endPost = Ptr (plusAddr# addr (o# +# l#))- - startIsAscii <- isAsciiPtrW8 startPre endPre - if startIsAscii- then do- endIsAscii <- isAsciiPtrW8 startPost endPost- if endIsAscii- then isAsciiPtrW64 startMid endMid- else pure False- else pure False
+ Data/ByteString/Encodings.hs view
@@ -0,0 +1,10 @@+module Data.ByteString.Encodings+ ( -- * ASCII + isAscii+ -- * UTF-8 + , isUtf8 + , isUtf8'+ ) where++import Data.ByteString.Internal.Ascii+import Data.ByteString.Internal.Utf8
+ Data/ByteString/Internal/Ascii.hs view
@@ -0,0 +1,106 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE MagicHash #-}+{-# LANGUAGE NoImplicitPrelude #-}++{-# OPTIONS_GHC -O2 #-}++module Data.ByteString.Internal.Ascii+ ( isAscii+ ) where++import Data.Bits ((.&.))+import Data.ByteString.Internal (ByteString(..), accursedUnutterablePerformIO)+import Data.Word (Word8, Word64)+import GHC.Base+import GHC.Ptr+import Foreign.ForeignPtr (withForeignPtr)+import Foreign.Storable (peek)++import qualified Data.ByteString as B++-- | Binary: 1000000010000000100000001000000010000000100000001000000010000000+-- Decimal: 9259542123273814144+m64 :: Word64+{-# inline m64 #-}+m64 = 0x8080808080808080++-- | Binary: 10000000+-- Decimal: 128+m8 :: Word8+{-# inline m8 #-}+m8 = 0x80++isAsciiPtrW64 :: Ptr Word64 -> Ptr Word64 -> IO Bool+isAsciiPtrW64 !p !q+ | p == q = pure True+ | otherwise = do+ c <- peek p+ if isAsciiW64 c+ then isAsciiPtrW64 (p `plusPtr` 8) q+ else pure False++isAsciiPtrW8 :: Ptr Word8 -> Ptr Word8 -> IO Bool+isAsciiPtrW8 !p !q+ | p == q = pure True+ | otherwise = do+ c <- peek p+ if isAsciiW8 c+ then isAsciiPtrW8 (p `plusPtr` 1) q+ else pure False++isAsciiW64 :: Word64 -> Bool+{-# inline isAsciiW64 #-}+isAsciiW64 !w = w .&. m64 == 0++isAsciiW8 :: Word8 -> Bool+{-# inline isAsciiW8 #-}+isAsciiW8 !w = w .&. m8 == 0++isAsciiSmall :: ByteString -> Bool+{-# inline isAsciiSmall #-}+isAsciiSmall !b = B.all (< m8) b++alignPtrPos :: Ptr a -> Ptr a+{-# inline alignPtrPos #-}+alignPtrPos addr@(Ptr a) + = case remAddr# a 8# of+ 0# -> addr+ n -> Ptr (plusAddr# a (8# -# n))++alignPtrNeg :: Ptr a -> Ptr a+{-# inline alignPtrNeg #-}+alignPtrNeg addr@(Ptr a)+ = case remAddr# a 8# of+ 0# -> addr+ n -> Ptr (plusAddr# a (negateInt# n))++-- | 'isAscii' can tell if a given 'ByteString' is ASCII-encoded.+isAscii :: ByteString -> Bool+{-# inline isAscii #-}+isAscii (PS _ _ 0) = True+isAscii b@(PS fp (I# o#) len@(I# l#)) =+ accursedUnutterablePerformIO+ $ withForeignPtr fp+ $ \(Ptr addr) ->+ if len < 8+ then pure (isAsciiSmall b)+ else do+ let+ startPre, endPre, startPost, endPost :: Ptr Word8+ startMid, endMid :: Ptr Word64+ + startPre = Ptr (plusAddr# addr o#)+ endPre = alignPtrPos startPre+ startMid = castPtr endPre+ endMid = castPtr startPost+ startPost = alignPtrNeg endPost+ endPost = Ptr (plusAddr# addr (o# +# l#))+ + startIsAscii <- isAsciiPtrW8 startPre endPre + if startIsAscii+ then do+ endIsAscii <- isAsciiPtrW8 startPost endPost+ if endIsAscii+ then isAsciiPtrW64 startMid endMid+ else pure False+ else pure False
+ Data/ByteString/Internal/Utf8.hs view
@@ -0,0 +1,131 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE MagicHash #-}+{-# LANGUAGE NoImplicitPrelude #-}++{-# OPTIONS_GHC -O2 #-}++module Data.ByteString.Internal.Utf8+ ( isUtf8 + , isUtf8' + ) where++import Data.Bits ((.&.))+import Data.ByteString.Internal (ByteString(..), accursedUnutterablePerformIO)+import Data.ByteString.Internal.Ascii (isAscii)+import Data.Word (Word8)+import GHC.Base+import GHC.Ptr+import Foreign.ForeignPtr (withForeignPtr)+import Foreign.Storable (peek)++data Utf8 = U8 | U16 | U24 | U32 | UNot+ deriving Eq++which :: Word8 -> Utf8+{-# inline which #-}+which !c+ | isUtf8b8 c = U8+ | isUtf8b16 c = U16+ | isUtf8b24 c = U24+ | isUtf8b32 c = U32+ | otherwise = UNot++isUtf8Ptr :: Ptr Word8 -> Ptr Word8 -> IO Bool+isUtf8Ptr !p !q+ | p == q = pure True+ | otherwise = do+ c <- peek p++ case which c of+ U8 -> isUtf8Ptr (p `plusPtr` 1) q + U16 -> if q `minusPtr` p >= 2+ then + do d <- peek (p `plusPtr` 1) + if isUtf8OtherBytes d then isUtf8Ptr (p `plusPtr` 2) q else pure False+ else pure False+ U24 -> if q `minusPtr` p >= 3+ then+ do d <- peek (p `plusPtr` 1); e <- peek (p `plusPtr` 2)+ if (isUtf8OtherBytes d && isUtf8OtherBytes e) then isUtf8Ptr (p `plusPtr` 3) q else pure False+ else pure False+ U32 -> if q `minusPtr` p >= 4+ then+ do d <- peek (p `plusPtr` 1); e <- peek (p `plusPtr` 2); f <- peek (p `plusPtr` 3);+ if (isUtf8OtherBytes d && isUtf8OtherBytes e && isUtf8OtherBytes f) then isUtf8Ptr (p `plusPtr` 4) q else pure False+ else pure False+ UNot -> pure False++-- Hex: 0x80+-- Binary: 10000000+-- Decimal: 128+isUtf8b8 :: Word8 -> Bool+{-# inline isUtf8b8 #-}+isUtf8b8 !w = w .&. 0x80 == 0++-- Hex: 0xE0 +-- Binary: 11100000+-- Decimal: 224+isUtf8b16 :: Word8 -> Bool+{-# inline isUtf8b16 #-}+isUtf8b16 !w = w .&. 0xE0 == 0xC0++-- Hex: 0xF0+-- Binary: 11110000+-- Decimal: 240 +isUtf8b24 :: Word8 -> Bool+{-# inline isUtf8b24 #-}+isUtf8b24 !w = w .&. 0xF0 == 0xE0+++-- Hex: 0xF8+-- Binary: 11111000+-- Decimal: 248+isUtf8b32 :: Word8 -> Bool+{-# inline isUtf8b32 #-}+isUtf8b32 !w = w .&. 0xF8 == 0xF0++-- Hex: 0xC0+-- Binary: 11000000+-- Decimal: 192+isUtf8OtherBytes :: Word8 -> Bool+{-# inline isUtf8OtherBytes #-}+isUtf8OtherBytes !w = w .&. 0xC0 == 0x80++-- | 'isUtf8' firsts calls the very fast 'isAscii' to see if+-- the data is ASCII (and thus UTF-8). Use this if you know most of your+-- data is ASCII-encoded.+-- If you know that most of your data is probably not UTF8-encoded, it is probably+-- best to use 'isUtf8'' to avoid this check.+isUtf8 :: ByteString -> Bool+isUtf8 (PS _ _ 0) = True+isUtf8 b@(PS fp (I# o#) (I# l#)) = if isAscii b then True else+ accursedUnutterablePerformIO+ $ withForeignPtr fp+ $ \(Ptr addr) ->+ do+ let+ start, end :: Ptr Word8+ start = Ptr (plusAddr# addr o#)+ end = Ptr (plusAddr# addr (o# +# l#))+ + isUtf8Ptr start end+{-# inline isUtf8 #-}++-- | 'isUtf8'' does not call 'isAscii'. Use this if+-- you know most of your data is not ASCII-encoded.+-- If you know that most of your data is probably ASCII-encoded, it is+-- probably best to use 'isUtf8'.+isUtf8' :: ByteString -> Bool+isUtf8' (PS _ _ 0) = True+isUtf8' (PS fp (I# o#) (I# l#)) =+ accursedUnutterablePerformIO+ $ withForeignPtr fp+ $ \(Ptr addr) ->+ do+ let+ start, end :: Ptr Word8+ start = Ptr (plusAddr# addr o#)+ end = Ptr (plusAddr# addr (o# +# l#))+ + isUtf8Ptr start end+{-# inline isUtf8' #-}
− Data/ByteString/Utf8.hs
@@ -1,125 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE EmptyCase #-}-{-# LANGUAGE MagicHash #-}-{-# LANGUAGE LambdaCase #-}-{-# LANGUAGE NoImplicitPrelude #-}--{-# OPTIONS_GHC -O2 #-}--module Data.ByteString.Utf8- ( isUtf8 - , isUtf8' - ) where--import Data.Bits ( (.&.) )-import Data.ByteString.Ascii (isAscii)-import Data.ByteString.Internal (ByteString(..), accursedUnutterablePerformIO)-import Data.Word (Word8)-import GHC.Base-import GHC.Ptr-import Foreign.ForeignPtr (withForeignPtr)-import Foreign.Storable (peek)--data Utf8 = U8 | U16 | U24 | U32 | UNot- deriving Eq--which :: Word8 -> Utf8-which c- | isUtf8b8 c = U8- | isUtf8b16 c = U16- | isUtf8b24 c = U24- | isUtf8b32 c = U32- | otherwise = UNot--isUtf8Ptr :: Ptr Word8 -> Ptr Word8 -> IO Bool-isUtf8Ptr !p !q- | p == q = pure True- | otherwise = do- c <- peek p-- case which c of- U8 -> isUtf8Ptr (p `plusPtr` 1) q - U16 -> if q `minusPtr` p >= 2- then - do d <- peek (p `plusPtr` 1) - if isUtf8OtherBytes d then isUtf8Ptr (p `plusPtr` 2) q else pure False- else pure False- U24 -> if q `minusPtr` p >= 3- then- do d <- peek (p `plusPtr` 1); e <- peek (p `plusPtr` 2)- if (isUtf8OtherBytes d && isUtf8OtherBytes e) then isUtf8Ptr (p `plusPtr` 3) q else pure False- else pure False- U32 -> if q `minusPtr` p >= 4- then- do d <- peek (p `plusPtr` 1); e <- peek (p `plusPtr` 2); f <- peek (p `plusPtr` 3);- if (isUtf8OtherBytes d && isUtf8OtherBytes e && isUtf8OtherBytes f) then isUtf8Ptr (p `plusPtr` 4) q else pure False- else pure False- UNot -> pure False---- Hex: 0x80--- Binary: 10000000--- Decimal: 128-isUtf8b8 :: Word8 -> Bool-isUtf8b8 !w = w .&. 0x80 == 0---- Hex: 0xE0 --- Binary: 11100000--- Decimal: 224-isUtf8b16 :: Word8 -> Bool-isUtf8b16 !w = w .&. 0xE0 == 0xC0---- Hex: 0xF0--- Binary: 11110000--- Decimal: 240 -isUtf8b24 :: Word8 -> Bool-isUtf8b24 !w = w .&. 0xF0 == 0xE0----- Hex: 0xF8--- Binary: 11111000--- Decimal: 248-isUtf8b32 :: Word8 -> Bool-isUtf8b32 !w = w .&. 0xF8 == 0xF0---- Hex: 0xC0--- Binary: 11000000--- Decimal: 192-isUtf8OtherBytes :: Word8 -> Bool-isUtf8OtherBytes !w = w .&. 0xC0 == 0x80---- | 'isUtf8' firsts calls the very fast 'Data.ByteString.Ascii.isAscii' to see if--- the data is ASCII (and thus UTF-8). Use this if you know most of your--- data is ASCII-encoded.--- If you know that most of your data is probably not UTF8-encoded, it is probably--- best to use 'isUtf8'' to avoid this check.-isUtf8 :: ByteString -> Bool-isUtf8 (PS _ _ 0) = True-isUtf8 b@(PS fp (I# o#) (I# l#)) = if isAscii b then True else- accursedUnutterablePerformIO- $ withForeignPtr fp- $ \(Ptr addr) ->- do- let- start, end :: Ptr Word8- start = Ptr (plusAddr# addr o#)- end = Ptr (plusAddr# addr (o# +# l#))- - isUtf8Ptr start end---- | 'isUtf8'' does not call 'Data.ByteString.Ascii.isAscii. Use this if--- you know most of your data is not ASCII-encoded.--- If you know that most of your data is probably ASCII-encoded, it is--- probably best to use 'isUtf8'.-isUtf8' :: ByteString -> Bool-isUtf8' (PS _ _ 0) = True-isUtf8' (PS fp (I# o#) (I# l#)) =- accursedUnutterablePerformIO- $ withForeignPtr fp- $ \(Ptr addr) ->- do- let- start, end :: Ptr Word8- start = Ptr (plusAddr# addr o#)- end = Ptr (plusAddr# addr (o# +# l#))- - isUtf8Ptr start end
bytestring-encodings.cabal view
@@ -1,13 +1,13 @@ name: bytestring-encodings version:- 0.1.0.1+ 0.2.0.1 synopsis: checks to see if a given bytestring adheres to a certain encoding description:- 'Data.ByteString.X' modules provide simple, efficient function 'isX :: ByteString -> Bool'- which returns 'True' if a given ByteString adheres to a certain encoding X,- and 'False' otherwise. + This module provides a family of functions, 'isX', of type @'ByteString' -> 'Bool'@,+ which return 'True' if the input ByteString adheres to a certain encoding X,+ and 'False' otherwise. license: MIT license-file:@@ -31,22 +31,16 @@ type: git location: https://github.com/chessai/bytestring-encodings -flag dev-wall-werror- description: turn on -Wall -Werror for development purposes- manual: True- default: False- library exposed-modules:- Data.ByteString.Ascii- , Data.ByteString.Utf8 + Data.ByteString.Encodings+ , Data.ByteString.Internal.Ascii+ , Data.ByteString.Internal.Utf8 build-depends:- base >=4.10 && <5.0+ base >= 4.7 && <5.0 , bytestring , ghc-prim default-language: Haskell2010- if flag(dev-wall-werror)- ghc-options: -Wall -Werror test-suite test type: exitcode-stdio-1.0