pcre-heavy 0.2.1 → 0.2.2
raw patch · 3 files changed
+60/−43 lines, 3 filesPVP ok
version bump matches the API change (PVP)
API changes (from Hackage documentation)
+ Text.Regex.PCRE.Heavy: (≈) :: Stringable a => a -> Regex -> Bool
Files
- README.md +7/−0
- library/Text/Regex/PCRE/Heavy.hs +51/−41
- pcre-heavy.cabal +2/−2
README.md view
@@ -25,6 +25,13 @@ True ``` +For `UnicodeSyntax` fans, it's also available as ≈ (U+2248 ALMOST EQUAL TO):++```haskell+>>> "https://unrelenting.technology" ≈ [re|^http.*|]+True+```+ ### Matching (Searching) (You can use any string type, not just String!)
library/Text/Regex/PCRE/Heavy.hs view
@@ -3,11 +3,13 @@ {-# LANGUAGE FlexibleInstances, BangPatterns #-} {-# LANGUAGE TemplateHaskell, QuasiQuotes #-} {-# LANGUAGE ForeignFunctionInterface #-}+{-# LANGUAGE UnicodeSyntax #-} -- | A usable regular expressions library on top of pcre-light. module Text.Regex.PCRE.Heavy ( -- * Matching (=~)+, (≈) , scan , scanO , scanRanges@@ -46,14 +48,14 @@ import System.IO.Unsafe (unsafePerformIO) import Foreign -substr :: BS.ByteString -> (Int, Int) -> BS.ByteString+substr ∷ BS.ByteString → (Int, Int) → BS.ByteString substr s (f, t) = BS.take (t - f) . BS.drop f $ s -behead :: [a] -> (a, [a])+behead ∷ [a] → (a, [a]) behead (h:t) = (h, t) behead [] = error "no head to behead" -reMatch :: Stringable a => Regex -> a -> Bool+reMatch ∷ Stringable a ⇒ Regex → a → Bool reMatch r s = isJust $ PCRE.match r (toByteString s) [] -- | Checks whether a string matches a regex.@@ -61,9 +63,13 @@ -- >>> :set -XQuasiQuotes -- >>> "https://unrelenting.technology" =~ [re|^http.*|] -- True-(=~) :: Stringable a => a -> Regex -> Bool+(=~) ∷ Stringable a ⇒ a → Regex → Bool (=~) = flip reMatch +-- | Same as =~.+(≈) ∷ Stringable a ⇒ a → Regex → Bool+(≈) = (=~)+ -- | Does raw PCRE matching (you probably shouldn't use this directly). -- -- >>> :set -XOverloadedStrings@@ -73,33 +79,33 @@ -- Just [(7,9)] -- >>> rawMatch [re|(\w)(\w)|] "a a ab abc ba" 0 [] -- Just [(4,6),(4,5),(5,6)]-rawMatch :: Regex -> BS.ByteString -> Int -> [PCREExecOption] -> Maybe [(Int, Int)]+rawMatch ∷ Regex → BS.ByteString → Int → [PCREExecOption] → Maybe [(Int, Int)] rawMatch r@(Regex pcreFp _) s offset opts = unsafePerformIO $ do- withForeignPtr pcreFp $ \pcrePtr -> do+ withForeignPtr pcreFp $ \pcrePtr → do let nCapt = PCRE.captureCount r ovecSize = (nCapt + 1) * 3 ovecBytes = ovecSize * size_of_cint- allocaBytes ovecBytes $ \ovec -> do+ allocaBytes ovecBytes $ \ovec → do let (strFp, off, len) = BS.toForeignPtr s- withForeignPtr strFp $ \strPtr -> do- results <- c_pcre_exec pcrePtr nullPtr (strPtr `plusPtr` off) (fromIntegral len) (fromIntegral offset)+ withForeignPtr strFp $ \strPtr → do+ results ← c_pcre_exec pcrePtr nullPtr (strPtr `plusPtr` off) (fromIntegral len) (fromIntegral offset) (combineExecOptions opts) ovec (fromIntegral ovecSize) if results < 0 then return Nothing else let loop n o acc = if n == results then return $ Just $ reverse acc else do- i <- peekElemOff ovec $! o- j <- peekElemOff ovec (o + 1)+ i ← peekElemOff ovec $! o+ j ← peekElemOff ovec (o + 1) loop (n + 1) (o + 2) ((fromIntegral i, fromIntegral j) : acc) in loop 0 0 [] -nextMatch :: Regex -> [PCREExecOption] -> BS.ByteString -> Int -> Maybe ([(Int, Int)], Int)+nextMatch ∷ Regex → [PCREExecOption] → BS.ByteString → Int → Maybe ([(Int, Int)], Int) nextMatch r opts str offset = case rawMatch r str offset opts of- Nothing -> Nothing- Just [] -> Nothing- Just ms -> Just (ms, maximum $ map snd ms)+ Nothing → Nothing+ Just [] → Nothing+ Just ms → Just (ms, maximum $ map snd ms) -- | Searches the string for all matches of a given regex. --@@ -111,11 +117,11 @@ -- -- >>> head $ scan [re|\s*entry (\d+) (\w+)\s*&?|] " entry 1 hello &entry 2 hi" -- (" entry 1 hello &",["1","hello"])-scan :: (Stringable a) => Regex -> a -> [(a, [a])]+scan ∷ (Stringable a) ⇒ Regex → a → [(a, [a])] scan r s = scanO r [] s -- | Exactly like 'scan', but passes runtime options to PCRE.-scanO :: (Stringable a) => Regex -> [PCREExecOption] -> a -> [(a, [a])]+scanO ∷ (Stringable a) ⇒ Regex → [PCREExecOption] → a → [(a, [a])] scanO r opts s = map behead $ map (fromByteString . substr str) <$> unfoldr (nextMatch r opts str) 0 where str = toByteString s @@ -126,37 +132,37 @@ -- [((0,17),[(7,8),(9,14)]),((17,27),[(23,24),(25,27)])] -- -- And just like 'scan', it's lazy.-scanRanges :: (Stringable a) => Regex -> a -> [((Int, Int), [(Int, Int)])]+scanRanges ∷ (Stringable a) ⇒ Regex → a → [((Int, Int), [(Int, Int)])] scanRanges r s = scanRangesO r [] s -- | Exactly like 'scanRanges', but passes runtime options to PCRE.-scanRangesO :: Stringable a => Regex -> [PCREExecOption] -> a -> [((Int, Int), [(Int, Int)])]+scanRangesO ∷ Stringable a ⇒ Regex → [PCREExecOption] → a → [((Int, Int), [(Int, Int)])] scanRangesO r opts s = map behead $ unfoldr (nextMatch r opts str) 0 where str = toByteString s class RegexReplacement a where- performReplacement :: BS.ByteString -> [BS.ByteString] -> a -> BS.ByteString+ performReplacement ∷ BS.ByteString → [BS.ByteString] → a → BS.ByteString -instance Stringable a => RegexReplacement a where+instance Stringable a ⇒ RegexReplacement a where performReplacement _ _ to = toByteString to -instance Stringable a => RegexReplacement (a -> [a] -> a) where+instance Stringable a ⇒ RegexReplacement (a → [a] → a) where performReplacement from groups replacer = toByteString $ replacer (fromByteString from) (map fromByteString groups) -instance Stringable a => RegexReplacement (a -> a) where+instance Stringable a ⇒ RegexReplacement (a → a) where performReplacement from _ replacer = toByteString $ replacer (fromByteString from) -instance Stringable a => RegexReplacement ([a] -> a) where+instance Stringable a ⇒ RegexReplacement ([a] → a) where performReplacement _ groups replacer = toByteString $ replacer (map fromByteString groups) -rawSub :: RegexReplacement r => Regex -> r -> BS.ByteString -> Int -> [PCREExecOption] -> Maybe (BS.ByteString, Int)+rawSub ∷ RegexReplacement r ⇒ Regex → r → BS.ByteString → Int → [PCREExecOption] → Maybe (BS.ByteString, Int) rawSub r t s offset opts = case rawMatch r s offset opts of- Just ((begin, end):groups) ->+ Just ((begin, end):groups) → Just (BS.concat [ substr s (0, begin) , performReplacement (substr s (begin, end)) (map (substr s) groups) t , substr s (end, BS.length s)], end)- _ -> Nothing+ _ → Nothing -- | Replaces the first occurence of a given regex. --@@ -169,15 +175,15 @@ -- You can use functions! -- A function of Stringable gets the full match. -- A function of [Stringable] gets the groups.--- A function of Stringable -> [Stringable] gets both.+-- A function of Stringable → [Stringable] gets both. -- -- >>> sub [re|%(\d+)(\w+)|] (\(d:w:_) -> "{" ++ d ++ " of " ++ w ++ "}" :: String) "Hello, %20thing" :: String -- "Hello, {20 of thing}"-sub :: (Stringable a, RegexReplacement r) => Regex -> r -> a -> a+sub ∷ (Stringable a, RegexReplacement r) ⇒ Regex → r → a → a sub r t s = subO r [] t s -- | Exactly like 'sub', but passes runtime options to PCRE.-subO :: (Stringable a, RegexReplacement r) => Regex -> [PCREExecOption] -> r -> a -> a+subO ∷ (Stringable a, RegexReplacement r) ⇒ Regex → [PCREExecOption] → r → a → a subO r opts t s = fromMaybe s $ fromByteString <$> fst <$> rawSub r t (toByteString s) 0 opts -- | Replaces all occurences of a given regex.@@ -186,17 +192,21 @@ -- -- >>> gsub [re|thing|] "world" "Hello, thing thing" :: String -- "Hello, world world"-gsub :: (Stringable a, RegexReplacement r) => Regex -> r -> a -> a+--+-- >>> gsub [re||] "" "Hello, world" :: String+-- "Hello, world"+gsub ∷ (Stringable a, RegexReplacement r) ⇒ Regex → r → a → a gsub r t s = gsubO r [] t s -- | Exactly like 'gsub', but passes runtime options to PCRE.-gsubO :: (Stringable a, RegexReplacement r) => Regex -> [PCREExecOption] -> r -> a -> a+gsubO ∷ (Stringable a, RegexReplacement r) ⇒ Regex → [PCREExecOption] → r → a → a gsubO r opts t s = fromByteString $ loop 0 str where str = toByteString s loop offset acc = case rawSub r t acc offset opts of- Just (result, newOffset) -> loop newOffset result- _ -> acc+ Just (result, newOffset) →+ if newOffset == offset then acc else loop newOffset result+ _ → acc -- | Splits the string using the given regex. --@@ -207,11 +217,11 @@ -- -- >>> split [re|%(begin|next|end)%|] "" -- [""]-split :: Stringable a => Regex -> a -> [a]+split ∷ Stringable a ⇒ Regex → a → [a] split r s = splitO r [] s -- | Exactly like 'split', but passes runtime options to PCRE.-splitO :: Stringable a => Regex -> [PCREExecOption] -> a -> [a]+splitO ∷ Stringable a ⇒ Regex → [PCREExecOption] → a → [a] splitO r opts s = map fromByteString $ map' (substr str) partRanges where map' f = foldr ((:) . f) [f (lastL, BS.length str)] -- avoiding the snoc operation (lastL, partRanges) = mapAccumL invRange 0 ranges@@ -221,14 +231,14 @@ instance Lift PCREOption where -- well, the constructor isn't exported, but at least it implements Read/Show :D- lift o = let o' = show o in [| read o' :: PCREOption |]+ lift o = let o' = show o in [| read o' ∷ PCREOption |] -quoteExpRegex :: [PCREOption] -> String -> ExpQ-quoteExpRegex opts txt = [| PCRE.compile (toByteString (txt :: String)) opts |]+quoteExpRegex ∷ [PCREOption] → String → ExpQ+quoteExpRegex opts txt = [| PCRE.compile (toByteString (txt ∷ String)) opts |] where !_ = PCRE.compile (toByteString txt) opts -- check at compile time -- | Returns a QuasiQuoter like 're', but with given PCRE options.-mkRegexQQ :: [PCREOption] -> QuasiQuoter+mkRegexQQ ∷ [PCREOption] → QuasiQuoter mkRegexQQ opts = QuasiQuoter { quoteExp = quoteExpRegex opts , quotePat = undefined@@ -236,5 +246,5 @@ , quoteDec = undefined } -- | A QuasiQuoter for regular expressions that does a compile time check.-re :: QuasiQuoter+re ∷ QuasiQuoter re = mkRegexQQ [utf8]
pcre-heavy.cabal view
@@ -1,5 +1,5 @@ name: pcre-heavy-version: 0.2.1+version: 0.2.2 synopsis: A regexp library on top of pcre-light you can actually use. description: A regular expressions library that does not suck.@@ -23,7 +23,7 @@ extra-source-files: README.md tested-with:- GHC == 7.8.2+ GHC == 7.8.3 source-repository head type: git