tokenizer-monad 0.1.0.0 → 0.2.0.0
raw patch · 8 files changed
+679/−32 lines, 8 filesdep +bytestringdep ~base
Dependencies added: bytestring
Dependency ranges changed: base
Files
- README.md +63/−0
- src/Control/Monad/Tokenizer.hs +127/−26
- src/Control/Monad/Tokenizer/Char8/Lazy.hs +103/−0
- src/Control/Monad/Tokenizer/Char8/Strict.hs +103/−0
- src/Control/Monad/Tokenizer/String.hs +89/−0
- src/Control/Monad/Tokenizer/Text/Lazy.hs +90/−0
- src/Control/Monad/Tokenizer/Text/Strict.hs +90/−0
- tokenizer-monad.cabal +14/−6
+ README.md view
@@ -0,0 +1,63 @@+# tokenizer-monad++**Motivation**: Before working with tokenizer-monad, I often implemented+tokenizers by recursively destroying Char lists. The resulting code was+purely functional, but hardly readable - even more so, if one destroys+Text instead of Char lists. In my mind, I usually imagine tokenization+algorithms like flow charts, hence I wanted to code them in a similar manner.++**Main idea**: You `walk` through the input string like a turtle, and everytime+you find a token boundary, you call `emit`. If some specific kinds of tokens+should be suppressed, you can 'discard' them instead (or filter afterwards).++This package supports Strings, strict and lazy Text, as well as strict and+lazy ASCII ByteStrings.++**Examples**:++This tokenizer is equivalent to `words` from Prelude:++ words' :: String -> [String]+ words' = runTokenizerCS $ untilEOT $ do+ c <- pop+ if c `elem` " \t\n\r"+ then discard+ else do+ walkWhile (not . isSpace)+ emit++ ...> words' "Dieses Haus ist blau."+ ["Dieses","Haus","ist","blau."]++This tokenizer is similar to `lines` from Prelude, but discards empty lines:++ lines' :: String -> [String]+ lines' = runTokenizerCS $ untilEOT $ do+ c <- pop+ if c `elem` "\n\r"+ then discard+ else do+ walkWhile (\c -> not (c `elem` "\r\n"))+ emit++ ...> lines' "Dieses Haus ist\n\nblau.\n"+ ["Dieses Haus ist","blau."]++A more advanced tokenizer, that can handle punctuation and HTTP URIs in text:++ t1Tokenize' :: Tokenizer Text ()+ t1Tokenize' = do+ http <- lookAhead "http://"+ https <- lookAhead "https://"+ if (http || https)+ then (walkWhile (not . isSpace) >> discard)+ else do+ c <- peek+ walk+ if isStopSym c+ then emit+ else if c `elem` (" \t\r\n" :: [Char])+ then discard+ else do+ walkWhile (\c -> (c=='_') || not (isSpace c || isPunctuation c))+ emit
src/Control/Monad/Tokenizer.hs view
@@ -1,23 +1,107 @@-{-# LANGUAGE OverloadedStrings, BangPatterns #-}+{-# LANGUAGE OverloadedStrings, BangPatterns, FlexibleInstances #-} -module Control.Monad.Tokenizer where+-- | A monad for writing pure tokenizers in an imperative-looking way.+--+-- Main idea: You 'walk' through the input string like a turtle, and everytime+-- you find a token boundary, you call 'emit'. If some specific kinds of tokens+-- should be suppressed, you can 'discard' them instead (or filter afterwards).+--+-- This module supports strict text, lazy text, and strings, though the package+-- also provides support for ASCII bytestrings in separate modules.+--+-- Example for a simple tokenizer, that splits words by whitespace and discards stop symbols:+--+-- > tokenizeWords :: T.Text -> [T.Text]+-- > tokenizeWords = runTokenizer $ untilEOT $ do+-- > c <- pop+-- > if isStopSym c+-- > then discard+-- > else if c `elem` (" \t\r\n" :: [Char])+-- > then discard+-- > else do+-- > walkWhile (\c -> (c=='_') || not (isSpace c || isPunctuation' c))+-- > emit +module Control.Monad.Tokenizer (+ -- * Monad+ Tokenizer,+ runTokenizer,+ runTokenizerCS,+ untilEOT,+ -- * Tests+ peek,+ isEOT,+ lookAhead,+ -- * Movement+ walk,+ walkBack,+ pop,+ walkWhile,+ walkFold,+ -- * Transactions+ emit,+ discard,+ restore,+ -- * Text types+ Tokenizable(..)+ ) where+ import Control.Monad import Data.Char import Data.Monoid import qualified Data.Text as T+import qualified Data.Text.Lazy as LT++-- | Text types that can be split by the Tokenizer monad. In this module,+-- instances are provided for String, strict Text, and lazy Text.+-- If you are dealing with ASCII ByteStrings, you can find instances in+-- the modules "Control.Monad.Tokenizer.Char8.Strict" and+-- "Control.Monad.Tokenizer.Char8.Lazy"+class Tokenizable t where+ tnull :: t -> Bool+ thead :: t -> Char+ ttail :: t -> t+ ttake :: Int -> t -> t+ tdrop :: Int -> t -> t+ tlower :: t -> t++instance Tokenizable T.Text where+ tnull = T.null+ thead = T.head+ ttail = T.tail+ ttake = T.take+ tdrop = T.drop+ tlower = T.toLower++instance Tokenizable LT.Text where+ tnull = LT.null+ thead = LT.head+ ttail = LT.tail+ ttake = LT.take . fromIntegral+ tdrop = LT.drop . fromIntegral+ tlower = LT.toLower++instance Tokenizable [Char] where+ tnull = null+ thead = head+ ttail = tail+ ttake = take+ tdrop = drop+ tlower = map toLower --- | Tokenizer monad. Use runTokenizer to run it-newtype Tokenizer a = Tokenizer { runTokenizer' :: (T.Text, Int, T.Text) -> (a,[T.Text] -> [T.Text],T.Text,Int,T.Text) }+-- | Tokenizer monad. Use runTokenizer or runTokenizerCS to run it+newtype Tokenizer t a = Tokenizer { runTokenizer' :: (t, Int, t) -> (a,[t] -> [t],t,Int,t) }+--- type explanation: (whole text since last emission, chars passed, remaining text)+--- -> (result, difference list of tokens, whole text, chars passed, remaining text) -instance Functor Tokenizer where+instance Functor (Tokenizer t) where fmap = liftM -instance Applicative Tokenizer where+instance Applicative (Tokenizer t) where pure = return (<*>) = ap -instance Monad Tokenizer where+instance Monad (Tokenizer t) where return a = Tokenizer $ \(whole,count,tail) -> (a,id,whole,count,tail) m >>= f = Tokenizer $ \(whole,count,tail) -> let (a1,o1,w1,!c1,t1) = runTokenizer' m (whole,count,tail)@@ -25,46 +109,63 @@ in (a2,o1.o2,w2,c2,t2) -- | Check if the next input chars agree with the given string-lookAhead :: [Char] -> Tokenizer Bool+lookAhead :: Tokenizable t => [Char] -> Tokenizer t Bool lookAhead chars = Tokenizer $ \(whole,count,tail) ->- let h = T.unpack $ T.take (length chars) tail+ let h = unpack $ ttake (length chars) tail in (h == chars, id, whole, count, tail)+ where unpack t | tnull t = []+ | otherwise = thead t : unpack (ttail t) -- | Proceed to the next character-walk :: Tokenizer ()+walk :: Tokenizable t => Tokenizer t () walk = Tokenizer $ \(whole,count,tail) ->- if T.null tail+ if tnull tail then ((),id,whole,count,tail)- else ((),id,whole,count+1,T.tail tail)+ else ((),id,whole,count+1,ttail tail) +-- | Walk back to the previous character, unless it was discarded/emitted.+walkBack :: Tokenizable t => Tokenizer t ()+walkBack = Tokenizer $ \(whole,count,_) ->+ if count > 0+ then ((),id,whole,count-1,tdrop (count-1) whole)+ else ((),id,whole,0,whole)++-- | Restore the state after the last emit/discard.+restore :: Tokenizer t ()+restore = Tokenizer $ \(whole,_,_) -> ((),id,whole,0,whole)+ -- | Peek the current character-peek :: Tokenizer Char+peek :: Tokenizable t => Tokenizer t Char peek = Tokenizer $ \(whole,count,tail) -> (th tail,id,whole,count,tail)- where th t | T.null t = '\0'- | otherwise = T.head t+ where th t | tnull t = '\0'+ | otherwise = thead t -- | Peek the current character and proceed-pop :: Tokenizer Char+pop :: Tokenizable t => Tokenizer t Char pop = peek <* walk -- | Break at the current position and emit the scanned token-emit :: Tokenizer ()-emit = Tokenizer $ \(whole,count,tail) -> ((),(T.take count whole:),tail,0,tail)+emit :: Tokenizable t => Tokenizer t ()+emit = Tokenizer $ \(whole,count,tail) -> ((),(ttake count whole:),tail,0,tail) -- | Break at the current position and discard the scanned token-discard :: Tokenizer ()+discard :: Tokenizer t () discard = Tokenizer $ \(whole,count,tail) -> ((),id,tail,0,tail) +-- | Have I reached the end of the input text?+isEOT :: Tokenizable t => Tokenizer t Bool+isEOT = Tokenizer $ \(whole, count, tail) -> (tnull tail, id, whole, count, tail)+ -- | Proceed as long as a given function succeeds-walkWhile :: (Char -> Bool) -> Tokenizer ()+walkWhile :: Tokenizable t => (Char -> Bool) -> Tokenizer t () walkWhile f = do c <- peek when (c /= '\0' && f c) $ walk >> walkWhile f -- | Proceed as long as a given fold returns Just (generalization of walkWhile)-walkFold :: a -> (Char -> a -> Maybe a) -> Tokenizer ()+walkFold :: Tokenizable t => a -> (Char -> a -> Maybe a) -> Tokenizer t () walkFold s0 f = do c <- peek unless (c == '\0') $ case f c s0 of@@ -72,20 +173,20 @@ Just s -> walk >> walkFold s f -- | Repeat a given tokenizer as long as the end of text is not reached-untilEOT :: Tokenizer () -> Tokenizer ()+untilEOT :: Tokenizable t => Tokenizer t () -> Tokenizer t () untilEOT f = do- eot <- Tokenizer $ \(whole,count,tail) -> (T.null tail,id, whole, count, tail)+ eot <- isEOT unless eot $ f >> untilEOT f -- | Split a string into tokens using the given tokenizer-runTokenizer :: Tokenizer () -> T.Text -> [T.Text]+runTokenizer :: Tokenizable t => Tokenizer t () -> t -> [t] runTokenizer m input =- let input' = T.toLower input+ let input' = tlower input in case runTokenizer' m (input',0,input') of (_, tokens, _, _, _) -> tokens [] -- | Split a string into tokens using the given tokenizer, case sensitive version-runTokenizerCS :: Tokenizer () -> T.Text -> [T.Text]+runTokenizerCS :: Tokenizable t => Tokenizer t () -> t -> [t] runTokenizerCS m input = case runTokenizer' m (input,0,input) of (_, tokens, _, _, _) -> tokens []
+ src/Control/Monad/Tokenizer/Char8/Lazy.hs view
@@ -0,0 +1,103 @@+-- | A monad for writing pure tokenizers in an imperative-looking way.+--+-- Main idea: You 'walk' through the input string like a turtle, and everytime+-- you find a token boundary, you call 'emit'. If some specific kinds of tokens+-- should be suppressed, you can 'discard' them instead (or filter afterwards).+--+-- This module supports is specialized for lazy 'ByteString's. The module+-- "Control.Monad.Tokenizer" provides more general types, but does not export a+-- 'Tokenizable' instance for 'ByteString's, as its implementation depends on the+-- encoding. This module assumes ASCII encoding (you have been warned!). +--+-- Example for a simple tokenizer, that splits words by whitespace and discards stop symbols:+--+-- > tokenizeWords :: ByteString -> [ByteString]+-- > tokenizeWords = runTokenizer $ untilEOT $ do+-- > c <- pop+-- > if isStopSym c+-- > then discard+-- > else if c `elem` (" \t\r\n" :: [Char])+-- > then discard+-- > else do+-- > walkWhile (\c -> (c=='_') || not (isSpace c || isPunctuation' c))+-- > emit++module Control.Monad.Tokenizer.Char8.Lazy (+ -- * Monad+ Tokenizer,+ runTokenizer,+ runTokenizerCS,+ untilEOT,+ -- * Tests+ peek,+ isEOT,+ lookAhead,+ -- * Movement+ walk,+ walkBack,+ pop,+ walkWhile,+ walkFold,+ -- * Transactions+ emit,+ discard,+ restore+ ) where++import qualified Control.Monad.Tokenizer as G+import Data.ByteString.Lazy.Char8 as C8+import Data.Char++-- | Tokenizer monad. Use runTokenizer or runTokenizerCS to run it+type Tokenizer = G.Tokenizer ByteString++-- | Check if the next input chars agree with the given string+lookAhead = G.lookAhead :: [Char] -> Tokenizer Bool++-- | Proceed to the next character+walk = G.walk :: Tokenizer ()++-- | Walk back to the previous character, unless it was discarded/emitted.+walkBack = G.walkBack :: Tokenizer ()++-- | Restore the state after the last emit/discard.+restore = G.restore :: Tokenizer ()++-- | Peek the current character+peek = G.peek :: Tokenizer Char++-- | Peek the current character and proceed+pop = G.pop :: Tokenizer Char++-- | Break at the current position and emit the scanned token+emit = G.emit :: Tokenizer ()++-- | Break at the current position and discard the scanned token+discard = G.discard :: Tokenizer ()++-- | Have I reached the end of the input text?+isEOT = G.isEOT :: Tokenizer Bool++-- | Proceed as long as a given function succeeds+walkWhile = G.walkWhile :: (Char -> Bool) -> Tokenizer ()++-- | Proceed as long as a given fold returns Just (generalization of walkWhile)+walkFold = G.walkFold :: a -> (Char -> a -> Maybe a) -> Tokenizer ()++-- | Repeat a given tokenizer as long as the end of text is not reached+untilEOT = G.untilEOT :: Tokenizer () -> Tokenizer ()++-- | Split a string into tokens using the given tokenizer+runTokenizer = G.runTokenizer :: Tokenizer () -> ByteString -> [ByteString]++-- | Split a string into tokens using the given tokenizer, case sensitive version+runTokenizerCS = G.runTokenizerCS :: Tokenizer () -> ByteString -> [ByteString]++-- | Assuming ASCII encoding+instance G.Tokenizable ByteString where+ tnull = C8.null+ thead = C8.head+ ttail = C8.tail+ ttake = C8.take . fromIntegral+ tdrop = C8.drop . fromIntegral+ tlower = C8.map toLower
+ src/Control/Monad/Tokenizer/Char8/Strict.hs view
@@ -0,0 +1,103 @@+-- | A monad for writing pure tokenizers in an imperative-looking way.+--+-- Main idea: You 'walk' through the input string like a turtle, and everytime+-- you find a token boundary, you call 'emit'. If some specific kinds of tokens+-- should be suppressed, you can 'discard' them instead (or filter afterwards).+--+-- This module supports is specialized for strict 'ByteString's. The module+-- "Control.Monad.Tokenizer" provides more general types, but does not export a+-- 'Tokenizable' instance for 'ByteString's, as its implementation depends on the+-- encoding. This module assumes ASCII encoding (you have been warned!). +--+-- Example for a simple tokenizer, that splits words by whitespace and discards stop symbols:+--+-- > tokenizeWords :: ByteString -> [ByteString]+-- > tokenizeWords = runTokenizer $ untilEOT $ do+-- > c <- pop+-- > if isStopSym c+-- > then discard+-- > else if c `elem` (" \t\r\n" :: [Char])+-- > then discard+-- > else do+-- > walkWhile (\c -> (c=='_') || not (isSpace c || isPunctuation' c))+-- > emit++module Control.Monad.Tokenizer.Char8.Strict (+ -- * Monad+ Tokenizer,+ runTokenizer,+ runTokenizerCS,+ untilEOT,+ -- * Tests+ peek,+ isEOT,+ lookAhead,+ -- * Movement+ walk,+ walkBack,+ pop,+ walkWhile,+ walkFold,+ -- * Transactions+ emit,+ discard,+ restore+ ) where++import qualified Control.Monad.Tokenizer as G+import Data.ByteString.Char8 as C8+import Data.Char++-- | Tokenizer monad. Use runTokenizer or runTokenizerCS to run it+type Tokenizer = G.Tokenizer ByteString++-- | Check if the next input chars agree with the given string+lookAhead = G.lookAhead :: [Char] -> Tokenizer Bool++-- | Proceed to the next character+walk = G.walk :: Tokenizer ()++-- | Walk back to the previous character, unless it was discarded/emitted.+walkBack = G.walkBack :: Tokenizer ()++-- | Restore the state after the last emit/discard.+restore = G.restore :: Tokenizer ()++-- | Peek the current character+peek = G.peek :: Tokenizer Char++-- | Peek the current character and proceed+pop = G.pop :: Tokenizer Char++-- | Break at the current position and emit the scanned token+emit = G.emit :: Tokenizer ()++-- | Break at the current position and discard the scanned token+discard = G.discard :: Tokenizer ()++-- | Have I reached the end of the input text?+isEOT = G.isEOT :: Tokenizer Bool++-- | Proceed as long as a given function succeeds+walkWhile = G.walkWhile :: (Char -> Bool) -> Tokenizer ()++-- | Proceed as long as a given fold returns Just (generalization of walkWhile)+walkFold = G.walkFold :: a -> (Char -> a -> Maybe a) -> Tokenizer ()++-- | Repeat a given tokenizer as long as the end of text is not reached+untilEOT = G.untilEOT :: Tokenizer () -> Tokenizer ()++-- | Split a string into tokens using the given tokenizer+runTokenizer = G.runTokenizer :: Tokenizer () -> ByteString -> [ByteString]++-- | Split a string into tokens using the given tokenizer, case sensitive version+runTokenizerCS = G.runTokenizerCS :: Tokenizer () -> ByteString -> [ByteString]++-- | Assuming ASCII encoding+instance G.Tokenizable ByteString where+ tnull = C8.null+ thead = C8.head+ ttail = C8.tail+ ttake = C8.take+ tdrop = C8.drop+ tlower = C8.map toLower
+ src/Control/Monad/Tokenizer/String.hs view
@@ -0,0 +1,89 @@+-- | A monad for writing pure tokenizers in an imperative-looking way.+--+-- Main idea: You 'walk' through the input string like a turtle, and everytime+-- you find a token boundary, you call 'emit'. If some specific kinds of tokens+-- should be suppressed, you can 'discard' them instead (or filter afterwards).+--+-- This module supports is specialized for 'String's. The module "Control.Monad.Tokenizer" provides more general types.+--+-- Example for a simple tokenizer, that splits words by whitespace and discards stop symbols:+--+-- > tokenizeWords :: String -> [String]+-- > tokenizeWords = runTokenizer $ untilEOT $ do+-- > c <- pop+-- > if isStopSym c+-- > then discard+-- > else if c `elem` (" \t\r\n" :: [Char])+-- > then discard+-- > else do+-- > walkWhile (\c -> (c=='_') || not (isSpace c || isPunctuation' c))+-- > emit++module Control.Monad.Tokenizer.String (+ -- * Monad+ Tokenizer,+ runTokenizer,+ runTokenizerCS,+ untilEOT,+ -- * Tests+ peek,+ isEOT,+ lookAhead,+ -- * Movement+ walk,+ walkBack,+ pop,+ walkWhile,+ walkFold,+ -- * Transactions+ emit,+ discard,+ restore+ ) where++import qualified Control.Monad.Tokenizer as G++-- | Tokenizer monad. Use runTokenizer or runTokenizerCS to run it+type Tokenizer = G.Tokenizer String++-- | Check if the next input chars agree with the given string+lookAhead = G.lookAhead :: [Char] -> Tokenizer Bool++-- | Proceed to the next character+walk = G.walk :: Tokenizer ()++-- | Walk back to the previous character, unless it was discarded/emitted.+walkBack = G.walkBack :: Tokenizer ()++-- | Restore the state after the last emit/discard.+restore = G.restore :: Tokenizer ()++-- | Peek the current character+peek = G.peek :: Tokenizer Char++-- | Peek the current character and proceed+pop = G.pop :: Tokenizer Char++-- | Break at the current position and emit the scanned token+emit = G.emit :: Tokenizer ()++-- | Break at the current position and discard the scanned token+discard = G.discard :: Tokenizer ()++-- | Have I reached the end of the input text?+isEOT = G.isEOT :: Tokenizer Bool++-- | Proceed as long as a given function succeeds+walkWhile = G.walkWhile :: (Char -> Bool) -> Tokenizer ()++-- | Proceed as long as a given fold returns Just (generalization of walkWhile)+walkFold = G.walkFold :: a -> (Char -> a -> Maybe a) -> Tokenizer ()++-- | Repeat a given tokenizer as long as the end of text is not reached+untilEOT = G.untilEOT :: Tokenizer () -> Tokenizer ()++-- | Split a string into tokens using the given tokenizer+runTokenizer = G.runTokenizer :: Tokenizer () -> String -> [String]++-- | Split a string into tokens using the given tokenizer, case sensitive version+runTokenizerCS = G.runTokenizerCS :: Tokenizer () -> String -> [String]
+ src/Control/Monad/Tokenizer/Text/Lazy.hs view
@@ -0,0 +1,90 @@+-- | A monad for writing pure tokenizers in an imperative-looking way.+--+-- Main idea: You 'walk' through the input string like a turtle, and everytime+-- you find a token boundary, you call 'emit'. If some specific kinds of tokens+-- should be suppressed, you can 'discard' them instead (or filter afterwards).+--+-- This module supports is specialized for lazy text. The module "Control.Monad.Tokenizer" provides more general types.+--+-- Example for a simple tokenizer, that splits words by whitespace and discards stop symbols:+--+-- > tokenizeWords :: LT.Text -> [LT.Text]+-- > tokenizeWords = runTokenizer $ untilEOT $ do+-- > c <- pop+-- > if isStopSym c+-- > then discard+-- > else if c `elem` (" \t\r\n" :: [Char])+-- > then discard+-- > else do+-- > walkWhile (\c -> (c=='_') || not (isSpace c || isPunctuation' c))+-- > emit++module Control.Monad.Tokenizer.Text.Lazy (+ -- * Monad+ Tokenizer,+ runTokenizer,+ runTokenizerCS,+ untilEOT,+ -- * Tests+ peek,+ isEOT,+ lookAhead,+ -- * Movement+ walk,+ walkBack,+ pop,+ walkWhile,+ walkFold,+ -- * Transactions+ emit,+ discard,+ restore+ ) where++import qualified Control.Monad.Tokenizer as G+import Data.Text.Lazy++-- | Tokenizer monad. Use runTokenizer or runTokenizerCS to run it+type Tokenizer = G.Tokenizer Text++-- | Check if the next input chars agree with the given string+lookAhead = G.lookAhead :: [Char] -> Tokenizer Bool++-- | Proceed to the next character+walk = G.walk :: Tokenizer ()++-- | Walk back to the previous character, unless it was discarded/emitted.+walkBack = G.walkBack :: Tokenizer ()++-- | Restore the state after the last emit/discard.+restore = G.restore :: Tokenizer ()++-- | Peek the current character+peek = G.peek :: Tokenizer Char++-- | Peek the current character and proceed+pop = G.pop :: Tokenizer Char++-- | Break at the current position and emit the scanned token+emit = G.emit :: Tokenizer ()++-- | Break at the current position and discard the scanned token+discard = G.discard :: Tokenizer ()++-- | Have I reached the end of the input text?+isEOT = G.isEOT :: Tokenizer Bool++-- | Proceed as long as a given function succeeds+walkWhile = G.walkWhile :: (Char -> Bool) -> Tokenizer ()++-- | Proceed as long as a given fold returns Just (generalization of walkWhile)+walkFold = G.walkFold :: a -> (Char -> a -> Maybe a) -> Tokenizer ()++-- | Repeat a given tokenizer as long as the end of text is not reached+untilEOT = G.untilEOT :: Tokenizer () -> Tokenizer ()++-- | Split a string into tokens using the given tokenizer+runTokenizer = G.runTokenizer :: Tokenizer () -> Text -> [Text]++-- | Split a string into tokens using the given tokenizer, case sensitive version+runTokenizerCS = G.runTokenizerCS :: Tokenizer () -> Text -> [Text]
+ src/Control/Monad/Tokenizer/Text/Strict.hs view
@@ -0,0 +1,90 @@+-- | A monad for writing pure tokenizers in an imperative-looking way.+--+-- Main idea: You 'walk' through the input string like a turtle, and everytime+-- you find a token boundary, you call 'emit'. If some specific kinds of tokens+-- should be suppressed, you can 'discard' them instead (or filter afterwards).+--+-- This module supports is specialized for strict text. The module "Control.Monad.Tokenizer" provides more general types.+--+-- Example for a simple tokenizer, that splits words by whitespace and discards stop symbols:+--+-- > tokenizeWords :: T.Text -> [T.Text]+-- > tokenizeWords = runTokenizer $ untilEOT $ do+-- > c <- pop+-- > if isStopSym c+-- > then discard+-- > else if c `elem` (" \t\r\n" :: [Char])+-- > then discard+-- > else do+-- > walkWhile (\c -> (c=='_') || not (isSpace c || isPunctuation' c))+-- > emit++module Control.Monad.Tokenizer.Text.Strict (+ -- * Monad+ Tokenizer,+ runTokenizer,+ runTokenizerCS,+ untilEOT,+ -- * Tests+ peek,+ isEOT,+ lookAhead,+ -- * Movement+ walk,+ walkBack,+ pop,+ walkWhile,+ walkFold,+ -- * Transactions+ emit,+ discard,+ restore+ ) where++import qualified Control.Monad.Tokenizer as G+import Data.Text++-- | Tokenizer monad. Use runTokenizer or runTokenizerCS to run it+type Tokenizer = G.Tokenizer Text++-- | Check if the next input chars agree with the given string+lookAhead = G.lookAhead :: [Char] -> Tokenizer Bool++-- | Proceed to the next character+walk = G.walk :: Tokenizer ()++-- | Walk back to the previous character, unless it was discarded/emitted.+walkBack = G.walkBack :: Tokenizer ()++-- | Restore the state after the last emit/discard.+restore = G.restore :: Tokenizer ()++-- | Peek the current character+peek = G.peek :: Tokenizer Char++-- | Peek the current character and proceed+pop = G.pop :: Tokenizer Char++-- | Break at the current position and emit the scanned token+emit = G.emit :: Tokenizer ()++-- | Break at the current position and discard the scanned token+discard = G.discard :: Tokenizer ()++-- | Have I reached the end of the input text?+isEOT = G.isEOT :: Tokenizer Bool++-- | Proceed as long as a given function succeeds+walkWhile = G.walkWhile :: (Char -> Bool) -> Tokenizer ()++-- | Proceed as long as a given fold returns Just (generalization of walkWhile)+walkFold = G.walkFold :: a -> (Char -> a -> Maybe a) -> Tokenizer ()++-- | Repeat a given tokenizer as long as the end of text is not reached+untilEOT = G.untilEOT :: Tokenizer () -> Tokenizer ()++-- | Split a string into tokens using the given tokenizer+runTokenizer = G.runTokenizer :: Tokenizer () -> Text -> [Text]++-- | Split a string into tokens using the given tokenizer, case sensitive version+runTokenizerCS = G.runTokenizerCS :: Tokenizer () -> Text -> [Text]
tokenizer-monad.cabal view
@@ -10,13 +10,13 @@ -- PVP summary: +-+------- breaking API changes -- | | +----- non-breaking API additions -- | | | +--- code changes with no API change-version: 0.1.0.0+version: 0.2.0.0 -- A short (one-line) description of the package. synopsis: An efficient and easy-to-use tokenizer monad. -- A longer description of the package.-description: This monad can be used for writing efficient and readable tokenizers.+description: This monad can be used for writing efficient and readable tokenizers in an imperative way. -- The license under which the package is released. license: GPL-3@@ -25,14 +25,14 @@ license-file: LICENSE -- The package author(s).-author: Marvin Cohrs+author: Enum Cohrs -- An email address to which users can send suggestions, bug reports, and -- patches.-maintainer: darcs@m.doomanddarkness.eu+maintainer: darcs@enumeration.eu -- A copyright notice.-copyright: (c) 2017, 2018 Marvin Cohrs+copyright: (c) 2017-2019 Enum Cohrs category: Text @@ -41,6 +41,7 @@ -- Extra files to be distributed with the package, such as examples or a -- README. extra-source-files: ChangeLog.md+ README.md -- Constraint on the version of Cabal needed to build this package. cabal-version: >=1.10@@ -49,6 +50,11 @@ library -- Modules exported by the library. exposed-modules: Control.Monad.Tokenizer+ Control.Monad.Tokenizer.Text.Strict+ Control.Monad.Tokenizer.Text.Lazy+ Control.Monad.Tokenizer.String+ Control.Monad.Tokenizer.Char8.Strict+ Control.Monad.Tokenizer.Char8.Lazy -- Modules included in this library but not exported. -- other-modules: @@ -57,7 +63,9 @@ other-extensions: OverloadedStrings, BangPatterns -- Other library packages from which modules are imported.- build-depends: base >=4.9 && <4.10, text >=1.2 && <1.3+ build-depends: base >=4.9 && <4.11,+ text >=1.2,+ bytestring -- Directories containing source files. hs-source-dirs: src