trifecta 0.51.0.1 → 0.52
raw patch · 16 files changed
+26/−1459 lines, 16 filesdep +charsetdep −kan-extensionsPVP ok
version bump matches the API change (PVP)
Dependencies added: charset
Dependencies removed: kan-extensions
API changes (from Hackage documentation)
- Text.Trifecta.CharSet: (\\) :: CharSet -> CharSet -> CharSet
- Text.Trifecta.CharSet: CharSet :: !Bool -> {-# UNPACK #-} !ByteSet -> !IntSet -> CharSet
- Text.Trifecta.CharSet: build :: (Char -> Bool) -> CharSet
- Text.Trifecta.CharSet: complement :: CharSet -> CharSet
- Text.Trifecta.CharSet: data CharSet
- Text.Trifecta.CharSet: delete :: Char -> CharSet -> CharSet
- Text.Trifecta.CharSet: difference :: CharSet -> CharSet -> CharSet
- Text.Trifecta.CharSet: empty :: CharSet
- Text.Trifecta.CharSet: filter :: (Char -> Bool) -> CharSet -> CharSet
- Text.Trifecta.CharSet: fold :: (Char -> b -> b) -> b -> CharSet -> b
- Text.Trifecta.CharSet: fromAscList :: String -> CharSet
- Text.Trifecta.CharSet: fromCharSet :: CharSet -> (Bool, IntSet)
- Text.Trifecta.CharSet: fromDistinctAscList :: String -> CharSet
- Text.Trifecta.CharSet: fromList :: String -> CharSet
- Text.Trifecta.CharSet: full :: CharSet
- Text.Trifecta.CharSet: insert :: Char -> CharSet -> CharSet
- Text.Trifecta.CharSet: instance Bounded CharSet
- Text.Trifecta.CharSet: instance Data CharSet
- Text.Trifecta.CharSet: instance Eq CharSet
- Text.Trifecta.CharSet: instance Monoid CharSet
- Text.Trifecta.CharSet: instance Ord CharSet
- Text.Trifecta.CharSet: instance Read CharSet
- Text.Trifecta.CharSet: instance Semigroup CharSet
- Text.Trifecta.CharSet: instance Show CharSet
- Text.Trifecta.CharSet: instance Typeable CharSet
- Text.Trifecta.CharSet: intersection :: CharSet -> CharSet -> CharSet
- Text.Trifecta.CharSet: isComplemented :: CharSet -> Bool
- Text.Trifecta.CharSet: isSubsetOf :: CharSet -> CharSet -> Bool
- Text.Trifecta.CharSet: map :: (Char -> Char) -> CharSet -> CharSet
- Text.Trifecta.CharSet: member :: Char -> CharSet -> Bool
- Text.Trifecta.CharSet: notMember :: Char -> CharSet -> Bool
- Text.Trifecta.CharSet: null :: CharSet -> Bool
- Text.Trifecta.CharSet: overlaps :: CharSet -> CharSet -> Bool
- Text.Trifecta.CharSet: partition :: (Char -> Bool) -> CharSet -> (CharSet, CharSet)
- Text.Trifecta.CharSet: range :: Char -> Char -> CharSet
- Text.Trifecta.CharSet: singleton :: Char -> CharSet
- Text.Trifecta.CharSet: size :: CharSet -> Int
- Text.Trifecta.CharSet: toArray :: CharSet -> UArray Char Bool
- Text.Trifecta.CharSet: toAscList :: CharSet -> String
- Text.Trifecta.CharSet: toCharSet :: IntSet -> CharSet
- Text.Trifecta.CharSet: toList :: CharSet -> String
- Text.Trifecta.CharSet: union :: CharSet -> CharSet -> CharSet
- Text.Trifecta.CharSet.Common: control, asciiLower, asciiUpper, latin1, ascii, separator, symbol, punctuation, number, mark, letter, octDigit, digit, print, alphaNum, alpha, upper, lower, space :: CharSet
- Text.Trifecta.CharSet.Posix: lookupPosixAsciiCharSet :: String -> Maybe CharSet
- Text.Trifecta.CharSet.Posix: lookupPosixUnicodeCharSet :: String -> Maybe CharSet
- Text.Trifecta.CharSet.Posix: posixAscii :: HashMap String CharSet
- Text.Trifecta.CharSet.Posix: posixUnicode :: HashMap String CharSet
- Text.Trifecta.CharSet.Posix.Ascii: alnum, xdigit, lower, upper, space, punct, word, print, graph, digit, cntrl, blank, ascii, alpha :: CharSet
- Text.Trifecta.CharSet.Posix.Ascii: lookupPosixAsciiCharSet :: String -> Maybe CharSet
- Text.Trifecta.CharSet.Posix.Ascii: posixAscii :: HashMap String CharSet
- Text.Trifecta.CharSet.Posix.Unicode: alnum, xdigit, lower, upper, space, punct, word, print, graph, digit, cntrl, blank, ascii, alpha :: CharSet
- Text.Trifecta.CharSet.Posix.Unicode: lookupPosixUnicodeCharSet :: String -> Maybe CharSet
- Text.Trifecta.CharSet.Posix.Unicode: posixUnicode :: HashMap String CharSet
- Text.Trifecta.CharSet.Unicode: UnicodeCategory :: String -> String -> CharSet -> String -> UnicodeCategory
- Text.Trifecta.CharSet.Unicode: control, other, notAssigned, surrogate, privateUse, format :: CharSet
- Text.Trifecta.CharSet.Unicode: dashPunctuation, punctuation, otherPunctuation, connectorPunctuation, finalQuote, initialQuote, closePunctuation, openPunctuation :: CharSet
- Text.Trifecta.CharSet.Unicode: data UnicodeCategory
- Text.Trifecta.CharSet.Unicode: decimalNumber, number, otherNumber, letterNumber :: CharSet
- Text.Trifecta.CharSet.Unicode: instance Data UnicodeCategory
- Text.Trifecta.CharSet.Unicode: instance Show UnicodeCategory
- Text.Trifecta.CharSet.Unicode: instance Typeable UnicodeCategory
- Text.Trifecta.CharSet.Unicode: lowercaseLetter, letter, otherLetter, modifierLetter, letterAnd, titlecaseLetter, uppercaseLetter :: CharSet
- Text.Trifecta.CharSet.Unicode: mathSymbol, symbol, otherSymbol, modifierSymbol, currencySymbol :: CharSet
- Text.Trifecta.CharSet.Unicode: nonSpacingMark, mark, enclosingMark, spacingCombiningMark :: CharSet
- Text.Trifecta.CharSet.Unicode: space, separator, paragraphSeparator, lineSeparator :: CharSet
- Text.Trifecta.CharSet.Unicode: unicodeCategories :: [UnicodeCategory]
- Text.Trifecta.CharSet.Unicode.Block: Block :: String -> CharSet -> Block
- Text.Trifecta.CharSet.Unicode.Block: alphabeticPresentationForms :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: arabic :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: arabicPresentationFormsA :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: arabicPresentationFormsB :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: armenian :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: arrows :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: basicLatin :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: bengali :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: blockCharSet :: Block -> CharSet
- Text.Trifecta.CharSet.Unicode.Block: blockElements :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: blockName :: Block -> String
- Text.Trifecta.CharSet.Unicode.Block: blocks :: [Block]
- Text.Trifecta.CharSet.Unicode.Block: bopomofo :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: bopomofoExtended :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: boxDrawing :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: braillePatterns :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: buhid :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: cherokee :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: cjkCompatibility :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: cjkCompatibilityForms :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: cjkCompatibilityIdeographs :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: cjkRadicalsSupplement :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: cjkSymbolsAndPunctuation :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: cjkUnifiedIdeographs :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: cjkUnifiedIdeographsExtensionA :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: combiningDiacriticalMarks :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: combiningDiacriticalMarksForSymbols :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: combiningHalfMarks :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: controlPictures :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: currencySymbols :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: cyrillic :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: cyrillicSupplementary :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: data Block
- Text.Trifecta.CharSet.Unicode.Block: devanagari :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: dingbats :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: enclosedAlphanumerics :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: enclosedCjkLettersAndMonths :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: ethiopic :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: generalPunctuation :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: geometricShapes :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: georgian :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: greekAndCoptic :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: greekExtended :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: gujarati :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: gurmukhi :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: halfwidthAndFullwidthForms :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: hangulCompatibilityJamo :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: hangulJamo :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: hangulSyllables :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: hanunoo :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: hebrew :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: highPrivateUseSurrogates :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: highSurrogates :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: hiragana :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: ideographicDescriptionCharacters :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: instance Data Block
- Text.Trifecta.CharSet.Unicode.Block: instance Show Block
- Text.Trifecta.CharSet.Unicode.Block: instance Typeable Block
- Text.Trifecta.CharSet.Unicode.Block: ipaExtensions :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: kanbun :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: kangxiRadicals :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: kannada :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: katakana :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: katakanaPhoneticExtensions :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: khmer :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: khmerSymbols :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: lao :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: latin1Supplement :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: latinExtendedA :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: latinExtendedAdditional :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: latinExtendedB :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: letterlikeSymbols :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: limbu :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: lookupBlock :: String -> Maybe Block
- Text.Trifecta.CharSet.Unicode.Block: lookupBlockCharSet :: String -> Maybe CharSet
- Text.Trifecta.CharSet.Unicode.Block: lowSurrogates :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: malayalam :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: mathematicalOperators :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: miscellaneousMathematicalSymbolsA :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: miscellaneousMathematicalSymbolsB :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: miscellaneousSymbols :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: miscellaneousSymbolsAndArrows :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: miscellaneousTechnical :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: mongolian :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: myanmar :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: numberForms :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: ogham :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: opticalCharacterRecognition :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: oriya :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: phoneticExtensions :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: privateUseArea :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: runic :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: sinhala :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: smallFormVariants :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: spacingModifierLetters :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: specials :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: superscriptsAndSubscripts :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: supplementalArrowsA :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: supplementalArrowsB :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: supplementalMathematicalOperators :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: syriac :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: tagalog :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: tagbanwa :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: taiLe :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: tamil :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: telugu :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: thaana :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: thai :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: tibetan :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: unifiedCanadianAboriginalSyllabics :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: variationSelectors :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: yiRadicals :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: yiSyllables :: CharSet
- Text.Trifecta.CharSet.Unicode.Block: yijingHexagramSymbols :: CharSet
- Text.Trifecta.CharSet.Unicode.Category: Category :: String -> String -> CharSet -> String -> Category
- Text.Trifecta.CharSet.Unicode.Category: categories :: [Category]
- Text.Trifecta.CharSet.Unicode.Category: categoryAbbreviation :: Category -> String
- Text.Trifecta.CharSet.Unicode.Category: categoryCharSet :: Category -> CharSet
- Text.Trifecta.CharSet.Unicode.Category: categoryDescription :: Category -> String
- Text.Trifecta.CharSet.Unicode.Category: categoryName :: Category -> String
- Text.Trifecta.CharSet.Unicode.Category: control, other, notAssigned, surrogate, privateUse, format :: CharSet
- Text.Trifecta.CharSet.Unicode.Category: dashPunctuation, punctuation, otherPunctuation, connectorPunctuation, finalQuote, initialQuote, closePunctuation, openPunctuation :: CharSet
- Text.Trifecta.CharSet.Unicode.Category: data Category
- Text.Trifecta.CharSet.Unicode.Category: decimalNumber, number, otherNumber, letterNumber :: CharSet
- Text.Trifecta.CharSet.Unicode.Category: instance Data Category
- Text.Trifecta.CharSet.Unicode.Category: instance Show Category
- Text.Trifecta.CharSet.Unicode.Category: instance Typeable Category
- Text.Trifecta.CharSet.Unicode.Category: lookupCategory :: String -> Maybe Category
- Text.Trifecta.CharSet.Unicode.Category: lookupCategoryCharSet :: String -> Maybe CharSet
- Text.Trifecta.CharSet.Unicode.Category: lowercaseLetter, letter, otherLetter, modifierLetter, letterAnd, titlecaseLetter, uppercaseLetter :: CharSet
- Text.Trifecta.CharSet.Unicode.Category: mathSymbol, symbol, otherSymbol, modifierSymbol, currencySymbol :: CharSet
- Text.Trifecta.CharSet.Unicode.Category: nonSpacingMark, mark, enclosingMark, spacingCombiningMark :: CharSet
- Text.Trifecta.CharSet.Unicode.Category: space, separator, paragraphSeparator, lineSeparator :: CharSet
- Text.Trifecta.Parser.Class: instance MonadParser m => MonadParser (Yoneda m)
- Text.Trifecta.Parser.Mark: instance MonadMark d m => MonadMark d (Yoneda m)
- Text.Trifecta.Parser.Prim: parseTest :: Show a => (forall r. Parser r String a) -> String -> IO ()
- Text.Trifecta.Util.ByteSet: ByteSet :: ByteString -> ByteSet
- Text.Trifecta.Util.ByteSet: fromList :: [Word8] -> ByteSet
- Text.Trifecta.Util.ByteSet: instance Eq ByteSet
- Text.Trifecta.Util.ByteSet: instance Ord ByteSet
- Text.Trifecta.Util.ByteSet: instance Show ByteSet
- Text.Trifecta.Util.ByteSet: member :: Word8 -> ByteSet -> Bool
- Text.Trifecta.Util.ByteSet: newtype ByteSet
+ Text.Trifecta.Parser.ByteString: parseByteString :: Show a => (forall r. Parser r String a) -> Delta -> ByteString -> Result TermDoc a
+ Text.Trifecta.Parser.ByteString: parseTest :: Show a => (forall r. Parser r String a) -> String -> IO ()
Files
- Text/Trifecta/CharSet.hs +0/−359
- Text/Trifecta/CharSet/Common.hs +0/−65
- Text/Trifecta/CharSet/Posix.hs +0/−21
- Text/Trifecta/CharSet/Posix/Ascii.hs +0/−61
- Text/Trifecta/CharSet/Posix/Unicode.hs +0/−64
- Text/Trifecta/CharSet/Unicode.hs +0/−177
- Text/Trifecta/CharSet/Unicode/Block.hs +0/−382
- Text/Trifecta/CharSet/Unicode/Category.hs +0/−212
- Text/Trifecta/Parser/ByteString.hs +19/−6
- Text/Trifecta/Parser/Char.hs +3/−3
- Text/Trifecta/Parser/Char8.hs +2/−2
- Text/Trifecta/Parser/Class.hs +0/−17
- Text/Trifecta/Parser/Mark.hs +0/−4
- Text/Trifecta/Parser/Prim.hs +0/−11
- Text/Trifecta/Util/ByteSet.hs +0/−64
- trifecta.cabal +2/−11
− Text/Trifecta/CharSet.hs
@@ -1,359 +0,0 @@-{-# LANGUAGE CPP #-}-{-# OPTIONS_GHC -fspec-constr #-}--------------------------------------------------------------------------------- |--- Module : Text.Trifecta.CharSet--- Copyright : (c) Edward Kmett 2010-2011--- License : BSD3--- Maintainer : ekmett@gmail.com--- Stability : experimental--- Portability : non-portable (Data, BangPatterns, MagicHash)------ Fast set membership tests for 'Char' values------ Stored as a (possibly negated) IntMap and a fast set used for the head byte.------ The set of valid (possibly negated) head bytes is stored unboxed as a 32-byte--- bytestring-based lookup table.------ Designed to be imported qualified:--- --- > import Text.Trifecta.CharSet (CharSet)--- > import qualified Text.Trifecta.CharSet as CharSet---- --------------------------------------------------------------------------------------------------------------------------------------------------------------------module Text.Trifecta.CharSet- (- -- * Set type- CharSet(..)- -- * Operators- , (\\)- -- * Query- , null- , size- , member- , notMember- , overlaps, isSubsetOf- , isComplemented- -- * Construction- , build- , empty- , singleton- , full- , insert- , delete- , complement- , range- -- * Combine- , union- , intersection- , difference- -- * Filter- , filter- , partition- -- * Map- , map- -- * Fold- , fold- -- * Conversion- -- ** List- , toList- , fromList- -- ** Ordered list- , toAscList- , fromAscList- , fromDistinctAscList- -- ** IntMaps- , fromCharSet- , toCharSet- -- ** Array- , toArray- ) where--import Data.Array.Unboxed hiding (range)-import Data.Data-import Data.Function (on)-import Data.IntSet (IntSet)-import Text.Trifecta.Util.ByteSet (ByteSet)-import qualified Text.Trifecta.Util.ByteSet as ByteSet-import Data.Bits hiding (complement)-import Data.Word-import Data.ByteString.Internal (c2w)-import Data.Semigroup-import qualified Data.IntSet as I-import qualified Data.List as L-import Prelude hiding (filter, map, null)-import qualified Prelude as P-import Text.Read--data CharSet = CharSet !Bool {-# UNPACK #-} !ByteSet !IntSet --charSet :: Bool -> IntSet -> CharSet-charSet b s = CharSet b (ByteSet.fromList (fmap headByte (I.toAscList s))) s--headByte :: Int -> Word8-headByte i - | i <= 0x7f = toEnum i - | i <= 0x7ff = toEnum $ 0x80 + (i `shiftR` 6)- | i <= 0xffff = toEnum $ 0xe0 + (i `shiftR` 12)- | otherwise = toEnum $ 0xf0 + (i `shiftR` 18)--pos :: IntSet -> CharSet-pos = charSet True--neg :: IntSet -> CharSet-neg = charSet False--(\\) :: CharSet -> CharSet -> CharSet-(\\) = difference--build :: (Char -> Bool) -> CharSet-build p = fromDistinctAscList $ P.filter p [minBound .. maxBound]-{-# INLINE build #-}--map :: (Char -> Char) -> CharSet -> CharSet-map f (CharSet True _ i) = pos (I.map (fromEnum . f . toEnum) i)-map f (CharSet False _ i) = fromList $ P.map f $ P.filter (\x -> fromEnum x `I.notMember` i) [ul..uh] -{-# INLINE map #-}--isComplemented :: CharSet -> Bool-isComplemented (CharSet True _ _) = False-isComplemented (CharSet False _ _) = True-{-# INLINE isComplemented #-}--toList :: CharSet -> String-toList (CharSet True _ i) = P.map toEnum (I.toList i)-toList (CharSet False _ i) = P.filter (\x -> fromEnum x `I.notMember` i) [ul..uh]-{-# INLINE toList #-}--toAscList :: CharSet -> String-toAscList (CharSet True _ i) = P.map toEnum (I.toAscList i)-toAscList (CharSet False _ i) = P.filter (\x -> fromEnum x `I.notMember` i) [ul..uh]-{-# INLINE toAscList #-}- -empty :: CharSet-empty = pos I.empty--singleton :: Char -> CharSet-singleton = pos . I.singleton . fromEnum-{-# INLINE singleton #-}--full :: CharSet-full = neg I.empty---- | /O(n)/ worst case-null :: CharSet -> Bool-null (CharSet True _ i) = I.null i-null (CharSet False _ i) = I.size i == numChars-{-# INLINE null #-}---- | /O(n)/-size :: CharSet -> Int-size (CharSet True _ i) = I.size i-size (CharSet False _ i) = numChars - I.size i-{-# INLINE size #-}--insert :: Char -> CharSet -> CharSet-insert c (CharSet True _ i) = pos (I.insert (fromEnum c) i)-insert c (CharSet False _ i) = neg (I.delete (fromEnum c) i)-{-# INLINE insert #-}--range :: Char -> Char -> CharSet-range a b - | a <= b = fromDistinctAscList [a..b]- | otherwise = empty--delete :: Char -> CharSet -> CharSet-delete c (CharSet True _ i) = pos (I.delete (fromEnum c) i)-delete c (CharSet False _ i) = neg (I.insert (fromEnum c) i)-{-# INLINE delete #-}--complement :: CharSet -> CharSet-complement (CharSet True s i) = CharSet False s i-complement (CharSet False s i) = CharSet True s i-{-# INLINE complement #-}--union :: CharSet -> CharSet -> CharSet-union (CharSet True _ i) (CharSet True _ j) = pos (I.union i j)-union (CharSet True _ i) (CharSet False _ j) = neg (I.difference j i)-union (CharSet False _ i) (CharSet True _ j) = neg (I.difference i j)-union (CharSet False _ i) (CharSet False _ j) = neg (I.intersection i j)-{-# INLINE union #-}--intersection :: CharSet -> CharSet -> CharSet-intersection (CharSet True _ i) (CharSet True _ j) = pos (I.intersection i j)-intersection (CharSet True _ i) (CharSet False _ j) = pos (I.difference i j)-intersection (CharSet False _ i) (CharSet True _ j) = pos (I.difference j i)-intersection (CharSet False _ i) (CharSet False _ j) = neg (I.union i j)-{-# INLINE intersection #-}--difference :: CharSet -> CharSet -> CharSet -difference (CharSet True _ i) (CharSet True _ j) = pos (I.difference i j)-difference (CharSet True _ i) (CharSet False _ j) = pos (I.intersection i j)-difference (CharSet False _ i) (CharSet True _ j) = neg (I.union i j)-difference (CharSet False _ i) (CharSet False _ j) = pos (I.difference j i)-{-# INLINE difference #-}--member :: Char -> CharSet -> Bool-member c (CharSet True b i)- | c <= toEnum 0x7f = ByteSet.member (c2w c) b- | otherwise = I.member (fromEnum c) i-member c (CharSet False b i) - | c <= toEnum 0x7f = not (ByteSet.member (c2w c) b)- | otherwise = I.notMember (fromEnum c) i-{-# INLINE member #-}--notMember :: Char -> CharSet -> Bool-notMember c s = not (member c s)-{-# INLINE notMember #-}--fold :: (Char -> b -> b) -> b -> CharSet -> b-fold f z (CharSet True _ i) = I.fold (f . toEnum) z i-fold f z (CharSet False _ i) = foldr f z $ P.filter (\x -> fromEnum x `I.notMember` i) [ul..uh]-{-# INLINE fold #-}--filter :: (Char -> Bool) -> CharSet -> CharSet -filter p (CharSet True _ i) = pos (I.filter (p . toEnum) i)-filter p (CharSet False _ i) = neg $ foldr (I.insert) i $ P.filter (\x -> (x `I.notMember` i) && not (p (toEnum x))) [ol..oh]-{-# INLINE filter #-}--partition :: (Char -> Bool) -> CharSet -> (CharSet, CharSet)-partition p (CharSet True _ i) = (pos l, pos r)- where (l,r) = I.partition (p . toEnum) i-partition p (CharSet False _ i) = (neg (foldr I.insert i l), neg (foldr I.insert i r))- where (l,r) = L.partition (p . toEnum) $ P.filter (\x -> x `I.notMember` i) [ol..oh]-{-# INLINE partition #-}--overlaps :: CharSet -> CharSet -> Bool-overlaps (CharSet True _ i) (CharSet True _ j) = not (I.null (I.intersection i j))-overlaps (CharSet True _ i) (CharSet False _ j) = not (I.isSubsetOf j i)-overlaps (CharSet False _ i) (CharSet True _ j) = not (I.isSubsetOf i j)-overlaps (CharSet False _ i) (CharSet False _ j) = any (\x -> I.notMember x i && I.notMember x j) [ol..oh] -- not likely-{-# INLINE overlaps #-}--isSubsetOf :: CharSet -> CharSet -> Bool-isSubsetOf (CharSet True _ i) (CharSet True _ j) = I.isSubsetOf i j-isSubsetOf (CharSet True _ i) (CharSet False _ j) = I.null (I.intersection i j)-isSubsetOf (CharSet False _ i) (CharSet True _ j) = all (\x -> I.member x i && I.member x j) [ol..oh] -- not bloody likely-isSubsetOf (CharSet False _ i) (CharSet False _ j) = I.isSubsetOf j i-{-# INLINE isSubsetOf #-}--fromList :: String -> CharSet -fromList = pos . I.fromList . P.map fromEnum-{-# INLINE fromList #-}--fromAscList :: String -> CharSet-fromAscList = pos . I.fromAscList . P.map fromEnum-{-# INLINE fromAscList #-}--fromDistinctAscList :: String -> CharSet-fromDistinctAscList = pos . I.fromDistinctAscList . P.map fromEnum-{-# INLINE fromDistinctAscList #-}---- isProperSubsetOf :: CharSet -> CharSet -> Bool--- isProperSubsetOf (P i) (P j) = I.isProperSubsetOf i j--- isProperSubsetOf (P i) (N j) = null (I.intersection i j) && ...--- isProperSubsetOf (N i) (N j) = I.isProperSubsetOf j i--ul, uh :: Char-ul = minBound-uh = maxBound-{-# INLINE ul #-}-{-# INLINE uh #-}--ol, oh :: Int-ol = fromEnum ul-oh = fromEnum uh-{-# INLINE ol #-}-{-# INLINE oh #-}--numChars :: Int-numChars = oh - ol + 1-{-# INLINE numChars #-}--instance Typeable CharSet where- typeOf _ = mkTyConApp charSetTyCon []--charSetTyCon :: TyCon-#if __GLASGOW_HASKELL__ < 704-charSetTyCon = mkTyCon "Text.Trifecta.CharSet.CharSet"-#else-charSetTyCon = mkTyCon3 "trifecta" "Text.Trifecta.CharSet" "CharSet"-#endif-{-# NOINLINE charSetTyCon #-}--instance Data CharSet where- gfoldl k z set - | isComplemented set = z complement `k` complement set- | otherwise = z fromList `k` toList set-- toConstr set - | isComplemented set = complementConstr- | otherwise = fromListConstr-- dataTypeOf _ = charSetDataType-- gunfold k z c = case constrIndex c of- 1 -> k (z fromList)- 2 -> k (z complement)- _ -> error "gunfold"--fromListConstr :: Constr-fromListConstr = mkConstr charSetDataType "fromList" [] Prefix-{-# NOINLINE fromListConstr #-}--complementConstr :: Constr-complementConstr = mkConstr charSetDataType "complement" [] Prefix-{-# NOINLINE complementConstr #-}--charSetDataType :: DataType-charSetDataType = mkDataType "Text.Trifecta.CharSet.CharSet" [fromListConstr, complementConstr]-{-# NOINLINE charSetDataType #-}---- returns an intset and if the charSet is positive-fromCharSet :: CharSet -> (Bool, IntSet)-fromCharSet (CharSet b _ i) = (b, i)-{-# INLINE fromCharSet #-}--toCharSet :: IntSet -> CharSet-toCharSet = pos-{-# INLINE toCharSet #-}--instance Eq CharSet where- (==) = (==) `on` toAscList--instance Ord CharSet where- compare = compare `on` toAscList--instance Bounded CharSet where- minBound = empty- maxBound = full---- TODO return a tighter bounded array perhaps starting from the least element present to the last element present?-toArray :: CharSet -> UArray Char Bool-toArray set = array (minBound, maxBound) $ fmap (\x -> (x, x `member` set)) [minBound .. maxBound]- -instance Show CharSet where- showsPrec d i- | isComplemented i = showParen (d > 10) $ showString "complement " . showsPrec 11 (complement i)- | otherwise = showParen (d > 10) $ showString "fromDistinctAscList " . showsPrec 11 (toAscList i)--instance Read CharSet where- readPrec = parens $ complemented +++ normal - where- complemented = prec 10 $ do - Ident "complement" <- lexP- complement `fmap` step readPrec- normal = prec 10 $ do- Ident "fromDistinctAscList" <- lexP- fromDistinctAscList `fmap` step readPrec--instance Semigroup CharSet where- (<>) = union--instance Monoid CharSet where- mempty = empty- mappend = union
− Text/Trifecta/CharSet/Common.hs
@@ -1,65 +0,0 @@--------------------------------------------------------------------------------- |--- Module : Text.Trifecta.CharSet.Common--- Copyright : (c) Edward Kmett 2010-2011--- License : BSD3--- Maintainer : ekmett@gmail.com--- Stability : experimental--- Portability : portable------ The various character classifications from "Data.Char" as 'CharSet's----------------------------------------------------------------------------------module Text.Trifecta.CharSet.Common- ( - -- ** Data.Char classes- control- , space- , lower- , upper- , alpha- , alphaNum- , print- , digit- , octDigit- , letter- , mark- , number- , punctuation- , symbol- , separator- , ascii- , latin1- , asciiUpper- , asciiLower- ) where--import Prelude ()-import Data.Char-import Text.Trifecta.CharSet---- Haskell character classes from Data.Char-control, space, lower, upper, alpha, alphaNum, - print, digit, octDigit, letter, mark, number, - punctuation, symbol, separator, ascii, latin1- , asciiUpper, asciiLower :: CharSet--control = build isControl-space = build isSpace-lower = build isLower-upper = build isUpper-alpha = build isAlpha-alphaNum = build isAlphaNum-print = build isPrint-digit = build isDigit-octDigit = build isOctDigit-letter = build isLetter-mark = build isMark-number = build isNumber-punctuation = build isPunctuation-symbol = build isSymbol-separator = build isSeparator-ascii = build isAscii-latin1 = build isLatin1-asciiUpper = build isAsciiUpper-asciiLower = build isAsciiLower
− Text/Trifecta/CharSet/Posix.hs
@@ -1,21 +0,0 @@--------------------------------------------------------------------------------- |--- Module : Text.Trifecta.CharSet.Posix--- Copyright : (c) Edward Kmett 2011--- License : BSD3--- Maintainer : ekmett@gmail.com--- Stability : experimental--- Portability : portable-------------------------------------------------------------------------------------module Text.Trifecta.CharSet.Posix- ( posixAscii- , lookupPosixAsciiCharSet- , posixUnicode- , lookupPosixUnicodeCharSet- ) where--import Text.Trifecta.CharSet.Posix.Ascii-import Text.Trifecta.CharSet.Posix.Unicode-import Prelude ()
− Text/Trifecta/CharSet/Posix/Ascii.hs
@@ -1,61 +0,0 @@--------------------------------------------------------------------------------- |--- Module : Text.Trifecta.CharSet.Posix.Ascii--- Copyright : (c) Edward Kmett 2010--- License : BSD3--- Maintainer : ekmett@gmail.com--- Stability : experimental--- Portability : portable-------------------------------------------------------------------------------------module Text.Trifecta.CharSet.Posix.Ascii- ( posixAscii- , lookupPosixAsciiCharSet- -- * Traditional POSIX ASCII \"classes\"- , alnum, alpha, ascii, blank, cntrl, digit, graph, print, word, punct, space, upper, lower, xdigit- ) where--import Prelude hiding (print)-import Data.Char-import Text.Trifecta.CharSet-import Data.HashMap.Lazy (HashMap)-import qualified Data.HashMap.Lazy as HashMap--alnum, alpha, ascii, blank, cntrl, digit, graph, print, word, punct, space, upper, lower, xdigit :: CharSet-alnum = alpha `union` digit-alpha = lower `union` upper-ascii = range '\x00' '\x7f'-blank = fromList " \t"-cntrl = insert '\x7f' $ range '\x00' '\x1f'-digit = range '0' '9'-lower = range 'a' 'z'-upper = range 'A' 'Z'-graph = range '\x21' '\x7e'-print = insert '\x20' graph-word = insert '_' alnum-punct = fromList "-!\"#$%&'()*+,./:;<=>?@[\\]^_`{|}~"-space = fromList " \t\r\n\v\f"-xdigit = digit `union` range 'a' 'f' `union` range 'A' 'F'---- :digit:, etc.-posixAscii :: HashMap String CharSet-posixAscii = HashMap.fromList- [ ("alnum", alnum)- , ("alpha", alpha)- , ("ascii", ascii)- , ("blank", blank)- , ("cntrl", cntrl)- , ("digit", digit)- , ("graph", graph) - , ("print", print)- , ("word", word)- , ("punct", punct)- , ("space", space)- , ("upper", upper)- , ("lower", lower)- , ("xdigit", xdigit)- ]--lookupPosixAsciiCharSet :: String -> Maybe CharSet-lookupPosixAsciiCharSet s = HashMap.lookup (Prelude.map toLower s) posixAscii
− Text/Trifecta/CharSet/Posix/Unicode.hs
@@ -1,64 +0,0 @@--------------------------------------------------------------------------------- |--- Module : Text.Trifecta.CharSet.Posix.Unicode--- Copyright : (c) Edward Kmett 2010--- License : BSD3--- Maintainer : ekmett@gmail.com--- Stability : experimental--- Portability : portable-------------------------------------------------------------------------------------module Text.Trifecta.CharSet.Posix.Unicode- ( posixUnicode- , lookupPosixUnicodeCharSet- -- * POSIX ASCII \"classes\"- , alnum, alpha, ascii, blank, cntrl, digit, graph, print, word, punct, space, upper, lower, xdigit- ) where--import Prelude hiding (print)-import Data.Char-import Text.Trifecta.CharSet-import qualified Text.Trifecta.CharSet.Unicode.Category as Category-import qualified Text.Trifecta.CharSet.Unicode.Block as Block-import Data.HashMap.Lazy (HashMap)-import qualified Data.HashMap.Lazy as HashMap--alnum, alpha, ascii, blank, cntrl, digit, graph, print, word, punct, space, upper, lower, xdigit :: CharSet-alnum = alpha `union` digit-ascii = Block.basicLatin-alpha = Category.letterAnd-blank = insert '\t' Category.space -cntrl = Category.control-digit = Category.decimalNumber-lower = Category.lowercaseLetter-upper = Category.uppercaseLetter-graph = complement (Category.separator `union` Category.other)-print = complement (Category.other)-word = Category.letter `union` Category.number `union` Category.connectorPunctuation-punct = Category.punctuation `union` Category.symbol-space = fromList " \t\r\n\v\f" `union` Category.separator-xdigit = digit `union` range 'a' 'f' `union` range 'A' 'F'---- :digit:, etc.-posixUnicode :: HashMap String CharSet-posixUnicode = HashMap.fromList- [ ("alnum", alnum)- , ("alpha", alpha)- , ("ascii", ascii)- , ("blank", blank)- , ("cntrl", cntrl)- , ("digit", digit)- , ("graph", graph) - , ("print", print)- , ("word", word)- , ("punct", punct)- , ("space", space)- , ("upper", upper)- , ("lower", lower)- , ("xdigit", xdigit)- ]--lookupPosixUnicodeCharSet :: String -> Maybe CharSet-lookupPosixUnicodeCharSet s = HashMap.lookup (Prelude.map toLower s) posixUnicode-
− Text/Trifecta/CharSet/Unicode.hs
@@ -1,177 +0,0 @@-{-# LANGUAGE DeriveDataTypeable #-}--------------------------------------------------------------------------------- |--- Module : Text.Trifecta.CharSet.Unicode--- Copyright : (c) Edward Kmett 2010--- License : BSD3--- Maintainer : ekmett@gmail.com--- Stability : experimental--- Portability : portable------ Provides unicode general categories, which are typically connoted by --- @\p{Ll}@ or @\p{Modifier_Letter}@. Lookups can be constructed using 'categories'--- or individual character sets can be used directly.----------------------------------------------------------------------------------module Text.Trifecta.CharSet.Unicode- ( - -- * Unicode General Category- UnicodeCategory(..)- -- * Lookup- , unicodeCategories- -- * CharSets by UnicodeCategory- -- ** Letter- , modifierLetter, otherLetter, letter- -- *** Letter\&- , lowercaseLetter, uppercaseLetter, titlecaseLetter, letterAnd- -- ** Mark- , nonSpacingMark, spacingCombiningMark, enclosingMark, mark- -- ** Separator- , space, lineSeparator, paragraphSeparator, separator- -- ** Symbol- , mathSymbol, currencySymbol, modifierSymbol, otherSymbol, symbol- -- ** Number- , decimalNumber, letterNumber, otherNumber, number- -- ** Punctuation- , dashPunctuation, openPunctuation, closePunctuation, initialQuote- , finalQuote, connectorPunctuation, otherPunctuation, punctuation - -- ** Other- , control, format, privateUse, surrogate, notAssigned, other- ) where--import Data.Char-import Data.Data-import Text.Trifecta.CharSet--data UnicodeCategory = UnicodeCategory String String CharSet String- deriving (Show, Data, Typeable)---- \p{Letter} or \p{Mc}-unicodeCategories :: [UnicodeCategory]-unicodeCategories =- [ UnicodeCategory "Letter" "L" letter "any kind of letter from any language."- , UnicodeCategory "Lowercase_Letter" "Ll" lowercaseLetter "a lowercase letter that has an uppercase variant"- , UnicodeCategory "Uppercase_Letter" "Lu" uppercaseLetter "an uppercase letter that has a lowercase variant"- , UnicodeCategory "Titlecase_Letter" "Lt" titlecaseLetter "a letter that appears at the start of a word when only the first letter of the word is capitalized"- , UnicodeCategory "Letter&" "L&" letterAnd "a letter that exists in lowercase and uppercase variants (combination of Ll, Lu and Lt)"- , UnicodeCategory "Modifier_Letter" "Lm" modifierLetter "a special character that is used like a letter"- , UnicodeCategory "Other_Letter" "Lo" otherLetter "a letter or ideograph that does not have lowercase and uppercase variants"- , UnicodeCategory "Mark" "M" mark "a character intended to be combined with another character (e.g. accents, umlauts, enclosing boxes, etc.)"- , UnicodeCategory "Non_Spacing_Mark" "Mn" nonSpacingMark "a character intended to be combined with another character without taking up extra space (e.g. accents, umlauts, etc.)"- , UnicodeCategory "Spacing_Combining_Mark" "Mc" spacingCombiningMark "a character intended to be combined with another character that takes up extra space (vowel signs in many Eastern languages)"- , UnicodeCategory "Enclosing_Mark" "Me" enclosingMark "a character that encloses the character is is combined with (circle, square, keycap, etc.)"- , UnicodeCategory "Separator" "Z" separator "any kind of whitespace or invisible separator"- , UnicodeCategory "Space_Separator" "Zs" space "a whitespace character that is invisible, but does take up space"- , UnicodeCategory "Line_Separator" "Zl" lineSeparator "line separator character U+2028"- , UnicodeCategory "Paragraph_Separator" "Zp" paragraphSeparator "paragraph separator character U+2029"- , UnicodeCategory "Symbol" "S" symbol "math symbols, currency signs, dingbats, box-drawing characters, etc."- , UnicodeCategory "Math_Symbol" "Sm" mathSymbol "any mathematical symbol"- , UnicodeCategory "Currency_Symbol" "Sc" currencySymbol "any currency sign"- , UnicodeCategory "Modifier_Symbol" "Sk" modifierSymbol "a combining character (mark) as a full character on its own"- , UnicodeCategory "Other_Symbol" "So" otherSymbol "various symbols that are not math symbols, currency signs, or combining characters"- , UnicodeCategory "Number" "N" number "any kind of numeric character in any script"- , UnicodeCategory "Decimal_Digit_Number" "Nd" decimalNumber "a digit zero through nine in any script except ideographic scripts"- , UnicodeCategory "Letter_Number" "Nl" letterNumber "a number that looks like a letter, such as a Roman numeral"- , UnicodeCategory "Other_Number" "No" otherNumber "a superscript or subscript digit, or a number that is not a digit 0..9 (excluding numbers from ideographic scripts)"- , UnicodeCategory "Punctuation" "P" punctuation "any kind of punctuation character"- , UnicodeCategory "Dash_Punctuation" "Pd" dashPunctuation "any kind of hyphen or dash"- , UnicodeCategory "Open_Punctuation" "Ps" openPunctuation "any kind of opening bracket"- , UnicodeCategory "Close_Punctuation" "Pe" closePunctuation "any kind of closing bracket"- , UnicodeCategory "Initial_Punctuation" "Pi" initialQuote "any kind of opening quote"- , UnicodeCategory "Final_Punctuation" "Pf" finalQuote "any kind of closing quote"- , UnicodeCategory "Connector_Punctuation" "Pc" connectorPunctuation "a punctuation character such as an underscore that connects words"- , UnicodeCategory "Other_Punctuation" "Po" otherPunctuation "any kind of punctuation character that is not a dash, bracket, quote or connector"- , UnicodeCategory "Other" "C" other "invisible control characters and unused code points"- , UnicodeCategory "Control" "Cc" control "an ASCII 0x00..0x1F or Latin-1 0x80..0x9F control character"- , UnicodeCategory "Format" "Cf" format "invisible formatting indicator"- , UnicodeCategory "Private_Use" "Co" privateUse "any code point reserved for private use"- , UnicodeCategory "Surrogate" "Cs" surrogate "one half of a surrogate pair in UTF-16 encoding"- , UnicodeCategory "Unassigned" "Cn" notAssigned "any code point to which no character has been assigned.properties" ]--cat :: GeneralCategory -> CharSet-cat category = build ((category ==) . generalCategory)---- Letter-lowercaseLetter, uppercaseLetter, titlecaseLetter, letterAnd, modifierLetter, otherLetter, letter :: CharSet-lowercaseLetter = cat LowercaseLetter-uppercaseLetter = cat UppercaseLetter-titlecaseLetter = cat TitlecaseLetter-letterAnd = lowercaseLetter - `union` uppercaseLetter - `union` titlecaseLetter-modifierLetter = cat ModifierLetter-otherLetter = cat OtherLetter-letter - = letterAnd - `union` modifierLetter - `union` otherLetter---- Marks-nonSpacingMark, spacingCombiningMark, enclosingMark, mark :: CharSet-nonSpacingMark = cat NonSpacingMark-spacingCombiningMark = cat SpacingCombiningMark-enclosingMark = cat EnclosingMark-mark - = nonSpacingMark - `union` spacingCombiningMark - `union` enclosingMark--space, lineSeparator, paragraphSeparator, separator :: CharSet-space = cat Space-lineSeparator = cat LineSeparator-paragraphSeparator = cat ParagraphSeparator-separator - = space - `union` lineSeparator - `union` paragraphSeparator--mathSymbol, currencySymbol, modifierSymbol, otherSymbol, symbol :: CharSet-mathSymbol = cat MathSymbol-currencySymbol = cat CurrencySymbol-modifierSymbol = cat ModifierSymbol-otherSymbol = cat OtherSymbol-symbol - = mathSymbol - `union` currencySymbol - `union` modifierSymbol - `union` otherSymbol--decimalNumber, letterNumber, otherNumber, number :: CharSet-decimalNumber = cat DecimalNumber-letterNumber = cat LetterNumber-otherNumber = cat OtherNumber-number - = decimalNumber - `union` letterNumber - `union` otherNumber--dashPunctuation, openPunctuation, closePunctuation, initialQuote, - finalQuote, connectorPunctuation, otherPunctuation, punctuation :: CharSet--dashPunctuation = cat DashPunctuation-openPunctuation = cat OpenPunctuation-closePunctuation = cat ClosePunctuation-initialQuote = cat InitialQuote-finalQuote = cat FinalQuote-connectorPunctuation = cat ConnectorPunctuation-otherPunctuation = cat OtherPunctuation-punctuation - = dashPunctuation - `union` openPunctuation - `union` closePunctuation - `union` initialQuote - `union` finalQuote - `union` connectorPunctuation - `union` otherPunctuation--control, format, privateUse, surrogate, notAssigned, other :: CharSet-control = cat Control-format = cat Format-privateUse = cat PrivateUse-surrogate = cat Surrogate-notAssigned = cat NotAssigned-other = control - `union` format - `union` privateUse - `union` surrogate - `union` notAssigned
− Text/Trifecta/CharSet/Unicode/Block.hs
@@ -1,382 +0,0 @@-{-# LANGUAGE DeriveDataTypeable #-}-{-# OPTIONS_GHC -fno-warn-missing-signatures #-}--------------------------------------------------------------------------------- |--- Module : Text.Trifecta.CharSet.Unicode.Block--- Copyright : (c) Edward Kmett 2010-2011--- License : BSD3--- Maintainer : ekmett@gmail.com--- Stability : experimental--- Portability : portable------ Provides unicode general categories, which are typically connoted by --- @\p{InBasicLatin}@ or @\p{InIPA_Extensions}@. Lookups can be constructed using 'categories'--- or individual character sets can be used directly.----------------------------------------------------------------------------------module Text.Trifecta.CharSet.Unicode.Block- ( - -- * Unicode General Category- Block(..)- -- * Lookup- , blocks- , lookupBlock- , lookupBlockCharSet- -- * CharSets by Block- , basicLatin- , latin1Supplement- , latinExtendedA- , latinExtendedB- , ipaExtensions- , spacingModifierLetters- , combiningDiacriticalMarks- , greekAndCoptic- , cyrillic- , cyrillicSupplementary- , armenian- , hebrew- , arabic- , syriac- , thaana- , devanagari- , bengali- , gurmukhi- , gujarati- , oriya- , tamil- , telugu- , kannada- , malayalam- , sinhala- , thai- , lao- , tibetan- , myanmar- , georgian- , hangulJamo- , ethiopic- , cherokee- , unifiedCanadianAboriginalSyllabics- , ogham- , runic- , tagalog- , hanunoo- , buhid- , tagbanwa- , khmer- , mongolian- , limbu- , taiLe- , khmerSymbols- , phoneticExtensions- , latinExtendedAdditional- , greekExtended- , generalPunctuation- , superscriptsAndSubscripts- , currencySymbols- , combiningDiacriticalMarksForSymbols- , letterlikeSymbols- , numberForms- , arrows- , mathematicalOperators- , miscellaneousTechnical- , controlPictures- , opticalCharacterRecognition- , enclosedAlphanumerics- , boxDrawing- , blockElements- , geometricShapes- , miscellaneousSymbols- , dingbats- , miscellaneousMathematicalSymbolsA- , supplementalArrowsA- , braillePatterns- , supplementalArrowsB- , miscellaneousMathematicalSymbolsB- , supplementalMathematicalOperators- , miscellaneousSymbolsAndArrows- , cjkRadicalsSupplement- , kangxiRadicals- , ideographicDescriptionCharacters- , cjkSymbolsAndPunctuation- , hiragana- , katakana- , bopomofo- , hangulCompatibilityJamo- , kanbun- , bopomofoExtended- , katakanaPhoneticExtensions- , enclosedCjkLettersAndMonths- , cjkCompatibility- , cjkUnifiedIdeographsExtensionA- , yijingHexagramSymbols- , cjkUnifiedIdeographs- , yiSyllables- , yiRadicals- , hangulSyllables- , highSurrogates- , highPrivateUseSurrogates- , lowSurrogates- , privateUseArea- , cjkCompatibilityIdeographs- , alphabeticPresentationForms- , arabicPresentationFormsA- , variationSelectors- , combiningHalfMarks- , cjkCompatibilityForms- , smallFormVariants- , arabicPresentationFormsB- , halfwidthAndFullwidthForms- , specials- ) where--import Data.Char-import Text.Trifecta.CharSet-import Data.Data-import Data.HashMap.Lazy (HashMap)-import qualified Data.HashMap.Lazy as HashMap--data Block = Block - { blockName :: String- , blockCharSet :: CharSet- } deriving (Show, Data, Typeable)--blocks :: [Block]-blocks =- [ Block "Basic_Latin" basicLatin- , Block "Latin-1_Supplement" latin1Supplement- , Block "Latin_Extended-A" latinExtendedA- , Block "IPA_Extensions" ipaExtensions- , Block "Spacing_Modifier_Letters" spacingModifierLetters-- , Block "Latin_Extended-A" latinExtendedA- , Block "Latin_Extended-B" latinExtendedB- , Block "IPA_Extensions" ipaExtensions- , Block "Spacing_Modifier_Letters" spacingModifierLetters- , Block "Combining_Diacritical_Marks" combiningDiacriticalMarks- , Block "Greek_and_Coptic" greekAndCoptic- , Block "Cyrillic" cyrillic- , Block "Cyrillic_Supplementary" cyrillicSupplementary- , Block "Armenian" armenian- , Block "Hebrew" hebrew- , Block "Arabic" arabic- , Block "Syriac" syriac- , Block "Thaana" thaana- , Block "Devanagari" devanagari- , Block "Bengali" bengali- , Block "Gurmukhi" gurmukhi- , Block "Gujarati" gujarati- , Block "Oriya" oriya- , Block "Tamil" tamil- , Block "Telugu" telugu- , Block "Kannada" kannada- , Block "Malayalam" malayalam- , Block "Sinhala" sinhala- , Block "Thai" thai- , Block "Lao" lao- , Block "Tibetan" tibetan- , Block "Myanmar" myanmar- , Block "Georgian" georgian- , Block "Hangul_Jamo" hangulJamo- , Block "Ethiopic" ethiopic- , Block "Cherokee" cherokee- , Block "Unified_Canadian_Aboriginal_Syllabics" unifiedCanadianAboriginalSyllabics- , Block "Ogham" ogham- , Block "Runic" runic- , Block "Tagalog" tagalog- , Block "Hanunoo" hanunoo- , Block "Buhid" buhid- , Block "Tagbanwa" tagbanwa- , Block "Khmer" khmer- , Block "Mongolian" mongolian- , Block "Limbu" limbu- , Block "Tai_Le" taiLe- , Block "Khmer_Symbols" khmerSymbols- , Block "Phonetic_Extensions" phoneticExtensions- , Block "Latin_Extended_Additional" latinExtendedAdditional- , Block "Greek_Extended" greekExtended- , Block "General_Punctuation" generalPunctuation- , Block "Superscripts_and_Subscripts" superscriptsAndSubscripts- , Block "Currency_Symbols" currencySymbols- , Block "Combining_Diacritical_Marks_for_Symbols" combiningDiacriticalMarksForSymbols- , Block "Letterlike_Symbols" letterlikeSymbols- , Block "Number_Forms" numberForms- , Block "Arrows" arrows- , Block "Mathematical_Operators" mathematicalOperators- , Block "Miscellaneous_Technical" miscellaneousTechnical- , Block "Control_Pictures" controlPictures- , Block "Optical_Character_Recognition" opticalCharacterRecognition- , Block "Enclosed_Alphanumerics" enclosedAlphanumerics- , Block "Box_Drawing" boxDrawing- , Block "Block_Elements" blockElements- , Block "Geometric_Shapes" geometricShapes- , Block "Miscellaneous_Symbols" miscellaneousSymbols- , Block "Dingbats" dingbats- , Block "Miscellaneous_Mathematical_Symbols-A" miscellaneousMathematicalSymbolsA- , Block "Supplemental_Arrows-A" supplementalArrowsA- , Block "Braille_Patterns" braillePatterns- , Block "Supplemental_Arrows-B" supplementalArrowsB- , Block "Miscellaneous_Mathematical_Symbols-B" miscellaneousMathematicalSymbolsB- , Block "Supplemental_Mathematical_Operators" supplementalMathematicalOperators- , Block "Miscellaneous_Symbols_and_Arrows" miscellaneousSymbolsAndArrows- , Block "CJK_Radicals_Supplement" cjkRadicalsSupplement- , Block "Kangxi_Radicals" kangxiRadicals- , Block "Ideographic_Description_Characters" ideographicDescriptionCharacters- , Block "CJK_Symbols_and_Punctuation" cjkSymbolsAndPunctuation- , Block "Hiragana" hiragana- , Block "Katakana" katakana- , Block "Bopomofo" bopomofo- , Block "Hangul_Compatibility_Jamo" hangulCompatibilityJamo- , Block "Kanbun" kanbun- , Block "Bopomofo_Extended" bopomofoExtended- , Block "Katakana_Phonetic_Extensions" katakanaPhoneticExtensions- , Block "Enclosed_CJK_Letters_and_Months" enclosedCjkLettersAndMonths- , Block "CJK_Compatibility" cjkCompatibility- , Block "CJK_Unified_Ideographs_Extension_A" cjkUnifiedIdeographsExtensionA- , Block "Yijing_Hexagram_Symbols" yijingHexagramSymbols- , Block "CJK_Unified_Ideographs" cjkUnifiedIdeographs- , Block "Yi_Syllables" yiSyllables- , Block "Yi_Radicals" yiRadicals- , Block "Hangul_Syllables" hangulSyllables- , Block "High_Surrogates" highSurrogates- , Block "High_Private_Use_Surrogates" highPrivateUseSurrogates- , Block "Low_Surrogates" lowSurrogates- , Block "Private_Use_Area" privateUseArea- , Block "CJK_Compatibility_Ideographs" cjkCompatibilityIdeographs- , Block "Alphabetic_Presentation_Forms" alphabeticPresentationForms- , Block "Arabic_Presentation_Forms-A" arabicPresentationFormsA- , Block "Variation_Selectors" variationSelectors- , Block "Combining_Half_Marks" combiningHalfMarks- , Block "CJK_Compatibility_Forms" cjkCompatibilityForms- , Block "Small_Form_Variants" smallFormVariants- , Block "Arabic_Presentation_Forms-B" arabicPresentationFormsB- , Block "Halfwidth_and_Fullwidth_Forms" halfwidthAndFullwidthForms- , Block "Specials" specials ]--lookupTable :: HashMap String Block-lookupTable = HashMap.fromList $ - Prelude.map (\y@(Block x _) -> (canonicalize x, y))- blocks--canonicalize :: String -> String-canonicalize s = case Prelude.map toLower s of- 'i': 'n' : xs -> go xs- xs -> go xs- where- go ('-':xs) = go xs- go ('_':xs) = go xs- go (' ':xs) = go xs- go (x:xs) = x : go xs- go [] = []--lookupBlock :: String -> Maybe Block-lookupBlock s = HashMap.lookup (canonicalize s) lookupTable--lookupBlockCharSet :: String -> Maybe CharSet-lookupBlockCharSet = fmap blockCharSet . lookupBlock--basicLatin = range '\x0000' '\x007f'-latin1Supplement = range '\x0080' '\x00ff'-latinExtendedA = range '\x0100' '\x017F'-latinExtendedB = range '\x0180' '\x024F'-ipaExtensions = range '\x0250' '\x02AF'-spacingModifierLetters = range '\x02B0' '\x02FF'-combiningDiacriticalMarks = range '\x0300' '\x036F'-greekAndCoptic = range '\x0370' '\x03FF'-cyrillic = range '\x0400' '\x04FF'-cyrillicSupplementary = range '\x0500' '\x052F'-armenian = range '\x0530' '\x058F'-hebrew = range '\x0590' '\x05FF'-arabic = range '\x0600' '\x06FF'-syriac = range '\x0700' '\x074F'-thaana = range '\x0780' '\x07BF'-devanagari = range '\x0900' '\x097F'-bengali = range '\x0980' '\x09FF'-gurmukhi = range '\x0A00' '\x0A7F'-gujarati = range '\x0A80' '\x0AFF'-oriya = range '\x0B00' '\x0B7F'-tamil = range '\x0B80' '\x0BFF'-telugu = range '\x0C00' '\x0C7F'-kannada = range '\x0C80' '\x0CFF'-malayalam = range '\x0D00' '\x0D7F'-sinhala = range '\x0D80' '\x0DFF'-thai = range '\x0E00' '\x0E7F'-lao = range '\x0E80' '\x0EFF'-tibetan = range '\x0F00' '\x0FFF'-myanmar = range '\x1000' '\x109F'-georgian = range '\x10A0' '\x10FF'-hangulJamo = range '\x1100' '\x11FF'-ethiopic = range '\x1200' '\x137F'-cherokee = range '\x13A0' '\x13FF'-unifiedCanadianAboriginalSyllabics = range '\x1400' '\x167F'-ogham = range '\x1680' '\x169F'-runic = range '\x16A0' '\x16FF'-tagalog = range '\x1700' '\x171F'-hanunoo = range '\x1720' '\x173F'-buhid = range '\x1740' '\x175F'-tagbanwa = range '\x1760' '\x177F'-khmer = range '\x1780' '\x17FF'-mongolian = range '\x1800' '\x18AF'-limbu = range '\x1900' '\x194F'-taiLe = range '\x1950' '\x197F'-khmerSymbols = range '\x19E0' '\x19FF'-phoneticExtensions = range '\x1D00' '\x1D7F'-latinExtendedAdditional = range '\x1E00' '\x1EFF'-greekExtended = range '\x1F00' '\x1FFF'-generalPunctuation = range '\x2000' '\x206F'-superscriptsAndSubscripts = range '\x2070' '\x209F'-currencySymbols = range '\x20A0' '\x20CF'-combiningDiacriticalMarksForSymbols = range '\x20D0' '\x20FF'-letterlikeSymbols = range '\x2100' '\x214F'-numberForms = range '\x2150' '\x218F'-arrows = range '\x2190' '\x21FF'-mathematicalOperators = range '\x2200' '\x22FF'-miscellaneousTechnical = range '\x2300' '\x23FF'-controlPictures = range '\x2400' '\x243F'-opticalCharacterRecognition = range '\x2440' '\x245F'-enclosedAlphanumerics = range '\x2460' '\x24FF'-boxDrawing = range '\x2500' '\x257F'-blockElements = range '\x2580' '\x259F'-geometricShapes = range '\x25A0' '\x25FF'-miscellaneousSymbols = range '\x2600' '\x26FF'-dingbats = range '\x2700' '\x27BF'-miscellaneousMathematicalSymbolsA = range '\x27C0' '\x27EF'-supplementalArrowsA = range '\x27F0' '\x27FF'-braillePatterns = range '\x2800' '\x28FF'-supplementalArrowsB = range '\x2900' '\x297F'-miscellaneousMathematicalSymbolsB = range '\x2980' '\x29FF'-supplementalMathematicalOperators = range '\x2A00' '\x2AFF'-miscellaneousSymbolsAndArrows = range '\x2B00' '\x2BFF'-cjkRadicalsSupplement = range '\x2E80' '\x2EFF'-kangxiRadicals = range '\x2F00' '\x2FDF'-ideographicDescriptionCharacters = range '\x2FF0' '\x2FFF'-cjkSymbolsAndPunctuation = range '\x3000' '\x303F'-hiragana = range '\x3040' '\x309F'-katakana = range '\x30A0' '\x30FF'-bopomofo = range '\x3100' '\x312F'-hangulCompatibilityJamo = range '\x3130' '\x318F'-kanbun = range '\x3190' '\x319F'-bopomofoExtended = range '\x31A0' '\x31BF'-katakanaPhoneticExtensions = range '\x31F0' '\x31FF'-enclosedCjkLettersAndMonths = range '\x3200' '\x32FF'-cjkCompatibility = range '\x3300' '\x33FF'-cjkUnifiedIdeographsExtensionA = range '\x3400' '\x4DBF'-yijingHexagramSymbols = range '\x4DC0' '\x4DFF'-cjkUnifiedIdeographs = range '\x4E00' '\x9FFF'-yiSyllables = range '\xA000' '\xA48F'-yiRadicals = range '\xA490' '\xA4CF'-hangulSyllables = range '\xAC00' '\xD7AF'-highSurrogates = range '\xD800' '\xDB7F'-highPrivateUseSurrogates = range '\xDB80' '\xDBFF'-lowSurrogates = range '\xDC00' '\xDFFF'-privateUseArea = range '\xE000' '\xF8FF'-cjkCompatibilityIdeographs = range '\xF900' '\xFAFF'-alphabeticPresentationForms = range '\xFB00' '\xFB4F'-arabicPresentationFormsA = range '\xFB50' '\xFDFF'-variationSelectors = range '\xFE00' '\xFE0F'-combiningHalfMarks = range '\xFE20' '\xFE2F'-cjkCompatibilityForms = range '\xFE30' '\xFE4F'-smallFormVariants = range '\xFE50' '\xFE6F'-arabicPresentationFormsB = range '\xFE70' '\xFEFF'-halfwidthAndFullwidthForms = range '\xFF00' '\xFFEF'-specials = range '\xFFF0' '\xFFFF'
− Text/Trifecta/CharSet/Unicode/Category.hs
@@ -1,212 +0,0 @@-{-# LANGUAGE DeriveDataTypeable #-}--------------------------------------------------------------------------------- |--- Module : Text.Trifecta.CharSet.Unicode.Category--- Copyright : (c) Edward Kmett 2010-2011--- License : BSD3--- Maintainer : ekmett@gmail.com--- Stability : experimental--- Portability : portable------ Provides unicode general categories, which are typically connoted by --- @\p{Ll}@ or @\p{Modifier_Letter}@. Lookups can be constructed using 'categories'--- or individual character sets can be used directly.--- --- A case, @_@ and @-@ insensitive lookup is provided by 'lookupCategory'--- and can be used to provide behavior similar to that of Perl or PCRE.----------------------------------------------------------------------------------module Text.Trifecta.CharSet.Unicode.Category- ( - -- * Unicode General Category- Category(..)- -- * Lookup- , categories- , lookupCategory- , lookupCategoryCharSet- -- * CharSets by Category- -- ** Letter- , modifierLetter, otherLetter, letter- -- *** Letter\&- , lowercaseLetter, uppercaseLetter, titlecaseLetter, letterAnd- -- ** Mark- , nonSpacingMark, spacingCombiningMark, enclosingMark, mark- -- ** Separator- , space, lineSeparator, paragraphSeparator, separator- -- ** Symbol- , mathSymbol, currencySymbol, modifierSymbol, otherSymbol, symbol- -- ** Number- , decimalNumber, letterNumber, otherNumber, number- -- ** Punctuation- , dashPunctuation, openPunctuation, closePunctuation, initialQuote- , finalQuote, connectorPunctuation, otherPunctuation, punctuation - -- ** Other- , control, format, privateUse, surrogate, notAssigned, other- ) where--import Data.Char-import Text.Trifecta.CharSet-import Data.Data-import Data.HashMap.Lazy (HashMap)-import qualified Data.HashMap.Lazy as HashMap--data Category = Category - { categoryName :: String- , categoryAbbreviation :: String- , categoryCharSet :: CharSet- , categoryDescription :: String- } deriving (Show, Data, Typeable)---- \p{Letter} or \p{Mc}-categories :: [Category]-categories =- [ Category "Letter" "L" letter "any kind of letter from any language."- , Category "Lowercase_Letter" "Ll" lowercaseLetter "a lowercase letter that has an uppercase variant"- , Category "Uppercase_Letter" "Lu" uppercaseLetter "an uppercase letter that has a lowercase variant"- , Category "Titlecase_Letter" "Lt" titlecaseLetter "a letter that appears at the start of a word when only the first letter of the word is capitalized"- , Category "Letter&" "L&" letterAnd "a letter that exists in lowercase and uppercase variants (combination of Ll, Lu and Lt)"- , Category "Modifier_Letter" "Lm" modifierLetter "a special character that is used like a letter"- , Category "Other_Letter" "Lo" otherLetter "a letter or ideograph that does not have lowercase and uppercase variants"- , Category "Mark" "M" mark "a character intended to be combined with another character (e.g. accents, umlauts, enclosing boxes, etc.)"- , Category "Non_Spacing_Mark" "Mn" nonSpacingMark "a character intended to be combined with another character without taking up extra space (e.g. accents, umlauts, etc.)"- , Category "Spacing_Combining_Mark" "Mc" spacingCombiningMark "a character intended to be combined with another character that takes up extra space (vowel signs in many Eastern languages)"- , Category "Enclosing_Mark" "Me" enclosingMark "a character that encloses the character is is combined with (circle, square, keycap, etc.)"- , Category "Separator" "Z" separator "any kind of whitespace or invisible separator"- , Category "Space_Separator" "Zs" space "a whitespace character that is invisible, but does take up space"- , Category "Line_Separator" "Zl" lineSeparator "line separator character U+2028"- , Category "Paragraph_Separator" "Zp" paragraphSeparator "paragraph separator character U+2029"- , Category "Symbol" "S" symbol "math symbols, currency signs, dingbats, box-drawing characters, etc."- , Category "Math_Symbol" "Sm" mathSymbol "any mathematical symbol"- , Category "Currency_Symbol" "Sc" currencySymbol "any currency sign"- , Category "Modifier_Symbol" "Sk" modifierSymbol "a combining character (mark) as a full character on its own"- , Category "Other_Symbol" "So" otherSymbol "various symbols that are not math symbols, currency signs, or combining characters"- , Category "Number" "N" number "any kind of numeric character in any script"- , Category "Decimal_Digit_Number" "Nd" decimalNumber "a digit zero through nine in any script except ideographic scripts"- , Category "Letter_Number" "Nl" letterNumber "a number that looks like a letter, such as a Roman numeral"- , Category "Other_Number" "No" otherNumber "a superscript or subscript digit, or a number that is not a digit 0..9 (excluding numbers from ideographic scripts)"- , Category "Punctuation" "P" punctuation "any kind of punctuation character"- , Category "Dash_Punctuation" "Pd" dashPunctuation "any kind of hyphen or dash"- , Category "Open_Punctuation" "Ps" openPunctuation "any kind of opening bracket"- , Category "Close_Punctuation" "Pe" closePunctuation "any kind of closing bracket"- , Category "Initial_Punctuation" "Pi" initialQuote "any kind of opening quote"- , Category "Final_Punctuation" "Pf" finalQuote "any kind of closing quote"- , Category "Connector_Punctuation" "Pc" connectorPunctuation "a punctuation character such as an underscore that connects words"- , Category "Other_Punctuation" "Po" otherPunctuation "any kind of punctuation character that is not a dash, bracket, quote or connector"- , Category "Other" "C" other "invisible control characters and unused code points"- , Category "Control" "Cc" control "an ASCII 0x00..0x1F or Latin-1 0x80..0x9F control character"- , Category "Format" "Cf" format "invisible formatting indicator"- , Category "Private_Use" "Co" privateUse "any code point reserved for private use"- , Category "Surrogate" "Cs" surrogate "one half of a surrogate pair in UTF-16 encoding"- , Category "Unassigned" "Cn" notAssigned "any code point to which no character has been assigned.properties" ]--lookupTable :: HashMap String Category-lookupTable = HashMap.fromList - [ (canonicalize x, category) - | category@(Category l s _ _) <- categories- , x <- [l,s] - ]--lookupCategory :: String -> Maybe Category-lookupCategory s = HashMap.lookup (canonicalize s) lookupTable--lookupCategoryCharSet :: String -> Maybe CharSet-lookupCategoryCharSet = fmap categoryCharSet . lookupCategory--canonicalize :: String -> String-canonicalize s = case Prelude.map toLower s of- 'i' : 's' : xs -> go xs- xs -> go xs- where- go ('-':xs) = go xs- go ('_':xs) = go xs- go (' ':xs) = go xs- go (x:xs) = x : go xs- go [] = []--cat :: GeneralCategory -> CharSet-cat category = build ((category ==) . generalCategory)---- Letter-lowercaseLetter, uppercaseLetter, titlecaseLetter, letterAnd, modifierLetter, otherLetter, letter :: CharSet-lowercaseLetter = cat LowercaseLetter-uppercaseLetter = cat UppercaseLetter-titlecaseLetter = cat TitlecaseLetter-letterAnd = lowercaseLetter - `union` uppercaseLetter - `union` titlecaseLetter-modifierLetter = cat ModifierLetter-otherLetter = cat OtherLetter-letter - = letterAnd - `union` modifierLetter - `union` otherLetter---- Marks-nonSpacingMark, spacingCombiningMark, enclosingMark, mark :: CharSet-nonSpacingMark = cat NonSpacingMark-spacingCombiningMark = cat SpacingCombiningMark-enclosingMark = cat EnclosingMark-mark - = nonSpacingMark - `union` spacingCombiningMark - `union` enclosingMark--space, lineSeparator, paragraphSeparator, separator :: CharSet-space = cat Space-lineSeparator = cat LineSeparator-paragraphSeparator = cat ParagraphSeparator-separator - = space - `union` lineSeparator - `union` paragraphSeparator--mathSymbol, currencySymbol, modifierSymbol, otherSymbol, symbol :: CharSet-mathSymbol = cat MathSymbol-currencySymbol = cat CurrencySymbol-modifierSymbol = cat ModifierSymbol-otherSymbol = cat OtherSymbol-symbol - = mathSymbol - `union` currencySymbol - `union` modifierSymbol - `union` otherSymbol--decimalNumber, letterNumber, otherNumber, number :: CharSet-decimalNumber = cat DecimalNumber-letterNumber = cat LetterNumber-otherNumber = cat OtherNumber-number - = decimalNumber - `union` letterNumber - `union` otherNumber--dashPunctuation, openPunctuation, closePunctuation, initialQuote, - finalQuote, connectorPunctuation, otherPunctuation, punctuation :: CharSet--dashPunctuation = cat DashPunctuation-openPunctuation = cat OpenPunctuation-closePunctuation = cat ClosePunctuation-initialQuote = cat InitialQuote-finalQuote = cat FinalQuote-connectorPunctuation = cat ConnectorPunctuation-otherPunctuation = cat OtherPunctuation-punctuation - = dashPunctuation - `union` openPunctuation - `union` closePunctuation - `union` initialQuote - `union` finalQuote - `union` connectorPunctuation - `union` otherPunctuation--control, format, privateUse, surrogate, notAssigned, other :: CharSet-control = cat Control-format = cat Format-privateUse = cat PrivateUse-surrogate = cat Surrogate-notAssigned = cat NotAssigned-other = control - `union` format - `union` privateUse - `union` surrogate - `union` notAssigned
Text/Trifecta/Parser/ByteString.hs view
@@ -18,6 +18,8 @@ module Text.Trifecta.Parser.ByteString ( parseFromFile , parseFromFileEx+ , parseByteString+ , parseTest ) where import Control.Applicative@@ -33,8 +35,6 @@ import Text.Trifecta.Parser.Result import Data.Sequence as Seq import qualified Data.ByteString.UTF8 as UTF8-import Text.Trifecta.Rope.Prim-import qualified Data.FingerTree as F -- | @parseFromFile p filePath@ runs a parser @p@ on the@@ -68,13 +68,26 @@ -- > parseFromFileEx :: Show a => (forall r. Parser r String a) -> String -> IO (Result TermDoc a)-parseFromFileEx p fn = k <$> B.readFile fn where- k i = starve- $ feed (rope (F.fromList [LineDirective (UTF8.fromString fn) 0, strand i]))+parseFromFileEx p fn = parseByteString p (Directed (UTF8.fromString fn) 0 0 0 0) <$> B.readFile fn++-- | @parseByteString p delta i@ runs a parser @p@ on @i@.++parseByteString :: Show a => (forall r. Parser r String a) -> Delta -> UTF8.ByteString -> Result TermDoc a+parseByteString p d inp = starve+ $ feed inp $ stepParser (fmap prettyTerm) (why prettyTerm)- (release (Directed (UTF8.fromString fn) 0 0 0 0) *> p)+ (release d *> p) mempty True mempty mempty+++parseTest :: Show a => (forall r. Parser r String a) -> String -> IO ()+parseTest p s = case parseByteString p mempty (UTF8.fromString s) of+ Failure xs -> displayLn $ toList xs+ Success xs a -> do+ unless (Seq.null xs) $ displayLn $ toList xs+ print a+
Text/Trifecta/Parser/Char.hs view
@@ -40,9 +40,9 @@ import Text.Trifecta.Parser.Class import Text.Trifecta.Rope.Delta import qualified Data.IntSet as IntSet-import Text.Trifecta.CharSet (CharSet(..))-import qualified Text.Trifecta.CharSet as CharSet-import qualified Text.Trifecta.Util.ByteSet as ByteSet+import Data.CharSet (CharSet(..))+import qualified Data.CharSet as CharSet+import qualified Data.CharSet.ByteSet as ByteSet import qualified Data.ByteString as Strict import Data.ByteString.Internal (w2c,c2w) import Data.ByteString.UTF8 as UTF8
Text/Trifecta/Parser/Char8.hs view
@@ -49,8 +49,8 @@ import Control.Monad (guard) import Text.Trifecta.Parser.Class hiding (satisfy) import Text.Trifecta.Rope.Delta-import Text.Trifecta.Util.ByteSet (ByteSet(..))-import qualified Text.Trifecta.Util.ByteSet as ByteSet+import Data.CharSet.ByteSet (ByteSet(..))+import qualified Data.CharSet.ByteSet as ByteSet import qualified Data.ByteString as Strict import Data.ByteString.Internal (w2c,c2w) import qualified Data.ByteString.Char8 as Char8
Text/Trifecta/Parser/Class.hs view
@@ -32,7 +32,6 @@ import Control.Monad.Trans.RWS.Strict as Strict import Control.Monad.Trans.Reader import Control.Monad.Trans.Identity-import Data.Functor.Yoneda import Data.Word import Data.ByteString as Strict import Data.Char (isSpace)@@ -230,22 +229,6 @@ position = lift position slicedWith f (IdentityT m) = IdentityT $ slicedWith f m lookAhead (IdentityT m) = IdentityT $ lookAhead m--instance MonadParser m => MonadParser (Yoneda m) where- try = lift . try . lowerYoneda- labels m ss = lift $ labels (lowerYoneda m) ss- line = lift line- unexpected = lift . unexpected- satisfy = lift . satisfy- satisfy8 = lift . satisfy8- someSpace = lift someSpace- semi = lift semi- highlightInterval h s e = lift $ highlightInterval h s e- nesting (Yoneda m) = Yoneda $ \f -> nesting (m f)- skipping = lift . skipping- position = lift position- slicedWith f = lift . slicedWith f . lowerYoneda- lookAhead = lift . lookAhead . lowerYoneda -- | Skip zero or more bytes worth of white space. More complex parsers are -- free to consider comments as white space.
Text/Trifecta/Parser/Mark.hs view
@@ -23,7 +23,6 @@ import Control.Monad.Trans.RWS.Strict as Strict import Control.Monad.Trans.Reader import Control.Monad.Trans.Identity-import Data.Functor.Yoneda import Data.Monoid import Text.Trifecta.Rope.Delta import Text.Trifecta.Parser.Class@@ -66,6 +65,3 @@ mark = lift mark release = lift . release -instance MonadMark d m => MonadMark d (Yoneda m) where- mark = lift mark- release = lift . release
Text/Trifecta/Parser/Prim.hs view
@@ -14,7 +14,6 @@ ( Parser(..) , why , stepParser- , parseTest , manyAccum ) where @@ -55,7 +54,6 @@ import Text.Trifecta.Rope.Delta as Delta import Text.Trifecta.Rope.Prim import Text.Trifecta.Rope.Bytes-import System.Console.Terminfo.PrettyPrint data Parser r e a = Parser { unparser ::@@ -322,12 +320,3 @@ errLoc (PanicErr r _) = Just $ delta r errLoc (Err (Diagnostic (Left _) _ _ _)) = Nothing errLoc (Err (Diagnostic (Right r) _ _ _)) = Just $ delta r--parseTest :: Show a => (forall r. Parser r String a) -> String -> IO ()-parseTest p s = case starve- $ feed (UTF8.fromString s)- $ stepParser (fmap prettyTerm) (why prettyTerm) (release mempty *> p) mempty True mempty mempty of- Failure xs -> displayLn $ toList xs- Success xs a -> do- unless (Seq.null xs) $ displayLn $ toList xs- print a
− Text/Trifecta/Util/ByteSet.hs
@@ -1,64 +0,0 @@-{-# LANGUAGE BangPatterns, MagicHash #-}--------------------------------------------------------------------------------- |--- Module : Text.Trifecta.Util.ByteSet--- Copyright : Edward Kmett 2011--- Bryan O'Sullivan 2008--- License : BSD3--- --- Maintainer : ekmett@gmail.com--- Stability : experimental--- Portability : unknown------ Fast set membership tests for byte values, The set representation is --- unboxed for efficiency and uses a lookup table. This is a fairly minimal--- API. You probably want to use CharSet.-------------------------------------------------------------------------------module Text.Trifecta.Util.ByteSet- (- -- * Data type- ByteSet(..)- -- * Construction- , fromList- -- * Lookup- , member- ) where--import Data.Bits ((.&.), (.|.))-import Foreign.Storable (peekByteOff, pokeByteOff)-import GHC.Base (Int(I#), iShiftRA#, narrow8Word#, shiftL#)-import GHC.Word (Word8(W8#))-import qualified Data.ByteString as B-import qualified Data.ByteString.Internal as I-import qualified Data.ByteString.Unsafe as U--newtype ByteSet = ByteSet B.ByteString deriving (Eq, Ord, Show)--data I = I {-# UNPACK #-} !Int {-# UNPACK #-} !Word8--shiftR :: Int -> Int -> Int-shiftR (I# x#) (I# i#) = I# (x# `iShiftRA#` i#)--shiftL :: Word8 -> Int -> Word8-shiftL (W8# x#) (I# i#) = W8# (narrow8Word# (x# `shiftL#` i#))--index :: Int -> I-index i = I (i `shiftR` 3) (1 `shiftL` (i .&. 7))-{-# INLINE index #-}--fromList :: [Word8] -> ByteSet-fromList s0 = ByteSet $ I.unsafeCreate 32 $ \t -> do- _ <- I.memset t 0 32- let go [] = return ()- go (c:cs) = do- prev <- peekByteOff t byte :: IO Word8- pokeByteOff t byte (prev .|. bit)- go cs- where I byte bit = index (fromIntegral c)- go s0 ---- | Check the set for membership.-member :: Word8 -> ByteSet -> Bool-member w (ByteSet t) = U.unsafeIndex t byte .&. bit /= 0- where - I byte bit = index (fromIntegral w)
trifecta.cabal view
@@ -1,6 +1,6 @@ name: trifecta category: Text, Parsing, Diagnostics, Pretty Printer, Logging-version: 0.51.0.1+version: 0.52 license: BSD3 cabal-version: >= 1.6 license-file: LICENSE@@ -25,14 +25,6 @@ library exposed-modules: Text.Trifecta- Text.Trifecta.CharSet- Text.Trifecta.CharSet.Common- Text.Trifecta.CharSet.Posix- Text.Trifecta.CharSet.Posix.Ascii- Text.Trifecta.CharSet.Posix.Unicode- Text.Trifecta.CharSet.Unicode- Text.Trifecta.CharSet.Unicode.Block- Text.Trifecta.CharSet.Unicode.Category Text.Trifecta.IntervalMap Text.Trifecta.Rope Text.Trifecta.Rope.Bytes@@ -94,7 +86,6 @@ Text.Trifecta.Parser.Identifier Text.Trifecta.Parser.Identifier.Style Text.Trifecta.Util.Array- Text.Trifecta.Util.ByteSet other-modules: Text.Trifecta.Util.Combinators@@ -104,6 +95,7 @@ build-depends: base >= 4 && < 5, array >= 0.3.0.2 && < 0.5,+ charset >= 0.3.2 && < 0.4, containers >= 0.3 && < 0.6, unordered-containers >= 0.2.1 && < 0.3, blaze-builder >= 0.3.0.1 && < 0.4,@@ -123,7 +115,6 @@ semigroupoids >= 1.3.1.2 && < 1.4, pointed >= 2.1.0.1 && < 2.2, transformers >= 0.2 && < 0.4,- kan-extensions >= 2.4.0.1 && < 2.5, comonad >= 1.1.1.3 && < 1.2, terminfo >= 0.3.2 && < 0.4, keys >= 2.1.3.1 && < 2.2,