delimited-text (empty) → 0.0.1
raw patch · 6 files changed
+152/−0 lines, 6 filesdep +arraydep +attoparsecdep +basesetup-changed
Dependencies added: array, attoparsec, base, binary, bytestring, bytestring-show
Files
- LICENSE +24/−0
- Setup.lhs +4/−0
- Text/Delimited.hs +12/−0
- Text/Delimited/ByteString/Lazy.hs +63/−0
- Text/Delimited/Types.hs +20/−0
- delimited-text.cabal +29/−0
+ LICENSE view
@@ -0,0 +1,24 @@+Copyright (c) Stefan Kersten 2008++All rights reserved.++Redistribution and use in source and binary forms, with or without+modification, are permitted provided that the following conditions+are met:+1. Redistributions of source code must retain the above copyright+ notice, this list of conditions and the following disclaimer.+2. Redistributions in binary form must reproduce the above copyright+ notice, this list of conditions and the following disclaimer in the+ documentation and/or other materials provided with the distribution.++THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND+ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE+IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE+ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHORS OR CONTRIBUTORS BE LIABLE+FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL+DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS+OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)+HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT+LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY+OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF+SUCH DAMAGE.
+ Setup.lhs view
@@ -0,0 +1,4 @@+#!/usr/bin/env runhaskell++> import Distribution.Simple+> main = defaultMain
+ Text/Delimited.hs view
@@ -0,0 +1,12 @@+-- | Parse text files containing lines with records separated by character+-- delimiters.+--+-- At this time parsing is only supported for lazy 'ByteString's.+--+module Text.Delimited (+ module Text.Delimited.ByteString.Lazy,+ module Text.Delimited.Types+) where++import Text.Delimited.ByteString.Lazy+import Text.Delimited.Types
+ Text/Delimited/ByteString/Lazy.hs view
@@ -0,0 +1,63 @@+module Text.Delimited.ByteString.Lazy (+ encode, decode, interact+) where++import Data.Binary.Put (Put, putByteString, runPut)+import Data.ByteString.Lazy (ByteString, toChunks)+import qualified Data.ByteString as BS+import Prelude hiding (interact)+import Text.Show.ByteString (putAscii, showp, unlinesP)+import qualified Data.ParserCombinators.Attoparsec.Char8 as P+import Text.Delimited.Types++-- | Intersperse a list of 'Put's with a delimiter.+intersperseP :: Put -> [Put] -> Put+intersperseP _ [] = return ()+intersperseP _ (x:[]) = showp x+intersperseP d (x:xs) = x >> d >> intersperseP d xs++-- | Convert 'Content' delimited by 'delim' to a 'Put'.+putContent :: Char -> Content -> Put+putContent delim = unlinesP . map (intersperseP (putAscii delim) . map putByteString)++-- | Encode records separated by newlines to a ByteString.+-- Record fields are separated by 'delim'.+encode :: Char -> Content -> ByteString+encode delim = runPut . putContent delim++-- | Construct a strict ByteString from a lazy one.+fromLazy :: ByteString -> BS.ByteString+fromLazy = BS.concat . toChunks++-- | Construct a parser from 'a' terminated by 'b'.+endBy :: P.Parser a -> P.Parser b -> P.Parser a+a `endBy` b = do+ r <- a+ b+ return r++-- | A lazy 'ByteString' parser for delimited text.+parser :: [Char] -> P.Parser Content+parser delims = line `P.manyTill` P.eof+ where+ line = (field `P.sepBy` sep) `endBy` eol+ field = fromLazy `fmap` P.takeWhile (P.notInClass (delims ++ nls))+ sep = P.skipMany1 (P.satisfy $ P.inClass delims)+ eol = P.skipMany1 (P.satisfy $ P.inClass nls)+ nls = "\n\r"++-- | Parse records separated by newlines from a ByteString.+-- Record fields are separated by any of the characters in 'delims'. There is+-- no way of escaping delimiters, so record fields may not contain any of the+-- characters in 'delims'.+decode :: [Char] -> ByteString -> Result Content+decode delims = snd . P.parse (parser delims)++-- | Decode a ByteString, apply a function to each 'Record' and encode the content.+-- Delimiters may contain multiple characters but only the first is used for+-- encoding.+interact :: (Record -> Record) -> [Char] -> ByteString -> Result ByteString+interact f delims s =+ case decode delims s of+ Right c -> Right (encode (head delims) (map f c))+ Left e -> Left e
+ Text/Delimited/Types.hs view
@@ -0,0 +1,20 @@+module Text.Delimited.Types (+ Content, Record, Field,+ Result+) where++import Data.ByteString (ByteString)+import Data.ParserCombinators.Attoparsec.Char8 (ParseError)++-- | A delimited file is a series of variable length records.+type Content = [Record]++-- | A record is a series of fields.+-- Each record is located on a separate line, delimited by a line break (CRLF).+type Record = [Field]++-- | A field is a strict ByteString.+type Field = ByteString++-- | Result type.+type Result a = Either ParseError a
+ delimited-text.cabal view
@@ -0,0 +1,29 @@+name: delimited-text+version: 0.0.1+synopsis: Parse character delimited textual data+description: Parse character delimited textual data+license: BSD3+license-file: LICENSE+category: Data+copyright: Copyright (c) Stefan Kersten 2008+author: Stefan Kersten+maintainer: Stefan Kersten+stability: provisional+homepage: http://code.haskell.org/~StefanKersten/code/delimited-text+tested-with: GHC == 6.10.1+build-type: Simple+cabal-version: >= 1.2++library+ exposed-modules: Text.Delimited+ other-modules: Text.Delimited.ByteString.Lazy+ Text.Delimited.Types+ + build-depends: array,+ base >= 3,+ binary >= 0.4,+ bytestring,+ attoparsec >= 0.5,+ bytestring-show >= 0.2++ ghc-options: -O2 -funbox-strict-fields