csv-enumerator 0.10.1.1 → 0.10.2.0
raw patch · 2 files changed
+131/−87 lines, 2 filesPVP: major bump suggested
API removals or changes: PVP suggests a major version bump
API changes (from Hackage documentation)
+ Data.CSV.Enumerator: mapCSVFileM :: CSVeable r => FilePath -> CSVSettings -> (r -> IO [r]) -> FilePath -> IO (Either SomeException Int)
+ Data.CSV.Enumerator: mapCSVFileM_ :: CSVeable r => FilePath -> CSVSettings -> (r -> IO a) -> IO (Either SomeException Int)
- Data.CSV.Enumerator: CSVS :: !Char -> !Maybe Char -> !Maybe Char -> !Char -> CSVSettings
+ Data.CSV.Enumerator: CSVS :: !Char -> !(Maybe Char) -> !(Maybe Char) -> !Char -> CSVSettings
- Data.CSV.Enumerator: csvOutputQuoteChar :: CSVSettings -> !Maybe Char
+ Data.CSV.Enumerator: csvOutputQuoteChar :: CSVSettings -> !(Maybe Char)
- Data.CSV.Enumerator: csvQuoteChar :: CSVSettings -> !Maybe Char
+ Data.CSV.Enumerator: csvQuoteChar :: CSVSettings -> !(Maybe Char)
Files
- csv-enumerator.cabal +1/−1
- src/Data/CSV/Enumerator.hs +130/−86
csv-enumerator.cabal view
@@ -1,5 +1,5 @@ Name: csv-enumerator-Version: 0.10.1.1+Version: 0.10.2.0 Synopsis: A flexible, fast, enumerator-based CSV parser library for Haskell. Homepage: http://github.com/ozataman/csv-enumerator License: BSD3
src/Data/CSV/Enumerator.hs view
@@ -1,16 +1,18 @@-{-# LANGUAGE OverloadedStrings, BangPatterns #-}-{-# LANGUAGE PackageImports #-}-{-# LANGUAGE TypeSynonymInstances, FlexibleInstances #-}+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE FlexibleInstances #-}+{-# LANGUAGE OverloadedStrings #-}+{-# LANGUAGE PackageImports #-}+{-# LANGUAGE TypeSynonymInstances #-} -module Data.CSV.Enumerator - ( +module Data.CSV.Enumerator+ ( -- * CSV Data types Row -- Simply @[ByteString]@ , Field -- Simply @ByteString@- , MapRow + , MapRow , CSVeable(..)- , ParsedRow(..) + , ParsedRow(..) -- * CSV Setttings , CSVSettings(..)@@ -25,7 +27,7 @@ -- * Very Basic CSV Operations (for Debugging or Quick&Dirty Needs) , parseCSV , parseRow- + -- * Generic Folds Over CSV Files -- | These operations enable you to do whatever you want with CSV files; -- including interleaved IO, etc.@@ -36,6 +38,8 @@ -- * Mapping Over CSV Files , mapCSVFile+ , mapCSVFileM+ , mapCSVFileM_ , mapAccumCSVFile , mapIntoHandle @@ -53,30 +57,30 @@ where -import Control.Applicative hiding (many)-import Control.Exception (bracket, SomeException)-import Control.Monad (mzero, mplus, foldM, when, liftM)-import Control.Monad.IO.Class (liftIO, MonadIO)-import qualified Data.ByteString as B-import qualified Data.ByteString.Char8 as B8-import Data.ByteString.Char8 (ByteString)-import Data.ByteString.Internal (c2w)-import qualified Data.Map as M-import System.Directory-import System.IO-import System.PosixCompat.Files (getFileStatus, fileSize)+import Control.Applicative hiding (many)+import Control.Exception (SomeException, bracket)+import Control.Monad (foldM, liftM, mplus, mzero, when)+import Control.Monad.IO.Class (MonadIO, liftIO)+import qualified Data.ByteString as B+import Data.ByteString.Char8 (ByteString)+import qualified Data.ByteString.Char8 as B8+import Data.ByteString.Internal (c2w)+import qualified Data.Map as M+import System.Directory+import System.IO+import System.PosixCompat.Files (fileSize, getFileStatus) -import Data.Attoparsec as P hiding (take)-import qualified Data.Attoparsec.Char8 as C8-import Data.Attoparsec.Enumerator-import qualified Data.Enumerator as E-import Data.Enumerator (($$), yield, continue)-import Data.Enumerator.Binary (enumFile)-import Data.Word (Word8)-import Safe (headMay)+import Data.Attoparsec as P hiding (take)+import qualified Data.Attoparsec.Char8 as C8+import Data.Attoparsec.Enumerator+import Data.Enumerator (continue, yield, ($$))+import qualified Data.Enumerator as E+import Data.Enumerator.Binary (enumFile)+import Data.Word (Word8)+import Safe (headMay) -import Data.CSV.Enumerator.Types-import Data.CSV.Enumerator.Parser+import Data.CSV.Enumerator.Parser+import Data.CSV.Enumerator.Types class CSVeable r where@@ -95,7 +99,7 @@ -- | Iteratee to push rows into a given file- fileSink + fileSink :: CSVSettings -> FilePath -> (Maybe Handle, Int)@@ -110,38 +114,38 @@ -> CSVSettings -- ^ CSV Settings -> (r -> [r]) -- ^ A function to map a row onto rows -> FilePath -- ^ Output file- -> IO (Either SomeException Int) -- ^ Number of rows processed + -> IO (Either SomeException Int) -- ^ Number of rows processed ------------------------------------------------------------------------------ -- | 'Row' instance for 'CSVeable' instance CSVeable Row where- rowToStr s !r = - let - sep = B.pack [c2w (csvOutputColSep s)] + rowToStr s !r =+ let+ sep = B.pack [c2w (csvOutputColSep s)] wrapField !f = case (csvOutputQuoteChar s) of- Just !x -> x `B8.cons` escape x f `B8.snoc` x+ Just !x -> (x `B8.cons` escape x f) `B8.snoc` x otherwise -> f escape c str = B8.intercalate (B8.pack [c,c]) $ B8.split c str in B.intercalate sep . map wrapField $ r- + fileHeaders _ = Nothing iterCSV csvs f acc = loop acc where loop !acc' = do- eof <- E.isEOF + eof <- E.isEOF case eof of True -> f acc' EOF False -> comboIter acc' procRow acc' = rowParser csvs >>= f acc' . ParsedRow comboIter acc' = procRow acc' >>= loop- - fileSink csvs fo = iter ++ fileSink csvs fo = iter where- iter :: (Maybe Handle, Int) - -> ParsedRow Row + iter :: (Maybe Handle, Int)+ -> ParsedRow Row -> E.Iteratee B.ByteString IO (Maybe Handle, Int) iter acc@(oh, i) EOF = case oh of@@ -154,26 +158,26 @@ oh <- liftIO $ openFile fo WriteMode iter (Just oh, i) r - iter (Just oh, !i) (ParsedRow (Just r)) = do - outputRowIter csvs oh r + iter (Just oh, !i) (ParsedRow (Just r)) = do+ outputRowIter csvs oh r yield (Just oh, i+1) (E.Chunks []) mapCSVFiles fis s f fo = foldM stepFile (Right 0) fis where- stepFile :: (Either SomeException Int) - -> FilePath + stepFile :: (Either SomeException Int)+ -> FilePath -> IO (Either SomeException Int)- stepFile res0 fi = do + stepFile res0 fi = do case res0 of Left x -> return $ Left x- Right i -> do + Right i -> do res <- foldCSVFile fi s (iter fi) (Nothing, i) return $ fmap snd res iter :: FilePath- -> (Maybe Handle, Int) - -> ParsedRow Row + -> (Maybe Handle, Int)+ -> ParsedRow Row -> E.Iteratee B.ByteString IO (Maybe Handle, Int) iter fi acc@(oh, i) EOF = case oh of Just oh' -> liftIO (hClose oh') >> yield (Nothing, i) E.EOF@@ -183,8 +187,8 @@ let row' = f r oh <- liftIO $ openFile fo AppendMode iter fi (Just oh, i) (ParsedRow (Just r))- iter fi (Just oh, !i) (ParsedRow (Just r)) = do - outputRowsIter s oh (f r) + iter fi (Just oh, !i) (ParsedRow (Just r)) = do+ outputRowsIter s oh (f r) return (Just oh, i+1) @@ -199,30 +203,34 @@ iterCSV csvs f !acc = loop ([], acc) where loop (headers, !acc') = do- eof <- E.isEOF + eof <- E.isEOF case eof of True -> f acc' EOF False -> comboIter headers acc' - comboIter !headers !acc' = do - a <- procRow headers acc' + comboIter !headers !acc' = do+ a <- procRow headers acc' loop (headers, a) -- Fill headers if not yet filled- procRow [] !acc' = rowParser csvs >>= (\(Just hs) -> loop (hs, acc'))+ procRow [] !acc' = do+ r <- rowParser csvs+ case r of+ Nothing -> loop ([], acc')+ Just hs -> loop (hs, acc') -- Process starting w/ the second row- procRow !headers !acc' = rowParser csvs >>= - toMapCSV headers >>= - f acc' . ParsedRow + procRow !headers !acc' = rowParser csvs >>=+ toMapCSV headers >>=+ f acc' . ParsedRow toMapCSV !headers !fs = yield (fs >>= (Just . M.fromList . zip headers)) (E.Chunks []) fileSink s fo = mapIter where- mapIter :: (Maybe Handle, Int) - -> ParsedRow MapRow + mapIter :: (Maybe Handle, Int)+ -> ParsedRow MapRow -> E.Iteratee B.ByteString IO (Maybe Handle, Int) mapIter acc@(oh, !i) EOF = case oh of Just oh' -> liftIO (hClose oh') >> yield (Nothing, i) E.EOF@@ -235,24 +243,24 @@ return oh' mapIter (Just oh, i) (ParsedRow (Just r)) mapIter (Just oh, !i) (ParsedRow (Just (!r))) = do- outputRowIter s oh r + outputRowIter s oh r return (Just oh, i+1) mapCSVFiles fis s f fo = foldM stepFile (Right 0) fis where- stepFile res0 fi = do + stepFile res0 fi = do case res0 of Left x -> return $ Left x- Right i -> do + Right i -> do res <- foldCSVFile fi s (iter fi) (Nothing, i) return $ fmap snd res addFileSource fi r = M.insert "FromFile" (B8.pack fi) r iter :: FilePath- -> (Maybe Handle, Int) - -> ParsedRow MapRow + -> (Maybe Handle, Int)+ -> ParsedRow MapRow -> E.Iteratee B.ByteString IO (Maybe Handle, Int) iter fi acc@(oh, i) EOF = case oh of Just oh' -> liftIO (hClose oh') >> yield (Nothing, i) E.EOF@@ -270,19 +278,19 @@ False -> B8.hPutStrLn oh' . rowToStr s . M.keys . (addFileSource fi) $ x return oh' iter fi (Just oh, i) (ParsedRow (Just r))- iter fi (Just oh, !i) (ParsedRow (Just r)) = + iter fi (Just oh, !i) (ParsedRow (Just r)) = let rows = map (addFileSource fi) $ f r in do- outputRowsIter s oh rows + outputRowsIter s oh rows return (Just oh, i+1) --------------------------------------------------------------------------------- | Open & fold over the CSV file. +-- | Open & fold over the CSV file. -- -- Processing starts on row 2 for MapRow instance to use first row as column -- headers.-foldCSVFile +foldCSVFile :: (CSVeable r) => FilePath -- ^ File to open as a CSV file -> CSVSettings -- ^ CSV settings to use on the input file@@ -297,23 +305,59 @@ -- resulting rows into a new file. -- -- Each row is simply a list of fields.-mapCSVFile +mapCSVFile :: (CSVeable r) => FilePath -- ^ Input file -> CSVSettings -- ^ CSV Settings -> (r -> [r]) -- ^ A function to map a row onto rows -> FilePath -- ^ Output file- -> IO (Either SomeException Int) -- ^ Number of rows processed + -> IO (Either SomeException Int) -- ^ Number of rows processed mapCSVFile fi s f fo = do res <- foldCSVFile fi s iter (Nothing, 0) return $ snd `fmap` res where- iter !acc (ParsedRow (Just !r)) = foldM chain acc (f r) + iter !acc (ParsedRow (Just !r)) = foldM chain acc (f r) iter !acc x = fileSink s fo acc x chain !acc !r = fileSink s fo acc (ParsedRow (Just r)) ------------------------------------------------------------------------------+-- | Take a CSV file, apply an IO action to each of its rows and save the+-- resulting rows into a new file.+--+-- Each row is simply a list of fields.+mapCSVFileM+ :: (CSVeable r)+ => FilePath -- ^ Input file+ -> CSVSettings -- ^ CSV Settings+ -> (r -> IO [r]) -- ^ A function to map a row onto rows+ -> FilePath -- ^ Output file+ -> IO (Either SomeException Int) -- ^ Number of rows processed+mapCSVFileM fi s f fo = do+ res <- foldCSVFile fi s iter (Nothing, 0)+ return $ snd `fmap` res+ where+ iter !acc (ParsedRow (Just !r)) = foldM chain acc =<< liftIO (f r)+ iter !acc x = fileSink s fo acc x+ chain !acc !r = fileSink s fo acc (ParsedRow (Just r))++++------------------------------------------------------------------------------+-- | Take a CSV file, apply an IO action to each of its rows and discard the results.+--+mapCSVFileM_+ :: (CSVeable r)+ => FilePath -- ^ Input file+ -> CSVSettings -- ^ CSV Settings+ -> (r -> IO a) -- ^ A function to process rows+ -> IO (Either SomeException Int) -- ^ Number of rows processed+mapCSVFileM_ fi s f = foldCSVFile fi s iter 0+ where+ iter !acc (ParsedRow (Just !r)) = liftIO (f r) >> return (acc+1)+++------------------------------------------------------------------------------ -- | Map-accumulate over a CSV file. Similar to 'mapAccumL' in 'Data.List'. mapAccumCSVFile :: (CSVeable r)@@ -353,7 +397,7 @@ -> FilePath -- ^ Target file path -> [r] -- ^ Data to be output -> IO Int -- ^ Number of rows written-writeCSVFile s fp rs = +writeCSVFile s fp rs = let doOutput h = writeHeaders s h rs >> outputRowsIter h outputRowsIter h = foldM (step h) 0 . map (rowToStr s) $ rs step h acc x = (B8.hPutStrLn h x) >> return (acc+1)@@ -367,13 +411,13 @@ -> FilePath -- ^ Target file path -> [r] -- ^ Data to be output -> IO Int -- ^ Number of rows written-appendCSVFile s fp rs = +appendCSVFile s fp rs = let doOutput (c,h) = when c (writeHeaders s h rs >> return ()) >> outputRowsIter h outputRowsIter h = foldM (step h) 0 . map (rowToStr s) $ rs step h acc x = (B8.hPutStrLn h x) >> return (acc+1) chkOpen = do wrHeader <- do- fe <- doesFileExist fp + fe <- doesFileExist fp if fe then do fs <- getFileStatus fp >>= return . fileSize@@ -400,7 +444,7 @@ -- columns and then write the row into the given 'Handle'. -- -- This is helpful in filtering the columns or perhaps combining a number of--- files that don't have the same columns. +-- files that don't have the same columns. -- -- Missing columns will be left empty. outputColumns :: CSVSettings -> Handle -> [ByteString] -> MapRow -> IO ()@@ -473,7 +517,7 @@ ------------------------------------------------------------------------------ -- | Create an iteratee that can map over a CSV stream and output results to--- a handle in an interleaved fashion. +-- a handle in an interleaved fashion. -- -- Example use: Let's map over a CSV file coming in through 'stdin' and push -- results to 'stdout'.@@ -487,7 +531,7 @@ -- > pv inputFile.csv | myApp > output.CSV -- -- And monitor the ongoing progress of processing.-mapIntoHandle +mapIntoHandle :: (CSVeable r) => CSVSettings -- ^ 'CSVSettings' -> Bool -- ^ Whether to write headers@@ -502,7 +546,7 @@ f' (False, i) r'@(ParsedRow (Just r)) = do rs <- f r headerDone <- if outh then writeHeaders csvs h rs else return True- if headerDone + if headerDone then f' (headerDone, 0) r' -- Headers are done, now process row else return (False, i+1) -- Problem in this row, move on to next f' (True, !i) (ParsedRow (Just r)) = do@@ -516,7 +560,7 @@ -- nature of this library. collectRows :: CSVeable r => CSVAction r [r] collectRows acc EOF = yield acc (E.Chunks [])-collectRows acc (ParsedRow (Just r)) = let a' = (r:acc) +collectRows acc (ParsedRow (Just r)) = let a' = (r:acc) in a' `seq` yield a' (E.Chunks []) collectRows acc (ParsedRow Nothing) = yield acc (E.Chunks []) @@ -524,14 +568,14 @@ ------------------------------------------------------------------------------ -- Parsers -rowParser - :: (Monad m, MonadIO m) +rowParser+ :: (Monad m, MonadIO m) => CSVSettings -> E.Iteratee B.ByteString m (Maybe Row)-rowParser csvs = E.catchError p handler - where +rowParser csvs = E.catchError p handler+ where p = iterParser $ row csvs handler e = do liftIO $ putStrLn ("Error in parsing: " ++ show e) yield Nothing (E.Chunks [])- +