packages feed

pdf-toolbox-core 0.0.4.1 → 0.1.1

raw patch · 46 files changed

+2887/−1698 lines, 46 filesdep +base16-bytestringdep +cipher-aesdep +cipher-rc4dep −errorsdep −transformersdep −zlib-bindingsdep ~attoparsecdep ~basedep ~bytestringPVP ok

version bump matches the API change (PVP)

Dependencies added: base16-bytestring, cipher-aes, cipher-rc4, crypto-api, cryptohash, hashable, hspec, pdf-toolbox-core, unordered-containers, vector

Dependencies removed: errors, transformers, zlib-bindings

Dependency ranges changed: attoparsec, base, bytestring

API changes (from Hackage documentation)

- Pdf.Toolbox.Core.Error: AnnotatedError :: String -> PdfError -> PdfError
- Pdf.Toolbox.Core.Error: IOError :: IOError -> PdfError
- Pdf.Toolbox.Core.Error: ParseError :: [String] -> String -> PdfError
- Pdf.Toolbox.Core.Error: UnexpectedError :: String -> PdfError
- Pdf.Toolbox.Core.Error: annotateError :: Monad m => String -> PdfE m a -> PdfE m a
- Pdf.Toolbox.Core.Error: annotatingError :: Monad m => PdfE m a -> String -> PdfE m a
- Pdf.Toolbox.Core.Error: data PdfError
- Pdf.Toolbox.Core.Error: instance GHC.Show.Show Pdf.Toolbox.Core.Error.PdfError
- Pdf.Toolbox.Core.Error: tryPdfIO :: MonadIO m => IO a -> PdfE m a
- Pdf.Toolbox.Core.Error: type PdfE m = ExceptT PdfError m
- Pdf.Toolbox.Core.IO: class Monad m => MonadIO (m :: * -> *)
- Pdf.Toolbox.Core.IO: data RIS
- Pdf.Toolbox.Core.IO: dropExactly :: MonadIO m => Int -> IS -> PdfE m ()
- Pdf.Toolbox.Core.IO: fromHandle :: Handle -> IO RIS
- Pdf.Toolbox.Core.IO: fromHandle' :: Handle -> Int -> IO RIS
- Pdf.Toolbox.Core.IO: inputStream :: MonadIO m => RIS -> PdfE m IS
- Pdf.Toolbox.Core.IO: liftIO :: MonadIO m => forall a. IO a -> m a
- Pdf.Toolbox.Core.IO: parse :: MonadIO m => Parser r -> IS -> PdfE m r
- Pdf.Toolbox.Core.IO: readExactly :: MonadIO m => Int -> IS -> PdfE m ByteString
- Pdf.Toolbox.Core.IO: seek :: MonadIO m => RIS -> Int64 -> PdfE m ()
- Pdf.Toolbox.Core.IO: size :: MonadIO m => RIS -> PdfE m Int64
- Pdf.Toolbox.Core.IO: takeBytes :: MonadIO m => Int64 -> IS -> PdfE m IS
- Pdf.Toolbox.Core.IO: tell :: MonadIO m => RIS -> PdfE m Int64
- Pdf.Toolbox.Core.IO: type IS = InputStream ByteString
- Pdf.Toolbox.Core.IO.RIS: RIS :: (IORef RIS') -> RIS
- Pdf.Toolbox.Core.IO.RIS: RIS' :: (Int64 -> IO (IO (Maybe ByteString))) -> IS -> IO Int64 -> Int64 -> RIS'
- Pdf.Toolbox.Core.IO.RIS: [risInputStream] :: RIS' -> IS
- Pdf.Toolbox.Core.IO.RIS: [risPos] :: RIS' -> IO Int64
- Pdf.Toolbox.Core.IO.RIS: [risSeek] :: RIS' -> Int64 -> IO (IO (Maybe ByteString))
- Pdf.Toolbox.Core.IO.RIS: [risSize] :: RIS' -> Int64
- Pdf.Toolbox.Core.IO.RIS: data RIS'
- Pdf.Toolbox.Core.IO.RIS: fromHandle :: Handle -> IO RIS
- Pdf.Toolbox.Core.IO.RIS: fromHandle' :: Handle -> Int -> IO RIS
- Pdf.Toolbox.Core.IO.RIS: inputStream :: RIS -> IO IS
- Pdf.Toolbox.Core.IO.RIS: newtype RIS
- Pdf.Toolbox.Core.IO.RIS: seek :: RIS -> Int64 -> IO ()
- Pdf.Toolbox.Core.IO.RIS: size :: RIS -> IO Int64
- Pdf.Toolbox.Core.IO.RIS: tell :: RIS -> IO Int64
- Pdf.Toolbox.Core.IO.RIS: type IS = InputStream ByteString
- Pdf.Toolbox.Core.Object.Builder: buildArray :: Array -> Builder
- Pdf.Toolbox.Core.Object.Builder: buildBoolean :: Boolean -> Builder
- Pdf.Toolbox.Core.Object.Builder: buildDict :: Dict -> Builder
- Pdf.Toolbox.Core.Object.Builder: buildIndirectObject :: Ref -> Object ByteString -> Builder
- Pdf.Toolbox.Core.Object.Builder: buildName :: Name -> Builder
- Pdf.Toolbox.Core.Object.Builder: buildNumber :: Number -> Builder
- Pdf.Toolbox.Core.Object.Builder: buildObject :: Object a -> Builder
- Pdf.Toolbox.Core.Object.Builder: buildRef :: Ref -> Builder
- Pdf.Toolbox.Core.Object.Builder: buildStr :: Str -> Builder
- Pdf.Toolbox.Core.Object.Builder: buildStream :: Stream ByteString -> Builder
- Pdf.Toolbox.Core.Object.Types: Array :: [Object ()] -> Array
- Pdf.Toolbox.Core.Object.Types: Boolean :: Bool -> Boolean
- Pdf.Toolbox.Core.Object.Types: Dict :: [(Name, Object ())] -> Dict
- Pdf.Toolbox.Core.Object.Types: Name :: ByteString -> Name
- Pdf.Toolbox.Core.Object.Types: NumInt :: Int -> Number
- Pdf.Toolbox.Core.Object.Types: NumReal :: Double -> Number
- Pdf.Toolbox.Core.Object.Types: OArray :: Array -> Object a
- Pdf.Toolbox.Core.Object.Types: OBoolean :: Boolean -> Object a
- Pdf.Toolbox.Core.Object.Types: ODict :: Dict -> Object a
- Pdf.Toolbox.Core.Object.Types: OName :: Name -> Object a
- Pdf.Toolbox.Core.Object.Types: ONull :: Object a
- Pdf.Toolbox.Core.Object.Types: ONumber :: Number -> Object a
- Pdf.Toolbox.Core.Object.Types: ORef :: Ref -> Object a
- Pdf.Toolbox.Core.Object.Types: OStr :: Str -> Object a
- Pdf.Toolbox.Core.Object.Types: OStream :: (Stream a) -> Object a
- Pdf.Toolbox.Core.Object.Types: Ref :: Int -> Int -> Ref
- Pdf.Toolbox.Core.Object.Types: Str :: ByteString -> Str
- Pdf.Toolbox.Core.Object.Types: Stream :: Dict -> a -> Stream a
- Pdf.Toolbox.Core.Object.Types: data Number
- Pdf.Toolbox.Core.Object.Types: data Object a
- Pdf.Toolbox.Core.Object.Types: data Ref
- Pdf.Toolbox.Core.Object.Types: data Stream a
- Pdf.Toolbox.Core.Object.Types: instance Data.String.IsString Pdf.Toolbox.Core.Object.Types.Name
- Pdf.Toolbox.Core.Object.Types: instance Data.String.IsString Pdf.Toolbox.Core.Object.Types.Str
- Pdf.Toolbox.Core.Object.Types: instance GHC.Base.Monoid Pdf.Toolbox.Core.Object.Types.Name
- Pdf.Toolbox.Core.Object.Types: instance GHC.Classes.Eq Pdf.Toolbox.Core.Object.Types.Array
- Pdf.Toolbox.Core.Object.Types: instance GHC.Classes.Eq Pdf.Toolbox.Core.Object.Types.Boolean
- Pdf.Toolbox.Core.Object.Types: instance GHC.Classes.Eq Pdf.Toolbox.Core.Object.Types.Dict
- Pdf.Toolbox.Core.Object.Types: instance GHC.Classes.Eq Pdf.Toolbox.Core.Object.Types.Name
- Pdf.Toolbox.Core.Object.Types: instance GHC.Classes.Eq Pdf.Toolbox.Core.Object.Types.Number
- Pdf.Toolbox.Core.Object.Types: instance GHC.Classes.Eq Pdf.Toolbox.Core.Object.Types.Ref
- Pdf.Toolbox.Core.Object.Types: instance GHC.Classes.Eq Pdf.Toolbox.Core.Object.Types.Str
- Pdf.Toolbox.Core.Object.Types: instance GHC.Classes.Eq a => GHC.Classes.Eq (Pdf.Toolbox.Core.Object.Types.Object a)
- Pdf.Toolbox.Core.Object.Types: instance GHC.Classes.Eq a => GHC.Classes.Eq (Pdf.Toolbox.Core.Object.Types.Stream a)
- Pdf.Toolbox.Core.Object.Types: instance GHC.Classes.Ord Pdf.Toolbox.Core.Object.Types.Name
- Pdf.Toolbox.Core.Object.Types: instance GHC.Classes.Ord Pdf.Toolbox.Core.Object.Types.Ref
- Pdf.Toolbox.Core.Object.Types: instance GHC.Show.Show Pdf.Toolbox.Core.Object.Types.Array
- Pdf.Toolbox.Core.Object.Types: instance GHC.Show.Show Pdf.Toolbox.Core.Object.Types.Boolean
- Pdf.Toolbox.Core.Object.Types: instance GHC.Show.Show Pdf.Toolbox.Core.Object.Types.Dict
- Pdf.Toolbox.Core.Object.Types: instance GHC.Show.Show Pdf.Toolbox.Core.Object.Types.Name
- Pdf.Toolbox.Core.Object.Types: instance GHC.Show.Show Pdf.Toolbox.Core.Object.Types.Number
- Pdf.Toolbox.Core.Object.Types: instance GHC.Show.Show Pdf.Toolbox.Core.Object.Types.Ref
- Pdf.Toolbox.Core.Object.Types: instance GHC.Show.Show Pdf.Toolbox.Core.Object.Types.Str
- Pdf.Toolbox.Core.Object.Types: instance GHC.Show.Show a => GHC.Show.Show (Pdf.Toolbox.Core.Object.Types.Object a)
- Pdf.Toolbox.Core.Object.Types: instance GHC.Show.Show a => GHC.Show.Show (Pdf.Toolbox.Core.Object.Types.Stream a)
- Pdf.Toolbox.Core.Object.Types: newtype Array
- Pdf.Toolbox.Core.Object.Types: newtype Boolean
- Pdf.Toolbox.Core.Object.Types: newtype Dict
- Pdf.Toolbox.Core.Object.Types: newtype Name
- Pdf.Toolbox.Core.Object.Types: newtype Str
- Pdf.Toolbox.Core.Object.Util: class FromObject c
- Pdf.Toolbox.Core.Object.Util: deleteValueForKey :: Name -> Dict -> Dict
- Pdf.Toolbox.Core.Object.Util: fromObject :: (FromObject c, Show a, Monad m) => Object a -> PdfE m c
- Pdf.Toolbox.Core.Object.Util: instance Pdf.Toolbox.Core.Object.Util.FromObject Pdf.Toolbox.Core.Object.Types.Array
- Pdf.Toolbox.Core.Object.Util: instance Pdf.Toolbox.Core.Object.Util.FromObject Pdf.Toolbox.Core.Object.Types.Boolean
- Pdf.Toolbox.Core.Object.Util: instance Pdf.Toolbox.Core.Object.Util.FromObject Pdf.Toolbox.Core.Object.Types.Dict
- Pdf.Toolbox.Core.Object.Util: instance Pdf.Toolbox.Core.Object.Util.FromObject Pdf.Toolbox.Core.Object.Types.Name
- Pdf.Toolbox.Core.Object.Util: instance Pdf.Toolbox.Core.Object.Util.FromObject Pdf.Toolbox.Core.Object.Types.Number
- Pdf.Toolbox.Core.Object.Util: instance Pdf.Toolbox.Core.Object.Util.FromObject Pdf.Toolbox.Core.Object.Types.Ref
- Pdf.Toolbox.Core.Object.Util: instance Pdf.Toolbox.Core.Object.Util.FromObject Pdf.Toolbox.Core.Object.Types.Str
- Pdf.Toolbox.Core.Object.Util: intValue :: Monad m => Number -> PdfE m Int
- Pdf.Toolbox.Core.Object.Util: lookupDict :: Monad m => Name -> Dict -> PdfE m (Object ())
- Pdf.Toolbox.Core.Object.Util: lookupDict' :: Name -> Dict -> Maybe (Object ())
- Pdf.Toolbox.Core.Object.Util: mapObject :: (a -> b) -> Object a -> Object b
- Pdf.Toolbox.Core.Object.Util: realValue :: Monad m => Number -> PdfE m Double
- Pdf.Toolbox.Core.Object.Util: setValueForKey :: Name -> Object () -> Dict -> Dict
- Pdf.Toolbox.Core.Object.Util: toArray :: (Show a, Monad m) => Object a -> PdfE m Array
- Pdf.Toolbox.Core.Object.Util: toBoolean :: (Show a, Monad m) => Object a -> PdfE m Boolean
- Pdf.Toolbox.Core.Object.Util: toDict :: (Show a, Monad m) => Object a -> PdfE m Dict
- Pdf.Toolbox.Core.Object.Util: toName :: (Show a, Monad m) => Object a -> PdfE m Name
- Pdf.Toolbox.Core.Object.Util: toNumber :: (Show a, Monad m) => Object a -> PdfE m Number
- Pdf.Toolbox.Core.Object.Util: toRef :: (Show a, Monad m) => Object a -> PdfE m Ref
- Pdf.Toolbox.Core.Object.Util: toStr :: (Show a, Monad m) => Object a -> PdfE m Str
- Pdf.Toolbox.Core.Object.Util: toStream :: (Show a, Monad m) => Object a -> PdfE m (Stream a)
- Pdf.Toolbox.Core.Parsers.Object: isRegularChar :: Char -> Bool
- Pdf.Toolbox.Core.Parsers.Object: parseArray :: Parser Array
- Pdf.Toolbox.Core.Parsers.Object: parseBoolean :: Parser Boolean
- Pdf.Toolbox.Core.Parsers.Object: parseDict :: Parser Dict
- Pdf.Toolbox.Core.Parsers.Object: parseHexStr :: Parser Str
- Pdf.Toolbox.Core.Parsers.Object: parseIndirectObject :: Parser (Ref, Object ())
- Pdf.Toolbox.Core.Parsers.Object: parseName :: Parser Name
- Pdf.Toolbox.Core.Parsers.Object: parseNumber :: Parser Number
- Pdf.Toolbox.Core.Parsers.Object: parseObject :: Parser (Object ())
- Pdf.Toolbox.Core.Parsers.Object: parseRef :: Parser Ref
- Pdf.Toolbox.Core.Parsers.Object: parseStr :: Parser Str
- Pdf.Toolbox.Core.Parsers.Object: parseTillStreamData :: Parser ()
- Pdf.Toolbox.Core.Parsers.Util: endOfLine :: Parser ()
- Pdf.Toolbox.Core.Parsers.XRef: parseSubsectionHeader :: Parser (Int, Int)
- Pdf.Toolbox.Core.Parsers.XRef: parseTableEntry :: Parser (Int64, Int, Bool)
- Pdf.Toolbox.Core.Parsers.XRef: parseTrailerAfterTable :: Parser Dict
- Pdf.Toolbox.Core.Parsers.XRef: startXRef :: Parser Int64
- Pdf.Toolbox.Core.Parsers.XRef: tableXRef :: Parser ()
- Pdf.Toolbox.Core.Stream: data StreamFilter
- Pdf.Toolbox.Core.Stream: decodeStream :: MonadIO m => [StreamFilter] -> (IS -> IO IS) -> Stream IS -> PdfE m (Stream IS)
- Pdf.Toolbox.Core.Stream: decodedStreamContent :: MonadIO m => RIS -> [StreamFilter] -> (IS -> IO IS) -> Int -> Stream Int64 -> PdfE m (Stream IS)
- Pdf.Toolbox.Core.Stream: knownFilters :: [StreamFilter]
- Pdf.Toolbox.Core.Stream: rawStreamContent :: MonadIO m => RIS -> Int -> Stream Int64 -> PdfE m (Stream IS)
- Pdf.Toolbox.Core.Stream: readStream :: MonadIO m => RIS -> PdfE m (Stream Int64)
- Pdf.Toolbox.Core.Stream.Filter.FlateDecode: flateDecode :: StreamFilter
- Pdf.Toolbox.Core.Stream.Filter.Type: DecodeException :: (SomeException) -> DecodeException
- Pdf.Toolbox.Core.Stream.Filter.Type: StreamFilter :: Name -> (Maybe Dict -> IS -> IO IS) -> StreamFilter
- Pdf.Toolbox.Core.Stream.Filter.Type: [filterDecode] :: StreamFilter -> Maybe Dict -> IS -> IO IS
- Pdf.Toolbox.Core.Stream.Filter.Type: [filterName] :: StreamFilter -> Name
- Pdf.Toolbox.Core.Stream.Filter.Type: data DecodeException
- Pdf.Toolbox.Core.Stream.Filter.Type: data StreamFilter
- Pdf.Toolbox.Core.Stream.Filter.Type: instance GHC.Exception.Exception Pdf.Toolbox.Core.Stream.Filter.Type.DecodeException
- Pdf.Toolbox.Core.Stream.Filter.Type: instance GHC.Show.Show Pdf.Toolbox.Core.Stream.Filter.Type.DecodeException
- Pdf.Toolbox.Core.Util: readCompressedObject :: MonadIO m => IS -> Int64 -> Int -> PdfE m (Object ())
- Pdf.Toolbox.Core.Util: readObjectAtOffset :: MonadIO m => RIS -> Int64 -> Int -> PdfE m (Object Int64)
- Pdf.Toolbox.Core.Writer: data PdfWriter m a
- Pdf.Toolbox.Core.Writer: deleteObject :: MonadIO m => Ref -> Int64 -> PdfWriter m ()
- Pdf.Toolbox.Core.Writer: instance Control.Monad.IO.Class.MonadIO m => Control.Monad.IO.Class.MonadIO (Pdf.Toolbox.Core.Writer.PdfWriter m)
- Pdf.Toolbox.Core.Writer: instance Control.Monad.Trans.Class.MonadTrans Pdf.Toolbox.Core.Writer.PdfWriter
- Pdf.Toolbox.Core.Writer: instance GHC.Base.Functor m => GHC.Base.Functor (Pdf.Toolbox.Core.Writer.PdfWriter m)
- Pdf.Toolbox.Core.Writer: instance GHC.Base.Monad m => GHC.Base.Applicative (Pdf.Toolbox.Core.Writer.PdfWriter m)
- Pdf.Toolbox.Core.Writer: instance GHC.Base.Monad m => GHC.Base.Monad (Pdf.Toolbox.Core.Writer.PdfWriter m)
- Pdf.Toolbox.Core.Writer: instance GHC.Classes.Eq Pdf.Toolbox.Core.Writer.Elem
- Pdf.Toolbox.Core.Writer: instance GHC.Classes.Ord Pdf.Toolbox.Core.Writer.Elem
- Pdf.Toolbox.Core.Writer: runPdfWriter :: MonadIO m => OutputStream ByteString -> PdfWriter m a -> m a
- Pdf.Toolbox.Core.Writer: writeObject :: MonadIO m => Ref -> Object ByteString -> PdfWriter m ()
- Pdf.Toolbox.Core.Writer: writePdfHeader :: MonadIO m => PdfWriter m ()
- Pdf.Toolbox.Core.Writer: writeXRefTable :: MonadIO m => Int64 -> Dict -> PdfWriter m ()
- Pdf.Toolbox.Core.XRef: StreamEntryCompressed :: Int -> Int -> StreamEntry
- Pdf.Toolbox.Core.XRef: StreamEntryFree :: Int -> Int -> StreamEntry
- Pdf.Toolbox.Core.XRef: StreamEntryUsed :: Int64 -> Int -> StreamEntry
- Pdf.Toolbox.Core.XRef: TableEntry :: Int64 -> Int -> Bool -> TableEntry
- Pdf.Toolbox.Core.XRef: XRefStream :: Int64 -> (Stream Int64) -> XRef
- Pdf.Toolbox.Core.XRef: XRefStreamEntry :: StreamEntry -> XRefEntry
- Pdf.Toolbox.Core.XRef: XRefTable :: Int64 -> XRef
- Pdf.Toolbox.Core.XRef: XRefTableEntry :: TableEntry -> XRefEntry
- Pdf.Toolbox.Core.XRef: [teGen] :: TableEntry -> Int
- Pdf.Toolbox.Core.XRef: [teIsFree] :: TableEntry -> Bool
- Pdf.Toolbox.Core.XRef: [teOffset] :: TableEntry -> Int64
- Pdf.Toolbox.Core.XRef: data StreamEntry
- Pdf.Toolbox.Core.XRef: data TableEntry
- Pdf.Toolbox.Core.XRef: data XRef
- Pdf.Toolbox.Core.XRef: data XRefEntry
- Pdf.Toolbox.Core.XRef: instance GHC.Show.Show Pdf.Toolbox.Core.XRef.StreamEntry
- Pdf.Toolbox.Core.XRef: instance GHC.Show.Show Pdf.Toolbox.Core.XRef.TableEntry
- Pdf.Toolbox.Core.XRef: instance GHC.Show.Show Pdf.Toolbox.Core.XRef.XRef
- Pdf.Toolbox.Core.XRef: instance GHC.Show.Show Pdf.Toolbox.Core.XRef.XRefEntry
- Pdf.Toolbox.Core.XRef: isTable :: MonadIO m => IS -> PdfE m Bool
- Pdf.Toolbox.Core.XRef: lastXRef :: MonadIO m => RIS -> PdfE m XRef
- Pdf.Toolbox.Core.XRef: lookupStreamEntry :: MonadIO m => Stream IS -> Ref -> PdfE m (Maybe StreamEntry)
- Pdf.Toolbox.Core.XRef: lookupTableEntry :: MonadIO m => RIS -> Ref -> PdfE m (Maybe TableEntry)
- Pdf.Toolbox.Core.XRef: prevXRef :: MonadIO m => RIS -> XRef -> PdfE m (Maybe XRef)
- Pdf.Toolbox.Core.XRef: trailer :: MonadIO m => RIS -> XRef -> PdfE m Dict
+ Pdf.Core: Array :: Array -> Object
+ Pdf.Core: Bool :: Bool -> Object
+ Pdf.Core: Decrypted :: EncryptionStatus
+ Pdf.Core: Dict :: Dict -> Object
+ Pdf.Core: Encrypted :: EncryptionStatus
+ Pdf.Core: Name :: Name -> Object
+ Pdf.Core: Null :: Object
+ Pdf.Core: Number :: Scientific -> Object
+ Pdf.Core: Plain :: EncryptionStatus
+ Pdf.Core: R :: Int -> Int -> Ref
+ Pdf.Core: Ref :: Ref -> Object
+ Pdf.Core: S :: Dict -> Int64 -> Stream
+ Pdf.Core: Stream :: Stream -> Object
+ Pdf.Core: String :: ByteString -> Object
+ Pdf.Core: data EncryptionStatus
+ Pdf.Core: data File
+ Pdf.Core: data Name
+ Pdf.Core: data Object
+ Pdf.Core: data Ref
+ Pdf.Core: data Stream
+ Pdf.Core: defaultUserPassword :: ByteString
+ Pdf.Core: encryptionStatus :: File -> IO EncryptionStatus
+ Pdf.Core: findObject :: File -> Ref -> IO Object
+ Pdf.Core: lastTrailer :: File -> IO Dict
+ Pdf.Core: setUserPassword :: File -> ByteString -> IO Bool
+ Pdf.Core: streamContent :: File -> Ref -> Stream -> IO (InputStream ByteString)
+ Pdf.Core: type Array = Vector Object
+ Pdf.Core: type Dict = HashMap Name Object
+ Pdf.Core: withPdfFile :: FilePath -> (File -> IO a) -> IO a
+ Pdf.Core.Encryption: DecryptStream :: DecryptorScope
+ Pdf.Core.Encryption: DecryptString :: DecryptorScope
+ Pdf.Core.Encryption: data DecryptorScope
+ Pdf.Core.Encryption: decryptObject :: Decryptor -> Ref -> Object -> IO Object
+ Pdf.Core.Encryption: defaultUserPassword :: ByteString
+ Pdf.Core.Encryption: instance GHC.Show.Show Pdf.Core.Encryption.Algorithm
+ Pdf.Core.Encryption: mkStandardDecryptor :: Dict -> Dict -> ByteString -> Either String (Maybe Decryptor)
+ Pdf.Core.Encryption: type Decryptor = Ref -> DecryptorScope -> InputStream ByteString -> IO (InputStream ByteString)
+ Pdf.Core.Exception: Corrupted :: String -> [String] -> Corrupted
+ Pdf.Core.Exception: Unexpected :: String -> [String] -> Unexpected
+ Pdf.Core.Exception: data Corrupted
+ Pdf.Core.Exception: data Unexpected
+ Pdf.Core.Exception: instance GHC.Exception.Type.Exception Pdf.Core.Exception.Corrupted
+ Pdf.Core.Exception: instance GHC.Exception.Type.Exception Pdf.Core.Exception.Unexpected
+ Pdf.Core.Exception: instance GHC.Show.Show Pdf.Core.Exception.Corrupted
+ Pdf.Core.Exception: instance GHC.Show.Show Pdf.Core.Exception.Unexpected
+ Pdf.Core.Exception: message :: String -> IO a -> IO a
+ Pdf.Core.Exception: sure :: Either String a -> IO a
+ Pdf.Core.File: Decrypted :: EncryptionStatus
+ Pdf.Core.File: Encrypted :: EncryptionStatus
+ Pdf.Core.File: File :: XRef -> Buffer -> [StreamFilter] -> IORef (Maybe Decryptor) -> File
+ Pdf.Core.File: NotFound :: String -> NotFound
+ Pdf.Core.File: Plain :: EncryptionStatus
+ Pdf.Core.File: [fileBuffer] :: File -> Buffer
+ Pdf.Core.File: [fileDecryptor] :: File -> IORef (Maybe Decryptor)
+ Pdf.Core.File: [fileFilters] :: File -> [StreamFilter]
+ Pdf.Core.File: [fileLastXRef] :: File -> XRef
+ Pdf.Core.File: data EncryptionStatus
+ Pdf.Core.File: data File
+ Pdf.Core.File: data NotFound
+ Pdf.Core.File: encryptionStatus :: File -> IO EncryptionStatus
+ Pdf.Core.File: findObject :: File -> Ref -> IO Object
+ Pdf.Core.File: fromBuffer :: [StreamFilter] -> Buffer -> IO File
+ Pdf.Core.File: fromBytes :: [StreamFilter] -> ByteString -> IO File
+ Pdf.Core.File: fromHandle :: [StreamFilter] -> Handle -> IO File
+ Pdf.Core.File: instance GHC.Classes.Eq Pdf.Core.File.EncryptionStatus
+ Pdf.Core.File: instance GHC.Enum.Enum Pdf.Core.File.EncryptionStatus
+ Pdf.Core.File: instance GHC.Exception.Type.Exception Pdf.Core.File.NotFound
+ Pdf.Core.File: instance GHC.Show.Show Pdf.Core.File.EncryptionStatus
+ Pdf.Core.File: instance GHC.Show.Show Pdf.Core.File.NotFound
+ Pdf.Core.File: lastTrailer :: File -> IO Dict
+ Pdf.Core.File: rawStreamContent :: File -> Ref -> Stream -> IO (InputStream ByteString)
+ Pdf.Core.File: setDecryptor :: File -> Decryptor -> IO ()
+ Pdf.Core.File: setUserPassword :: File -> ByteString -> IO Bool
+ Pdf.Core.File: streamContent :: File -> Ref -> Stream -> IO (InputStream ByteString)
+ Pdf.Core.File: withPdfFile :: FilePath -> (File -> IO a) -> IO a
+ Pdf.Core.IO.Buffer: Buffer :: IO (Maybe ByteString) -> IO Int64 -> (Int64 -> IO ()) -> (Int64 -> IO ()) -> IO Int64 -> Buffer
+ Pdf.Core.IO.Buffer: [back] :: Buffer -> Int64 -> IO ()
+ Pdf.Core.IO.Buffer: [read] :: Buffer -> IO (Maybe ByteString)
+ Pdf.Core.IO.Buffer: [seek] :: Buffer -> Int64 -> IO ()
+ Pdf.Core.IO.Buffer: [size] :: Buffer -> IO Int64
+ Pdf.Core.IO.Buffer: [tell] :: Buffer -> IO Int64
+ Pdf.Core.IO.Buffer: data Buffer
+ Pdf.Core.IO.Buffer: dropExactly :: Int -> InputStream ByteString -> IO ()
+ Pdf.Core.IO.Buffer: fromBytes :: ByteString -> IO Buffer
+ Pdf.Core.IO.Buffer: fromHandle :: Handle -> IO Buffer
+ Pdf.Core.IO.Buffer: toInputStream :: Buffer -> InputStream ByteString
+ Pdf.Core.Name: data Name
+ Pdf.Core.Name: instance Data.Hashable.Class.Hashable Pdf.Core.Name.Name
+ Pdf.Core.Name: instance Data.String.IsString Pdf.Core.Name.Name
+ Pdf.Core.Name: instance GHC.Base.Monoid Pdf.Core.Name.Name
+ Pdf.Core.Name: instance GHC.Base.Semigroup Pdf.Core.Name.Name
+ Pdf.Core.Name: instance GHC.Classes.Eq Pdf.Core.Name.Name
+ Pdf.Core.Name: instance GHC.Classes.Ord Pdf.Core.Name.Name
+ Pdf.Core.Name: instance GHC.Show.Show Pdf.Core.Name.Name
+ Pdf.Core.Name: make :: ByteString -> Either String Name
+ Pdf.Core.Name: toByteString :: Name -> ByteString
+ Pdf.Core.Object: Array :: Array -> Object
+ Pdf.Core.Object: Bool :: Bool -> Object
+ Pdf.Core.Object: Dict :: Dict -> Object
+ Pdf.Core.Object: Name :: Name -> Object
+ Pdf.Core.Object: Null :: Object
+ Pdf.Core.Object: Number :: Scientific -> Object
+ Pdf.Core.Object: R :: Int -> Int -> Ref
+ Pdf.Core.Object: Ref :: Ref -> Object
+ Pdf.Core.Object: S :: Dict -> Int64 -> Stream
+ Pdf.Core.Object: Stream :: Stream -> Object
+ Pdf.Core.Object: String :: ByteString -> Object
+ Pdf.Core.Object: data Name
+ Pdf.Core.Object: data Object
+ Pdf.Core.Object: data Ref
+ Pdf.Core.Object: data Stream
+ Pdf.Core.Object: instance Data.Hashable.Class.Hashable Pdf.Core.Object.Ref
+ Pdf.Core.Object: instance GHC.Classes.Eq Pdf.Core.Object.Object
+ Pdf.Core.Object: instance GHC.Classes.Eq Pdf.Core.Object.Ref
+ Pdf.Core.Object: instance GHC.Classes.Eq Pdf.Core.Object.Stream
+ Pdf.Core.Object: instance GHC.Classes.Ord Pdf.Core.Object.Ref
+ Pdf.Core.Object: instance GHC.Show.Show Pdf.Core.Object.Object
+ Pdf.Core.Object: instance GHC.Show.Show Pdf.Core.Object.Ref
+ Pdf.Core.Object: instance GHC.Show.Show Pdf.Core.Object.Stream
+ Pdf.Core.Object: type Array = Vector Object
+ Pdf.Core.Object: type Dict = HashMap Name Object
+ Pdf.Core.Object.Builder: buildArray :: Array -> Builder
+ Pdf.Core.Object.Builder: buildBool :: Bool -> Builder
+ Pdf.Core.Object.Builder: buildDict :: Dict -> Builder
+ Pdf.Core.Object.Builder: buildIndirectObject :: Ref -> Object -> Builder
+ Pdf.Core.Object.Builder: buildIndirectStream :: Ref -> Dict -> ByteString -> Builder
+ Pdf.Core.Object.Builder: buildName :: Name -> Builder
+ Pdf.Core.Object.Builder: buildNumber :: Scientific -> Builder
+ Pdf.Core.Object.Builder: buildObject :: Object -> Builder
+ Pdf.Core.Object.Builder: buildRef :: Ref -> Builder
+ Pdf.Core.Object.Builder: buildStream :: Dict -> ByteString -> Builder
+ Pdf.Core.Object.Builder: buildString :: ByteString -> Builder
+ Pdf.Core.Object.Util: arrayValue :: Object -> Maybe Array
+ Pdf.Core.Object.Util: boolValue :: Object -> Maybe Bool
+ Pdf.Core.Object.Util: dictValue :: Object -> Maybe Dict
+ Pdf.Core.Object.Util: int64Value :: Object -> Maybe Int64
+ Pdf.Core.Object.Util: intValue :: Object -> Maybe Int
+ Pdf.Core.Object.Util: nameValue :: Object -> Maybe Name
+ Pdf.Core.Object.Util: realValue :: Object -> Maybe Double
+ Pdf.Core.Object.Util: refValue :: Object -> Maybe Ref
+ Pdf.Core.Object.Util: streamValue :: Object -> Maybe Stream
+ Pdf.Core.Object.Util: stringValue :: Object -> Maybe ByteString
+ Pdf.Core.Parsers.Object: isRegularChar :: Char -> Bool
+ Pdf.Core.Parsers.Object: parseArray :: Parser Array
+ Pdf.Core.Parsers.Object: parseBool :: Parser Bool
+ Pdf.Core.Parsers.Object: parseDict :: Parser Dict
+ Pdf.Core.Parsers.Object: parseHexString :: Parser ByteString
+ Pdf.Core.Parsers.Object: parseIndirectObject :: Parser (Ref, Object)
+ Pdf.Core.Parsers.Object: parseName :: Parser Name
+ Pdf.Core.Parsers.Object: parseNumber :: Parser Scientific
+ Pdf.Core.Parsers.Object: parseObject :: Parser Object
+ Pdf.Core.Parsers.Object: parseRef :: Parser Ref
+ Pdf.Core.Parsers.Object: parseString :: Parser ByteString
+ Pdf.Core.Parsers.Object: parseTillStreamData :: Parser ()
+ Pdf.Core.Parsers.Util: endOfLine :: Parser ()
+ Pdf.Core.Parsers.XRef: parseSubsectionHeader :: Parser (Int, Int)
+ Pdf.Core.Parsers.XRef: parseTableEntry :: Parser (Int64, Int, Bool)
+ Pdf.Core.Parsers.XRef: parseTrailerAfterTable :: Parser Dict
+ Pdf.Core.Parsers.XRef: startXRef :: Parser Int64
+ Pdf.Core.Parsers.XRef: tableXRef :: Parser ()
+ Pdf.Core.Stream: data StreamFilter
+ Pdf.Core.Stream: decodeStream :: [StreamFilter] -> Stream -> InputStream ByteString -> IO (InputStream ByteString)
+ Pdf.Core.Stream: decodedStreamContent :: Buffer -> [StreamFilter] -> (InputStream ByteString -> IO (InputStream ByteString)) -> Int -> Stream -> IO (InputStream ByteString)
+ Pdf.Core.Stream: knownFilters :: [StreamFilter]
+ Pdf.Core.Stream: rawStreamContent :: Buffer -> Int -> Int64 -> IO (InputStream ByteString)
+ Pdf.Core.Stream: readStream :: InputStream ByteString -> Int64 -> IO Stream
+ Pdf.Core.Stream.Filter.FlateDecode: flateDecode :: Maybe StreamFilter
+ Pdf.Core.Stream.Filter.Type: StreamFilter :: Name -> (Maybe Dict -> InputStream ByteString -> IO (InputStream ByteString)) -> StreamFilter
+ Pdf.Core.Stream.Filter.Type: [filterDecode] :: StreamFilter -> Maybe Dict -> InputStream ByteString -> IO (InputStream ByteString)
+ Pdf.Core.Stream.Filter.Type: [filterName] :: StreamFilter -> Name
+ Pdf.Core.Stream.Filter.Type: data StreamFilter
+ Pdf.Core.Types: Rectangle :: a -> a -> a -> a -> Rectangle a
+ Pdf.Core.Types: data Rectangle a
+ Pdf.Core.Types: instance GHC.Show.Show a => GHC.Show.Show (Pdf.Core.Types.Rectangle a)
+ Pdf.Core.Types: rectangleFromArray :: Array -> Either String (Rectangle Double)
+ Pdf.Core.Util: notice :: Maybe a -> String -> Either String a
+ Pdf.Core.Util: readCompressedObject :: InputStream ByteString -> Int64 -> Int -> IO Object
+ Pdf.Core.Util: readObjectAtOffset :: Buffer -> Int64 -> IO (Ref, Object)
+ Pdf.Core.Writer: data Writer
+ Pdf.Core.Writer: deleteObject :: Writer -> Ref -> Int64 -> IO ()
+ Pdf.Core.Writer: instance GHC.Classes.Eq Pdf.Core.Writer.Elem
+ Pdf.Core.Writer: instance GHC.Classes.Ord Pdf.Core.Writer.Elem
+ Pdf.Core.Writer: makeWriter :: OutputStream ByteString -> IO Writer
+ Pdf.Core.Writer: writeHeader :: Writer -> IO ()
+ Pdf.Core.Writer: writeObject :: Writer -> Ref -> Object -> IO ()
+ Pdf.Core.Writer: writeStream :: Writer -> Ref -> Dict -> ByteString -> IO ()
+ Pdf.Core.Writer: writeXRefStream :: Writer -> Int64 -> Ref -> Dict -> IO ()
+ Pdf.Core.Writer: writeXRefTable :: Writer -> Int64 -> Dict -> IO ()
+ Pdf.Core.XRef: EntryCompressed :: Int -> Int -> Entry
+ Pdf.Core.XRef: EntryFree :: Int -> Int -> Entry
+ Pdf.Core.XRef: EntryUsed :: Int64 -> Int -> Entry
+ Pdf.Core.XRef: UnknownXRefStreamEntryType :: Int -> UnknownXRefStreamEntryType
+ Pdf.Core.XRef: XRefStream :: Int64 -> Stream -> XRef
+ Pdf.Core.XRef: XRefTable :: Int64 -> XRef
+ Pdf.Core.XRef: data Entry
+ Pdf.Core.XRef: data UnknownXRefStreamEntryType
+ Pdf.Core.XRef: data XRef
+ Pdf.Core.XRef: instance GHC.Classes.Eq Pdf.Core.XRef.Entry
+ Pdf.Core.XRef: instance GHC.Classes.Eq Pdf.Core.XRef.XRef
+ Pdf.Core.XRef: instance GHC.Exception.Type.Exception Pdf.Core.XRef.UnknownXRefStreamEntryType
+ Pdf.Core.XRef: instance GHC.Show.Show Pdf.Core.XRef.Entry
+ Pdf.Core.XRef: instance GHC.Show.Show Pdf.Core.XRef.UnknownXRefStreamEntryType
+ Pdf.Core.XRef: instance GHC.Show.Show Pdf.Core.XRef.XRef
+ Pdf.Core.XRef: isTable :: InputStream ByteString -> IO Bool
+ Pdf.Core.XRef: lastXRef :: Buffer -> IO XRef
+ Pdf.Core.XRef: lookupStreamEntry :: Dict -> InputStream ByteString -> Ref -> IO (Maybe Entry)
+ Pdf.Core.XRef: lookupTableEntry :: Buffer -> XRef -> Ref -> IO (Maybe Entry)
+ Pdf.Core.XRef: prevXRef :: Buffer -> XRef -> IO (Maybe XRef)
+ Pdf.Core.XRef: readXRef :: Buffer -> Int64 -> IO XRef
+ Pdf.Core.XRef: trailer :: Buffer -> XRef -> IO Dict

Files

changelog.md view
@@ -1,18 +1,10 @@-0.0.4.1--* fix compilation on ghc 7.4, 7.6 and 7.8--0.0.4.0--* switch to errors-2.0--0.0.3.3--* add upper bound to `errors` dependency-* support ghc-7.10.1+unreleased -0.0.3.2+0.1.1 +* rework API+* support ghc from 8.0 to 8.10 and drop older versions+* interpret unknown xref stream entry type as reference to null object * support 1- and 2-digit escapes sequence in literal string  0.0.3.0
compat/Prelude.hs view
@@ -5,24 +5,16 @@ (   module P, -#if MIN_VERSION_base(4,8,0)+#if MIN_VERSION_base(4,11,0) #else-  (<$>),-  Monoid(..),-  Applicative(..),+  Semigroup(..), #endif ) where -#if MIN_VERSION_base(4,6,0) import "base" Prelude as P-#else-import "base" Prelude as P hiding (catch)-#endif -#if MIN_VERSION_base(4,8,0)+#if MIN_VERSION_base(4,11,0) #else-import Data.Functor ((<$>))-import Data.Monoid(Monoid(..))-import Control.Applicative (Applicative(..))+import Data.Semigroup(Semigroup(..)) #endif
+ lib/Pdf/Core.hs view
@@ -0,0 +1,28 @@++-- | Low level API for parsing PDF file.+--+-- See "Pdf.Core.Writer" for basic API for writing new PDF file or+-- incrementally updating existing one.++module Pdf.Core+( File+, withPdfFile+, lastTrailer+, findObject+, Object(..)+, Name+, Dict+, Array+, Ref(..)+, Stream(..)+, streamContent+, EncryptionStatus(..)+, encryptionStatus+, setUserPassword+, defaultUserPassword+)+where++import Pdf.Core.Object+import Pdf.Core.File+import Pdf.Core.Encryption
+ lib/Pdf/Core/Encryption.hs view
@@ -0,0 +1,288 @@+{-# LANGUAGE OverloadedStrings #-}++-- | Basic support for encrypted PDF files++module Pdf.Core.Encryption+( Decryptor+, DecryptorScope(..)+, defaultUserPassword+, mkStandardDecryptor+, decryptObject+)+where++import Pdf.Core.Object+import Pdf.Core.Object.Util+import Pdf.Core.Util++import qualified Data.Traversable as Traversable+import Data.Bits (xor)+import Data.IORef+import Data.ByteString (ByteString)+import qualified Data.ByteString as BS+import qualified Data.ByteString.Lazy as BSL+import Data.ByteString.Builder+import qualified Data.Vector as Vector+import qualified Data.HashMap.Strict as HashMap+import Control.Monad+import System.IO.Streams (InputStream)+import qualified System.IO.Streams as Streams+import qualified Crypto.Cipher.RC4 as RC4+import qualified Crypto.Cipher.AES as AES+import qualified Crypto.Hash.MD5 as MD5+import qualified Crypto.Padding as Padding++-- | Encryption handler may specify different encryption keys for strings+-- and streams+data DecryptorScope+  = DecryptString+  | DecryptStream++-- | Decrypt input stream+type Decryptor+  =  Ref+  -> DecryptorScope+  -> InputStream ByteString+  -> IO (InputStream ByteString)++-- | Decrypt object with the decryptor+decryptObject :: Decryptor -> Ref -> Object -> IO Object+decryptObject decryptor ref (String str)+  = String <$> decryptStr decryptor ref str+decryptObject decryptor ref (Dict dict)+  = Dict <$> decryptDict decryptor ref dict+decryptObject decryptor ref (Array arr)+  = Array <$> decryptArray decryptor ref arr+decryptObject _ _ o = return o++decryptStr :: Decryptor -> Ref -> ByteString -> IO ByteString+decryptStr decryptor ref str = do+  is <- Streams.fromList [str]+  res <- decryptor ref DecryptString is >>= Streams.toList+  return $ BS.concat res++decryptDict :: Decryptor -> Ref -> Dict -> IO Dict+decryptDict decryptor ref vals = Traversable.forM vals $+  decryptObject decryptor ref++decryptArray :: Decryptor -> Ref -> Array -> IO Array+decryptArray decryptor ref vals = Vector.forM vals decr+  where+  decr = decryptObject decryptor ref++-- | The default user password+defaultUserPassword :: ByteString+defaultUserPassword = BS.pack [+  0x28, 0xBF, 0x4E, 0x5E, 0x4E, 0x75, 0x8A, 0x41, 0x64, 0x00, 0x4E,+  0x56, 0xFF, 0xFA, 0x01, 0x08, 0x2E, 0x2E, 0x00, 0xB6, 0xD0, 0x68,+  0x3E, 0x80, 0x2F, 0x0C, 0xA9, 0xFE, 0x64, 0x53, 0x69, 0x7A+  ]++-- | Standard decryptor, RC4+mkStandardDecryptor :: Dict+                    -- ^ document trailer+                    -> Dict+                    -- ^ encryption dictionary+                    -> ByteString+                    -- ^ user password (32 bytes exactly,+                    -- see 7.6.3.3 Encryption Key Algorithm)+                    -> Either String (Maybe Decryptor)+mkStandardDecryptor tr enc pass = do+  filterType <-+    case HashMap.lookup "Filter" enc of+      Just o -> nameValue o `notice` "Filter should be a name"+      _ -> Left "Filter missing"+  unless (filterType == "Standard") $+    Left ("Unsupported encryption handler: " ++ show filterType)++  v <-+    case HashMap.lookup "V" enc of+      Just n -> intValue n `notice` "V should be an integer"+      _ -> Left "V is missing"++  if v == 4+    then mk4+    else mk12 v++  where+  mk12 v = do+    n <-+      case v of+        1 -> Right 5+        2 -> do+          case HashMap.lookup "Length" enc of+            Just o -> fmap (`div` 8) (intValue o+                        `notice` "Length should be an integer")+            Nothing -> Left "Length is missing"+        _ -> Left ("Unsuported encryption handler version: " ++ show v)++    ekey <- mkKey tr enc pass n+    ok <- verifyKey tr enc ekey+    return $+      if not ok+        then Nothing+        else Just $ \ref _ is -> mkDecryptor V2 ekey n ref is++  mk4 = do+    cryptoFilters <-+      case HashMap.lookup "CF" enc of+        Nothing -> Left "CF is missing in crypt handler V4"+        Just o -> dictValue o `notice` "CF should be a dictionary"+    keysMap <- Traversable.forM cryptoFilters $ \obj -> do+      dict <- dictValue obj `notice` "Crypto filter should be a dictionary"+      n <-+        case HashMap.lookup "Length" dict of+          Nothing -> Left "Crypto filter without Length"+          Just o -> intValue o `notice` "Crypto filter length should be int"+      algName <-+        case HashMap.lookup "CFM" dict of+          Nothing -> Left "CFM is missing"+          Just o -> nameValue o `notice` "CFM should be a name"+      alg <-+        case algName of+          "V2" -> return V2+          "AESV2" -> return AESV2+          _ -> Left $ "Unknown crypto method: " ++ show algName+      ekey <- mkKey tr enc pass n+      return (ekey, n, alg)++    (stdCfKey, _, _) <- HashMap.lookup "StdCF" keysMap+      `notice` "StdCF is missing"+    ok <- verifyKey tr enc stdCfKey+    if not ok+      then return Nothing++      else do+        strFName <- (HashMap.lookup "StrF" enc >>= nameValue)+          `notice` "StrF is missing"+        (strFKey, strFN, strFAlg) <- HashMap.lookup strFName keysMap+          `notice` ("Crypto filter not found: " ++ show strFName)++        stmFName <- (HashMap.lookup "StmF" enc >>= nameValue)+          `notice` "StmF is missing"+        (stmFKey, stmFN, stmFAlg) <- HashMap.lookup stmFName keysMap+          `notice` ("Crypto filter not found: " ++ show stmFName)++        return $ Just $ \ref scope is ->+          case scope of+            DecryptString -> mkDecryptor strFAlg strFKey strFN ref is+            DecryptStream -> mkDecryptor stmFAlg stmFKey stmFN ref is++mkKey :: Dict -> Dict -> ByteString -> Int -> Either String ByteString+mkKey tr enc pass n = do+  oVal <- do+    o <- HashMap.lookup "O" enc `notice` "O is missing"+    stringValue o `notice` "o should be a string"++  pVal <- do+    o <- HashMap.lookup "P" enc `notice` "P is missing"+    i <- intValue o `notice` "P should be an integer"+    Right . BS.pack . BSL.unpack . toLazyByteString+          . word32LE . fromIntegral $ i++  idVal <- do+    ids <- (HashMap.lookup "ID" tr >>= arrayValue)+        `notice` "ID should be an array"+    case (Vector.toList ids) of+      [] -> Left "ID array is empty"+      (x:_) -> stringValue x+                  `notice` "The first element if ID should be a string"++  rVal <- (HashMap.lookup "R" enc >>= intValue)+      `notice` "R should be an integer"++  encMD <-+    case HashMap.lookup "EncryptMetadata" enc of+      Nothing -> return True+      Just o -> boolValue o `notice` "EncryptMetadata should be a bool"++  let ekey' = BS.take n $ MD5.hash $ BS.concat [pass, oVal, pVal, idVal, pad]+      pad =+        if rVal < 4 || encMD+          then BS.empty+          else BS.pack (replicate 4 255)+      ekey =+        if rVal < 3+           then ekey'+           else foldl (\bs _ -> BS.take n $ MD5.hash bs)+                      ekey'+                      [1 :: Int .. 50]++  return ekey++verifyKey :: Dict -> Dict -> ByteString -> Either String Bool+verifyKey tr enc ekey = do+  rVal <- (HashMap.lookup "R" enc >>= intValue)+      `notice` "R should be an integer"++  idVal <- do+    ids <- (HashMap.lookup "ID" tr >>= arrayValue)+        `notice` "ID should be an array"+    case (Vector.toList ids) of+      [] -> Left "ID array is empty"+      (x:_) -> stringValue x+                  `notice` "The first element if ID should be a string"++  uVal <- (HashMap.lookup "U" enc >>= stringValue)+      `notice` "U should be a string"++  return $+    case rVal of+      2 ->+        let uVal' = snd $ RC4.combine (RC4.initCtx ekey)+                                      defaultUserPassword+        in uVal == uVal'+      _ ->+        let pass1 = snd $ RC4.combine (RC4.initCtx ekey)+                        $ BS.take 16 $ MD5.hash+                        $ BS.concat [defaultUserPassword, idVal]+            uVal' = loop 1 pass1+            loop 20 input = input+            loop i input = loop (i + 1) $ snd $ RC4.combine (RC4.initCtx+                                        $ BS.map (`xor` i) ekey) input+        in BS.take 16 uVal == BS.take 16 uVal'++data Algorithm+  = V2+  | AESV2+  deriving (Show)++mkDecryptor+  :: Algorithm+  -> ByteString+  -> Int+  -> Ref+  -> InputStream ByteString+  -> IO (InputStream ByteString)+mkDecryptor alg ekey n (R index gen) is = do+  let key = BS.take (16 `min` n + 5) $ MD5.hash $ BS.concat+        [ ekey+        , BS.pack $ take 3 $ BSL.unpack $ toLazyByteString+                  $ int32LE $ fromIntegral index+        , BS.pack $ take 2 $ BSL.unpack $ toLazyByteString+                  $ int32LE $ fromIntegral gen+        , salt alg+        ]+      salt V2 = ""+      salt AESV2 = "sAlT"++  case alg of+    V2 -> do+      ioRef <- newIORef $ RC4.initCtx key+      let readNext = do+            chunk <- Streams.read is+            case chunk of+              Nothing -> return Nothing+              Just c -> do+                ctx' <- readIORef ioRef+                let (ctx'', res) = RC4.combine ctx' c+                writeIORef ioRef ctx''+                return (Just res)+      Streams.makeInputStream readNext++    AESV2 -> do+      content <- BS.concat <$> Streams.toList is+      let initV = BS.take 16 content+          aes = AES.initAES key+          decrypted = AES.decryptCBC aes initV $ BS.drop 16 content+      Streams.fromByteString $ Padding.unpadPKCS5 decrypted
+ lib/Pdf/Core/Exception.hs view
@@ -0,0 +1,41 @@+{-# LANGUAGE DeriveDataTypeable #-}++-- | Exceptions and utilities+module Pdf.Core.Exception+(+  Corrupted(..),+  Unexpected(..),+  sure,+  message+)+where++import Data.Typeable+import Control.Exception hiding (throw)++-- | File is corrupted+--+-- Contains general message and a list of details+data Corrupted = Corrupted String [String]+  deriving (Show, Typeable)++instance Exception Corrupted where++-- | Something unexpected occurs, probably API missuse+data Unexpected = Unexpected String [String]+  deriving (Show, Typeable)++instance Exception Unexpected where++-- | We are sure it is 'Right'. Otherwise 'Corripted' is thrown+sure :: Either String a -> IO a+sure (Right a) = return a+sure (Left err) = throwIO (Corrupted err [])++-- | Catch 'Corrupted' and 'Unexpected'+-- and add a message to it before rethrowing+message :: String -> IO a -> IO a+message msg a = a `catches`+  [ Handler $ \(Corrupted err msgs) -> throwIO (Corrupted msg (err : msgs))+  , Handler $ \(Unexpected err msgs) -> throwIO (Unexpected msg (err : msgs))+  ]
+ lib/Pdf/Core/File.hs view
@@ -0,0 +1,243 @@+{-# LANGUAGE OverloadedStrings #-}++-- | Pdf file as a set of objects++module Pdf.Core.File+( File(..)+, withPdfFile+, fromHandle+, fromBytes+, fromBuffer+, lastTrailer+, findObject+, streamContent+, rawStreamContent+, EncryptionStatus(..)+, encryptionStatus+, setUserPassword+, setDecryptor+, NotFound(..)+)+where++import Data.ByteString (ByteString)+import qualified Data.ByteString as ByteString+import Data.IORef+import qualified Data.HashMap.Strict as HashMap+import Control.Monad+import Control.Exception (Exception, throwIO, catch)+import System.IO (Handle)+import qualified System.IO as IO+import System.IO.Streams (InputStream)++import Pdf.Core.Object+import Pdf.Core.Object.Util+import Pdf.Core.Exception+import Pdf.Core.XRef+import Pdf.Core.Stream (StreamFilter)+import Pdf.Core.Util+import qualified Pdf.Core.Stream as Stream+import Pdf.Core.IO.Buffer (Buffer)+import qualified Pdf.Core.IO.Buffer as Buffer+import Pdf.Core.Encryption++-- | Pdf file is a collection of 'Object's+data File = File+  { fileLastXRef :: XRef+  , fileBuffer :: Buffer+  , fileFilters :: [StreamFilter]+  , fileDecryptor :: IORef (Maybe Decryptor)+  }++-- | The last trailer is an entry point to PDF file. All other objects+-- usually are referensed from it, directly or indirectly.+lastTrailer :: File -> IO Dict+lastTrailer file = trailer (fileBuffer file) (fileLastXRef file)++-- | Get an object with the specified ref.+findObject :: File -> Ref -> IO Object+findObject file ref = do+  mentry <- fmap Just (lookupEntryRec file ref)+    `catch` \(UnknownXRefStreamEntryType _) -> return Nothing+  case mentry of+    Nothing -> return Null+    Just entry -> readObjectForEntry file entry++-- | Get content of the stream+--+-- It's decrypted and decoded using registered 'StreamFilter's if necessary.+streamContent :: File -> Ref -> Stream -> IO (InputStream ByteString)+streamContent file ref s = do+  is <- rawStreamContent file ref s+  Stream.decodeStream (fileFilters file) s is++-- | Get content of the stream+--+-- Content would be decrypted if necessary.+rawStreamContent :: File -> Ref -> Stream -> IO (InputStream ByteString)+rawStreamContent file ref (S dict pos) = do+  len <- do+    obj <- sure $ HashMap.lookup "Length" dict+      `notice` "Length missing in stream"+    case obj of+      Number _ -> sure $ intValue obj+        `notice` "Length should be an integer"+      Ref r -> do+        o <- findObject file r+        sure $ intValue o `notice` "Length should be an integer"+      _ -> throwIO $ Corrupted "Length should be an integer" []+  is <- Stream.rawStreamContent (fileBuffer file) len pos+  mdecryptor <- readIORef (fileDecryptor file)+  case mdecryptor of+    Nothing -> return is+    Just decryptor -> decryptor ref DecryptStream is++-- | Describes wether PDF file is encrypted, plain or already decrypted+data EncryptionStatus+  = Encrypted  -- ^ requires decryption+  | Decrypted  -- ^ already decrypted+  | Plain      -- ^ doesn't require decryption+  deriving (Show, Eq, Enum)++-- | Get encryption status.+--+-- If it's 'Encrypted', you may want to 'setUserPassword' to decrypt it.+encryptionStatus :: File -> IO EncryptionStatus+encryptionStatus file = do+  tr <- lastTrailer file+  case HashMap.lookup "Encrypt" tr of+    Nothing -> return Plain+    Just _ -> do+      decr <- readIORef (fileDecryptor file)+      case decr of+        Nothing -> return Encrypted+        Just _ -> return Decrypted++-- | Set user password to decrypt PDF file.+--+-- Use empty bytestring to set the default password.+-- Returns @True@ on success.+-- See also 'setDecryptor'.+setUserPassword :: File -> ByteString -> IO Bool+setUserPassword file password = message "setUserPassword" $ do+  tr <- lastTrailer file+  enc <-+    case HashMap.lookup "Encrypt" tr of+      Nothing -> throwIO (Unexpected "document is not encrypted" [])+      Just o -> do+        o' <- deref file o+        case o' of+          Dict d -> return d+          Null -> throwIO (Corrupted "encryption encryption dict is null" [])+          _ -> throwIO (Corrupted "document Encrypt should be a dictionary" [])+  let either_decryptor = mkStandardDecryptor tr enc+        (ByteString.take 32 $ password `mappend` defaultUserPassword)+  case either_decryptor of+    Left err -> throwIO $ Corrupted err []+    Right Nothing -> return False+    Right (Just decryptor) -> do+      setDecryptor file decryptor+      return True+  where+  deref f (Ref ref) = findObject f ref+  deref _ o = return o++-- | Decrypt file using the specified decryptor.+--+-- Use it if 'setUserPassword' doesn't work for you.+setDecryptor :: File -> Decryptor -> IO ()+setDecryptor file decryptor =+  writeIORef (fileDecryptor file) (Just decryptor)++-- | Create file from a buffer.+--+-- You may use 'Stream.knownFilters' as the first argument.+fromBuffer :: [StreamFilter] -> Buffer -> IO File+fromBuffer filters buffer = do+  xref <- lastXRef buffer+  decryptor <- newIORef Nothing+  return File+    { fileLastXRef = xref+    , fileBuffer = buffer+    , fileFilters = filters+    , fileDecryptor = decryptor+    }++-- | Create file from a binary handle.+--+-- You may use 'Stream.knownFilters' as the first argument.+fromHandle :: [StreamFilter] -> Handle -> IO File+fromHandle filters handle = do+  buffer <- Buffer.fromHandle handle+  fromBuffer filters buffer++-- | Create file from a ByteString.+--+-- You may use 'Stream.knownFilters' as the first argument.+fromBytes :: [StreamFilter] -> ByteString -> IO File+fromBytes filters bytes = do+  buffer <- Buffer.fromBytes bytes+  fromBuffer filters buffer++-- | Open Pdf file+--+-- You may want to check 'encryptionStatus' and 'setUserPassword' if+-- file is encrypted.+withPdfFile :: FilePath -> (File -> IO a) -> IO a+withPdfFile path action =+  IO.withBinaryFile path IO.ReadMode $ \handle -> do+    file <- fromHandle Stream.knownFilters handle+    action file++lookupEntryRec :: File -> Ref -> IO Entry+lookupEntryRec file ref = loop (fileLastXRef file)+  where+  loop xref = do+    res <- lookupEntry file ref xref+    case res of+      Just e -> return e+      Nothing -> do+        prev <- prevXRef (fileBuffer file) xref+        case prev of+          Just p -> loop p+          Nothing -> throwIO (NotFound $ "The Ref not found: " ++ show ref)++lookupEntry :: File -> Ref -> XRef -> IO (Maybe Entry)+lookupEntry file ref xref@(XRefTable _) =+  lookupTableEntry (fileBuffer file) xref ref+lookupEntry file ref (XRefStream _ s@(S dict _)) = do+  content <- streamContent file ref s+  lookupStreamEntry dict content ref++readObjectForEntry :: File -> Entry -> IO Object++readObjectForEntry _ EntryFree{} = return Null++readObjectForEntry file (EntryUsed off gen) = do+  (ref, obj) <- readObjectAtOffset (fileBuffer file) off+  let R _ gen' = ref+  unless (gen' == gen) $+    throwIO (Corrupted "readObjectForEntry" ["object generation missmatch"])+  decrypt file ref obj++readObjectForEntry file (EntryCompressed index num) = do+  let ref= R index 0+  objStream@(S dict _) <- do+    o <- findObject file ref+    sure $ streamValue o `notice` "Compressed entry should be in stream"+  first <- sure $ (HashMap.lookup "First" dict >>= intValue)+      `notice` "First should be an integer"+  content <- streamContent file ref objStream+  readCompressedObject content (fromIntegral first) num++decrypt :: File -> Ref -> Object -> IO Object+decrypt file ref o = do+  maybe_decr <- readIORef (fileDecryptor file)+  case maybe_decr of+    Nothing -> return o+    Just decr -> decryptObject decr ref o++data NotFound = NotFound String+  deriving (Show)++instance Exception NotFound
+ lib/Pdf/Core/IO/Buffer.hs view
@@ -0,0 +1,83 @@++-- | Buffer abstracts from file IO++module Pdf.Core.IO.Buffer+(+  Buffer(..),+  toInputStream,+  fromHandle,+  fromBytes,+  dropExactly+)+where++import Prelude hiding (read)+import Data.Int+import Data.IORef+import Data.ByteString (ByteString)+import qualified Data.ByteString as ByteString+import Control.Monad+import System.IO+import qualified System.IO.Streams as Streams+import System.IO.Streams.Internal (InputStream(..))+import qualified System.IO.Streams.Internal as Streams++-- | Interface to file+data Buffer = Buffer+  { read :: IO (Maybe ByteString)+  , size :: IO Int64+  , seek :: Int64 -> IO ()+  , back :: Int64 -> IO ()+  , tell :: IO Int64+  }++-- | Convert buffer to 'InputStream'+toInputStream :: Buffer -> InputStream ByteString+toInputStream buf = InputStream+  { Streams._read = read buf+  , Streams._unRead = back buf . fromIntegral . ByteString.length+  }++-- | Make buffer from handle+--+-- Don't touch the handle while using buffer+fromHandle :: Handle -> IO Buffer+-- it is in IO in case we'll need to store intermediate state+fromHandle h = return $ Buffer+  { read = do+      bs <- ByteString.hGetSome h defaultSize+      if ByteString.null bs+        then return Nothing+        else return (Just bs)+  , size = fromIntegral <$> hFileSize h+  , seek = hSeek h AbsoluteSeek . fromIntegral+  , back = hSeek h RelativeSeek . negate . fromIntegral+  , tell = fromIntegral <$> hTell h+  }++-- | Buffer from strict 'ByteString'+--+-- That is mostly for testing+fromBytes :: ByteString -> IO Buffer+fromBytes bs = do+  ref <- newIORef 0+  return Buffer+    { read = do+        pos <- readIORef ref+        let chunk = ByteString.drop pos bs+        modifyIORef ref (+ ByteString.length chunk)+        if ByteString.null chunk+          then return Nothing+          else return (Just chunk)+    , seek = writeIORef ref . fromIntegral+    , size = return $ fromIntegral (ByteString.length bs)+    , back = modifyIORef ref . flip (-) . fromIntegral+    , tell = fromIntegral <$> readIORef ref+    }++-- | Drop specified number of bytes from input stream+dropExactly :: Int -> InputStream ByteString -> IO ()+dropExactly n = void . Streams.readExactly n++defaultSize :: Int+defaultSize = 32752
+ lib/Pdf/Core/Name.hs view
@@ -0,0 +1,41 @@+{-# LANGUAGE GeneralizedNewtypeDeriving #-}++-- | Name is an atomic symbol+--+-- Note that `IsString` instance is a partial function++module Pdf.Core.Name+(+  Name,+  make,+  toByteString+)+where++import Data.String+import Data.ByteString (ByteString)+import qualified Data.ByteString as ByteString+import Data.Hashable (Hashable)++-- | Names usually are used as keys in dictionaries+--+-- Byte 0 is not allowed inside names+newtype Name = Name ByteString+  deriving (Eq, Show, Ord, Monoid, Semigroup, Hashable)++-- | Make a name.+--+-- Throws if the bytestring contains 0+make :: ByteString -> Either String Name+make bs+  | ByteString.any (== 0) bs+  = Left "Name.make: 0 byte is not allowed"+  | otherwise+  = Right (Name bs)++-- | Unwrap name to bytestring+toByteString :: Name -> ByteString+toByteString (Name bs) = bs++instance IsString Name where+  fromString = either error id . make . fromString
+ lib/Pdf/Core/Object.hs view
@@ -0,0 +1,54 @@+{-# LANGUAGE GeneralizedNewtypeDeriving #-}++-- | Module contains definitions of pdf objects+--+-- See PDF1.7:7.3++module Pdf.Core.Object+( Object(..)+, Dict+, Array+, Stream(..)+, Ref(..)+, Name+)+where++import Pdf.Core.Name (Name)++import Data.Int+import Data.ByteString (ByteString)+import Data.Scientific (Scientific)+import Data.Vector (Vector)+import Data.Hashable+import Data.HashMap.Strict as HashMap++-- | Dictionary+type Dict = HashMap Name Object++-- | An array+type Array = Vector Object++-- | Contains stream dictionary and an offset in file+data Stream = S Dict Int64+  deriving (Eq, Show)++-- | Object reference, contains object index and generation+data Ref = R Int Int+  deriving (Eq, Show, Ord)++instance Hashable Ref where+  hashWithSalt salt (R a b) = hashWithSalt salt (a, b)++-- | Any pdf object+data Object =+  Number Scientific |+  Bool Bool |+  Name Name |+  Dict Dict |+  Array Array |+  String ByteString |+  Stream Stream |+  Ref Ref |+  Null+  deriving (Eq, Show)
+ lib/Pdf/Core/Object/Builder.hs view
@@ -0,0 +1,156 @@+{-# LANGUAGE OverloadedStrings #-}++-- | Render 'Object' to bytestring++module Pdf.Core.Object.Builder+( buildIndirectObject+, buildIndirectStream+, buildObject+, buildNumber+, buildBool+, buildName+, buildDict+, buildArray+, buildString+, buildRef+, buildStream+)+where++import Data.Char+import Data.ByteString (ByteString)+import qualified Data.ByteString.Char8 as Char8+import qualified Data.ByteString.Lazy as BSL+import Data.ByteString.Builder+import qualified Data.ByteString.Base16 as Base16+import Data.Scientific (Scientific)+import qualified Data.Scientific as Scientific+import qualified Data.Vector as Vector+import qualified Data.HashMap.Strict as HashMap+import Text.Printf++import Pdf.Core.Object+import qualified Pdf.Core.Name as Name++-- | Build indirect object except streams+buildIndirectObject :: Ref -> Object -> Builder+buildIndirectObject ref object =+  buildObjectWith ref $+    buildObject object++-- | Build indirect stream+buildIndirectStream :: Ref -> Dict -> BSL.ByteString -> Builder+buildIndirectStream ref dict dat =+  buildObjectWith ref $+    buildStream dict dat++buildObjectWith :: Ref -> Builder -> Builder+buildObjectWith (R i g) inner =+  char7 '\n' `mappend`+  intDec i `mappend`+  char7 ' ' `mappend`+  intDec g `mappend`+  byteString " obj\n" `mappend`+  inner `mappend`+  byteString "\nendobj\n"++-- | Render inline object (without \"obj/endobj\").+-- It is 'error' to supply 'Stream', because it could not+-- be inlined, but should always be an indirect object+buildObject :: Object -> Builder+buildObject (Number n) = buildNumber n+buildObject (Bool b) = buildBool b+buildObject (Name n) = buildName n+buildObject (Dict d) = buildDict d+buildObject (Array a) = buildArray a+buildObject (String s) = buildString s+buildObject (Ref r) = buildRef r+buildObject (Stream _) = error "buildObject: please don't pass streams to me"+buildObject Null = byteString "null"++-- | Build a stream+--+-- The function doesn't try to encode or encrypt the content+buildStream :: Dict -> BSL.ByteString -> Builder+buildStream dict content = mconcat+  [ buildDict dict+  , byteString "stream\n"+  , lazyByteString content+  , byteString "\nendstream"+  ]++-- | Build a number+buildNumber :: Scientific -> Builder+buildNumber+  = either bFloat intDec+  . Scientific.floatingOrInteger+  where+  bFloat d = string7 $ printf "%f" (d :: Double)++-- | Build a bool+buildBool :: Bool -> Builder+buildBool True = byteString "true"+buildBool False = byteString "false"++-- | Build a name+buildName :: Name -> Builder+-- XXX: escaping+buildName n = char7 '/' `mappend` byteString (Name.toByteString n)++intercalate :: Builder -> [Builder] -> Builder+intercalate _ [] = mempty+intercalate sep (x:xs) = x `mappend` go xs+  where+  go [] = mempty+  go (y:ys) = sep `mappend` y `mappend` go ys++-- | Build a dictionary+buildDict :: Dict -> Builder+buildDict dict =+  byteString "<<" `mappend`+  intercalate (char7 ' ') (concatMap build $ HashMap.toList dict) `mappend`+  byteString ">>"+  where+  build (key, val) = [buildName key, buildObject val]++-- | Build an array+buildArray :: Array -> Builder+buildArray xs =+  char7 '[' `mappend`+  intercalate (char7 ' ') (map buildObject $ Vector.toList xs) `mappend`+  char7 ']'++-- | Build a string+--+-- It may produce literal or hex string based on the context.+buildString :: ByteString -> Builder+buildString s =+  if Char8.all isPrint s+    then mconcat+      [ char7 '('+      , byteString . Char8.pack . concatMap escape . Char8.unpack $ s+      , char7 ')'+      ]+    else mconcat+      [ char7 '<'+      , byteString $ Base16.encode s+      , char7 '>'+      ]+  where+  escape '(' = "\\("+  escape ')' = "\\)"+  escape '\\' = "\\\\"+  escape '\n' = "\\n"+  escape '\r' = "\\r"+  escape '\t' = "\\t"+  escape '\b' = "\\b"+  escape ch = [ch]++-- | Build a reference+buildRef :: Ref -> Builder+buildRef (R i j) = mconcat+  [ intDec i+  , char7 ' '+  , intDec j+  , byteString " R"+  ]
+ lib/Pdf/Core/Object/Util.hs view
@@ -0,0 +1,90 @@++-- | Utils relayted to pdf objects++module Pdf.Core.Object.Util+( intValue+, int64Value+, boolValue+, realValue+, nameValue+, stringValue+, arrayValue+, streamValue+, refValue+, dictValue+)+where++import Pdf.Core.Object++import Data.ByteString (ByteString)+import Data.Scientific (Scientific)+import qualified Data.Scientific as Scientific+import Data.Int (Int64)++-- | Try to convert object to 'Int'+--+-- Floating value doesn't automatically get converted+intValue :: Object -> Maybe Int+intValue (Number n)+  = either (const Nothing) Just+  . floatingOrInteger+  $ n+intValue _ = Nothing++-- | Specialized to prevent defaulting warning+floatingOrInteger :: Scientific -> Either Double Int+floatingOrInteger = Scientific.floatingOrInteger++-- | Try to convert object to 'Int64'.+--+-- This is for cases, where according to the specs values above 2^29+-- (Int) have to be expected.+int64Value :: Object -> Maybe Int64+int64Value (Number n) = Scientific.toBoundedInteger n+int64Value _ = Nothing++-- | Try to convert object to 'Bool'+boolValue :: Object -> Maybe Bool+boolValue (Bool b) = Just b+boolValue _ = Nothing++-- | Try to convert object to 'Double'+--+-- Integral value automatically gets converted+realValue :: Object -> Maybe Double+realValue (Number n)+  = either Just (Just . fromIntegral)+  . floatingOrInteger+  $ n+realValue _ = Nothing++-- | Try to convert object to 'Name'+nameValue :: Object -> Maybe Name+nameValue (Name n) = Just n+nameValue _ = Nothing++-- | Try to convert object to 'ByteString'+stringValue :: Object -> Maybe ByteString+stringValue (String s) = Just s+stringValue _ = Nothing++-- | Try to convert object to array+arrayValue :: Object -> Maybe Array+arrayValue (Array arr) = Just arr+arrayValue _ = Nothing++-- | Try to convert object to stream+streamValue :: Object -> Maybe Stream+streamValue (Stream s) = Just s+streamValue _ = Nothing++-- | Try to convert object to reference+refValue :: Object -> Maybe Ref+refValue (Ref ref) = Just ref+refValue _ = Nothing++-- | Try to convert object to dictionary+dictValue :: Object -> Maybe Dict+dictValue (Dict d) = Just d+dictValue _ = Nothing
+ lib/Pdf/Core/Parsers/Object.hs view
@@ -0,0 +1,229 @@+{-# LANGUAGE OverloadedStrings #-}++-- | This module contains parsers for pdf objects++module Pdf.Core.Parsers.Object+( -- * Parse any object+  parseObject+  -- * Parse object of specific type+, parseDict+, parseArray+, parseName+, parseString+, parseHexString+, parseRef+, parseNumber+, parseBool+  -- * Other+, parseTillStreamData+, parseIndirectObject+, isRegularChar+)+where++import Pdf.Core.Object+import qualified Pdf.Core.Name as Name+import Pdf.Core.Parsers.Util++import Data.Char+import Data.List+import Data.ByteString (ByteString)+import qualified Data.ByteString as BS+import qualified Data.ByteString.Char8 as BS8+import Data.Attoparsec.ByteString (Parser)+import qualified Data.Attoparsec.ByteString.Char8 as P+import Data.Scientific (Scientific)+import qualified Data.Scientific as Scientific+import qualified Data.Vector as Vector+import qualified Data.HashMap.Strict as HashMap+import Control.Applicative+import Control.Monad++-- | Parse a dictionary+parseDict :: Parser Dict+parseDict = do+  void $ P.string "<<"+  dict <- many parseKey+  P.skipSpace+  void $ P.string ">>"+  return $ HashMap.fromList dict++parseKey :: Parser (Name, Object)+parseKey = do+  P.skipSpace+  key <- parseName+  val <- parseObject+  return (key, val)++-- | Parse an array+parseArray :: Parser Array+parseArray = do+  void $ P.char '['+  array <- many parseObject+  P.skipSpace+  void $ P.char ']'+  return $ Vector.fromList array++-- | Parse number+parseNumber :: Parser Scientific+parseNumber = P.choice [+  P.scientific,+  Scientific.fromFloatDigits <$>+    (P.signed+      $ read+      . ("0."++)+      . BS8.unpack <$>+        (P.char '.' >> P.takeWhile1 isDigit) :: Parser Double)+  ]++-- | Parse literal string+parseString :: Parser ByteString+parseString = do+  void $ P.char '('+  str <- takeStr 0 []+  return $ BS8.pack str+  where+  takeStr :: Int -> String -> Parser String+  takeStr lvl res = do+    ch <- P.anyChar+    case ch of+      '(' -> takeStr (lvl + 1) (ch : res)+      ')' -> if lvl == 0+               then return $ reverse res+               else takeStr (lvl - 1) (ch : res)+      '\\' -> do+        ch' <- P.anyChar+        if ch' `elem` ("()\\" :: String)+          then takeStr lvl (ch' : res)+          else case ch' of+                 'r' -> takeStr lvl ('\r' : res)+                 'n' -> takeStr lvl ('\n' : res)+                 'f' -> takeStr lvl ('\f' : res)+                 'b' -> takeStr lvl ('\b' : res)+                 't' -> takeStr lvl ('\t' : res)+                 '\r' -> takeStr lvl res+                 _ -> do+                   ds <- take3Digits [ch']+                   let i = toEnum+                         . foldl'+                             (\acc (a, b) -> acc + a * charToInt b)+                             0+                         . zip [1, 8, 64]+                         $ ds+                   takeStr lvl (i : res)+      _ -> takeStr lvl (ch : res)+  charToInt ch = fromEnum ch - 48+  take3Digits ds+    | length ds >= 3+    = return ds+    | otherwise+    = do+      d <- P.peekChar'+      if isDigit d+        then do+          void P.anyChar+          take3Digits (d : ds)+        else+          return (ds ++ repeat '0')++-- | Parse hex string+parseHexString :: Parser ByteString+parseHexString = do+  void $ P.char '<'+  str <- many takeHex+  void $ P.char '>'+  return $ BS.pack str+  where+  takeHex = do+    ch1 <- P.satisfy isHexDigit+    ch2 <- P.satisfy isHexDigit+    return $ fromIntegral $ digitToInt ch1 * 16 + digitToInt ch2++-- | Parse a reference+parseRef :: Parser Ref+parseRef = do+  obj <- P.decimal+  P.skipSpace+  gen <- P.decimal+  P.skipSpace+  void $ P.char 'R'+  return $ R obj gen++-- | Parse a name+parseName :: Parser Name+parseName = do+  void $ P.char '/'+  -- XXX: escaping+  bs <- P.takeWhile1 isRegularChar+  either fail return $+    Name.make bs++-- | Whether the character can appear in 'Name'+isRegularChar :: Char -> Bool+isRegularChar = (`notElem` ("[]()/<>{}% \n\r" :: String))++-- | Parse bool value+parseBool :: Parser Bool+parseBool = P.choice [+  P.string "true" >> return True,+  P.string "false" >> return False+  ]++-- | Consumes input till stream's data+--+-- Use 'parseDict' then 'parseTillStreamData'+-- to determine whether the object is dictionary or stream.+-- If 'parseTillStreamData' fails, then it is a dictionary.+-- Otherwise it is stream, and current position in input data+-- will point to stream's data start+--+-- >>> parse (parseDict >>= \dict -> parseTillStreamData >> return dict) "<</Key 123>>\nstream\n1234\nendstream"+-- Done "1234\nendstream" Dict [(Name "Key",ONumber (NumInt 123))]+parseTillStreamData :: Parser ()+parseTillStreamData = do+  P.skipSpace+  void $ P.string "stream"+  endOfLine++-- | Parse any 'Object' except 'Stream'+-- because for 'Stream' we need offset of data in file+--+-- >>> parseOnly parseObject "/Name"+-- Right (OName (Name "Name"))+parseObject :: Parser Object+parseObject = do+  P.skipSpace+  P.choice [+    const Null <$> P.string "null",+    Name <$> parseName,+    Bool <$> parseBool,+    Dict <$> parseDict,+    Array <$> parseArray,+    String <$> parseString,+    String <$> parseHexString,+    Ref <$> parseRef,+    Number <$> parseNumber+    ]++-- | Parse object. Input position should point+-- to offset defined in XRef+--+-- >>> parseOnly parseIndirectObject "1 2 obj\n12"+-- Right (Ref 1 2,ONumber (NumInt 12))+parseIndirectObject :: Parser (Ref, Object)+parseIndirectObject = do+  P.skipSpace+  index <- P.decimal :: Parser Int+  P.skipSpace+  gen <- P.decimal :: Parser Int+  P.skipSpace+  void $ P.string "obj"+  P.skipSpace+  obj <- parseObject+  let ref = R index gen+  case obj of+    Dict d -> P.choice [+      parseTillStreamData >> return (ref, Stream (S d 0)),+      return (ref, Dict d)+      ]+    _ -> return (ref, obj)
+ lib/Pdf/Core/Parsers/Util.hs view
@@ -0,0 +1,23 @@++-- | Utils++module Pdf.Core.Parsers.Util+(+  endOfLine+)+where++import Data.Attoparsec.ByteString (Parser)+import qualified Data.Attoparsec.ByteString.Char8 as P+import Control.Applicative (many)++-- | In pdf file EOL could be \"\\n\", \"\\r\" or \"\\n\\r\"+--+-- Also space (0x20) is usually ok before EOL+endOfLine :: Parser ()+endOfLine = do+  _ <- many $ P.char ' '+  P.choice [+    P.endOfLine, -- it already handles both the \n and \n\r+    P.char '\r' >>= const (return ())+    ]
+ lib/Pdf/Core/Parsers/XRef.hs view
@@ -0,0 +1,96 @@+{-# LANGUAGE OverloadedStrings #-}++-- | Parsers for XRef++module Pdf.Core.Parsers.XRef+( startXRef+, tableXRef+, parseSubsectionHeader+, parseTrailerAfterTable+, parseTableEntry+)+where++import Pdf.Core.Object+import Pdf.Core.Parsers.Object+import Pdf.Core.Parsers.Util++import Data.Int+import Data.Attoparsec.ByteString (Parser)+import qualified Data.Attoparsec.ByteString.Char8 as P+import Control.Applicative (many)++-- for doctest+-- $setup+-- >>> :set -XOverloadedStrings+-- >>> import Data.Attoparsec.ByteString.Char8++-- | Offset of the very last xref table+--+-- Before calling it, make sure your are currently somewhere near+-- the end of pdf file. Otherwice it can eat all the memory.+-- E.g. examine only the last 1KB+--+-- >>> parseOnly startXRef "anything...startxref\n222\n%%EOF...blah\nstartxref\n123\n%%EOF"+-- Right 123+startXRef :: Parser Int64+startXRef = do+  res <- many $ do+    _ <- P.manyTill P.anyChar $ P.string "startxref"+    P.skipSpace+    offset <- P.decimal+    P.skipSpace+    _ <- P.string "%%EOF"+    return offset+  case res of+    [] -> fail "Trailer not found"+    xs -> return $ last xs++-- | When current input position points to xref stream+-- (or doesn't point to xref at all), the parser will fail.+-- When it points to xref table, the parser will succeed+-- and input position will point to the first xref subsection+--+-- >>> parseOnly tableXRef "xref\n"+-- Right ()+-- >>> parseOnly tableXRef "not xref"+-- Left "Failed reading: takeWith"+tableXRef :: Parser ()+tableXRef = do+  _ <- P.string "xref"+  endOfLine++-- | Parse subsection header, return (the first object index, number of object)+--+-- Input position will point to the first object+parseSubsectionHeader :: Parser (Int, Int)+parseSubsectionHeader = do+  start <- P.decimal+  P.skipSpace+  count <- P.decimal+  endOfLine+  return (start, count)++-- | Parse trailer located after XRef table+--+-- Input position should point to the \"trailer\" keyword+parseTrailerAfterTable :: Parser Dict+parseTrailerAfterTable = do+  P.skipSpace+  _ <- P.string "trailer"+  endOfLine+  P.skipSpace+  parseDict++-- | Parse XRef table entry. Returns offset, generation and whether the object is free.+parseTableEntry :: Parser (Int64, Int, Bool)+parseTableEntry = do+  offset <- P.decimal+  P.skipSpace+  generation <- P.decimal+  P.skipSpace+  c <- P.anyChar+  case c of+    'n' -> return (offset, generation, False)+    'f' -> return (offset, generation, True)+    _ -> fail $ "error parsing XRef table entry: unknown char: " ++ [c]
+ lib/Pdf/Core/Stream.hs view
@@ -0,0 +1,140 @@+{-# LANGUAGE  OverloadedStrings #-}+{-# LANGUAGE PatternGuards #-}++-- | Stream related tools++module Pdf.Core.Stream+(+  StreamFilter,+  knownFilters,+  readStream,+  rawStreamContent,+  decodedStreamContent,+  decodeStream+)+where++import Pdf.Core.Exception+import Pdf.Core.Object+import Pdf.Core.Parsers.Object+import Pdf.Core.Stream.Filter.Type+import Pdf.Core.Stream.Filter.FlateDecode+import Pdf.Core.IO.Buffer (Buffer)+import qualified Pdf.Core.IO.Buffer as Buffer++import Data.Int+import Data.Maybe+import Data.ByteString (ByteString)+import qualified Data.Vector as Vector+import qualified Data.HashMap.Strict as HashMap+import Control.Monad+import Control.Exception hiding (throw)+import System.IO.Streams (InputStream)+import qualified System.IO.Streams as Streams+import qualified System.IO.Streams.Attoparsec as Streams++-- | Read 'Stream' from stream+--+-- We need to pass current position here to calculate stream data offset+readStream :: InputStream ByteString -> Int64 -> IO Stream+readStream is off = do+  (is', counter) <- Streams.countInput is+  (_, obj) <- Streams.parseFromStream parseIndirectObject is'+    `catch` \(Streams.ParseException msg) -> throwIO (Corrupted msg [])+  case obj of+    Stream (S dict _) -> do+      off' <- counter+      return (S dict (off + off'))+    _ -> throwIO $ Streams.ParseException ("stream expected, but got: "+                                          ++ show obj)++-- | All stream filters implemented by the toolbox+--+-- Right now it contains only FlateDecode filter+knownFilters :: [StreamFilter]+knownFilters = catMaybes [flateDecode]++-- | Raw stream content.+-- Filters are not applyed+--+-- The 'InputStream' returned is valid only until the next 'bufferSeek'+--+-- Note: \"Length\" could be an indirect object, but we don't want+-- to read indirect objects here. So we require length to be provided+rawStreamContent :: Buffer+                 -> Int           -- ^ stream length+                 -> Int64         -- ^ stream offset+                                  -- The payload is offset of stream data+                 -> IO (InputStream ByteString)+rawStreamContent buf len off = do+  Buffer.seek buf off+  Streams.takeBytes (fromIntegral len) (Buffer.toInputStream buf)++-- | Decode stream content+--+-- It should be already decrypted+--+-- The 'InputStream' is valid only until the next 'bufferSeek'+decodeStream :: [StreamFilter]+             -> Stream -> InputStream ByteString+             -> IO (InputStream ByteString)+decodeStream filters (S dict _) istream =+  buildFilterList dict >>= foldM decode istream+  where+  decode is (name, params) = do+    f <- findFilter name+    filterDecode f params is+  findFilter name =+    case filter ((== name) . filterName) filters of+      [] -> throwIO $ Corrupted "Filter not found" []+      (f : _) -> return f++buildFilterList :: Dict -> IO [(Name, Maybe Dict)]+buildFilterList dict = do+  let f = fromMaybe Null $ HashMap.lookup "Filter" dict+      p = fromMaybe Null $ HashMap.lookup "DecodeParms" dict+  case (f, p) of+    (Null, _) -> return []+    (Name fd, Null) -> return [(fd, Nothing)]+    (Name fd, Dict pd) -> return [(fd, Just pd)]+    (Name fd, Array arr)+      | [Dict pd] <- Vector.toList arr+      -> return [(fd, Just pd)]+    (Array fa, Null) -> do+      fa' <- forM (Vector.toList fa) $ \o ->+        case o of+          Name n -> return n+          _ -> throwIO $ Corrupted ("Filter should be a Name") []+      return $ zip fa' (repeat Nothing)+    (Array fa, Array pa) | Vector.length fa == Vector.length pa -> do+      fa' <- forM (Vector.toList fa) $ \o ->+        case o of+          Name n -> return n+          _ -> throwIO $ Corrupted ("Filter should be a Name") []+      pa' <- forM (Vector.toList pa) $ \o ->+        case o of+          Dict d -> return d+          _ -> throwIO $ Corrupted ("DecodeParams should be a dictionary") []+      return $ zip fa' (map Just pa')+    _ -> throwIO $ Corrupted ("Can't handle Filter and DecodeParams: ("+                            ++ show f ++ ", " ++ show p ++ ")") []++-- | Decoded stream content+--+-- The 'InputStream' is valid only until the next 'bufferSeek'+--+-- Note: \"Length\" could be an indirect object, that is why+-- we cann't read it ourself+decodedStreamContent :: Buffer+                     -> [StreamFilter]+                     -> (InputStream ByteString -> IO (InputStream ByteString))+                     -- ^ decryptor+                     -> Int+                     -- ^ stream length+                     -> Stream+                     -- ^ stream with offset+                     -> IO (InputStream ByteString)+decodedStreamContent buf filters decryptor len s@(S _ off) =+  rawStreamContent buf len off >>=+  decryptor >>=+  decodeStream filters s
+ lib/Pdf/Core/Stream/Filter/Type.hs view
@@ -0,0 +1,23 @@++-- | Stream filter++module Pdf.Core.Stream.Filter.Type+(+  StreamFilter(..)+)+where++import Data.ByteString (ByteString)+import System.IO.Streams (InputStream)++import Pdf.Core.Object++-- | Stream filter+data StreamFilter = StreamFilter {+  -- | as \"Filter\" key value in stream dictionary+  filterName :: Name,+  -- | decode params -> content -> decoded content+  filterDecode :: Maybe Dict+               -> InputStream ByteString+               -> IO (InputStream ByteString)+  }
+ lib/Pdf/Core/Types.hs view
@@ -0,0 +1,28 @@++-- | Compound data structures from sec. 7.9 of PDF32000:2008++module Pdf.Core.Types+(+  Rectangle(..),+  rectangleFromArray+)+where++import Pdf.Core+import Pdf.Core.Util+import Pdf.Core.Object.Util++import qualified Data.Vector as Vector++-- | Rectangle+data Rectangle a = Rectangle a a a a+  deriving Show++-- | Create rectangle form an array of 4 numbers+rectangleFromArray :: Array -> Either String (Rectangle Double)+rectangleFromArray arr = do+  res <- mapM realValue (Vector.toList arr)+      `notice` "Rectangle should contain real values"+  case res of+    [a, b, c, d] -> return $ Rectangle a b c d+    _ -> Left ("rectangleFromArray: " ++ show arr)
+ lib/Pdf/Core/Util.hs view
@@ -0,0 +1,84 @@++-- | Unclassified tools++module Pdf.Core.Util+( notice+, readObjectAtOffset+, readCompressedObject+)+where++import Pdf.Core.IO.Buffer (Buffer)+import qualified Pdf.Core.IO.Buffer as Buffer+import Pdf.Core.Exception+import Pdf.Core.Object+import Pdf.Core.Parsers.Object++import Data.Int+import Data.ByteString (ByteString)+import Data.Attoparsec.ByteString.Char8 (Parser)+import qualified Data.Attoparsec.ByteString.Char8 as Parser+import Control.Monad+import Control.Exception hiding (throw)+import System.IO.Streams (InputStream)+import qualified System.IO.Streams as Streams+import qualified System.IO.Streams.Attoparsec as Streams++-- | Add a message to 'Maybe'+notice :: Maybe a -> String -> Either String a+notice Nothing = Left+notice (Just a) = const (Right a)++-- | Read indirect object at the specified offset+--+-- Returns the object and the 'Ref'. The payload for stream+-- will be an offset of stream content+readObjectAtOffset :: Buffer+                   -> Int64   -- ^ object offset+                   -> IO (Ref, Object)+readObjectAtOffset buf off = message "readObjectAtOffset" $ do+  Buffer.seek buf off+  (ref, o) <- Streams.parseFromStream parseIndirectObject+    (Buffer.toInputStream buf)+      `catch` \(Streams.ParseException msg) -> throwIO (Corrupted msg [])+  case o of+    Stream (S dict _) -> do+      pos <- Buffer.tell buf+      return (ref, Stream (S dict pos))+    Ref _ -> throwIO $ Corrupted "Indirect object can't be a Ref" []+    _ -> return (ref, o)++-- | Read object from object stream+--+-- Never returns 'Stream'+readCompressedObject :: InputStream ByteString+                     -- ^ decoded object stream+                     -> Int64+                     -- ^ an offset of the first object+                     -- (\"First\" key in dictionary)+                     -> Int+                     -- ^ object number to read+                     -> IO Object+readCompressedObject is first num = do+  (is', counter) <- Streams.countInput is+  off <- do+    res <- Streams.parseFromStream (replicateM (num + 1) headerP) is'+      `catch` \(Streams.ParseException msg) -> throwIO $ Corrupted+        "Object stream" [msg]+    when (null res) $+      error "readCompressedObject: imposible"+    case last res of+      (_, off) -> return off+  pos <- counter+  Buffer.dropExactly (fromIntegral $ first + off - pos) is+  Streams.parseFromStream parseObject is+    `catch` \(Streams.ParseException msg) -> throwIO $ Corrupted+      "Object in object stream" [msg]+  where+  headerP :: Parser (Int, Int64)+  headerP = do+    n <- Parser.decimal+    Parser.skipSpace+    off <- Parser.decimal+    Parser.skipSpace+    return (n, off)
+ lib/Pdf/Core/Writer.hs view
@@ -0,0 +1,262 @@+{-# LANGUAGE GeneralizedNewtypeDeriving #-}+{-# LANGUAGE OverloadedStrings #-}++-- | Write PDF files+--+-- It could be used to generate new PDF file+-- or to incrementally update the existent one+--+-- To generate new file, first call 'writeHeader',+-- then a number of 'writeObject' and finally 'writeXRefTable'+-- or `writeXRefStream`.+--+-- To incrementally update PDF file just omit the+-- `writeHeader` and append the result to the existent file.+-- Make sure to use `writeXRefTable` if the original file uses xref table,+-- or use `writeXRefStream` if it uses xref stream.++module Pdf.Core.Writer+( Writer+, makeWriter+, writeHeader+, writeObject+, writeStream+, deleteObject+, writeXRefTable+, writeXRefStream+)+where++import Pdf.Core.Object+import Pdf.Core.Object.Builder++import Data.IORef+import Data.Int+import qualified Data.Vector as Vector+import Data.Set (Set)+import qualified Data.Set as Set+import qualified Data.HashMap.Strict as HashMap+import Data.ByteString (ByteString)+import qualified Data.ByteString.Lazy as BSL+import Data.ByteString.Builder+import Data.Function+import Control.Monad+import System.IO.Streams (OutputStream)+import qualified System.IO.Streams as Streams++newtype Writer = Writer {toStateRef :: IORef State}++makeWriter :: OutputStream ByteString -> IO Writer+makeWriter output = do+  (out, count) <- Streams.countOutput output+  let emptyState = State {+        stOutput = out,+        stObjects = Set.empty,+        stCount = count,+        stOffset = 0+        }+  Writer <$> newIORef emptyState++data Elem = Elem {+  elemIndex :: {-# UNPACK #-} !Int,+  elemGen :: {-# UNPACK #-} !Int,+  elemOffset :: {-# UNPACK #-} !Int64,+  elemFree :: !Bool+  }++instance Eq Elem where+  (==) = (==) `on` elemIndex++instance Ord Elem where+  compare = compare `on` elemIndex++data State = State {+  stOutput :: OutputStream ByteString,+  stObjects :: !(Set Elem),+  stCount :: IO Int64,+  stOffset :: {-# UNPACK #-} !Int64+  }++-- | Write PDF header. Used for generating new PDF files.+-- Should be the first call. Not used fo incremental updates+writeHeader :: Writer -> IO ()+writeHeader writer = do+  st <- readIORef (toStateRef writer)+  Streams.write (Just "%PDF-1.7\n") (stOutput st)++-- | Write object+writeObject :: Writer -> Ref -> Object -> IO ()+writeObject writer ref@(R index gen) obj = do+  pos <- countWritten writer+  st <- readIORef (toStateRef writer)+  addElem writer $ Elem index gen pos False+  dumpObject (stOutput st) ref obj++-- | Write stream+writeStream :: Writer -> Ref -> Dict -> BSL.ByteString -> IO ()+writeStream writer ref@(R index gen) dict dat = do+  pos <- countWritten writer+  st <- readIORef (toStateRef writer)+  addElem writer $ Elem index gen pos False+  dumpStream (stOutput st) ref dict dat++-- | Delete object+deleteObject :: Writer -> Ref -> Int64 -> IO ()+deleteObject writer (R index gen) nextFree =+  addElem writer $ Elem index gen nextFree True++-- | Write xref table. Should be the last call.+-- Used for generating and incremental updates.+--+-- Note that when doing incremental update you should use this function+-- only if the original PDF file has xref table. If it has xref stream,+-- then use `writeXRefStream`.+writeXRefTable+  :: Writer+  -> Int64    -- ^ size of the original PDF file. Should be 0 for new file+  -> Dict     -- ^ trailer+  -> IO ()+writeXRefTable writer offset tr = do+  off <- (+ offset) <$> countWritten writer+  st <- readIORef (toStateRef writer)+  let elems = Set.mapMonotonic (\e -> e {elemOffset = elemOffset e + offset})+            $ stObjects st+      content = mconcat+        [ byteString "xref\n"+        , buildXRefTable (Set.toAscList elems)+        , byteString "trailer\n"+        , buildDict tr+        , byteString "\nstartxref\n"+        , int64Dec off+        , byteString "\n%%EOF\n"+        ]+  Streams.writeLazyByteString (toLazyByteString content) (stOutput st)++-- | Write xref stream. Should be the last call.+-- Used for generating and incremental updates.+--+-- Note that when doing incremental update you should use this function+-- only if the original PDF file has xref stream. If it has xref table,+-- then use `writeXRefTable`.+--+-- This function will update/delete the following keys in the trailer:+-- Type, W, Index, Filter, Length.+writeXRefStream+  :: Writer+  -> Int64    -- ^ size of the original PDF file. Should be 0 for new file+  -> Ref+  -> Dict     -- ^ trailer+  -> IO ()+writeXRefStream writer offset ref@(R index gen) tr = do+  pos <- countWritten writer+  addElem writer $ Elem index gen pos False+  st <- readIORef (toStateRef writer)+  let elems = Set.mapMonotonic (\e -> e {elemOffset = elemOffset e + offset})+            $ stObjects st+      off = pos + offset+      content = toLazyByteString $ buildXRefStream (Set.toAscList elems)+      dict+        = HashMap.insert "Type" (Name "XRef")+        . HashMap.insert "W" (Array $ Vector.fromList $ map Number [1, 8, 8])+        . HashMap.insert "Index" (Array $ Vector.fromList $ map Number trIndex)+        . HashMap.insert "Length" (Number $ fromIntegral $ BSL.length content)+        . HashMap.delete "Filter"+        $ tr+      trIndex = concatMap sectionIndex (xrefSections (Set.toAscList elems))+      sectionIndex [] = error "impossible"+      sectionIndex s@(e:_) = map fromIntegral [elemIndex e, length s]+      end = mconcat+        [ "\nstartxref\n"+        , int64Dec off+        , "\n%%EOF\n"+        ]+  dumpStream (stOutput st) ref dict content+  Streams.writeLazyByteString (toLazyByteString end) (stOutput st)++countWritten :: Writer -> IO Int64+countWritten writer = do+  st <- readIORef (toStateRef writer)+  c <- (stOffset st +) <$> stCount st+  writeIORef (toStateRef writer) st{stOffset = c}+  return $! c++addElem :: Writer -> Elem -> IO ()+addElem writer e = do+  st <- readIORef (toStateRef writer)+  when (Set.member e $ stObjects st) $+    error $ "Writer: attempt to write object with the same index: " ++ show (elemIndex e)+  writeIORef (toStateRef writer) $ st+    { stObjects = Set.insert e $ stObjects st+    }++dumpObject :: OutputStream ByteString -> Ref -> Object -> IO ()+dumpObject out ref o =+  Streams.writeLazyByteString+    (toLazyByteString $ buildIndirectObject ref o)+    out++dumpStream :: OutputStream ByteString -> Ref -> Dict -> BSL.ByteString -> IO ()+dumpStream out ref dict dat =+  Streams.writeLazyByteString+    (toLazyByteString $ buildIndirectStream ref dict dat) out++buildXRefTable :: [Elem] -> Builder+buildXRefTable entries =+  mconcat (map buildXRefTableSection $ xrefSections entries)++xrefSections :: [Elem] -> [[Elem]]+xrefSections [] = []+xrefSections xs = let (s, rest) = xrefSection xs in s : xrefSections rest++xrefSection :: [Elem] -> ([Elem], [Elem])+xrefSection [] = error "impossible"+xrefSection (x:xs) = go (elemIndex x + 1) [x] xs+  where+  go _ res [] = (reverse res, [])+  go i res (y:ys) =+    if i == elemIndex y+      then go (i + 1) (y : res) ys+      else (reverse res, y:ys)++buildXRefTableSection :: [Elem] -> Builder+buildXRefTableSection [] = error "impossible"+buildXRefTableSection s@(e:_) = mconcat+  [ intDec (elemIndex e)+  , char7 ' '+  , intDec (length s)+  , char7 '\n'+  , loop s+  ]+  where+  loop (x:xs) = mconcat+    [ buildFixed 10 '0' (elemOffset x)+    , char7 ' '+    , buildFixed 5 '0' (elemGen x)+    , char7 ' '+    , char7 (if elemFree x then 'f' else 'n')+    , string7 "\r\n"+    ] `mappend` loop xs+  loop [] = mempty++buildXRefStream :: [Elem] -> Builder+buildXRefStream entries =+  mconcat (map buildXRefStreamSection $ xrefSections entries)++buildXRefStreamSection :: [Elem] -> Builder+buildXRefStreamSection = mconcat . map buildOne+  where+  buildOne e =+    let (tp, field1, field2) = if elemFree e+          then (0, 0, succ (elemGen e))+          else (1, elemOffset e, elemGen e)+    in mconcat+      [ int8 tp+      , int64BE field1+      , int64BE (fromIntegral field2)+      ]++buildFixed :: Show a => Int -> Char -> a -> Builder+buildFixed len c i =+  let v = take len $ show i+      l = length v+  in string7 $ replicate (len - l) c ++ v
+ lib/Pdf/Core/XRef.hs view
@@ -0,0 +1,252 @@+{-# LANGUAGE OverloadedStrings #-}+{-# LANGUAGE DeriveDataTypeable #-}++-- | Cross reference++module Pdf.Core.XRef+(+  XRef(..),+  Entry(..),+  readXRef,+  lastXRef,+  prevXRef,+  trailer,+  lookupTableEntry,+  lookupStreamEntry,+  isTable,+  UnknownXRefStreamEntryType(..),+)+where++import Pdf.Core.Object+import Pdf.Core.Object.Util+import Pdf.Core.Parsers.XRef+import Pdf.Core.Stream+import Pdf.Core.Exception+import Pdf.Core.Util+import Pdf.Core.IO.Buffer (Buffer)+import qualified Pdf.Core.IO.Buffer as Buffer++import Data.Typeable+import Data.Int+import Data.ByteString (ByteString)+import qualified Data.ByteString as ByteString+import qualified Data.Vector as Vector+import qualified Data.HashMap.Strict as HashMap+import Control.Monad+import Control.Exception hiding (throw)+import System.IO.Streams (InputStream)+import qualified System.IO.Streams as Streams+import qualified System.IO.Streams.Attoparsec as Streams++-- | Entry in cross reference stream+data Entry =+  -- | Object number and generation+  EntryFree Int Int |+  -- | Object offset (in bytes from the beginning of file) and generation+  EntryUsed Int64 Int |+  -- | Object number of object stream and index within the object stream+  EntryCompressed Int Int+  deriving (Eq, Show)++-- | Cross reference+data XRef =+  -- | Offset+  XRefTable Int64 |+  -- | Offset and stream+  XRefStream Int64 Stream+  deriving (Eq, Show)++-- | Check whether the stream starts with \"xref\" keyword.+-- The keyword itself and newline after it are consumed+isTable :: InputStream ByteString -> IO Bool+isTable is = (Streams.parseFromStream tableXRef is >> return True)+  `catch` \(Streams.ParseException _) -> return False++-- | Find the last cross reference+lastXRef :: Buffer -> IO XRef+lastXRef buf = do+  sz <- Buffer.size buf+  Buffer.seek buf $ max 0 (sz - 1024)+  (Streams.parseFromStream startXRef (Buffer.toInputStream buf)+    >>= readXRef buf+    ) `catch` \(Streams.ParseException msg) ->+                  throwIO (Corrupted "lastXRef" [msg])++-- | Read XRef at specified offset+readXRef :: Buffer -> Int64 -> IO XRef+readXRef buf off = do+  Buffer.seek buf off+  let is = Buffer.toInputStream buf+  table <- isTable is+  if table+    then return (XRefTable off)+    else do+      s <- readStream is off+      return (XRefStream off s)++-- | Find prev cross reference+prevXRef :: Buffer -> XRef -> IO (Maybe XRef)+prevXRef buf xref = message "prevXRef" $ do+  tr <- trailer buf xref+  case HashMap.lookup "Prev" tr of+    Just prev -> do+      off <- sure $ intValue prev+        `notice` "Prev in trailer should be an integer"+      Just <$> readXRef buf (fromIntegral off)+    _ -> return Nothing++-- | Read trailer for the xref+trailer :: Buffer -> XRef -> IO Dict+trailer buf (XRefTable off) = do+  Buffer.seek buf off+  let is = Buffer.toInputStream buf+  table <- isTable is+  unless table $+    throwIO (Unexpected "trailer" ["table not found"])+  ( skipTable is >>+    Streams.parseFromStream parseTrailerAfterTable is+    ) `catch` \(Streams.ParseException msg) ->+                  throwIO (Corrupted "trailer" [msg])+trailer _ (XRefStream _ (S dict _)) = return dict++skipTable :: InputStream ByteString -> IO ()+skipTable is = message "skipTable" $+  (subsectionHeader is+    `catch` \(Streams.ParseException msg) ->+      throwIO (Corrupted msg []))+    >>= go . snd+  where+  go count = nextSubsectionHeader is count >>= maybe (return ()) (go . snd)++subsectionHeader :: InputStream ByteString -> IO (Int, Int)+subsectionHeader = Streams.parseFromStream parseSubsectionHeader++nextSubsectionHeader :: InputStream ByteString -> Int -> IO (Maybe (Int, Int))+nextSubsectionHeader is count = message "nextSubsectionHeader" $ do+  skipSubsection is count+  fmap Just (subsectionHeader is)+    `catch` \(Streams.ParseException _) -> return Nothing++skipSubsection :: InputStream ByteString -> Int -> IO ()+skipSubsection is count = Buffer.dropExactly (count * 20) is++-- | Read xref entry for the indirect object from xref table+lookupTableEntry :: Buffer+                 -> XRef  -- ^ should be xref table+                 -> Ref   -- ^ indirect object to look for+                 -> IO (Maybe Entry)+lookupTableEntry buf (XRefTable tableOff) (R index gen)+  = message "lookupTableEntry" $ do+  Buffer.seek buf tableOff+  table <- isTable (Buffer.toInputStream buf)+  unless table $+    throwIO $ Unexpected "Not a table" []+  (subsectionHeader (Buffer.toInputStream buf) >>= go)+    `catch` \(Streams.ParseException err) -> throwIO (Corrupted err [])+  where+  go (start, count) = do+    if index >= start && index < start + count+      then do+        -- that is our section, lets seek to the row+        Buffer.tell buf+          >>= Buffer.seek buf . (+ (fromIntegral $ index - start) * 20)+        (off, gen', free) <-+          Streams.parseFromStream parseTableEntry (Buffer.toInputStream buf)+            `catch` \(Streams.ParseException msg) ->+              throwIO (Corrupted "parseTableEntry failed" [msg])+        unless (free || gen == gen') $ do+          print (index, gen, off, gen', free)+          throwIO $ Corrupted "Generation mismatch" []+        let entry = if free+              then EntryFree (fromIntegral off) gen+              else EntryUsed off gen+        return (Just entry)+      else+        -- go to the next section if any+        nextSubsectionHeader (Buffer.toInputStream buf) count+        >>= maybe (return Nothing) go+lookupTableEntry _ XRefStream{} _ =+  throwIO $ Unexpected "lookupTableEntry" ["Only xref table allowed"]++-- | Read xref entry for the indirect object from xref stream+--+-- See pdf1.7 spec: 7.5.8 Cross-Reference Streams.+-- May throw 'UnknownXRefStreamEntryType'+lookupStreamEntry+  :: Dict                    -- ^ xref stream dictionary+  -> InputStream ByteString  -- ^ decoded xref stream content+  -> Ref                     -- ^ indirect object+  -> IO (Maybe Entry)+lookupStreamEntry dict is (R objNumber _) =+  message "lookupStreamEntry" $ do++  index <- sure $ do+    sz <- (HashMap.lookup "Size" dict >>= intValue)+      `notice` "Size should be an integer"+    i <-+      case HashMap.lookup "Index" dict of+        Nothing           -> Right [Number 0, Number (fromIntegral sz)]+        Just (Array arr) -> Right (Vector.toList arr)+        _                 -> Left "Index should be an array"++    let convertIndex res [] = Right (reverse res)+        convertIndex res (x1:x2:xs) = do+          from <- intValue x1 `notice` "from index should be an integer"+          count <- intValue x2 `notice` "count should be an integer"+          convertIndex ((from, count) : res) xs+        convertIndex _ _ = Left $ "Malformed Index in xref stream: " ++ show i++    convertIndex [] i++  width <- sure $ do+    ws <-+      case HashMap.lookup "W" dict of+        Just (Array ws) -> Right (Vector.toList ws)+        _ -> Left "W should be an array"+    mapM intValue ws+      `notice` "W should contains integers"++  unless (length width == 3) $+    throwIO $ Corrupted ("Malformed With array in xref stream: "+                        ++ show width) []++  values <- do+    let position = loop 0 index+        totalWidth = sum width+        loop _ [] = Nothing+        loop pos ((from, count) : xs) =+          if objNumber < from || objNumber >= from + count+            then loop (pos + totalWidth * count) xs+            else Just (pos + totalWidth * (objNumber - from))+    case position of+      Nothing -> return Nothing+      Just p -> do+        Buffer.dropExactly p is+        Just . ByteString.unpack <$> Streams.readExactly totalWidth is++  case values of+    Nothing -> return Nothing+    Just vs -> do+      let [v1, v2, v3] = map conv $ collect [] width vs :: [Int64]+            where+            conv l = conv' (length l - 1) 0 l+            conv' _ res [] = res+            conv' power res (x:xs) =+              conv' (power-1) (res + (fromIntegral x * 256 ^ power)) xs+            collect res [] [] = reverse res+            collect res (x:xs) ys = collect (take x ys : res) xs (drop x ys)+            collect _ _ _ = error "readStreamEntry: collect: impossible"+      case v1 of+        0 -> return $ Just $ EntryFree (fromIntegral v2)+                                             (fromIntegral v3)+        1 -> return $ Just $ EntryUsed v2 (fromIntegral v3)+        2 -> return $ Just $ EntryCompressed (fromIntegral v2)+                                                   (fromIntegral v3)+        _ -> throwIO $ UnknownXRefStreamEntryType (fromIntegral v1)++-- | Unknown entry type should be interpreted as reference to null object+data UnknownXRefStreamEntryType = UnknownXRefStreamEntryType Int+  deriving (Show, Typeable)++instance Exception UnknownXRefStreamEntryType
− lib/Pdf/Toolbox/Core.hs
@@ -1,26 +0,0 @@---- | Low level tools for processing PDF file--module Pdf.Toolbox.Core-(-  module Pdf.Toolbox.Core.Error,-  module Pdf.Toolbox.Core.IO,-  module Pdf.Toolbox.Core.Stream,-  module Pdf.Toolbox.Core.Object.Types,-  module Pdf.Toolbox.Core.Object.Builder,-  module Pdf.Toolbox.Core.Object.Util,-  module Pdf.Toolbox.Core.XRef,-  module Pdf.Toolbox.Core.Util,-  module Pdf.Toolbox.Core.Writer-)-where--import Pdf.Toolbox.Core.Error-import Pdf.Toolbox.Core.IO-import Pdf.Toolbox.Core.Stream-import Pdf.Toolbox.Core.Object.Types-import Pdf.Toolbox.Core.Object.Builder-import Pdf.Toolbox.Core.Object.Util-import Pdf.Toolbox.Core.XRef-import Pdf.Toolbox.Core.Util-import Pdf.Toolbox.Core.Writer
− lib/Pdf/Toolbox/Core/Error.hs
@@ -1,54 +0,0 @@-{-# LANGUAGE  ScopedTypeVariables #-}---- | Error used by API--module Pdf.Toolbox.Core.Error-(-  PdfError(..),-  PdfE,-  annotateError,-  annotatingError,-  tryPdfIO,-  module Control.Error-)-where--import Control.Error-import Control.Exception-import Control.Monad-import Control.Monad.IO.Class---- | Errors-data PdfError =-  -- | Parser error-  ParseError [String] String |-  IOError IOError |-  AnnotatedError String PdfError |-  -- | Something unexpected-  UnexpectedError String-  deriving Show---- | API uses this for error handling-type PdfE m = ExceptT PdfError m---- | Wrap any 'PdfError' into 'AnnotatedError'------ Usefull when you want to add high-level description to--- error, returned by low-level function-annotateError :: Monad m => String -> PdfE m a -> PdfE m a-annotateError ann = fmapLT (AnnotatedError ann')-  where-  ann' = if length ann <= 50 then ann else take 47 ann ++ "..."---- | 'annotateError' with fliped arguments-annotatingError :: Monad m => PdfE m a -> String -> PdfE m a-annotatingError = flip annotateError---- | Catch exception if any and convert to 'IOError'-tryPdfIO :: MonadIO m => IO a -> PdfE m a-tryPdfIO action = do-  res <- liftIO $ Right `liftM` action-      `catch` (\(e :: IOError) -> return $ Left $ IOError e)-  case res of-    Right a -> return a-    Left e -> throwE e
− lib/Pdf/Toolbox/Core/IO.hs
@@ -1,72 +0,0 @@-{-# LANGUAGE ScopedTypeVariables #-}---- | Basic IO operations for PDF--module Pdf.Toolbox.Core.IO-(-  IS,-  RIS,-  RIS.fromHandle,-  RIS.fromHandle',-  MonadIO,-  liftIO,-  size,-  seek,-  tell,-  parse,-  inputStream,-  takeBytes,-  readExactly,-  dropExactly-)-where--import Data.Int-import Data.Attoparsec.ByteString (Parser)-import Data.ByteString (ByteString)-import Control.Monad.IO.Class-import Control.Exception-import qualified System.IO.Streams as Streams-import qualified System.IO.Streams.Attoparsec as Streams--import Pdf.Toolbox.Core.Error-import qualified Pdf.Toolbox.Core.IO.RIS as RIS-import Pdf.Toolbox.Core.IO.RIS (RIS, IS)---- | Total number of bytes in 'RIS'-size :: MonadIO m => RIS -> PdfE m Int64-size = tryPdfIO . RIS.size---- | Change input position in 'RIS'-seek :: MonadIO m => RIS -> Int64 -> PdfE m ()-seek ris = tryPdfIO . RIS.seek ris---- | Current input position-tell :: MonadIO m => RIS -> PdfE m Int64-tell = tryPdfIO . RIS.tell---- | Parse from 'IS'-parse :: MonadIO m => Parser r -> IS -> PdfE m r-parse p is = do-  res <- liftIO $ (Right <$> Streams.parseFromStream p is)-    `catch` (\(Streams.ParseException str) -> return $ Left $ ParseError [] str)-    `catch` (\(e :: IOError) -> return $ Left $ IOError e)-  case res of-    Left e -> throwE e-    Right r -> return r---- | Convert random access stream to sequential-inputStream :: MonadIO m => RIS -> PdfE m IS-inputStream = tryPdfIO . RIS.inputStream---- | See 'Streams.takeBytes'-takeBytes :: MonadIO m => Int64 -> IS -> PdfE m IS-takeBytes n = tryPdfIO . Streams.takeBytes n---- | See 'Streams.readExactly'-readExactly :: MonadIO m => Int -> IS -> PdfE m ByteString-readExactly n = tryPdfIO . Streams.readExactly n---- | Same as 'readExactly', but ignores the result-dropExactly :: MonadIO m => Int -> IS -> PdfE m ()-dropExactly n is = readExactly n is >> return ()
− lib/Pdf/Toolbox/Core/IO/RIS.hs
@@ -1,92 +0,0 @@---- | Input stream with random access--module Pdf.Toolbox.Core.IO.RIS-(-  IS,-  RIS(..),-  RIS'(..),-  seek,-  size,-  tell,-  inputStream,-  fromHandle,-  fromHandle'-)-where--import Data.Int (Int64)-import Data.ByteString (ByteString)-import qualified Data.ByteString as BS-import Data.IORef-import System.IO-import System.IO.Streams (InputStream)-import qualified System.IO.Streams as Streams---- | Sequential input stream-type IS = InputStream ByteString---- | Internal state of 'RIS'-data RIS' = RIS' {-  risSeek :: Int64 -> IO (IO (Maybe ByteString)),-  risInputStream :: IS,-  risPos :: IO Int64,-  risSize :: Int64-  }---- | Random access Input Stream-newtype RIS = RIS (IORef RIS')---- | Seek the stream-seek :: RIS -> Int64 -> IO ()-seek (RIS ref) pos = do-  ris <- readIORef ref-  source <- risSeek ris pos-  stream <- Streams.makeInputStream source-  (s, c) <- Streams.countInput stream-  writeIORef ref ris {-    risInputStream = s,-    risPos = (+ pos) <$> c-    }---- | Create RIS from 'Handle' with the specified chunk size-fromHandle' :: Handle -> Int -> IO RIS-fromHandle' h buf = do-  sz <- hFileSize h-  posRef <- newIORef 0-  stream <- Streams.makeInputStream (f posRef)-  (s, c) <- Streams.countInput stream-  RIS <$> newIORef RIS' {-    risSeek = \pos -> do-      hSeek h AbsoluteSeek (fromIntegral pos)-      ref <- newIORef $ fromIntegral pos-      return $ f ref,-    risInputStream = s,-    risPos = c,-    risSize = fromIntegral sz-    }-  where-  f ref = do-    prevPos <- readIORef ref-    curtPos <- hTell h-    hSeek h AbsoluteSeek prevPos-    chunk <- BS.hGetSome h buf-    hSeek h AbsoluteSeek curtPos-    writeIORef ref $! prevPos + fromIntegral (BS.length chunk)-    return $! if BS.null chunk then Nothing else Just chunk---- | Create RIS from 'Handle' with default chunk size-fromHandle :: Handle -> IO RIS-fromHandle h = fromHandle' h 32752---- | Number of bytes in the stream-size :: RIS -> IO Int64-size (RIS ref) = risSize <$> readIORef ref---- | Current position in bytes-tell :: RIS -> IO Int64-tell (RIS ref) = readIORef ref >>= risPos---- | Get sequential input stream, that is valid until the next 'seek'-inputStream :: RIS -> IO IS-inputStream (RIS ref) = risInputStream <$> readIORef ref
− lib/Pdf/Toolbox/Core/Object/Builder.hs
@@ -1,120 +0,0 @@-{-# LANGUAGE OverloadedStrings #-}-{-# LANGUAGE CPP #-}---- | Render 'Object' to bytestring--module Pdf.Toolbox.Core.Object.Builder-(-  buildIndirectObject,-  buildObject,-  buildNumber,-  buildBoolean,-  buildName,-  buildDict,-  buildArray,-  buildStr,-  buildRef,-  buildStream-)-where--import Data.Char-import qualified Data.ByteString as BS-import qualified Data.ByteString.Char8 as BS8-import qualified Data.ByteString.Lazy as BSL-import Data.ByteString.Lazy.Builder--#if MIN_VERSION_bytestring(0, 10, 4)-#else-import Data.ByteString.Lazy.Builder.ASCII-#endif--import Text.Printf--import Pdf.Toolbox.Core.Object.Types---- | Build indirect object-buildIndirectObject :: Ref -> Object BSL.ByteString -> Builder-buildIndirectObject (Ref i g) object =-  char7 '\n' `mappend`-  intDec i `mappend`-  char7 ' ' `mappend`-  intDec g `mappend`-  byteString " obj\n" `mappend`-  build object `mappend`-  byteString "\nendobj\n"-  where-  build (OStream s) = buildStream s-  build o = buildObject o---- | Render inline object (without \"obj/endobj\").--- It is 'error' to supply 'Stream', because it could not--- be inlined, but should always be an indirect object-buildObject :: Object a -> Builder-buildObject (ONumber n) = buildNumber n-buildObject (OBoolean b) = buildBoolean b-buildObject (OName n) = buildName n-buildObject (ODict d) = buildDict d-buildObject (OArray a) = buildArray a-buildObject (OStr s) = buildStr s-buildObject (ORef r) = buildRef r-buildObject (OStream _) = error "buildObject: please don't pass streams to me"-buildObject ONull = byteString "null"--buildStream :: Stream BSL.ByteString -> Builder-buildStream (Stream dict content) =-  buildDict dict `mappend`-  byteString "stream\n" `mappend`-  lazyByteString content `mappend`-  byteString "\nendstream"--buildNumber :: Number -> Builder-buildNumber (NumInt i) = intDec i-buildNumber (NumReal d) = string7 $ printf "%f" d--buildBoolean :: Boolean -> Builder-buildBoolean (Boolean True) = byteString "true"-buildBoolean (Boolean False) = byteString "false"--buildName :: Name -> Builder-buildName (Name n) = char7 '/' `mappend` byteString n--intercalate :: Builder -> [Builder] -> Builder-intercalate _ [] = mempty-intercalate sep (x:xs) = x `mappend` go xs-  where-  go [] = mempty-  go (y:ys) = sep `mappend` y `mappend` go ys--buildDict :: Dict -> Builder-buildDict (Dict xs) =-  byteString "<<" `mappend`-  intercalate (char7 ' ') (concatMap build xs) `mappend`-  byteString ">>"-  where-  build (key, val) = [buildName key, buildObject val]--buildArray :: Array -> Builder-buildArray (Array xs) =-  char7 '[' `mappend`-  intercalate (char7 ' ') (map buildObject xs) `mappend`-  char7 ']'--buildStr :: Str -> Builder-buildStr (Str s) =-  if BS8.all isPrint s-    then char7 '(' `mappend` (byteString . BS8.pack . concatMap escape . BS8.unpack $ s) `mappend` char7 ')'-    else char7 '<' `mappend` (byteString . BS.pack . concatMap toHex . BS.unpack $ s) `mappend` char7 '>'-  where-  toHex w = map (\a -> if a < 10 then a + 48 else a + 55) [w `div` 16, w `mod` 16]-  escape '(' = "\\("-  escape ')' = "\\)"-  escape '\\' = "\\\\"-  escape '\n' = "\\n"-  escape '\r' = "\\r"-  escape '\t' = "\\t"-  escape '\b' = "\\b"-  escape ch = [ch]--buildRef :: Ref -> Builder-buildRef (Ref i j) = intDec i `mappend` char7 ' ' `mappend` intDec j `mappend` byteString " R"
− lib/Pdf/Toolbox/Core/Object/Types.hs
@@ -1,84 +0,0 @@-{-# LANGUAGE GeneralizedNewtypeDeriving #-}---- | Module contains definitions of pdf objects------ See PDF1.7:7.3--module Pdf.Toolbox.Core.Object.Types-(-  Object(..),-  Number(..),-  Boolean(..),-  Name(..),-  Dict(..),-  Array(..),-  Str(..),-  Stream(..),-  Ref(..)-)-where--import Data.String-import Data.ByteString (ByteString)---- | Integer or real -data Number =-  NumInt Int |-  NumReal Double-  deriving (Eq, Show)---- | \"true\" or \"false\"-newtype Boolean = Boolean Bool-  deriving (Eq, Show)---- | Names usually are used as keys in dictionaries------ They starts with \'/\', but we strip it out, see 'Pdf.Toolbox.Core.Parsers.Object.parseName'-newtype Name = Name ByteString-  deriving (Eq, Show, Ord, Monoid)---- | Set of key/value pairs-newtype Dict = Dict [(Name, Object ())]-  deriving (Eq, Show)---- | An array-newtype Array = Array [Object ()]-  deriving (Eq, Show)---- | Sequence of zero or more bytes------ Represents both the literal and hexadecimal strings-newtype Str = Str ByteString-  deriving (Eq, Show)---- | Contains stream dictionary and a payload------ The payload could be offset within pdf file, actual content,--- content stream or nothing-data Stream a = Stream Dict a-  deriving (Eq, Show)---- | Object reference, contains object index and generation-data Ref = Ref Int Int-  deriving (Eq, Show, Ord)---- | Any pdf object------ It is parameterized by 'Stream' content-data Object a =-  ONumber Number |-  OBoolean Boolean |-  OName Name |-  ODict Dict |-  OArray Array |-  OStr Str |-  OStream (Stream a) |-  ORef Ref |-  ONull-  deriving (Eq, Show)--instance IsString Name where-  fromString = Name . fromString--instance IsString Str where-  fromString = Str . fromString
− lib/Pdf/Toolbox/Core/Object/Util.hs
@@ -1,130 +0,0 @@---- | Utils relayted to pdf objects--module Pdf.Toolbox.Core.Object.Util-(-  -- * Casting pdf objects-  FromObject(..),-  toNumber,-  toBoolean,-  toName,-  toDict,-  toArray,-  toStr,-  toRef,-  toStream,-  mapObject,-  -- * Dictionary-  lookupDict,-  lookupDict',-  setValueForKey,-  deleteValueForKey,-  -- * Number-  intValue,-  realValue-)-where--import Pdf.Toolbox.Core.Object.Types-import Pdf.Toolbox.Core.Error--lookupDict :: Monad m => Name -> Dict -> PdfE m (Object ())-lookupDict key (Dict d) =-  case lookup key d of-    Just o -> return o-    Nothing -> throwE $ UnexpectedError $ "Key not found: " ++ show key ++ " " ++ show d--lookupDict' :: Name -> Dict -> Maybe (Object ())-lookupDict' key (Dict d) = lookup key d--deleteValueForKey :: Name -> Dict -> Dict-deleteValueForKey key (Dict vals) = Dict $ go vals-  where-  go [] = []-  go (x@(k, _) : xs)-    | k == key = go xs-    | otherwise = x : go xs--setValueForKey :: Name -> Object () -> Dict -> Dict-setValueForKey key val dict = Dict $ (key, val) : vals-  where-  Dict vals = deleteValueForKey key dict--intValue :: Monad m => Number -> PdfE m Int-intValue (NumInt i) = return i-intValue (NumReal r) = throwE $ UnexpectedError $ "Integer expected, but real received: " ++ show r--realValue :: Monad m => Number -> PdfE m Double-realValue (NumReal r) = return r-realValue (NumInt i) = return $ fromIntegral i--toNumber :: (Show a, Monad m) => Object a -> PdfE m Number-toNumber (ONumber n) = return n-toNumber o = throwE $ UnexpectedError $ "Can't cast object to Number: " ++ show o--toBoolean :: (Show a, Monad m) => Object a -> PdfE m Boolean-toBoolean (OBoolean b) = return b-toBoolean o = throwE $ UnexpectedError $ "Can't cast object to Boolean: " ++ show o--toName :: (Show a, Monad m) => Object a -> PdfE m Name-toName (OName n) = return n-toName o = throwE $ UnexpectedError $ "Can't cast object to Name: " ++ show o--toDict :: (Show a, Monad m) => Object a -> PdfE m Dict-toDict (ODict d) = return d-toDict o = throwE $ UnexpectedError $ "Can't cast object to Dict: " ++ show o--toStr :: (Show a, Monad m) => Object a -> PdfE m Str-toStr (OStr s) = return s-toStr o = throwE $ UnexpectedError $ "Can't cast object to Str: " ++ show o--toRef :: (Show a, Monad m) => Object a -> PdfE m Ref-toRef (ORef r) = return r-toRef o = throwE $ UnexpectedError $ "Can't cast object to Ref: " ++ show o--toArray :: (Show a, Monad m) => Object a -> PdfE m Array-toArray (OArray a) = return a-toArray o = throwE $ UnexpectedError $ "Can't cast object to Array: " ++ show o--toStream :: (Show a, Monad m) => Object a -> PdfE m (Stream a)-toStream (OStream s) = return s-toStream o = throwE $ UnexpectedError $ "Can't cast object to Stream: " ++ show o---- | Apply function to all stream contents-mapObject :: (a -> b) -> Object a -> Object b-mapObject f o =-  case o of-    ONumber n -> ONumber n-    OBoolean b -> OBoolean b-    OName n -> OName n-    ODict d -> ODict d-    OArray a -> OArray a-    OStr s -> OStr s-    OStream (Stream d a) -> OStream (Stream d $ f a)-    ORef r -> ORef r-    ONull -> ONull---- | Allows you to cast 'Object' to specific type-class FromObject c where-  fromObject :: (Show a, Monad m) => Object a -> PdfE m c--instance FromObject Number where-  fromObject = toNumber--instance FromObject Boolean where-  fromObject = toBoolean--instance FromObject Name where-  fromObject = toName--instance FromObject Dict where-  fromObject = toDict--instance FromObject Str where-  fromObject = toStr--instance FromObject Ref where-  fromObject = toRef--instance FromObject Array where-  fromObject = toArray
− lib/Pdf/Toolbox/Core/Parsers/Object.hs
@@ -1,261 +0,0 @@-{-# LANGUAGE OverloadedStrings #-}-{-# LANGUAGE CPP #-}---- | This module contains parsers for pdf objects--module Pdf.Toolbox.Core.Parsers.Object-(-  -- * Parse any object-  parseObject,-  -- * Parse object of specific type-  parseDict,-  parseArray,-  parseName,-  parseStr,-  parseHexStr,-  parseRef,-  parseNumber,-  parseBoolean,-  -- * Other-  parseTillStreamData,-  parseIndirectObject,-  isRegularChar-)-where--import Data.List-import Data.Char-import qualified Data.ByteString as BS-import qualified Data.ByteString.Char8 as BS8-import Data.Attoparsec.ByteString (Parser)-import qualified Data.Attoparsec.ByteString.Char8 as P--#if MIN_VERSION_attoparsec(0, 12, 0)-import qualified Data.Scientific as Scientific-#endif--import Control.Applicative-import Control.Monad--import Pdf.Toolbox.Core.Object.Types-import Pdf.Toolbox.Core.Parsers.Util---- for doctest--- $setup--- >>> :set -XOverloadedStrings--- >>> import Data.Attoparsec.ByteString.Char8---- |--- >>> parseOnly parseDict "<</Key1(some string)/Key2 123>>"--- Right (Dict [(Name "Key1",OStr (Str "some string")),(Name "Key2",ONumber (NumInt 123))])-parseDict :: Parser Dict-parseDict = do-  _ <- P.string "<<"-  dict <- many parseKey-  P.skipSpace-  _ <- P.string ">>"-  return $ Dict dict--parseKey :: Parser (Name, Object ())-parseKey = do-  P.skipSpace-  key <- parseName-  val <- parseObject-  return (key, val)---- |--- >>> parseOnly parseArray "[1 (string) /Name []]"--- Right (Array [ONumber (NumInt 1),OStr (Str "string"),OName (Name "Name"),OArray (Array [])])-parseArray :: Parser Array-parseArray = do-  _ <- P.char '['-  array <- many parseObject-  P.skipSpace-  _ <- P.char ']'-  return $ Array array---- |--- >>> parseOnly parseNumber "123"--- Right (NumInt 123)--- >>> parseOnly parseNumber "12.3"--- Right (NumReal 12.3)--- >>> parseOnly parseNumber ".01"--- Right (NumReal 1.0e-2)-parseNumber :: Parser Number-parseNumber = P.choice [-  number,-  NumReal <$> (P.signed $ read . ("0."++) . BS8.unpack <$> (P.char '.' >> P.takeWhile1 isDigit))-  ]-  where-#if MIN_VERSION_attoparsec(0, 12, 0)-  number = toNum <$> P.scientific-  toNum = either NumReal NumInt . Scientific.floatingOrInteger-#else-  number = toNum <$> P.number-  toNum (P.I i) = NumInt $ fromIntegral i-  toNum (P.D d) = NumReal d-#endif---- |--- >>> parseOnly parseStr "(hello)"--- Right (Str "hello")-parseStr :: Parser Str-parseStr = do-  _ <- P.char '('-  str <- takeStr 0 []-  return $ Str $ BS8.pack str-  where-  takeStr :: Int -> String -> Parser String-  takeStr lvl res = do-    ch <- P.anyChar-    case ch of-      '(' -> takeStr (lvl + 1) (ch : res)-      ')' -> if lvl == 0-               then return $ reverse res-               else takeStr (lvl - 1) (ch : res)-      '\\' -> do-        ch' <- P.anyChar-        if ch' `elem` ("()\\" :: String)-          then takeStr lvl (ch' : res)-          else case ch' of-                 'r' -> takeStr lvl ('\r' : res)-                 'n' -> takeStr lvl ('\n' : res)-                 'f' -> takeStr lvl ('\f' : res)-                 'b' -> takeStr lvl ('\b' : res)-                 't' -> takeStr lvl ('\t' : res)-                 '\r' -> takeStr lvl res-                 _ -> do-                   ds <- take3Digits [ch']-                   let i = toEnum-                         . foldl'-                             (\acc (a, b) -> acc + a * charToInt b)-                             0-                         . zip [1, 8, 64]-                         $ ds-                   takeStr lvl (i : res)-      _ -> takeStr lvl (ch : res)-  charToInt ch = fromEnum ch - 48-  take3Digits ds-    | length ds >= 3-    = return ds-    | otherwise-    = do-      d <- P.peekChar'-      if isDigit d-        then do-          void P.anyChar-          take3Digits (d : ds)-        else-          return (ds ++ repeat '0')---- |--- >>> parseOnly parseHexStr "<68656C6C6F>"--- Right (Str "hello")-parseHexStr :: Parser Str-parseHexStr = do-  _ <- P.char '<'-  str <- many takeHex-  _ <- P.char '>'-  return $ Str $ BS.pack str-  where-  takeHex = do-    ch1 <- P.satisfy isHexDigit-    ch2 <- P.satisfy isHexDigit-    return $ fromIntegral $ digitToInt ch1 * 16 + digitToInt ch2---- |--- >>> parseOnly parseRef "0 2 R"--- Right (Ref 0 2)-parseRef :: Parser Ref-parseRef = do-  obj <- P.decimal-  P.skipSpace-  gen <- P.decimal-  P.skipSpace-  _ <- P.char 'R'-  return $ Ref obj gen---- |--- >>> parseOnly parseName "/Name"--- Right (Name "Name")-parseName :: Parser Name-parseName = do-  _ <- P.char '/'-  Name <$> P.takeWhile1 isRegularChar---- | Whether the character can appear in 'Name'-isRegularChar :: Char -> Bool-isRegularChar = (`notElem` ("[]()/<>{}% \n\r" :: String))---- |--- >>> parseOnly parseBoolean "true"--- Right (Boolean True)--- >>> parseOnly parseBoolean "false"--- Right (Boolean False)-parseBoolean :: Parser Boolean-parseBoolean = Boolean <$> P.choice [-  P.string "true" >> return True,-  P.string "false" >> return False-  ]---- | Consumes input till stream's data------ Use 'parseDict' then 'parseTillStreamData'--- to determine whether the object is dictionary or stream.--- If 'parseTillStreamData' fails, then it is a dictionary.--- Otherwise it is stream, and current position in input data--- will point to stream's data start------ >>> parse (parseDict >>= \dict -> parseTillStreamData >> return dict) "<</Key 123>>\nstream\n1234\nendstream"--- Done "1234\nendstream" Dict [(Name "Key",ONumber (NumInt 123))]-parseTillStreamData :: Parser ()-parseTillStreamData = do-  P.skipSpace-  _ <- P.string "stream"-  endOfLine---- | It parses any 'Object' except 'Stream'--- cos for 'Stream' we need offset of data in file------ >>> parseOnly parseObject "/Name"--- Right (OName (Name "Name"))-parseObject :: Parser (Object ())-parseObject = do-  P.skipSpace-  P.choice [-    const ONull <$> P.string "null",-    OName <$> parseName,-    OBoolean <$> parseBoolean,-    ODict <$> parseDict,-    OArray <$> parseArray,-    OStr <$> parseStr,-    OStr <$> parseHexStr,-    ORef <$> parseRef,-    ONumber <$> parseNumber-    ]---- | Parse object. Input position should point--- to offset defined in XRef------ >>> parseOnly parseIndirectObject "1 2 obj\n12"--- Right (Ref 1 2,ONumber (NumInt 12))-parseIndirectObject :: Parser (Ref, Object ())-parseIndirectObject = do-  P.skipSpace-  index <- P.decimal :: Parser Int-  P.skipSpace-  gen <- P.decimal :: Parser Int-  P.skipSpace-  _ <- P.string "obj"-  P.skipSpace-  obj <- parseObject-  let ref = Ref index gen-  case obj of-    ODict d -> P.choice [-      parseTillStreamData >> return (ref, OStream $ Stream d ()),-      return (ref, ODict d)-      ]-    _ -> return (ref, obj)---- |--- More tests
− lib/Pdf/Toolbox/Core/Parsers/Util.hs
@@ -1,23 +0,0 @@---- | Utils--module Pdf.Toolbox.Core.Parsers.Util-(-  endOfLine-)-where--import Data.Attoparsec.ByteString (Parser)-import qualified Data.Attoparsec.ByteString.Char8 as P-import Control.Applicative (many)---- | In pdf file EOL could be \"\\n\", \"\\r\" or \"\\n\\r\"------ Also space (0x20) is usually ok before EOL-endOfLine :: Parser ()-endOfLine = do-  _ <- many $ P.char ' '-  P.choice [-    P.endOfLine, -- it already handles both the \n and \n\r-    P.char '\r' >>= const (return ())-    ]
− lib/Pdf/Toolbox/Core/Parsers/XRef.hs
@@ -1,96 +0,0 @@-{-# LANGUAGE OverloadedStrings #-}---- | Parsers for XRef--module Pdf.Toolbox.Core.Parsers.XRef-(-  startXRef,-  tableXRef,-  parseSubsectionHeader,-  parseTrailerAfterTable,-  parseTableEntry-)-where--import Data.Int-import Data.Attoparsec.ByteString (Parser)-import qualified Data.Attoparsec.ByteString.Char8 as P-import Control.Applicative (many)--import Pdf.Toolbox.Core.Object.Types-import Pdf.Toolbox.Core.Parsers.Object-import Pdf.Toolbox.Core.Parsers.Util---- for doctest--- $setup--- >>> :set -XOverloadedStrings--- >>> import Data.Attoparsec.ByteString.Char8---- | Offset of the very last xref table------ Before calling it, make sure your are currently somewhere near--- the end of pdf file. Otherwice it can eat all the memory.--- E.g. examine only the last 1KB------ >>> parseOnly startXRef "anything...startxref\n222\n%%EOF...blah\nstartxref\n123\n%%EOF"--- Right 123-startXRef :: Parser Int64-startXRef = do-  res <- many $ do-    _ <- P.manyTill P.anyChar $ P.string "startxref"-    P.skipSpace-    offset <- P.decimal-    P.skipSpace-    _ <- P.string "%%EOF"-    return offset-  case res of-    [] -> fail "Trailer not found"-    xs -> return $ last xs---- | When current input position points to xref stream--- (or doesn't point to xref at all), the parser will fail.--- When it points to xref table, the parser will succeed--- and input position will point to the first xref subsection------ >>> parseOnly tableXRef "xref\n"--- Right ()--- >>> parseOnly tableXRef "not xref"--- Left "Failed reading: takeWith"-tableXRef :: Parser ()-tableXRef = do-  _ <- P.string "xref"-  endOfLine---- | Parse subsection header, return (the first object index, number of object)------ Input position will point to the first object-parseSubsectionHeader :: Parser (Int, Int)-parseSubsectionHeader = do-  start <- P.decimal-  P.skipSpace-  count <- P.decimal-  endOfLine-  return (start, count)---- | Parse trailer located after XRef table------ Input position should point to the \"trailer\" keyword-parseTrailerAfterTable :: Parser Dict-parseTrailerAfterTable = do-  _ <- P.string "trailer"-  endOfLine-  P.skipSpace-  parseDict---- | Parse XRef table entry. Returns offset, generation and whether the object is free.-parseTableEntry :: Parser (Int64, Int, Bool)-parseTableEntry = do-  offset <- P.decimal-  P.skipSpace-  generation <- P.decimal-  P.skipSpace-  c <- P.anyChar-  case c of-    'n' -> return (offset, generation, False)-    'f' -> return (offset, generation, True)-    _ -> fail $ "error parsing XRef table entry: unknown char: " ++ [c]
− lib/Pdf/Toolbox/Core/Stream.hs
@@ -1,103 +0,0 @@-{-# LANGUAGE  OverloadedStrings #-}---- | Stream related tools--module Pdf.Toolbox.Core.Stream-(-  StreamFilter,-  knownFilters,-  rawStreamContent,-  decodedStreamContent,-  readStream,-  decodeStream-)-where--import Data.Int-import Control.Monad--import Pdf.Toolbox.Core.Object.Types-import Pdf.Toolbox.Core.Object.Util-import Pdf.Toolbox.Core.IO-import Pdf.Toolbox.Core.Parsers.Object-import Pdf.Toolbox.Core.Stream.Filter.Type-import Pdf.Toolbox.Core.Stream.Filter.FlateDecode-import Pdf.Toolbox.Core.Error---- | All stream filters implemented by the toolbox------ Right now it contains only FlateDecode filter-knownFilters :: [StreamFilter]-knownFilters = [flateDecode]---- | Raw content of stream.--- Filters are not applyed------ The 'IS' is valid only until the next 'seek'------ Note: \"Length\" could be an indirect object, but we don't want--- to read indirect objects here. So we require length to be provided-rawStreamContent :: MonadIO m-                 => RIS                 -- ^ random access input stream to read from-                 -> Int                 -- ^ stream length-                 -> Stream Int64        -- ^ stream object to read content for.-                                        -- The payload is offset of stream data-                 -> PdfE m (Stream IS)  -- ^ resulting stream object-rawStreamContent ris len (Stream dict off) = annotateError ("reading raw stream content at offset: " ++ show off) $ do-  seek ris off-  is <- inputStream ris >>= takeBytes (fromIntegral len)-  return $ Stream dict is---- | Decoded stream content------ The 'IS' is valid only until the next 'seek'------ Note: \"Length\" could be an indirect object, that is why--- we cann't read it ourself-decodedStreamContent :: MonadIO m-                     => RIS                -- ^ random input stream to read from-                     -> [StreamFilter]     -- ^ stream filters-                     -> (IS -> IO IS)      -- ^ decryptor-                     -> Int                -- ^ stream length-                     -> Stream Int64       -- ^ stream with offset-                     -> PdfE m (Stream IS)-decodedStreamContent ris filters decryptor len s = rawStreamContent ris len s >>= decodeStream filters decryptor---- | Read 'Stream' at the current position in the 'RIS'-readStream :: MonadIO m => RIS -> PdfE m (Stream Int64)-readStream ris = do-  Stream dict _ <- inputStream ris >>= parse parseIndirectObject >>= toStream . snd-  Stream dict `liftM` tell ris---- | Decode stream content------ The 'IS' is valid only until the next 'RIS' operation-decodeStream :: MonadIO m => [StreamFilter] -> (IS -> IO IS) -> Stream IS -> PdfE m (Stream IS)-decodeStream filters decryptor (Stream dict istream) = annotateError "Can't decode stream" $ do-  is <- liftIO $ decryptor istream-  list <- buildFilterList dict-  Stream dict `liftM` foldM decode is list-  where-  decode is (name, params) = do-    f <- findFilter name-    tryPdfIO $ filterDecode f params is-  findFilter name = tryHead (UnexpectedError $ "Filter not found: " ++ show name) $-    filter ((== name) . filterName) filters--buildFilterList :: Monad m => Dict -> PdfE m [(Name, Maybe Dict)]-buildFilterList dict = do-  f <- lookupDict "Filter" dict `catchE` (const $ return ONull)-  p <- lookupDict "DecodeParms" dict `catchE` (const $ return ONull)-  case (f, p) of-    (ONull, _) -> return []-    (OName fd, ONull) -> return [(fd, Nothing)]-    (OName fd, ODict pd) -> return [(fd, Just pd)]-    (OName fd, OArray (Array [ODict pd])) -> return [(fd, Just pd)]-    (OArray (Array fa), ONull) -> do-      fa' <- mapM fromObject fa-      return $ zip fa' (repeat Nothing)-    (OArray (Array fa), OArray (Array pa)) | length fa == length pa -> do-      fa' <- mapM fromObject fa-      pa' <- mapM fromObject pa-      return $ zip fa' (map Just pa')-    _ -> throwE $ UnexpectedError $ "Can't handle Filter and DecodeParams: (" ++ show f ++ ", " ++ show p ++ ")"
− lib/Pdf/Toolbox/Core/Stream/Filter/FlateDecode.hs
@@ -1,68 +0,0 @@-{-# LANGUAGE OverloadedStrings #-}-{-# LANGUAGE ScopedTypeVariables #-}---- | Flate decode filter--module Pdf.Toolbox.Core.Stream.Filter.FlateDecode-(-  flateDecode-)-where--import Data.Word-import qualified Data.ByteString as BS-import Codec.Zlib-import Control.Error-import Control.Exception-import qualified System.IO.Streams as Streams--import Pdf.Toolbox.Core.IO-import Pdf.Toolbox.Core.Object.Types-import Pdf.Toolbox.Core.Object.Util-import Pdf.Toolbox.Core.Stream.Filter.Type---- | Vary basic implementation. Only PNG-UP prediction is implemented-flateDecode :: StreamFilter-flateDecode = StreamFilter {-  filterName = "FlateDecode",-  filterDecode = \params is -> decode params is >>= catchZlibExceptions-  }--catchZlibExceptions :: IS -> IO IS-catchZlibExceptions is =-  Streams.makeInputStream $-    Streams.read is-    `catch` (\(e :: ZlibException) -> throwIO $ DecodeException $ toException e)--decode :: Maybe Dict -> IS -> IO IS-decode Nothing is = Streams.decompress is-decode (Just dict) is = do-  predictor <- runExceptT $ lookupDict "Predictor" dict-  case predictor of-    Left _ -> Streams.decompress is-    Right p -> do-      p' <- runExceptT $ fromObject p >>= intValue-      case p' of-        Left e -> fail $ "Malformed predictor: " ++ show e-        Right val -> Streams.decompress is >>= unpredict dict val--unpredict :: Dict -> Int -> IS -> IO IS-unpredict _ 1 is = return is-unpredict dict 12 is = do-  c <- runExceptT $ lookupDict "Columns" dict >>= fromObject >>= intValue-  case c of-    Left e -> fail $ "flateDecode: malformed Columns value: " ++ show e-    Right cols -> unpredict12 (cols + 1) is-unpredict _ p _ = fail $ "Unsupported predictor: " ++ show p---- | PGN-UP prediction------ TODO: Hacky solution, rewrite it-unpredict12 :: Int -> IS -> IO IS-unpredict12 cols is = Streams.toList is >>= Streams.fromList . return . BS.pack . step (replicate cols 0) [] . concatMap BS.unpack-  where-  step :: [Word8] -> [Word8] -> [Word8] -> [Word8]-  step _ _ [] = []-  step (c:cs) [] (_:xs) = step cs [c] xs-  step (c:cs) (p:ps) (x:xs) = (x + p) : step cs (c:(x + p):ps) xs-  step [] ps xs = step (reverse ps) [] xs
− lib/Pdf/Toolbox/Core/Stream/Filter/Type.hs
@@ -1,29 +0,0 @@-{-# LANGUAGE DeriveDataTypeable #-}---- | Stream filter--module Pdf.Toolbox.Core.Stream.Filter.Type-(-  StreamFilter(..),-  DecodeException(..)-)-where--import Data.Typeable-import Control.Exception--import Pdf.Toolbox.Core.Object.Types-import Pdf.Toolbox.Core.IO---- | Stream filter-data StreamFilter = StreamFilter {-  filterName :: Name,      -- ^ as \"Filter\" key value in stream dictionary-  filterDecode :: Maybe Dict -> IS -> IO IS    -- ^ decode params -> content -> decoded content-}---- | Exception that should be thrown by the decoder in case of any error--- User code could catch it when reading from decoded stream content-data DecodeException = DecodeException (SomeException)-  deriving (Show, Typeable)--instance Exception DecodeException
− lib/Pdf/Toolbox/Core/Util.hs
@@ -1,61 +0,0 @@---- | Unclassified tools--module Pdf.Toolbox.Core.Util-(-  readObjectAtOffset,-  readCompressedObject-)-where--import Data.Int-import qualified Data.Attoparsec.ByteString.Char8 as Parser-import Control.Monad-import qualified System.IO.Streams as Streams--import Pdf.Toolbox.Core.Object.Types-import Pdf.Toolbox.Core.Parsers.Object-import Pdf.Toolbox.Core.Error-import Pdf.Toolbox.Core.IO---- | Read indirect object at the specified offset-readObjectAtOffset :: MonadIO m-                   => RIS              -- ^ input stream to read from-                   -> Int64            -- ^ object offset-                   -> Int              -- ^ object generation-                   -> PdfE m (Object Int64)-readObjectAtOffset ris off gen = do-  seek ris off-  (Ref _ gen', o) <- inputStream ris >>= parse parseIndirectObject-  unless (gen == gen') $ throwE $ UnexpectedError $ "Generation mismatch, expected: " ++ show gen ++ ", found: " ++ show gen'-  case o of-    ONumber val -> return $ ONumber val-    OBoolean val -> return $ OBoolean val-    OName val -> return $ OName val-    ODict val -> return $ ODict val-    OArray val -> return $ OArray val-    OStr val -> return $ OStr val-    OStream (Stream dict _) -> (OStream . Stream dict) `liftM` tell ris-    ORef _ -> throwE $ UnexpectedError "Indirect object can't be ORef"-    ONull -> return ONull---- | Read object from object stream-readCompressedObject :: MonadIO m-                     => IS         -- ^ input object stream decoded content-                     -> Int64      -- ^ an offset of the first object (\"First\" key in dictionary)-                     -> Int        -- ^ object number to read-                     -> PdfE m (Object ())-readCompressedObject is first num = do-  (is', countConsumed) <- liftIO $ Streams.countInput is-  res <- replicateM (num + 1) $ parse headerP is' :: MonadIO m => PdfE m [(Int, Int64)]-  (_, off) <- tryLast (UnexpectedError $ "readCompressedObject: tryLast: impossible") res-  pos <- liftIO $ countConsumed-  dropExactly (fromIntegral $ first + off - pos) is-  parse parseObject is-  where-  headerP = do-    n <- Parser.decimal-    Parser.skipSpace-    off <- Parser.decimal-    Parser.skipSpace-    return (n, off)
− lib/Pdf/Toolbox/Core/Writer.hs
@@ -1,230 +0,0 @@-{-# LANGUAGE GeneralizedNewtypeDeriving #-}-{-# LANGUAGE OverloadedStrings #-}-{-# LANGUAGE CPP #-}---- | Write PDF files------ It could be used to generate new PDF file--- or to incrementally update the existent one------ To generate new file, first call 'writePdfHeader',--- then a number of 'writeObject' and finally 'writeXRefTable'------ To incrementally update PDF file just ommit the--- `writePdfHeader` and append the result to the existent file--module Pdf.Toolbox.Core.Writer-(-  PdfWriter,-  runPdfWriter,-  writePdfHeader,-  writeObject,-  deleteObject,-  writeXRefTable-)-where--import Data.Int-import Data.Set (Set)-import qualified Data.Set as Set-import Data.ByteString (ByteString)-import qualified Data.ByteString.Lazy as BSL-import Data.ByteString.Lazy.Builder--#if MIN_VERSION_bytestring(0, 10, 4)-#else-import Data.ByteString.Lazy.Builder.ASCII-#endif--import Data.Function-import Control.Monad-import Control.Monad.Trans.Class-import Control.Monad.Trans.State-import Control.Monad.IO.Class-import System.IO.Streams (OutputStream)-import qualified System.IO.Streams as Streams--import Pdf.Toolbox.Core.Object.Types-import Pdf.Toolbox.Core.Object.Builder---- | The monad-newtype PdfWriter m a = PdfWriter (StateT PdfState m a)-  deriving (Functor, Applicative, Monad, MonadIO, MonadTrans)---- | Execute writer action-runPdfWriter :: MonadIO m-             => OutputStream ByteString    -- ^ streams to write to-             -> PdfWriter m a              -- ^ action to run-             -> m a-runPdfWriter output (PdfWriter action) = do-  (out, count) <- liftIO $ Streams.countOutput output-  let emptyState = PdfState {-        stOutput = out,-        stObjects = Set.empty,-        stCount = count,-        stOffset = 0-        }-  evalStateT action emptyState--data Elem = Elem {-  elemIndex :: {-# UNPACK #-} !Int,-  elemGen :: {-# UNPACK #-} !Int,-  elemOffset :: {-# UNPACK #-} !Int64,-  elemFree :: !Bool-  }--instance Eq Elem where-  (==) = (==) `on` elemIndex--instance Ord Elem where-  compare = compare `on` elemIndex--data PdfState = PdfState {-  stOutput :: OutputStream ByteString,-  stObjects :: !(Set Elem),-  stCount :: IO Int64,-  stOffset :: {-# UNPACK #-} !Int64-  }---- | Write PDF header. Used for generating new PDF files.--- Should be the first call. Not used fo incremental updates-writePdfHeader :: MonadIO m => PdfWriter m ()-writePdfHeader = do-  output <- PdfWriter $ gets stOutput-  liftIO $ Streams.write (Just "%PDF-1.7\n") output---- | Write object-writeObject :: MonadIO m => Ref -> Object BSL.ByteString -> PdfWriter m ()-writeObject ref@(Ref index gen) obj = do-  st <- PdfWriter get-  pos <- countWritten-  addElem $ Elem index gen pos False-  dumpObject (stOutput st) ref obj-  return ()---- | Delete object-deleteObject :: MonadIO m => Ref -> Int64 -> PdfWriter m ()-deleteObject (Ref index gen) nextFree =-  addElem $ Elem index gen nextFree True---- | Write xref table. Should be the last call.--- Used for generating and incremental updates.-writeXRefTable :: MonadIO m-               => Int64           -- ^ size of the original PDF file. Should be 0 for new file-               -> Dict            -- ^ trailer-               -> PdfWriter m ()-writeXRefTable offset tr = do-  st <- PdfWriter get-  off <- (+ offset) `liftM` countWritten-  let elems = Set.mapMonotonic (\e -> e {elemOffset = elemOffset e + offset})  $ stObjects st-      content = byteString "xref\n" `mappend`-                buildXRefTable (Set.toAscList elems) `mappend`-                byteString "trailer\n" `mappend`-                buildDict tr `mappend`-                byteString "\nstartxref\n" `mappend`-                int64Dec off `mappend`-                byteString "\n%%EOF\n"-  liftIO $ Streams.writeLazyByteString (toLazyByteString content) (stOutput st)--countWritten :: MonadIO m => PdfWriter m Int64-countWritten = do-  st <- PdfWriter get-  c <- (stOffset st +) `liftM` liftIO (stCount st)-  PdfWriter $ put $ st {stOffset = c}-  return $! c--addElem :: Monad m => Elem -> PdfWriter m ()-addElem e = do-  st <- PdfWriter get-  when (Set.member e $ stObjects st) $ error $ "PdfWriter: attempt to write object with the same index: " ++ show (elemIndex e)-  PdfWriter $ put st {stObjects = Set.insert e $ stObjects st}--dumpObject :: MonadIO m => OutputStream ByteString -> Ref -> Object BSL.ByteString -> m ()-dumpObject out ref o = liftIO $ Streams.writeLazyByteString (toLazyByteString $ buildIndirectObject ref o) out--buildXRefTable :: [Elem] -> Builder-buildXRefTable entries =-  mconcat (map buildXRefSection $ sections entries)-  where-  sections :: [Elem] -> [[Elem]]-  sections [] = []-  sections xs = let (s, rest) = section xs in s : sections rest-  section [] = error "impossible"-  section (x:xs) = go (elemIndex x + 1) [x] xs-    where-    go _ res [] = (reverse res, [])-    go i res (y:ys) =-      if i == elemIndex y-        then go (i + 1) (y : res) ys-        else (reverse res, y:ys)--buildXRefSection :: [Elem] -> Builder-buildXRefSection [] = error "impossible"-buildXRefSection s@(e:_) =-  intDec (elemIndex e) `mappend`-  char7 ' ' `mappend`-  intDec (length s) `mappend`-  char7 '\n' `mappend`-  loop s-  where-  loop (x:xs) =-    buildFixed 10 '0' (elemOffset x) `mappend`-    char7 ' ' `mappend`-    buildFixed 5 '0' (elemGen x) `mappend`-    char7 ' ' `mappend`-    char7 (if elemFree x then 'f' else 'n') `mappend`-    string7 "\r\n" `mappend`-    loop xs-  loop [] = mempty--buildFixed :: Show a => Int -> Char -> a -> Builder-buildFixed len c i =-  let v = take len $ show i-      l = length v-  in string7 $ replicate (len - l) c ++ v--{---- At attempt to do it directly with Set.--- Actually uses 2x memory...-buildXRefTable :: Set Elem -> Builder-buildXRefTable elems-  | Set.null elems = mempty-  | otherwise = buildXRefSection elems--buildXRefSection :: Set Elem -> Builder-buildXRefSection elems =-  intDec (elemIndex start) `mappend`-  char7 ' ' `mappend`-  intDec len `mappend`-  char7 '\n' `mappend`-  section `mappend`-  buildXRefTable rest-  where-  (start, len, rest) = sectionLength elems-  section = buildSection len elems--buildSection :: Int -> Set Elem -> Builder-buildSection 0 _ = mempty-buildSection l els =-  let (x, xs) = Set.deleteFindMin els-  in buildFixed 10 '0' (elemOffset x) `mappend`-     char7 ' ' `mappend`-     buildFixed 5 '0' (elemGen x) `mappend`-     char7 ' ' `mappend`-     char7 (if elemFree x then 'f' else 'n') `mappend`-     string7 "\r\n" `mappend`-     buildSection (l - 1) xs--sectionLength :: Set Elem -> (Elem, Int, Set Elem)-sectionLength els =-  let (x, xs) = Set.deleteFindMin els-      (count, rest) = go 1 (elemIndex x) xs-  in (x, count, rest)-  where-  go n val xs-    | Set.null xs = (n, xs)-    | otherwise = let (next, rest) = Set.deleteFindMin xs-                  in if elemIndex next == val + 1-                       then go (n + 1) (val + 1) rest-                       else (n, xs)--}
− lib/Pdf/Toolbox/Core/XRef.hs
@@ -1,201 +0,0 @@-{-# LANGUAGE OverloadedStrings #-}---- | Cross reference--module Pdf.Toolbox.Core.XRef-(-  XRef(..),-  XRefEntry(..),-  TableEntry(..),-  StreamEntry(..),-  lastXRef,-  prevXRef,-  trailer,-  lookupTableEntry,-  lookupStreamEntry,-  isTable-)-where--import Data.Int-import qualified Data.ByteString as BS-import Control.Monad--import Pdf.Toolbox.Core.Object.Types-import Pdf.Toolbox.Core.Object.Util-import Pdf.Toolbox.Core.IO-import Pdf.Toolbox.Core.Parsers.XRef-import Pdf.Toolbox.Core.Stream-import Pdf.Toolbox.Core.Error---- | Entry in cross reference table-data TableEntry = TableEntry {-  teOffset :: Int64,-  teGen :: Int,-  teIsFree :: Bool-  } deriving Show---- | Entry in cross reference stream-data StreamEntry =-  -- | Object number and generation-  StreamEntryFree Int Int |-  -- | Object offset (in bytes from the beginning of file) and generation-  StreamEntryUsed Int64 Int |-  -- | Object number of object stream and index within the object stream-  StreamEntryCompressed Int Int-  deriving Show---- | Entry in cross reference-data XRefEntry =-  XRefTableEntry TableEntry |-  XRefStreamEntry StreamEntry-  deriving Show---- | Cross reference-data XRef =-  -- | Offset-  XRefTable Int64 |-  -- | Offset and stream with content offset-  XRefStream Int64 (Stream Int64)-  deriving Show---- | Find the last cross reference-lastXRef :: MonadIO m => RIS -> PdfE m XRef-lastXRef ris = annotateError "Can't find the last xref" $ do-  sz <- size ris-  seek ris $ max 0 (sz - 1024)-  off <- inputStream ris >>= parse startXRef-  readXRef ris off--readXRef :: MonadIO m => RIS -> Int64 -> PdfE m XRef-readXRef ris off = do-  seek ris off-  table <- inputStream ris >>= isTable-  if table-    then return $ XRefTable off-    else XRefStream off `liftM` readStream ris---- | Check whether the stream starts with \"xref\" keyword.--- The keyword iyself is consumed-isTable :: MonadIO m => IS -> PdfE m Bool-isTable is = do-  res <- runExceptT (parse tableXRef is)-  case res of-    Right _ -> return True-    Left _ -> return False---- | Find prev cross reference-prevXRef :: MonadIO m => RIS -> XRef -> PdfE m (Maybe XRef)-prevXRef ris xref = annotateError "Can't find prev xref" $ do-  tr <- trailer ris xref-  prev <- runExceptT $ lookupDict "Prev" tr-  case prev of-    Right p -> do-      off <- fromObject p >>= intValue-      Just `liftM` readXRef ris (fromIntegral off)-    Left _ -> return Nothing---- | Read trailer for the xref-trailer :: MonadIO m => RIS -> XRef -> PdfE m Dict-trailer ris (XRefTable off) = annotateError ("Reading trailer for xref table: " ++ show off) $ do-  seek ris off-  inputStream ris >>= \is -> do-    _ <- isTable is-    skipTable is-    parse parseTrailerAfterTable is-trailer _ (XRefStream _ (Stream dict _)) = return dict--skipTable :: MonadIO m => IS -> PdfE m ()-skipTable is =-  subsectionHeader is >>= go . snd-  where-  go count = nextSubsectionHeader is count >>= maybe (return ()) (go . snd)--subsectionHeader :: MonadIO m => IS -> PdfE m (Int, Int)-subsectionHeader = parse parseSubsectionHeader--nextSubsectionHeader :: MonadIO m => IS -> Int -> PdfE m (Maybe (Int, Int))-nextSubsectionHeader is count = do-  skipSubsection is count-  hush `liftM` (runExceptT $ subsectionHeader is)--skipSubsection :: MonadIO m => IS -> Int -> PdfE m ()-skipSubsection is count = dropExactly (count * 20) is---- | Read xref entry for the indirect object from xref table------ RIS position should point to the begining of the next--- line after \"xref\" keyword-lookupTableEntry :: MonadIO m-               => RIS             -- ^ input stream to read from-               -> Ref             -- ^ indirect object to look for-               -> PdfE m (Maybe TableEntry)-lookupTableEntry ris (Ref index gen) = annotateError "Can't read entry from xref table" $-  inputStream ris >>= subsectionHeader >>= go-  where-  go (start, count) = do-    if index >= start && index < start + count-      then do-        tell ris >>= seek ris . (+ (fromIntegral $ index - start) * 20)-        (off, gen', free) <- inputStream ris >>= parse parseTableEntry-        unless (gen == gen') $ throwE $ UnexpectedError "Generation mismatch"-        return $ Just $ TableEntry off gen free-      else do-        is <- inputStream ris-        nextSubsectionHeader is count >>= maybe (return Nothing) go---- | Read xref entry for the indirect object from xref stream------ See pdf1.7 spec: 7.5.8 Cross-Reference Streams-lookupStreamEntry :: MonadIO m-                => Stream IS                -- ^ decoded xref stream content-                -> Ref                      -- ^ indirect object-                -> PdfE m (Maybe StreamEntry)-lookupStreamEntry (Stream dict is) (Ref objNumber _) = annotateError "Can't parse xref stream" $ do-  sz <- lookupDict "Size" dict >>= fromObject >>= intValue--  index <- do-    Array i <- (lookupDict "Index" dict >>= fromObject)-      `catchE`-      const (return $ Array [ONumber $ NumInt 0, ONumber $ NumInt sz])-    let convertIndex res [] = return $ reverse res-        convertIndex res (x1:x2:xs) = do-          from <- fromObject x1 >>= intValue-          count <- fromObject x2 >>= intValue-          convertIndex ((from, count) : res) xs-        convertIndex _ _ = throwE $ UnexpectedError $ "Malformed Index in xref stream: " ++ show i-    convertIndex [] i--  width <- do-    Array w <- lookupDict "W" dict >>= fromObject-    mapM (fromObject >=> intValue) w-  unless (length width == 3) $ throwE $ UnexpectedError $ "Malformed With array in xref stream: " ++ show width--  values <- do-    let position = loop 0 index-        totalWidth = sum width-        loop _ [] = Nothing-        loop pos ((from, count) : xs) =-          if objNumber < from || objNumber >= from + count-            then loop (pos + totalWidth * count) xs-            else Just (pos + totalWidth * (objNumber - from))-    case position of-      Nothing -> return Nothing-      Just p -> dropExactly p is >> (Just . BS.unpack) `liftM` readExactly totalWidth is--  case values of-    Nothing -> return Nothing-    Just vs -> do-      let [v1, v2, v3] = map conv $ collect [] width vs :: [Int64]-            where-            conv l = conv' (length l - 1) 0 l-            conv' _ res [] = res-            conv' power res (x:xs) = conv' (power-1) (res + (fromIntegral x * 256 ^ power)) xs-            collect res [] [] = reverse res-            collect res (x:xs) ys = collect (take x ys : res) xs (drop x ys)-            collect _ _ _ = error "readStreamEntry: collect: impossible"-      case v1 of-        0 -> return $ Just $ StreamEntryFree (fromIntegral v2) (fromIntegral v3)-        1 -> return $ Just $ StreamEntryUsed v2 (fromIntegral v3)-        2 -> return $ Just $ StreamEntryCompressed (fromIntegral v2) (fromIntegral v3)-        _ -> throwE $ UnexpectedError $ "Unexpected xret stream entry type: " ++ show v1
+ no-zlib/Pdf/Core/Stream/Filter/FlateDecode.hs view
@@ -0,0 +1,16 @@++-- | Flate decode filter++module Pdf.Core.Stream.Filter.FlateDecode+(+  flateDecode+)+where++import Pdf.Core.Stream.Filter.Type++-- | Vary basic implementation. Only PNG-UP prediction is implemented+--+-- Nothing when zlib is disabled via cabal flag+flateDecode :: Maybe StreamFilter+flateDecode = Nothing
pdf-toolbox-core.cabal view
@@ -1,5 +1,5 @@ name:                pdf-toolbox-core-version:             0.0.4.1+version:             0.1.1 synopsis:            A collection of tools for processing PDF files. license:             BSD3 license-file:        LICENSE@@ -8,7 +8,7 @@ copyright:           Copyright (c) Yuras Shumovich 2012-2016 category:            PDF build-type:          Simple-cabal-version:       >=1.8+cabal-version:       >=1.10 homepage:            https://github.com/Yuras/pdf-toolbox extra-source-files:  changelog.md description:@@ -28,32 +28,76 @@   type:                git   location:            git://github.com/Yuras/pdf-toolbox.git +flag zlib+  description: Enable deflate support via zlib; requires that the Zlib flag be set on io-streams+  default:             True+  manual:              True+ library   hs-source-dirs:      lib                        compat-  exposed-modules:     Pdf.Toolbox.Core-                       Pdf.Toolbox.Core.IO-                       Pdf.Toolbox.Core.IO.RIS-                       Pdf.Toolbox.Core.Object.Types-                       Pdf.Toolbox.Core.Object.Builder-                       Pdf.Toolbox.Core.Object.Util-                       Pdf.Toolbox.Core.Error-                       Pdf.Toolbox.Core.Parsers.Object-                       Pdf.Toolbox.Core.Parsers.XRef-                       Pdf.Toolbox.Core.Parsers.Util-                       Pdf.Toolbox.Core.Stream.Filter.Type-                       Pdf.Toolbox.Core.Stream.Filter.FlateDecode-                       Pdf.Toolbox.Core.Stream-                       Pdf.Toolbox.Core.XRef-                       Pdf.Toolbox.Core.Util-                       Pdf.Toolbox.Core.Writer+  if flag(zlib)+    hs-source-dirs:    zlib+  else+    hs-source-dirs:    no-zlib+  exposed-modules:     Pdf.Core+                       Pdf.Core.File+                       Pdf.Core.IO.Buffer+                       Pdf.Core.Encryption+                       Pdf.Core.Exception+                       Pdf.Core.Name+                       Pdf.Core.Object+                       Pdf.Core.Object.Builder+                       Pdf.Core.Object.Util+                       Pdf.Core.Parsers.Object+                       Pdf.Core.Parsers.XRef+                       Pdf.Core.Parsers.Util+                       Pdf.Core.Stream+                       Pdf.Core.Stream.Filter.Type+                       Pdf.Core.Stream.Filter.FlateDecode+                       Pdf.Core.Types+                       Pdf.Core.XRef+                       Pdf.Core.Util+                       Pdf.Core.Writer   other-modules:       Prelude   build-depends:       base >= 4.5 && < 5,-                       bytestring,+                       bytestring >= 0.10.4 && < 0.12,+                       base16-bytestring >= 1,                        io-streams,-                       attoparsec >= 0.10,+                       attoparsec >= 0.12,                        scientific,-                       errors >=2.0 && <3.0,-                       transformers,+                       vector,+                       hashable,+                       unordered-containers,                        containers,-                       zlib-bindings+                       cipher-rc4,+                       cipher-aes,+                       crypto-api,+                       cryptohash+  if impl(ghc >= 8.10)+    ghc-options:       -Wunused-packages+  default-language:    Haskell2010++test-suite test+  type:                exitcode-stdio-1.0+  hs-source-dirs:      test+                       compat+  main-is:             test.hs+  other-modules:       Test.XRef+                       Test.Stream+                       Test.Parsers.Object+                       Test.Object.Builder+                       Test.Object.Util+                       Test.Name+                       Prelude+  build-depends:       base,+                       pdf-toolbox-core,+                       bytestring,+                       vector,+                       unordered-containers,+                       attoparsec,+                       io-streams,+                       hspec+  if impl(ghc >= 8.10)+    ghc-options:       -Wunused-packages+  default-language:    Haskell2010
+ test/Test/Name.hs view
@@ -0,0 +1,24 @@+{-# LANGUAGE OverloadedStrings #-}++module Test.Name+(+  spec+)+where++import qualified Pdf.Core.Name as Name++import Data.Either+import Test.Hspec++spec :: Spec+spec = describe "Name" $ do+  makeSpec++makeSpec :: Spec+makeSpec = describe "make" $ do+  it "should wrap bytestring to name" $ do+    Name.make "hello" `shouldSatisfy` isRight++  it "should not allow 0 byte inside a name" $ do+    Name.make "hello\0" `shouldSatisfy` isLeft
+ test/Test/Object/Builder.hs view
@@ -0,0 +1,96 @@+{-# LANGUAGE OverloadedStrings #-}++module Test.Object.Builder+(+  spec,+)+where++import Pdf.Core.Object+import Pdf.Core.Object.Builder++import qualified Data.ByteString.Builder as Builder+import qualified Data.Vector as Vector+import qualified Data.HashMap.Strict as HashMap+import Test.Hspec++spec :: Spec+spec = describe "Object.Builder" $ do+  buildBoolSpec+  buildStringSpec+  buildNameSpec+  buildNumberSpec+  buildArraySpec+  buildDictSpec+  buildRefSpec+  buildStreamSpec++buildBoolSpec :: Spec+buildBoolSpec = describe "buildBool" $ do+  it "should build 'true' for True" $ do+    let res = buildBool True +    Builder.toLazyByteString res `shouldBe` "true"++  it "should build 'false' for False" $ do+    let res = buildBool False+    Builder.toLazyByteString res `shouldBe` "false"++buildStringSpec :: Spec+buildStringSpec = describe "buildString" $ do+  it "should produce literal string when all chars are printable" $ do+    let res = buildString "hello"+    Builder.toLazyByteString res `shouldBe` "(hello)"++  it "should produce hex string when there are not printable chars" $ do+    let res = buildString "\NUL\255"+    Builder.toLazyByteString res `shouldBe` "<00ff>"++  it "should escape special chars" $ do+    let res = buildString "()\\"+    Builder.toLazyByteString res `shouldBe` "(\\(\\)\\\\)"++buildNameSpec :: Spec+buildNameSpec = describe "buildName" $ do+  it "should build a name" $ do+    let res = buildName "hello"+    Builder.toLazyByteString res `shouldBe` "/hello"++buildNumberSpec :: Spec+buildNumberSpec = describe "buildNumber" $ do+  it "should build int" $ do+    let res = buildNumber 42+    Builder.toLazyByteString res `shouldBe` "42"++  it "should build float" $ do+    let res = buildNumber 42.4+    Builder.toLazyByteString res `shouldBe` "42.4"++buildArraySpec :: Spec+buildArraySpec = describe "buildArray" $ do+  it "should build an array" $ do+    let res = buildArray (Vector.fromList [Number 42, Bool False])+    Builder.toLazyByteString res `shouldBe` "[42 false]"++  it "should build empty array" $ do+    let res = buildArray Vector.empty+    Builder.toLazyByteString res `shouldBe` "[]"++buildDictSpec :: Spec+buildDictSpec = describe "buildDict" $ do+  it "should build a dictionary" $ do+    let res = buildDict (HashMap.fromList [("hello", Bool False)])+    Builder.toLazyByteString res `shouldBe` "<</hello false>>"++buildRefSpec :: Spec+buildRefSpec = describe "buildRef" $ do+  it "should build a ref" $ do+    let res = buildRef (R 42 24)+    Builder.toLazyByteString res `shouldBe` "42 24 R"++buildStreamSpec :: Spec+buildStreamSpec = describe "buildStream" $ do+  it "should build a stream" $ do+    let res = buildStream dict "hello"+        dict = HashMap.fromList [("a", String "b")]+    Builder.toLazyByteString res+      `shouldBe` "<</a (b)>>stream\nhello\nendstream"
+ test/Test/Object/Util.hs view
@@ -0,0 +1,109 @@+{-# LANGUAGE OverloadedStrings #-}++module Test.Object.Util+(+  spec,+)+where++import Pdf.Core.Object+import Pdf.Core.Object.Util++import qualified Data.Vector as Vector+import qualified Data.HashMap.Strict as HashMap+import Test.Hspec++spec :: Spec+spec = describe "Object.Util" $ do+  boolValueSpec+  stringValueSpec+  intValueSpec+  realValueSpec+  arrayValueSpec+  dictValueSpec+  nameValueSpec+  refValueSpec+  streamValueSpec++boolValueSpec :: Spec+boolValueSpec = describe "boolValue" $ do+  it "should convert boolean value to Bool" $ do+    boolValue (Bool True) `shouldBe` Just True++  it "should return Nothing for other values" $ do+    boolValue (String "hello") `shouldBe` Nothing++stringValueSpec :: Spec+stringValueSpec = describe "stringValue" $ do+  it "should convert string value to ByteString" $ do+    stringValue (String "hello") `shouldBe` Just "hello"++  it "should return Nothing for other values" $ do+    stringValue (Bool True) `shouldBe` Nothing++intValueSpec :: Spec+intValueSpec = describe "intValue" $ do+  it "should convert int value to Int" $ do+    intValue (Number 42) `shouldBe` Just 42++  it "should not convert float value" $ do+    intValue (Number 42.6) `shouldBe` Nothing++  it "should not convert any other value" $ do+    intValue (Bool True) `shouldBe` Nothing++realValueSpec :: Spec+realValueSpec = describe "realValue" $ do+  it "should convert int value to Float" $ do+    realValue (Number 42) `shouldBe` Just 42.0++  it "should convert float value to Float" $ do+    realValue (Number 42.4) `shouldBe` Just 42.4++  it "should not convert any other value" $ do+    realValue (Bool True) `shouldBe` Nothing++arrayValueSpec :: Spec+arrayValueSpec = describe "arrayValue" $ do+  it "should convert array value to Array" $ do+    let arr = Vector.fromList [Bool True]+    arrayValue (Array arr) `shouldBe` Just arr++  it "should return Nothing for any other value" $ do+    arrayValue (Bool True) `shouldBe` Nothing++dictValueSpec :: Spec+dictValueSpec = describe "dictValue" $ do+  it "should convert dict value to Dict" $ do+    let dict = HashMap.fromList [("hello", Bool True)]+    dictValue (Dict dict) `shouldBe` Just dict++  it "should return Nothing for any other value" $ do+    dictValue (Bool True) `shouldBe` Nothing++nameValueSpec :: Spec+nameValueSpec = describe "nameValue" $ do+  it "should convert name value to Name" $ do+    nameValue (Name "hello") `shouldBe` Just "hello"++  it "should return Nothing for any other value" $ do+    nameValue (Bool True) `shouldBe` Nothing++refValueSpec :: Spec+refValueSpec = describe "refValue" $ do+  it "should convert ref value to Ref" $ do+    let ref = R 42 24+    refValue (Ref ref) `shouldBe` Just ref++  it "should return Nothing for any other value" $ do+    refValue (Bool True) `shouldBe` Nothing++streamValueSpec :: Spec+streamValueSpec = describe "streamValue" $ do+  it "should convert stream value to Stream" $ do+    let stream = S dict 42+        dict = HashMap.fromList [("a", String "b")]+    streamValue (Stream stream) `shouldBe` Just stream++  it "should return Nothing for any other value" $ do+    streamValue (Bool True) `shouldBe` (Nothing :: Maybe Stream)
+ test/Test/Parsers/Object.hs view
@@ -0,0 +1,102 @@+{-# LANGUAGE OverloadedStrings #-}++module Test.Parsers.Object+(+  spec+)+where++import Pdf.Core.Object+import Pdf.Core.Parsers.Object++import Data.Attoparsec.ByteString+import qualified Data.Vector as Vector+import qualified Data.HashMap.Strict as HashMap+import Test.Hspec++spec :: Spec+spec = describe "Parsers.Object" $ do+  parseStringSpec+  parseHexStringSpec+  parseBoolSpec+  parseNameSpec+  parseNumberSpec+  parseArraySpec+  parseDictSpec+  parseRefSpec++parseStringSpec :: Spec+parseStringSpec = describe "parseString" $ do+  it "should unescape 3-digit character" $ do+    parseOnly parseString "(hello\\040world)"+      `shouldBe` Right "hello world"++  it "should unescape 2-digit character" $ do+    parseOnly parseString "(hello\\40world)"+      `shouldBe` Right "hello world"++  it "should unescape 1-digit character" $ do+    parseOnly parseString "(hello\\0world)"+      `shouldBe` Right "hello\NULworld"++  it "should accept nested parens" $ do+    parseOnly parseString "(hello( )world)"+      `shouldBe` Right "hello( )world"++  it "should unescape special chars" $ do+    parseOnly parseString "(\\(\\)\\\\\\n\\f\\r\\t\\b)"+      `shouldBe` Right "()\\\n\f\r\t\b"++parseHexStringSpec :: Spec+parseHexStringSpec = describe "parseHexString" $ do+  it "should parse hex string" $ do+    parseOnly parseHexString "<00FFff>"+      `shouldBe` Right "\NUL\255\255"++parseBoolSpec :: Spec+parseBoolSpec = describe "parseBool" $ do+  it "should parse 'true' as True" $ do+    parseOnly parseBool "true"+      `shouldBe` Right True++  it "should parse 'false' as False" $ do+    parseOnly parseBool "false"+      `shouldBe` Right False++parseNameSpec :: Spec+parseNameSpec = describe "parseName" $ do+  it "should parse a name" $ do+    parseOnly parseName "/hello"+      `shouldBe` Right "hello"++parseNumberSpec :: Spec+parseNumberSpec = describe "parseNumber" $ do+  it "should parse int" $ do+    parseOnly parseNumber "42"+      `shouldBe` Right 42++  it "should parse float" $ do+    parseOnly parseNumber "42.4"+      `shouldBe` Right 42.4++  it "should parse float without leading 0." $ do+    parseOnly parseNumber ".4"+      `shouldBe` Right 0.4++parseArraySpec :: Spec+parseArraySpec = describe "parseArray" $ do+  it "should parse array" $ do+    parseOnly parseArray "[42 true]"+      `shouldBe` Right (Vector.fromList [Number 42, Bool True])++parseDictSpec :: Spec+parseDictSpec = describe "parseDict" $ do+  it "should parse a dictionary" $ do+    parseOnly parseDict "<</hello true>>"+      `shouldBe` Right (HashMap.fromList [("hello", Bool True)])++parseRefSpec :: Spec+parseRefSpec = describe "parseRef" $ do+  it "should parse a reference" $ do+    parseOnly parseRef "42 24 R"+      `shouldBe` Right (R 42 24)
+ test/Test/Stream.hs view
@@ -0,0 +1,23 @@+{-# LANGUAGE OverloadedStrings #-}++module Test.Stream+(+  spec+)+where++import Control.Monad+import qualified System.IO.Streams as Streams+import qualified System.IO.Streams.Attoparsec as Streams++import Pdf.Core.Stream++import Test.Hspec++spec :: Spec+spec = describe "Stream" $ do+  describe "readStream" $ do+    it "should throw ParseException when indirect object is not a stream" $ (do+        is <- Streams.fromByteString "1 1 obj\r(hello)\nendobj"+        void $ readStream is 0+      ) `shouldThrow` \(Streams.ParseException _) -> True
+ test/Test/XRef.hs view
@@ -0,0 +1,183 @@+{-# LANGUAGE OverloadedStrings #-}++module Test.XRef+( spec+)+where++import Pdf.Core.Object+import Pdf.Core.XRef+import Pdf.Core.Exception+import qualified Pdf.Core.IO.Buffer as Buffer++import qualified Data.ByteString as ByteString+import qualified Data.Vector as Vector+import qualified Data.HashMap.Strict as HashMap+import Control.Monad+import qualified System.IO.Streams as Streams+import Test.Hspec++spec :: Spec+spec = describe "XRef" $ do+  describe "isTable" $ do+    it "should return True when the stream starts from \"xref\\n\" string" $+      (Streams.fromByteString "xref\n" >>= isTable)+        `shouldReturn` True++    it "should return False when the stream doesn't start from \"xref\\n\"" $+      (Streams.fromByteString "_xref\n" >>= isTable)+        `shouldReturn` False++    it "should consume \"xref\\n\" prefix" $ (do+      is <- Streams.fromByteString "xref\nhello"+      void $ isTable is+      Streams.readExactly 5 is+      ) `shouldReturn` "hello"+++  describe "readXRef" $ do+    it "should support xref table" $ (do+      buf <- Buffer.fromBytes "helloxref\nworld"+      readXRef buf 5+      ) `shouldReturn` XRefTable 5++    it "should support xref stream" $ (do+      buf <- Buffer.fromBytes "hello1 1 obj\n<<>>stream\r\ncontent"+      readXRef buf 5+      ) `shouldReturn` XRefStream 5 (S HashMap.empty 25)++    it "should throw exception if xref not found" $ (do+      buf <- Buffer.fromBytes "hello\n"+      readXRef buf 0+      ) `shouldThrow` anyException+++  describe "lastXRef" $ do+    it "should find the latest xref" $ (+      Buffer.fromBytes "helloxref\nxref\nstartxref\n10\n%%EOF\+        \worldstartxref\n5\n%%EOF"+      >>= lastXRef+      ) `shouldReturn` XRefTable 5++    it "should throw Corrupted when xref not found" $ (+      Buffer.fromBytes "helloxref\n%%EOF"+      >>= lastXRef+      ) `shouldThrow` \Corrupted{} -> True+++  describe "trailer" $ do+    it "should return the dictionary for xref stream" $+      let dict = HashMap.fromList [("Hello", String "World")]+      in trailer undefined (XRefStream 0 (S dict 0))+        `shouldReturn` dict++    it "should parse trailer after xref table" $ (do+      buf <- Buffer.fromBytes "helloxref\n1 1\n0000000001 00000 n\r\n\+        \trailer\n<</Hello(world)>>"+      trailer buf (XRefTable 5)+      ) `shouldReturn` HashMap.fromList [("Hello", String "world")]++    it "should handle multisection table" $ (do+      buf <- Buffer.fromBytes "helloxref\n1 1\n0000000001 00000 n\r\n\+        \1 1\n0000000002 00000 n\r\ntrailer\n<</Hello(world)>>"+      trailer buf (XRefTable 5)+      ) `shouldReturn` HashMap.fromList [("Hello", String "world")]++    it "should throw Corrupted exception if can't parse" $ (do+      buf <- Buffer.fromBytes "helloxref\n1 Hello(world)>>"+      trailer buf (XRefTable 5)+      ) `shouldThrow` \Corrupted{} -> True+++  describe "prevXRef" $ do+    it "should read xref located at offset from\+        \ Prev entry in current trailer" $ (do+      let dict = HashMap.fromList [("Prev", Number 5)]+      buf <- Buffer.fromBytes "helloxref\n"+      prevXRef buf (XRefStream undefined (S dict undefined))+      ) `shouldReturn` Just (XRefTable 5)++    it "should return Nothing for the last xref" $ (do+      let dict = HashMap.fromList []+      buf <- Buffer.fromBytes "helloxref\n"+      prevXRef buf (XRefStream undefined (S dict undefined))+      ) `shouldReturn` Nothing++    it "should throw Corrupted when Prev is not an int" $ (do+      let dict = HashMap.fromList [("Prev", String "hello")]+      buf <- Buffer.fromBytes "helloxref\n"+      prevXRef buf (XRefStream undefined (S dict undefined))+      ) `shouldThrow` \Corrupted{} -> True++  describe "lookupTableEntry" $ do+    it "should look for the entry in subsections" $ (do+      buf <- Buffer.fromBytes "helloxref\n\+        \1 2\n\+        \0000000011 00000 n\r\n\+        \0000000022 00000 n\r\n\+        \3 2\n\+        \0000000033 00000 n\r\n\+        \0000000044 00000 n\r\n\+        \trailer"+      lookupTableEntry buf (XRefTable 5) (R 4 0)+      ) `shouldReturn` Just (EntryUsed 44 0)++    it "should return free entry" $ (do+      buf <- Buffer.fromBytes "helloxref\n\+        \1 2\n\+        \0000000011 00000 n\r\n\+        \0000000022 00001 f\r\n\+        \trailer"+      lookupTableEntry buf (XRefTable 5) (R 2 0)+      ) `shouldReturn` Just (EntryFree 22 0)++    it "should return Nothing when not found" $ (do+      buf <- Buffer.fromBytes "helloxref\n\+        \1 2\n\+        \0000000011 00000 n\r\n\+        \0000000022 00000 n\r\n\+        \trailer"+      lookupTableEntry buf (XRefTable 5) (R 4 0)+      ) `shouldReturn` Nothing++  describe "lookupStreamEntry" $ do+    let bytes = ByteString.pack+          [ 0,  0, 1,  2+          , 1,  0, 2,  3+          , 2,  0, 3,  4+          , 0,  0, 4,  0+          ]+        dict = HashMap.fromList+          [ ("Index", Array $ Vector.fromList $ map Number [3, 4])+          , ("W", Array $ Vector.fromList $ map Number [1, 2, 1])+          , ("Size", Number 4)+          ]+    it "should handle free objects" $ (do+      is <- Streams.fromByteString bytes+      lookupStreamEntry dict is (R 6 0)+      ) `shouldReturn` Just (EntryFree 4 0)++    it "should handle used objects" $ (do+      is <- Streams.fromByteString bytes+      lookupStreamEntry dict is (R 4 0)+      ) `shouldReturn` Just (EntryUsed 2 3)++    it "should handle compressed objects" $ (do+      is <- Streams.fromByteString bytes+      lookupStreamEntry dict is (R 5 0)+      ) `shouldReturn` Just (EntryCompressed 3 4)++    it "should return Nothing when object to found" $ (do+      is <- Streams.fromByteString bytes+      lookupStreamEntry dict is (R 7 0)+      ) `shouldReturn` Nothing++    it "should handle multiple sections" $ (do+      let dict' = HashMap.fromList+            [ ("Index", Array $ Vector.fromList $ map Number [3, 2, 10, 2])+            , ("W", Array $ Vector.fromList $ map Number [1, 2, 1])+            , ("Size", Number 4)+            ]+      is <- Streams.fromByteString bytes+      lookupStreamEntry dict' is (R 11 0)+      ) `shouldReturn` Just (EntryFree 4 0)
+ test/test.hs view
@@ -0,0 +1,24 @@++module Main+(+  main+)+where++import qualified Test.XRef+import qualified Test.Stream+import qualified Test.Parsers.Object+import qualified Test.Object.Builder+import qualified Test.Object.Util+import qualified Test.Name++import Test.Hspec++main :: IO ()+main = hspec $ do+  Test.XRef.spec+  Test.Stream.spec+  Test.Parsers.Object.spec+  Test.Object.Builder.spec+  Test.Object.Util.spec+  Test.Name.spec
+ zlib/Pdf/Core/Stream/Filter/FlateDecode.hs view
@@ -0,0 +1,73 @@+{-# LANGUAGE OverloadedStrings #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE PatternGuards #-}++-- | Flate decode filter++module Pdf.Core.Stream.Filter.FlateDecode+(+  flateDecode+)+where++import Data.Word+import Data.ByteString (ByteString)+import qualified Data.ByteString as ByteString+import qualified Data.HashMap.Strict as HashMap+import Control.Exception hiding (throw)+import System.IO.Streams (InputStream)+import qualified System.IO.Streams as Streams++import Pdf.Core.Exception+import Pdf.Core.Object+import Pdf.Core.Object.Util+import Pdf.Core.Stream.Filter.Type++-- | Vary basic implementation. Only PNG-UP prediction is implemented+--+-- Nothing when zlib is disabled via cabal flag+flateDecode :: Maybe StreamFilter+flateDecode = Just StreamFilter+  { filterName = "FlateDecode"+  , filterDecode = decode+  }++decode :: Maybe Dict -> InputStream ByteString -> IO (InputStream ByteString)+decode Nothing is = Streams.decompress is+decode (Just dict) is =+  case HashMap.lookup "Predictor" dict of+    Nothing -> Streams.decompress is+    Just o | Just val <- intValue o ->+      Streams.decompress is >>= unpredict dict val+    _ -> throwIO $ Corrupted "Predictor should be an integer" []++unpredict :: Dict+          -> Int+          -> InputStream ByteString+          -> IO (InputStream ByteString)+unpredict _ 1 is = return is+unpredict dict 12 is = message "unpredict" $+  case HashMap.lookup "Columns" dict of+    Nothing -> throwIO $ Corrupted "Column is missing" []+    Just o+      | Just cols <- intValue o+      -> unpredict12 (cols + 1) is+    _ -> throwIO $ Corrupted "Column should be an integer" []+unpredict _ p _ = throwIO $ Unexpected ("Unsupported predictor: " ++ show p) []++-- | PGN-UP prediction+--+-- TODO: Hacky solution, rewrite it+unpredict12 :: Int -> InputStream ByteString -> IO (InputStream ByteString)+unpredict12 cols is+  = Streams.toList is+  >>= Streams.fromList . return+                       . ByteString.pack+                       . step (replicate cols 0) []+                       . concatMap ByteString.unpack+  where+  step :: [Word8] -> [Word8] -> [Word8] -> [Word8]+  step _ _ [] = []+  step (c:cs) [] (_:xs) = step cs [c] xs+  step (c:cs) (p:ps) (x:xs) = (x + p) : step cs (c:(x + p):ps) xs+  step [] ps xs = step (reverse ps) [] xs