lz4-frame-conduit 0.1.0.1 → 0.1.0.2
raw patch · 5 files changed
+80/−53 lines, 5 filesPVP: major bump suggested
API removals or changes: PVP suggests a major version bump
API changes (from Hackage documentation)
- Codec.Compression.LZ4.Conduit: decompress :: (MonadUnliftIO m, MonadResource m) => ConduitT ByteString ByteString m ()
+ Codec.Compression.LZ4.Conduit: decompress :: forall m. (MonadUnliftIO m, MonadResource m) => ConduitT ByteString ByteString m ()
Files
- CHANGELOG.md +4/−0
- README.md +4/−0
- lz4-frame-conduit.cabal +2/−1
- src/Codec/Compression/LZ4/CTypes.hsc +11/−0
- src/Codec/Compression/LZ4/Conduit.hsc +59/−52
CHANGELOG.md view
@@ -1,3 +1,7 @@+# 0.1.0.2++* GHC 9.8.4 compatibility.+ # 0.1.0.1 * Fix incorrect compression logic (compressing input Bytes repeatedly).
README.md view
@@ -2,6 +2,10 @@ Haskell Conduit implementing the official LZ4 frame streaming format. +## Building++Use `git clone --recursive`, because `lz4/` is a git submodule.+ ## Rationale and comparison to non-`lz4`-compatible libraries There exist two `lz4` formats:
lz4-frame-conduit.cabal view
@@ -1,5 +1,5 @@ name: lz4-frame-conduit-version: 0.1.0.1+version: 0.1.0.2 synopsis: Conduit implementing the official LZ4 frame streaming format description: Conduit implementing the official LZ4 frame streaming format.@@ -14,6 +14,7 @@ category: Web build-type: Simple cabal-version: >=1.10+tested-with: GHC == 8.6.5, GHC == 9.8.4 extra-source-files: README.md CHANGELOG.md
src/Codec/Compression/LZ4/CTypes.hsc view
@@ -3,6 +3,17 @@ {-# LANGUAGE QuasiQuotes #-} {-# LANGUAGE TemplateHaskell #-} +{-|+Module : Codec.Compression.LZ4.CTypes+Description : C type definitions for the lz4 compression codec.+Copyright : (c) Niklas Hambüchen, 2020+License : MIT+Maintainer : mail@nh2.me+Stability : stable++-}++ module Codec.Compression.LZ4.CTypes ( Lz4FrameException(..) , BlockSizeID(..)
src/Codec/Compression/LZ4/Conduit.hsc view
@@ -1,15 +1,60 @@ {-# LANGUAGE LambdaCase #-} {-# LANGUAGE MultiWayIf #-}-{-# LANGUAGE PartialTypeSignatures #-} {-# LANGUAGE QuasiQuotes #-} {-# LANGUAGE ScopedTypeVariables #-} {-# LANGUAGE TemplateHaskell #-} {-# OPTIONS_GHC -fno-warn-partial-type-signatures #-} --- | TODO: Implement:------ * Block checksumming--- * Dictionary support+{-|+Module : Codec.Compression.LZ4.Conduit+Description : Conduit for the lz4 compression codec.+Copyright : (c) Niklas Hambüchen, 2020+License : MIT+Maintainer : mail@nh2.me+Stability : stable+++Help Wanted / TODOs++Please feel free to send me a pull request for any of the following items:++* TODO Block checksumming++* TODO Dictionary support++* TODO Performance:+ Write a version of `compress` that emits ByteStrings of known+ constant length. That will allow us to do compression in a zero-copy+ fashion, writing compressed bytes directly into a the ByteStrings+ (e.g using `unsafePackMallocCString` or equivalent).+ We currently don't do that (instead, use allocaBytes + copying packCStringLen)+ to ensure that the ByteStrings generated are as compact as possible+ (for the case that `written < size`), since the current `compress`+ conduit directly yields the outputs of LZ4F_compressUpdate()+ (unless they are of 0 length when they are buffered in the context+ tmp buffer).++* TODO Try enabling checksums, then corrupt a bit and see if lz4c detects it.++* TODO Add `with*` style bracketed functions for creating the+ LZ4F_createCompressionContext and Lz4FramePreferencesPtr+ for prompt resource release,+ in addition to the GC'd variants below.+ This would replace our use of `finalizeForeignPtr` in the conduit.+ `finalizeForeignPtr` seems almost as good, but note that it+ doesn't guarantee prompt resource release on exceptions;+ a `with*` style function that uses `bracket` does.+ However, it isn't clear yet which one would be faster+ (what the cost of `mask` is compared to foreign pointer finalizers).+ Also note that prompt freeing has side benefits,+ such as reduced malloc() fragmentation (the closer malloc()+ and free() are to each other, the smaller is the chance to+ have malloc()s on top of the our malloc() in the heap,+ thus the smaller the chance that we cannot decrease the+ heap pointer upon free() (because "mallocs on top" render+ heap memory unreturnable to the OS; memory fragmentation).+-}+ module Codec.Compression.LZ4.Conduit ( Lz4FrameException(..) , BlockSizeID(..)@@ -167,43 +212,6 @@ return () ---- TODO Performance:--- Write a version of `compress` that emits ByteStrings of known--- constant length. That will allow us to do compression in a zero-copy--- fashion, writing compressed bytes directly into a the ByteStrings--- (e.g using `unsafePackMallocCString` or equivalent).--- We currently don't do that (instead, use allocaBytes + copying packCStringLen)--- to ensure that the ByteStrings generated are as compact as possible--- (for the case that `written < size`), since the current `compress`--- conduit directly yields the outputs of LZ4F_compressUpdate()--- (unless they are of 0 length when they are buffered in the context--- tmp buffer).---- TODO Try enabling checksums, then corrupt a bit and see if lz4c detects it.---- TODO Add `with*` style bracketed functions for creating the--- LZ4F_createCompressionContext and Lz4FramePreferencesPtr--- for prompt resource release,--- in addition to the GC'd variants below.--- This would replace our use of `finalizeForeignPtr` in the conduit.--- `finalizeForeignPtr` seems almost as good, but note that it--- doesn't guarantee prompt resource release on exceptions;--- a `with*` style function that uses `bracket` does.--- However, it isn't clear yet which one would be faster--- (what the cost of `mask` is compared to foreign pointer finalizers).--- Also note that prompt freeing has side benefits,--- such as reduced malloc() fragmentation (the closer malloc()--- and free() are to each other, the smaller is the chance to--- have malloc()s on top of the our malloc() in the heap,--- thus the smaller the chance that we cannot decrease the--- heap pointer upon free() (because "mallocs on top" render--- heap memory unreturnable to the OS; memory fragmentation).----- TODO Turn the above TODO into documentation-- withScopedLz4fCompressionContext :: (HasCallStack) => (ScopedLz4FrameCompressionContext -> IO a) -> IO a withScopedLz4fCompressionContext f = bracket@@ -296,9 +304,9 @@ } |] +-- | TODO allow passing in cOptPtr instead of NULL. lz4fCompressUpdate :: (HasCallStack) => ScopedLz4FrameCompressionContext -> Ptr CChar -> CSize -> Ptr CChar -> CSize -> IO CSize lz4fCompressUpdate (ScopedLz4FrameCompressionContext ctx) destBuf destBufLen srcBuf srcBufLen = do- -- TODO allow passing in cOptPtr instead of NULL. written <- handleLz4Error [C.block| size_t { size_t err_or_written = LZ4F_compressUpdate($(LZ4F_cctx* ctx), $(char* destBuf), $(size_t destBufLen), $(char* srcBuf), $(size_t srcBufLen), NULL); return err_or_written;@@ -306,9 +314,9 @@ return written +-- | TODO allow passing in cOptPtr instead of NULL. lz4fCompressEnd :: (HasCallStack) => ScopedLz4FrameCompressionContext -> Ptr CChar -> CSize -> IO CSize lz4fCompressEnd (ScopedLz4FrameCompressionContext ctx) footerBuf footerBufLen = do- -- TODO allow passing in cOptPtr instead of NULL. footerWritten <- handleLz4Error [C.block| size_t { size_t err_or_footerWritten = LZ4F_compressEnd($(LZ4F_cctx* ctx), $(char* footerBuf), $(size_t footerBufLen), NULL); return err_or_footerWritten;@@ -316,7 +324,7 @@ return footerWritten --- Note [Single call to LZ4F_compressUpdate() can create multiple blocks]+-- | Note [Single call to LZ4F_compressUpdate() can create multiple blocks] -- A single call to LZ4F_compressUpdate() can create multiple blocks, -- and handles buffers > 32-bit sizes; see: -- https://github.com/lz4/lz4/blob/52cac9a97342641315c76cfb861206d6acd631a8/lib/lz4frame.c#L601@@ -527,8 +535,7 @@ --- All notes that apply to `haskell_lz4_freeCompressionContext` apply--- here as well.+-- | All notes that apply to `haskell_lz4_freeCompressionContext` apply here as well. C.verbatim [r| void haskell_lz4_freeDecompressionContext(LZ4F_dctx** ctxPtr) {@@ -548,10 +555,9 @@ foreign import ccall "&haskell_lz4_freeDecompressionContext" haskell_lz4_freeDecompressionContext :: FunPtr (Ptr (Ptr LZ4F_dctx) -> IO ()) +-- | All notes that apply to `lz4fCreateCompressonContext` apply here as well. lz4fCreateDecompressionContext :: (HasCallStack) => IO Lz4FrameDecompressionContext lz4fCreateDecompressionContext = do- -- All notes that apply to `lz4fCreateCompressonContext` apply here- -- as well. ctxForeignPtr :: ForeignPtr (Ptr LZ4F_dctx) <- mallocForeignPtr withForeignPtr ctxForeignPtr $ \ptr -> poke ptr nullPtr @@ -576,9 +582,9 @@ return decompressSizeHint +-- TODO allow passing in dOptPtr instead of NULL. lz4fDecompress :: (HasCallStack) => Lz4FrameDecompressionContext -> Ptr CChar -> Ptr CSize -> Ptr CChar -> Ptr CSize -> IO CSize lz4fDecompress (Lz4FrameDecompressionContext ctxForeignPtr) dstBuffer dstSizePtr srcBuffer srcSizePtr = do- -- TODO allow passing in dOptPtr instead of NULL. decompressSizeHint <- handleLz4Error [C.block| size_t { LZ4F_dctx* ctxPtr = *$fptr-ptr:(LZ4F_dctx** ctxForeignPtr);@@ -588,7 +594,8 @@ return decompressSizeHint -decompress :: (MonadUnliftIO m, MonadResource m) => ConduitT ByteString ByteString m ()+-- | TODO check why decompressSizeHint is always 4+decompress :: forall m . (MonadUnliftIO m, MonadResource m) => ConduitT ByteString ByteString m () decompress = do ctx <- liftIO lz4fCreateDecompressionContext @@ -651,7 +658,7 @@ poke dstBufferSizePtr size peek dstBufferPtr - let loopSingleBs :: CSize -> ByteString -> _+ let loopSingleBs :: CSize -> ByteString -> ConduitT ByteString ByteString m CSize loopSingleBs decompressSizeHint bs = do (outBs, srcRead, newDecompressSizeHint) <- liftIO $ unsafeUseAsCStringLen bs $ \(srcBuffer, srcSize) -> do