hpdft 0.4.7.1 → 0.4.7.2
raw patch · 6 files changed
+65/−13 lines, 6 filesPVP: minor bump suggested
API additions: PVP suggests at least a minor version bump
API changes (from Hackage documentation)
+ PDF.Object: displayPdfHex :: Text -> String
Files
- CHANGELOG.md +10/−0
- README.md +2/−2
- hpdft.cabal +1/−1
- src/PDF/Object.hs +24/−1
- src/PDF/Outlines.hs +16/−9
- test/Unit.hs +12/−0
CHANGELOG.md view
@@ -1,5 +1,15 @@ # Changelog +## 0.4.7.2 (2026-09-29)++### Fixed++- Outlines: decode `/Title` stored as `PdfHex` (including via indirect objects) for `hpdft toc` / `-O`, using the same rules as hex string parsing instead of `Show` output.++### Added++- `PDF.Object.displayPdfHex` for human-readable text from a parsed hex string object.+ ## 0.4.7.1 (2026-08-10) ### Fixed
README.md view
@@ -128,5 +128,5 @@ ## Version -Released: **0.4.7.1** (2026-08-10) — Form/stream extraction fixes, layout heuristics, and aligned paragraph diff.-Previous release: **0.4.7.0**.+Released: **0.4.7.2** (2026-09-29) — Outline (`toc` / `-O`) titles from hex strings decode correctly.+Previous release: **0.4.7.1**.
hpdft.cabal view
@@ -1,6 +1,6 @@ cabal-version: 3.8 name: hpdft-version: 0.4.7.1+version: 0.4.7.2 synopsis: PDF parsing library and CLI for text, layout, diff, images, and forms description: hpdft is a Haskell library and command-line tool for parsing PDF files.
src/PDF/Object.hs view
@@ -13,6 +13,7 @@ module PDF.Object ( parsePdfLetters+ , displayPdfHex , parsePDFObj , parseRefsArray , pdfObj@@ -26,7 +27,7 @@ , xref ) where -import Data.Char (chr)+import Data.Char (chr, isHexDigit) import qualified Data.Map as M import qualified Data.ByteString.Char8 as BS import qualified Data.ByteString.Lazy.Char8 as BSL@@ -391,6 +392,28 @@ pdfhexletter s = case parseOnly (concat <$> many1 pdfhexutf16be) s of Right t -> utf16be t Left e -> BS.unpack s++-- | Human-readable form of a parsed 'PdfHex' value (matches 'pdfhex' / 'pdfhexSec').+displayPdfHex :: T.Text -> String+displayPdfHex h+ | T.all isHexDigit h = displayHexDigitString (BS.pack (T.unpack h))+ | otherwise = T.unpack h++displayHexDigitString :: BS.ByteString -> String+displayHexDigitString lets =+ case parseOnly+ ( (try $ string "feff" <|> string "FEFF")+ *> many1 (oneOf "0123456789abcdefABCDEF")+ )+ lets of+ Right s -> pdfhexletter (BS.pack s)+ Left _ -> displayPdfHexBytes (decodeHexBytes lets)++displayPdfHexBytes :: BS.ByteString -> String+displayPdfHexBytes decrypted =+ case parseOnly parsePdfLetters (BS.cons '(' (BS.snoc decrypted ')')) of+ Right t -> T.unpack t+ Left _ -> BS.unpack decrypted pdfhexutf16be :: Parser String pdfhexutf16be = do
src/PDF/Outlines.hs view
@@ -25,7 +25,7 @@ import PDF.Document (Document(..), openDocument, docRootRef) import PDF.DocumentStructure import PDF.Error (PdfError(..), PdfResult)-import PDF.Object (parseRefsArray, parsePdfLetters)+import PDF.Object (parseRefsArray, parsePdfLetters, displayPdfHex) import qualified Data.Text as T @@ -231,15 +231,22 @@ findTitle :: Dict -> PDFObjIndex -> PdfResult String findTitle dict objs = case findObjFromDict dict "/Title" of- Just (PdfText s) -> case parseOnly parsePdfLetters (BS.pack (T.unpack s)) of- Right t -> Right (T.unpack t)- Left _ -> Right (T.unpack s)- Just (ObjRef r) -> case findObjsByRef r objs of- Just [PdfText s] -> Right (T.unpack s)- Just s -> Left (ParseError ("Unknown title object: " ++ show s) BS.empty)- Nothing -> Left (MissingObject r)- Just x -> Right (show x) Nothing -> Left (MissingKey "/Title" "outline")+ Just o -> titleFromObj o objs++titleFromObj :: Obj -> PDFObjIndex -> PdfResult String+titleFromObj (PdfText s) _ =+ case parseOnly parsePdfLetters (BS.pack (T.unpack s)) of+ Right t -> Right (T.unpack t)+ Left _ -> Right (T.unpack s)+titleFromObj (PdfHex h) _ = Right (displayPdfHex h)+titleFromObj (ObjRef r) objs =+ case findObjsByRef r objs of+ Just (o : _) -> titleFromObj o objs+ Just [] -> Left (ParseError "Empty title object" BS.empty)+ Nothing -> Left (MissingObject r)+titleFromObj o _ =+ Left (ParseError ("Unknown title object: " ++ show o) BS.empty) listToMaybe :: [a] -> Maybe a listToMaybe (x:_) = Just x
test/Unit.hs view
@@ -22,6 +22,7 @@ , extractPageImages ) import PDF.FormExtract (pageFormNames, extractFormPdf)+import PDF.Object (displayPdfHex) import PDF.Text (pdfToTextTaggedBS, pdfToTextDoc, pdfToTextStreamDoc) import PDF.Error (PdfResult) @@ -200,6 +201,7 @@ ++ taggedEndToEndResults ++ textStreamResults ++ cmapEncodingResults+ ++ pdfHexTitleResults ++ normalizePdfNumberResults ++ heightSpecResults ++ encryptSpecResults@@ -1497,6 +1499,16 @@ , assertBool "bytesToCodes JISmap 2-byte fixed" (bytesToCodes jfi [0x46, 0x7C, 0x4B, 0x5C] == [0x467C, 0x4B5C]) ]++pdfHexTitleResults :: [Result]+pdfHexTitleResults =+ [ assertBool "displayPdfHex already decoded"+ (displayPdfHex (T.pack "Hello") == "Hello")+ , assertBool "displayPdfHex UTF-16BE hex digits"+ (displayPdfHex (T.pack "FEFF00480065006C006C006F") == "Hello")+ , assertBool "displayPdfHex latin1 hex digits"+ (displayPdfHex (T.pack "48656c6c6f") == "Hello")+ ] normalizePdfNumberResults :: [Result] normalizePdfNumberResults =