packages feed

hpdft 0.4.7.1 → 0.4.7.2

raw patch · 6 files changed

+65/−13 lines, 6 filesPVP: minor bump suggested

API additions: PVP suggests at least a minor version bump

API changes (from Hackage documentation)

+ PDF.Object: displayPdfHex :: Text -> String

Files

CHANGELOG.md view
@@ -1,5 +1,15 @@ # Changelog +## 0.4.7.2 (2026-09-29)++### Fixed++- Outlines: decode `/Title` stored as `PdfHex` (including via indirect objects) for `hpdft toc` / `-O`, using the same rules as hex string parsing instead of `Show` output.++### Added++- `PDF.Object.displayPdfHex` for human-readable text from a parsed hex string object.+ ## 0.4.7.1 (2026-08-10)  ### Fixed
README.md view
@@ -128,5 +128,5 @@  ## Version -Released: **0.4.7.1** (2026-08-10) — Form/stream extraction fixes, layout heuristics, and aligned paragraph diff.-Previous release: **0.4.7.0**.+Released: **0.4.7.2** (2026-09-29) — Outline (`toc` / `-O`) titles from hex strings decode correctly.+Previous release: **0.4.7.1**.
hpdft.cabal view
@@ -1,6 +1,6 @@ cabal-version:       3.8 name:                hpdft-version:             0.4.7.1+version:             0.4.7.2 synopsis:            PDF parsing library and CLI for text, layout, diff, images, and forms description:     hpdft is a Haskell library and command-line tool for parsing PDF files.
src/PDF/Object.hs view
@@ -13,6 +13,7 @@  module PDF.Object   ( parsePdfLetters+  , displayPdfHex   , parsePDFObj   , parseRefsArray   , pdfObj@@ -26,7 +27,7 @@   , xref   ) where -import Data.Char (chr)+import Data.Char (chr, isHexDigit) import qualified Data.Map as M import qualified Data.ByteString.Char8 as BS import qualified Data.ByteString.Lazy.Char8 as BSL@@ -391,6 +392,28 @@ pdfhexletter s = case parseOnly (concat <$> many1 pdfhexutf16be) s of   Right t -> utf16be t   Left e -> BS.unpack s++-- | Human-readable form of a parsed 'PdfHex' value (matches 'pdfhex' / 'pdfhexSec').+displayPdfHex :: T.Text -> String+displayPdfHex h+  | T.all isHexDigit h = displayHexDigitString (BS.pack (T.unpack h))+  | otherwise = T.unpack h++displayHexDigitString :: BS.ByteString -> String+displayHexDigitString lets =+  case parseOnly+         ( (try $ string "feff" <|> string "FEFF")+           *> many1 (oneOf "0123456789abcdefABCDEF")+         )+         lets of+    Right s -> pdfhexletter (BS.pack s)+    Left _  -> displayPdfHexBytes (decodeHexBytes lets)++displayPdfHexBytes :: BS.ByteString -> String+displayPdfHexBytes decrypted =+  case parseOnly parsePdfLetters (BS.cons '(' (BS.snoc decrypted ')')) of+    Right t -> T.unpack t+    Left _  -> BS.unpack decrypted  pdfhexutf16be :: Parser String pdfhexutf16be = do
src/PDF/Outlines.hs view
@@ -25,7 +25,7 @@ import PDF.Document (Document(..), openDocument, docRootRef) import PDF.DocumentStructure import PDF.Error (PdfError(..), PdfResult)-import PDF.Object (parseRefsArray, parsePdfLetters)+import PDF.Object (parseRefsArray, parsePdfLetters, displayPdfHex)  import qualified Data.Text as T @@ -231,15 +231,22 @@ findTitle :: Dict -> PDFObjIndex -> PdfResult String findTitle dict objs =   case findObjFromDict dict "/Title" of-    Just (PdfText s) -> case parseOnly parsePdfLetters (BS.pack (T.unpack s)) of-      Right t -> Right (T.unpack t)-      Left _  -> Right (T.unpack s)-    Just (ObjRef r) -> case findObjsByRef r objs of-      Just [PdfText s] -> Right (T.unpack s)-      Just s -> Left (ParseError ("Unknown title object: " ++ show s) BS.empty)-      Nothing -> Left (MissingObject r)-    Just x -> Right (show x)     Nothing -> Left (MissingKey "/Title" "outline")+    Just o  -> titleFromObj o objs++titleFromObj :: Obj -> PDFObjIndex -> PdfResult String+titleFromObj (PdfText s) _ =+  case parseOnly parsePdfLetters (BS.pack (T.unpack s)) of+    Right t -> Right (T.unpack t)+    Left _  -> Right (T.unpack s)+titleFromObj (PdfHex h) _ = Right (displayPdfHex h)+titleFromObj (ObjRef r) objs =+  case findObjsByRef r objs of+    Just (o : _) -> titleFromObj o objs+    Just []      -> Left (ParseError "Empty title object" BS.empty)+    Nothing      -> Left (MissingObject r)+titleFromObj o _ =+  Left (ParseError ("Unknown title object: " ++ show o) BS.empty)  listToMaybe :: [a] -> Maybe a listToMaybe (x:_) = Just x
test/Unit.hs view
@@ -22,6 +22,7 @@   , extractPageImages   ) import PDF.FormExtract (pageFormNames, extractFormPdf)+import PDF.Object (displayPdfHex) import PDF.Text (pdfToTextTaggedBS, pdfToTextDoc, pdfToTextStreamDoc) import PDF.Error (PdfResult) @@ -200,6 +201,7 @@           ++ taggedEndToEndResults           ++ textStreamResults           ++ cmapEncodingResults+          ++ pdfHexTitleResults           ++ normalizePdfNumberResults           ++ heightSpecResults           ++ encryptSpecResults@@ -1497,6 +1499,16 @@       , assertBool "bytesToCodes JISmap 2-byte fixed"           (bytesToCodes jfi [0x46, 0x7C, 0x4B, 0x5C] == [0x467C, 0x4B5C])       ]++pdfHexTitleResults :: [Result]+pdfHexTitleResults =+  [ assertBool "displayPdfHex already decoded"+      (displayPdfHex (T.pack "Hello") == "Hello")+  , assertBool "displayPdfHex UTF-16BE hex digits"+      (displayPdfHex (T.pack "FEFF00480065006C006C006F") == "Hello")+  , assertBool "displayPdfHex latin1 hex digits"+      (displayPdfHex (T.pack "48656c6c6f") == "Hello")+  ]  normalizePdfNumberResults :: [Result] normalizePdfNumberResults =