linebreak 1.0.0.3 → 1.1.0.0
raw patch · 5 files changed
+333/−190 lines, 5 filesdep +hspecdep ~basedep ~hyphenationPVP ok
version bump matches the API change (PVP)
Dependencies added: hspec
Dependency ranges changed: base, hyphenation
API changes (from Hackage documentation)
- Text.LineBreak: bfHyphenSymbol :: BreakFormat -> Char
- Text.LineBreak: bfHyphenator :: BreakFormat -> Maybe Hyphenator
- Text.LineBreak: bfMaxCol :: BreakFormat -> Int
- Text.LineBreak: bfTabRep :: BreakFormat -> Int
- Text.LineBreak: instance Show Element
+ Text.LineBreak: [bfHyphenSymbol] :: BreakFormat -> Char
+ Text.LineBreak: [bfHyphenator] :: BreakFormat -> Maybe Hyphenator
+ Text.LineBreak: [bfMaxCol] :: BreakFormat -> Int
+ Text.LineBreak: [bfTabRep] :: BreakFormat -> Int
+ Text.LineBreak: afrikaans :: Hyphenator
+ Text.LineBreak: armenian :: Hyphenator
+ Text.LineBreak: assamese :: Hyphenator
+ Text.LineBreak: basque :: Hyphenator
+ Text.LineBreak: bengali :: Hyphenator
+ Text.LineBreak: bulgarian :: Hyphenator
+ Text.LineBreak: catalan :: Hyphenator
+ Text.LineBreak: chinese :: Hyphenator
+ Text.LineBreak: coptic :: Hyphenator
+ Text.LineBreak: croatian :: Hyphenator
+ Text.LineBreak: czech :: Hyphenator
+ Text.LineBreak: danish :: Hyphenator
+ Text.LineBreak: data Hyphenator
+ Text.LineBreak: dutch :: Hyphenator
+ Text.LineBreak: english_GB :: Hyphenator
+ Text.LineBreak: english_US :: Hyphenator
+ Text.LineBreak: esperanto :: Hyphenator
+ Text.LineBreak: estonian :: Hyphenator
+ Text.LineBreak: ethiopic :: Hyphenator
+ Text.LineBreak: finnish :: Hyphenator
+ Text.LineBreak: french :: Hyphenator
+ Text.LineBreak: friulan :: Hyphenator
+ Text.LineBreak: galician :: Hyphenator
+ Text.LineBreak: georgian :: Hyphenator
+ Text.LineBreak: german_1901 :: Hyphenator
+ Text.LineBreak: german_1996 :: Hyphenator
+ Text.LineBreak: german_Swiss :: Hyphenator
+ Text.LineBreak: greek_Ancient :: Hyphenator
+ Text.LineBreak: greek_Mono :: Hyphenator
+ Text.LineBreak: greek_Poly :: Hyphenator
+ Text.LineBreak: gujarati :: Hyphenator
+ Text.LineBreak: hindi :: Hyphenator
+ Text.LineBreak: hungarian :: Hyphenator
+ Text.LineBreak: icelandic :: Hyphenator
+ Text.LineBreak: indonesian :: Hyphenator
+ Text.LineBreak: instance GHC.Show.Show Text.LineBreak.Element
+ Text.LineBreak: interlingua :: Hyphenator
+ Text.LineBreak: irish :: Hyphenator
+ Text.LineBreak: italian :: Hyphenator
+ Text.LineBreak: kannada :: Hyphenator
+ Text.LineBreak: kurmanji :: Hyphenator
+ Text.LineBreak: latin :: Hyphenator
+ Text.LineBreak: latin_Classic :: Hyphenator
+ Text.LineBreak: latvian :: Hyphenator
+ Text.LineBreak: lithuanian :: Hyphenator
+ Text.LineBreak: malayalam :: Hyphenator
+ Text.LineBreak: marathi :: Hyphenator
+ Text.LineBreak: mongolian :: Hyphenator
+ Text.LineBreak: norwegian_Bokmal :: Hyphenator
+ Text.LineBreak: norwegian_Nynorsk :: Hyphenator
+ Text.LineBreak: occitan :: Hyphenator
+ Text.LineBreak: oriya :: Hyphenator
+ Text.LineBreak: panjabi :: Hyphenator
+ Text.LineBreak: piedmontese :: Hyphenator
+ Text.LineBreak: polish :: Hyphenator
+ Text.LineBreak: portuguese :: Hyphenator
+ Text.LineBreak: romanian :: Hyphenator
+ Text.LineBreak: romansh :: Hyphenator
+ Text.LineBreak: russian :: Hyphenator
+ Text.LineBreak: sanskrit :: Hyphenator
+ Text.LineBreak: serbian_Cyrillic :: Hyphenator
+ Text.LineBreak: serbocroatian_Cyrillic :: Hyphenator
+ Text.LineBreak: serbocroatian_Latin :: Hyphenator
+ Text.LineBreak: slovak :: Hyphenator
+ Text.LineBreak: slovenian :: Hyphenator
+ Text.LineBreak: spanish :: Hyphenator
+ Text.LineBreak: swedish :: Hyphenator
+ Text.LineBreak: tamil :: Hyphenator
+ Text.LineBreak: telugu :: Hyphenator
+ Text.LineBreak: thai :: Hyphenator
+ Text.LineBreak: turkish :: Hyphenator
+ Text.LineBreak: turkmen :: Hyphenator
+ Text.LineBreak: ukrainian :: Hyphenator
+ Text.LineBreak: uppersorbian :: Hyphenator
+ Text.LineBreak: welsh :: Hyphenator
Files
- CHANGES +2/−0
- Text/LineBreak.hs +0/−187
- linebreak.cabal +22/−3
- src/Text/LineBreak.hs +213/−0
- test/hspec.hs +96/−0
+ CHANGES view
@@ -0,0 +1,2 @@+1.1.0.0:+- Reexport language hyphenators
− Text/LineBreak.hs
@@ -1,187 +0,0 @@------------------------------------------------------------------------------------ |--- Module : Text.LineBreak--- Copyright : (C) 2014 Francesco Ariis--- License : BSD3 (see LICENSE file)------ Maintainer : Francesco Ariis <fa-ml@ariis.it>--- Stability : provisional--- Portability : portable------ Simple functions to break a String to fit a maximum text width, using--- Knuth-Liang hyphenation algorithm.------ Example:------ > import Text.Hyphenation--- > import Text.LineBreak--- >--- > hyp = Just english_US--- > bf = BreakFormat 25 4 '-' hyp--- > cs = "Using hyphenation with gruesomely non parsimonious wording."--- > main = putStr $ breakString bf cs------ will output:------ > Using hyphenation with--- > gruesomely non parsimo---- > nious wording.-------------------------------------------------------------------------------------module Text.LineBreak ( breakString, breakStringLn, BreakFormat(..) ) where--import Text.Hyphenation-import Data.Char (isSpace)-import Data.List (find, inits, tails, span)----------------- TYPES ------------------ | How to break the strings: maximum width of the lines, number of spaces--- to replace tabs with (dumb replacement), symbol to use to hyphenate--- words, hypenator to use (language, exceptions, etc.; refer to--- "Text.Hyphenation" for usage instructions). To break lines without--- hyphenating, put @Nothing@ in @bfHyphenator@.-data BreakFormat = BreakFormat { bfMaxCol :: Int,- bfTabRep :: Int,- bfHyphenSymbol :: Char,- bfHyphenator :: Maybe Hyphenator }--data BrState = BrState { bsCurrCol :: Int, -- current column- bsBroken :: String } -- output string--data Element = ElWord String -- things we need to place- | ElSpace Int Bool -- n of spaces, presence of final breakline- deriving (Show)-------------------- FUNCTIONS ---------------------- | Breaks some text (String) to make it fit in a certain width. The output--- is a String, suitable for writing to screen or file.-breakString :: BreakFormat -> String -> String-breakString bf cs = hackClean out- where els = parseEls (subTabs (bfTabRep bf) cs)- out = bsBroken $ foldl (putElem bf) (BrState 0 "") els---- | Convenience for @lines $ breakString bf cs@-breakStringLn :: BreakFormat -> String -> [String]-breakStringLn bf cs = lines $ breakString bf cs----------------------- ANCILLARIES ------------------------ PARSING ------ fino a qui--- o word 'till ws--- o wspa 'till (\n | word). se \n, prendilo-parseEls :: String -> [Element]-parseEls [] = []-parseEls cs@(c:_) | isSpace c = let (p, r) = span isSpace cs- in parseWS p ++ parseEls r- | otherwise = let (p, r) = span (not . isSpace) cs- in parseWord p : parseEls r---- Signatures between the two |parse| are different because there can--- be more element in a single white-space string (newline newline), while--- that is not possible with parseWord-parseWS :: String -> [Element]-parseWS [] = []-parseWS ws = case span (/= '\n') ws of- (a, "") -> [elspace a False] -- no newlines- (a, '\n':rs) -> elspace a True : parseWS rs- where elspace cs b = ElSpace (length cs) b--parseWord :: String -> Element-parseWord wr = ElWord wr---- number of spaces to replace \t with, string-subTabs :: Int -> String -> String-subTabs i cs = cs >>= f i- where f n '\t' = replicate i ' '- f _ c = return c---- COMPUTATION ----putElem :: BreakFormat -> BrState -> Element -> BrState-putElem (BreakFormat maxc _ sym hyp)- bs@(BrState currc currstr) el =- if avspace >= elLenght el- then putString bs maxc (el2string el)- else case el of- (ElSpace _ _) -> putString bs maxc "\n"- (ElWord cs) -> putString bs maxc (broken cs)- where avspace = maxc - currc -- starting col: 1- fstcol = currc == 0- broken cs = breakWord hyp sym avspace cs fstcol--elLenght :: Element -> Int-elLenght (ElWord cs) = length cs-elLenght (ElSpace i _) = i---- convert element to string-el2string :: Element -> String-el2string (ElSpace i False) = replicate i ' '-el2string (ElSpace i True) = "\n"-el2string (ElWord cs) = cs---- put a string and updates the state--- (more than macol? new line, but no hyphenation!)-putString :: BrState -> Int -> String -> BrState-putString bs _ [] = bs-putString (BrState currc currstr) maxcol (c:cs) =- let currc' = if c == '\n'- then 0- else currc + 1- bs' = if currc' <= maxcol- then BrState currc' (currstr ++ [c])- else BrState 1 (currstr ++ "\n" ++ [c])- in putString bs' maxcol cs---- breaks a word given remaining space, using an hypenator--- the last bool is a "you are on the first col, can't start--- a new line-breakWord :: Maybe Hyphenator -> Char -> Int -> String -> Bool -> String-breakWord mhy ch avspace cs nlb = case find ((<= avspace) . hypLen) poss of- Just a -> a- Nothing -> cs -- don't find? return input- where hw = case mhy of- Just hy -> hyphenate hy cs- Nothing -> [cs]- poss = map cf $ reverse $ zip (inits hw) (tails hw)- -- poss ~= ["hyphenation\n","hyphen-\nation",- -- "hy-\nphenation","\nhyphenation"]-- -- crea hyphenated from two bits- cf ([], ew) = (if nlb then "" else "\n") ++ concat ew- cf (iw, []) = concat iw ++ "\n"- cf (iw, ew) = concat iw ++ [ch] ++ "\n" ++ concat ew-- hypLen cs = length . takeWhile (/= '\n') $ cs---- CLEAN ------ removes eof/eol whitespace-hackClean :: String -> String-hackClean cs = noEoflWs cs- where noEoflWs cs = f "" cs-- -- the ugliness- f acc [] = acc- f acc cs@(a:as) =- let (i, e) = span (== ' ') cs in- if i == ""- then f (acc ++ [a]) as- else case e of- ('\n':rest) -> f (acc ++ "\n") rest -- eol ws- [] -> f acc [] -- eof ws- _ -> f (acc ++ [a]) as -- normal--
linebreak.cabal view
@@ -1,5 +1,5 @@ name: linebreak-version: 1.0.0.3+version: 1.1.0.0 synopsis: breaks strings to fit width description: Simple functions to break a String to fit a maximum text width, using Knuth-Liang hyphenation algorhitm.@@ -10,7 +10,9 @@ maintainer: fa-ml@ariis.it category: Text build-type: Simple-cabal-version: >=1.8+cabal-version: >=1.10+tested-with: GHC==7.8.4+extra-source-files: CHANGES source-repository head type: darcs@@ -18,5 +20,22 @@ library -- Modules exported by the library.+ default-language: Haskell2010 exposed-modules: Text.LineBreak- build-depends: base >= 4.5 && < 5, hyphenation >= 0.4 && < 1+ build-depends: base == 4.*,+ hyphenation >= 0.8 && < 1+ hs-source-dirs: src+ ghc-options: -Wall++test-suite test+ default-language: Haskell2010+ ghc-options: -Wall+ HS-Source-Dirs: test, src+ main-is: hspec.hs+ other-modules: Text.LineBreak+ build-depends: base == 4.*,+ hyphenation >= 0.8 && < 1+ -- same as above, plus hspec+ , hspec >= 2.2 && < 2.9+ type: exitcode-stdio-1.0+ ghc-options: -Wall
+ src/Text/LineBreak.hs view
@@ -0,0 +1,213 @@+--------------------------------------------------------------------------------+-- |+-- Module : Text.LineBreak+-- Copyright : (C) 2014 Francesco Ariis+-- License : BSD3 (see LICENSE file)+--+-- Maintainer : Francesco Ariis <fa-ml@ariis.it>+-- Stability : provisional+-- Portability : portable+--+-- Simple functions to break a String to fit a maximum text width, using+-- Knuth-Liang hyphenation algorithm.+--+-- Example:+--+-- > import Text.LineBreak+-- >+-- > hyp = Just english_US+-- > bf = BreakFormat 25 4 '-' hyp+-- > cs = "Using hyphenation with gruesomely non parsimonious wording."+-- > main = putStr $ breakString bf cs+--+-- will output:+--+-- > Using hyphenation with+-- > gruesomely non parsimo-+-- > nious wording.+--+-------------------------------------------------------------------------------++module Text.LineBreak ( -- * Line breaking+ breakString, breakStringLn, BreakFormat(..),+ -- * Hypenators+ -- | Convenience reexport from+ -- "Text.Hyphenation.Language".+ Hyphenator,+ afrikaans, armenian, assamese, basque, bengali,+ bulgarian, catalan, chinese, coptic, croatian,+ czech, danish, dutch, english_US, english_GB,+ esperanto, estonian, ethiopic, finnish, french,+ friulan, galician, georgian, german_1901,+ german_1996, german_Swiss, greek_Ancient,+ greek_Mono, greek_Poly, gujarati, hindi, hungarian,+ icelandic, indonesian, interlingua, irish,+ italian, kannada, kurmanji, latin, latin_Classic,+ latvian, lithuanian, malayalam, marathi,+ mongolian, norwegian_Bokmal, norwegian_Nynorsk,+ occitan, oriya, panjabi, piedmontese, polish,+ portuguese, romanian, romansh, russian, sanskrit,+ serbian_Cyrillic, serbocroatian_Cyrillic,+ serbocroatian_Latin, slovak, slovenian, spanish,+ swedish, tamil, telugu, thai, turkish, turkmen,+ ukrainian, uppersorbian, welsh+ ) where++import Text.Hyphenation+import Data.Char (isSpace)+import Data.List (find, inits, tails)++-- TODO: tabs are broken (as it is just a plain substitution). Use a+-- smart sub method. [bug] [test]++-- TODO: [improvement] valid for Text, etc.+++-----------+-- TYPES --+-----------++-- | How to break the strings: maximum width of the lines, number of spaces+-- to replace tabs with (dumb replacement), symbol to use to hyphenate+-- words, hypenator to use (language, exceptions, etc.; refer to+-- "Text.Hyphenation" for usage instructions). To break lines without+-- hyphenating, put @Nothing@ in @bfHyphenator@.+data BreakFormat = BreakFormat { bfMaxCol :: Int,+ bfTabRep :: Int,+ bfHyphenSymbol :: Char,+ bfHyphenator :: Maybe Hyphenator }++data BrState = BrState { bsCurrCol :: Int, -- current column+ bsBroken :: String } -- output string++data Element = ElWord String -- things we need to place+ | ElSpace Int Bool -- n of spaces, presence of final breakline+ deriving (Show)+++---------------+-- FUNCTIONS --+---------------++-- | Breaks some text (String) to make it fit in a certain width. The output+-- is a String, suitable for writing to screen or file.+breakString :: BreakFormat -> String -> String+breakString bf cs = hackClean out+ where els = parseEls (subTabs (bfTabRep bf) cs)+ out = bsBroken $ foldl (putElem bf) (BrState 0 "") els+ -- todo horrible hack is horrible [benchmark] [refactor]++-- | Convenience for @lines $ breakString bf cs@.+breakStringLn :: BreakFormat -> String -> [String]+breakStringLn bf cs = lines $ breakString bf cs++-----------------+-- ANCILLARIES --+-----------------++-- PARSING --++-- fino a qui+-- o word 'till ws+-- o wspa 'till (\n | word). se \n, prendilo+parseEls :: String -> [Element]+parseEls [] = []+parseEls cs@(c:_) | isSpace c = let (p, r) = span isSpace cs+ in parseWS p ++ parseEls r+ | otherwise = let (p, r) = break isSpace cs+ in parseWord p : parseEls r++-- Signatures between the two |parse| are different because there can+-- be more element in a single white-space string (newline newline), while+-- that is not possible with parseWord+parseWS :: String -> [Element]+parseWS [] = []+parseWS ws = case span (/= '\n') ws of+ (a, "") -> [elspace a False] -- no newlines+ (a, '\n':rs) -> elspace a True : parseWS rs+ where elspace cs b = ElSpace (length cs) b++parseWord :: String -> Element+parseWord wr = ElWord wr++-- number of spaces to replace \t with, string+subTabs :: Int -> String -> String+subTabs i cs = cs >>= f i+ where f _ '\t' = replicate i ' '+ f _ c = return c++-- COMPUTATION --++putElem :: BreakFormat -> BrState -> Element -> BrState+putElem (BreakFormat maxc _ sym hyp)+ bs@(BrState currc _) el =+ if avspace >= elLenght el+ then putString bs maxc (el2string el)+ else case el of+ (ElSpace _ _) -> putString bs maxc "\n"+ (ElWord cs) -> putString bs maxc (broken cs)+ where avspace = maxc - currc -- starting col: 1+ fstcol = currc == 0+ broken cs = breakWord hyp sym avspace cs fstcol++elLenght :: Element -> Int+elLenght (ElWord cs) = length cs+elLenght (ElSpace i _) = i++-- convert element to string+el2string :: Element -> String+el2string (ElSpace i False) = replicate i ' '+el2string (ElSpace _ True) = "\n"+el2string (ElWord cs) = cs++-- put a string and updates the state+-- (more than macol? new line, but no hyphenation!)+putString :: BrState -> Int -> String -> BrState+putString bs _ [] = bs+putString (BrState currc currstr) maxcol (c:cs) =+ let currc' = if c == '\n'+ then 0+ else currc + 1+ bs' = if currc' <= maxcol+ then BrState currc' (currstr ++ [c])+ else BrState 1 (currstr ++ "\n" ++ [c])+ in putString bs' maxcol cs++-- breaks a word given remaining space, using an hypenator+-- the last bool is a "you are on the first col, can't start+-- a new line+breakWord :: Maybe Hyphenator -> Char -> Int -> String -> Bool -> String+breakWord mhy ch avspace cs nlb = case find ((<= avspace) . hypLen) poss of+ Just a -> a+ Nothing -> cs -- don't find? return input+ where hw = case mhy of+ Just hy -> hyphenate hy cs+ Nothing -> [cs]+ poss = map cf $ reverse $ zip (inits hw) (tails hw)++ -- crea hyphenated from two bits+ cf ([], ew) = (if nlb then "" else "\n") ++ concat ew+ cf (iw, []) = concat iw ++ "\n"+ cf (iw, ew) = concat iw ++ [ch] ++ "\n" ++ concat ew++ hypLen wcs = length . takeWhile (/= '\n') $ wcs++-- CLEAN --++-- removes eof/eol whitespace+hackClean :: String -> String+hackClean cs = noEoflWs cs+ where noEoflWs wcs = f "" wcs++ -- the ugliness+ f acc [] = acc+ f acc wcs@(a:as) =+ let (i, e) = span (== ' ') wcs in+ if i == ""+ then f (acc ++ [a]) as+ else case e of+ ('\n':rest) -> f (acc ++ "\n") rest -- eol ws+ [] -> f acc [] -- eof ws+ _ -> f (acc ++ [a]) as -- normal++
+ test/hspec.hs view
@@ -0,0 +1,96 @@+import Test.Hspec+import Text.LineBreak+++hyp :: Maybe Hyphenator+hyp = Just english_US++bf :: BreakFormat+bf = BreakFormat 25 4 '-' hyp++testBr :: String -> String+testBr = breakString bf++myUnlines :: [String]-> [Char]+myUnlines = init . unlines -- unlines adds an extra "\n" ad end-of-string+++main :: IO ()+main = hspec $ do++ describe "Text.Linebreak.breakString" $ do++ let sip = "Using hyphenation with gruesomely non parsimonious wording."+ sipout = myUnlines ["Using hyphenation with",+ "gruesomely non parsimo-",+ "nious wording." ]+ it "breaks a string to certain width" $+ testBr sip `shouldBe` sipout++ let sipoutNH = myUnlines ["Using hyphenation with",+ "gruesomely non",+ "parsimonious wording." ]+ bfnh = BreakFormat 25 4 '-' Nothing+ it "breaks a string to certain width (w/o hypenator)" $+ breakString bfnh sip `shouldBe` sipoutNH++ let postws = "word "+ postwsout = "word"+ it "ws at end of word should be truncated" $+ testBr postws `shouldBe` postwsout++ let postwscc = "aaa" ++ replicate (25 - 3 - 1) ' ' ++ "bbb"+ postwsccout = "aaa\nbbb"+ it "no end-of-line white-space, corner case (words added after it)" $+ testBr postwscc `shouldBe` postwsccout++ let broken = "broken\nstring"+ brokenout = myUnlines ["broken", "string"]+ it "should not remove \\n already in the string" $+ testBr broken `shouldBe` brokenout++ let brokenL = "broken\n\n\n\nstring"+ brokenoutL = "broken\n\n\n\nstring"+ it "should respect multiple '\\n' already in the string" $+ testBr brokenL `shouldBe` brokenoutL++ let spaced = "spaced string"+ spacedout = spaced+ it "should not remove blankspace already in the string" $+ testBr spaced `shouldBe` spacedout++ let simple = "simple"+ simpleout = simple+ it "should not add extra \\n at the end of the string" $+ testBr simple `shouldBe` simpleout++ let prews = " word"+ prewsout = prews+ it "should not remove whitespace @ beginning of string" $+ testBr prews `shouldBe` prewsout++ let muchws = "word second"+ muchwsout = myUnlines ["word", "second"]+ it "should truncate eol whitespace" $+ testBr muchws `shouldBe` muchwsout++ let toolong = replicate 24 'a' ++ "bbb"+ toolongout = replicate 24 'a' ++ "b\nbb"+ it "should truncate words longer than screenwidth" $+ testBr toolong `shouldBe` toolongout++ -- TODO what happens with really tiny width?+ let tinywd = "tomcat"+ tinywdout = "to\nmc\nat"+ tinybf = BreakFormat 2 4 '-' (Just english_US)+ tinyBr = breakString tinybf+ it "should wrap (but not hyphenate long words)" $+ tinyBr tinywd `shouldBe` tinywdout+++ describe "Text.Linebreak.breakStringLn" $ do++ let sip = "Using hyphenation with gruesomely non parsimonious wording."+ it "is the same as |lines $ breakString bf cs|" $ do+ breakStringLn bf sip `shouldBe` lines (breakString bf sip)+