scalpel 0.5.1 → 0.6.0
raw patch · 5 files changed
+139/−114 lines, 5 filesdep +case-insensitivedep +http-clientdep +http-client-tlsdep −curldep ~scalpel-corePVP ok
version bump matches the API change (PVP)
Dependencies added: case-insensitive, http-client, http-client-tls
Dependencies removed: curl
Dependency ranges changed: scalpel-core
API changes (from Hackage documentation)
- Text.HTML.Scalpel: [curlOpts] :: Config str -> [CurlOption]
- Text.HTML.Scalpel: scrapeURLWithOpts :: (Ord str, StringLike str) => [CurlOption] -> URL -> Scraper str a -> IO (Maybe a)
+ Text.HTML.Scalpel: [manager] :: Config str -> Maybe Manager
+ Text.HTML.Scalpel: atDepth :: Selector -> Int -> Selector
+ Text.HTML.Scalpel: data SerialScraper str a
+ Text.HTML.Scalpel: inSerial :: StringLike str => SerialScraper str a -> Scraper str a
+ Text.HTML.Scalpel: matches :: StringLike str => Selector -> Scraper str ()
+ Text.HTML.Scalpel: seekBack :: StringLike str => Scraper str a -> SerialScraper str a
+ Text.HTML.Scalpel: seekNext :: StringLike str => Scraper str a -> SerialScraper str a
+ Text.HTML.Scalpel: stepBack :: StringLike str => Scraper str a -> SerialScraper str a
+ Text.HTML.Scalpel: stepNext :: StringLike str => Scraper str a -> SerialScraper str a
+ Text.HTML.Scalpel: textSelector :: Selector
+ Text.HTML.Scalpel: untilBack :: StringLike str => Scraper str a -> SerialScraper str b -> SerialScraper str b
+ Text.HTML.Scalpel: untilNext :: StringLike str => Scraper str a -> SerialScraper str b -> SerialScraper str b
- Text.HTML.Scalpel: Config :: [CurlOption] -> Decoder str -> Config str
+ Text.HTML.Scalpel: Config :: Decoder str -> Maybe Manager -> Config str
- Text.HTML.Scalpel: attr :: (Ord str, Show str, StringLike str) => String -> Selector -> Scraper str str
+ Text.HTML.Scalpel: attr :: (Show str, StringLike str) => String -> Selector -> Scraper str str
- Text.HTML.Scalpel: attrs :: (Ord str, Show str, StringLike str) => String -> Selector -> Scraper str [str]
+ Text.HTML.Scalpel: attrs :: (Show str, StringLike str) => String -> Selector -> Scraper str [str]
- Text.HTML.Scalpel: chroot :: (Ord str, StringLike str) => Selector -> Scraper str a -> Scraper str a
+ Text.HTML.Scalpel: chroot :: StringLike str => Selector -> Scraper str a -> Scraper str a
- Text.HTML.Scalpel: chroots :: (Ord str, StringLike str) => Selector -> Scraper str a -> Scraper str [a]
+ Text.HTML.Scalpel: chroots :: StringLike str => Selector -> Scraper str a -> Scraper str [a]
- Text.HTML.Scalpel: data AttributeName :: *
+ Text.HTML.Scalpel: data AttributeName
- Text.HTML.Scalpel: data AttributePredicate :: *
+ Text.HTML.Scalpel: data AttributePredicate
- Text.HTML.Scalpel: data Scraper str a :: * -> * -> *
+ Text.HTML.Scalpel: data Scraper str a
- Text.HTML.Scalpel: data Selector :: *
+ Text.HTML.Scalpel: data Selector
- Text.HTML.Scalpel: data TagName :: *
+ Text.HTML.Scalpel: data TagName
- Text.HTML.Scalpel: html :: (Ord str, StringLike str) => Selector -> Scraper str str
+ Text.HTML.Scalpel: html :: StringLike str => Selector -> Scraper str str
- Text.HTML.Scalpel: htmls :: (Ord str, StringLike str) => Selector -> Scraper str [str]
+ Text.HTML.Scalpel: htmls :: StringLike str => Selector -> Scraper str [str]
- Text.HTML.Scalpel: innerHTML :: (Ord str, StringLike str) => Selector -> Scraper str str
+ Text.HTML.Scalpel: innerHTML :: StringLike str => Selector -> Scraper str str
- Text.HTML.Scalpel: innerHTMLs :: (Ord str, StringLike str) => Selector -> Scraper str [str]
+ Text.HTML.Scalpel: innerHTMLs :: StringLike str => Selector -> Scraper str [str]
- Text.HTML.Scalpel: position :: (Ord str, StringLike str) => Scraper str Int
+ Text.HTML.Scalpel: position :: StringLike str => Scraper str Int
- Text.HTML.Scalpel: scrape :: (Ord str, StringLike str) => Scraper str a -> [Tag str] -> Maybe a
+ Text.HTML.Scalpel: scrape :: StringLike str => Scraper str a -> [Tag str] -> Maybe a
- Text.HTML.Scalpel: scrapeStringLike :: (Ord str, StringLike str) => str -> Scraper str a -> Maybe a
+ Text.HTML.Scalpel: scrapeStringLike :: StringLike str => str -> Scraper str a -> Maybe a
- Text.HTML.Scalpel: scrapeURL :: (Ord str, StringLike str) => URL -> Scraper str a -> IO (Maybe a)
+ Text.HTML.Scalpel: scrapeURL :: StringLike str => URL -> Scraper str a -> IO (Maybe a)
- Text.HTML.Scalpel: scrapeURLWithConfig :: (Ord str, StringLike str) => Config str -> URL -> Scraper str a -> IO (Maybe a)
+ Text.HTML.Scalpel: scrapeURLWithConfig :: StringLike str => Config str -> URL -> Scraper str a -> IO (Maybe a)
- Text.HTML.Scalpel: text :: (Ord str, StringLike str) => Selector -> Scraper str str
+ Text.HTML.Scalpel: text :: StringLike str => Selector -> Scraper str str
- Text.HTML.Scalpel: texts :: (Ord str, StringLike str) => Selector -> Scraper str [str]
+ Text.HTML.Scalpel: texts :: StringLike str => Selector -> Scraper str [str]
- Text.HTML.Scalpel: type Decoder str = CurlResponse_ [(String, String)] ByteString -> str
+ Text.HTML.Scalpel: type Decoder str = Response ByteString -> str
Files
- CHANGELOG.md +27/−0
- README.md +26/−26
- scalpel.cabal +8/−6
- src/Text/HTML/Scalpel.hs +43/−28
- src/Text/HTML/Scalpel/Internal/Scrape/URL.hs +35/−54
CHANGELOG.md view
@@ -2,6 +2,33 @@ ## HEAD +## 0.6.0++### Breaking Changes++- `anySelector` now captures text nodes. This causes different results when used+ with a plural scraper (e.g. `chroots`). Usage with a singular scraper (e.g.+ `chroot`) should be unaffected.+- The dependency on `curl` has been replaced with `http-client` and+ `http-client-tls`. This has the following observable changes.+ - `scrapeURLWithOpts` is removed.+ - The `Config` type used with `scrapeURLWithConfig` no longer contains a list+ of curl options. Instead it now takes a `Maybe Manager` from `http-client`.+ - The `Decoder` function type now takes in a `Response` type from+ `http-client`.+ - `scrapeURL` will now throw an exception if there is a problem connecting to+ a URL.++### Other Changes++- Remove `Ord` constraint from public APIs.+- Add `atDepth` operator which allows for selecting nodes at a specified depth+ in relation to another node (#21).+- Fix issue selecting malformed HTML where `"a" // "c"` would not match+ `<a><b><c></c></a></b>`.+- Add `textSelector` for selecting text nodes.+- Add `SerialScraper` type and associated primitives (#48).+ ## 0.5.1 - Fix bug (#59, #54) in DFS traversal order.
README.md view
@@ -234,14 +234,13 @@ ### scalpel-core -The `scalpel` package relies on curl to provide networking support. For small-projects and one off scraping tasks this is likely sufficient. However when-using scalpel in existing projects or on platforms without curl this dependency-can be a hindrance.+The `scalpel` package depends on 'http-client' and 'http-client-tls' to provide+networking support. For projects with an existing HTTP client these dependencies+may be unnecessary. For these scenarios users can instead depend on [scalpel-core](https://hackage.haskell.org/package/scalpel-core) which does not-provide networking support and does not depend on curl.+provide networking support and has minimal dependencies. Troubleshooting ---------------@@ -252,40 +251,41 @@ with the request. In some cases, this even means returning no markup at all in an effort to prevent scraping. -To work around this, you can add your own user agent string with a curl option.+To work around this, you can add your own user agent string. ```haskell #!/usr/local/bin/stack--- stack runghc --resolver lts-6.24 --install-ghc --package scalpel-0.4.0+-- stack runghc --resolver lts-6.24 --install-ghc --package scalpel-0.6.0+{-# LANGUAGE NamedFieldPuns #-}+{-# LANGUAGE OverloadedStrings #-} -import Network.Curl import Text.HTML.Scalpel+import qualified Network.HTTP.Client as HTTP+import qualified Network.HTTP.Client.TLS as HTTP+import qualified Network.HTTP.Types.Header as HTTP ++-- Create a new manager settings based on the default TLS manager that updates+-- the request headers to include a custom user agent.+managerSettings :: HTTP.ManagerSettings+managerSettings = HTTP.tlsManagerSettings {+ HTTP.managerModifyRequest = \req -> do+ req' <- HTTP.managerModifyRequest HTTP.tlsManagerSettings req+ return $ req' {+ HTTP.requestHeaders = (HTTP.hUserAgent, "My Custom UA")+ : HTTP.requestHeaders req'+ }+}+ main = do- html <- scrapeURLWithOpts opts url $ htmls anySelector+ manager <- Just <$> HTTP.newManager managerSettings+ html <- scrapeURLWithConfig (def { manager }) url $ htmls anySelector maybe printError printHtml html where url = "https://www.google.com"- opts = [ CurlUserAgent "some user agent string" ] printError = putStrLn "Failed" printHtml = mapM_ putStrLn ``` A list of user agent strings can be found [here](http://www.useragentstring.com/pages/useragentstring.php).--### Building on Windows--Building scalpel on Windows can be a challenge because of the dependency on-curl. In order to successfully build scalpel you must download-[curl](http://curl.haxx.se/download.html) and add the following to your-stack.yaml file.--```yaml-extra-lib-dirs: ["C:/Program Files/cURL/dlls"]-extra-include-dirs: ["C:/Program Files/cURL/dlls"]-```--If you do not require network support, you can instead depend on-[scalpel-core](https://hackage.haskell.org/package/scalpel-core) which does not-does not depend on curl.
scalpel.cabal view
@@ -1,5 +1,5 @@ name: scalpel-version: 0.5.1+version: 0.6.0 synopsis: A high level web scraping library for Haskell. description: Scalpel is a web scraping library inspired by libraries like Parsec and@@ -24,7 +24,7 @@ source-repository this type: git location: https://github.com/fimad/scalpel.git- tag: v0.5.1+ tag: v0.6.0 library other-extensions:@@ -37,12 +37,14 @@ hs-source-dirs: src/ default-language: Haskell2010 build-depends:- base >= 4.6 && < 5- , scalpel-core == 0.5.1+ base >= 4.6 && < 5+ , scalpel-core == 0.6.0 , bytestring- , curl >= 1.3.4+ , case-insensitive , data-default- , tagsoup >= 0.12.2+ , http-client >= 0.4.30+ , http-client-tls >= 0.2.4+ , tagsoup >= 0.12.2 , text default-extensions: ParallelListComp
src/Text/HTML/Scalpel.hs view
@@ -66,34 +66,36 @@ -- The following snippet defines a function, @allComments@, that will download -- the web page, and extract all of the comments into a list: ----- > type Author = String--- >--- > data Comment--- > = TextComment Author String--- > | ImageComment Author URL--- > deriving (Show, Eq)--- >--- > allComments :: IO (Maybe [Comment])--- > allComments = scrapeURL "http://example.com/article.html" comments--- > where--- > comments :: Scraper String [Comment]--- > comments = chroots ("div" @: [hasClass "container"]) comment--- >--- > comment :: Scraper String Comment--- > comment = textComment <|> imageComment--- >--- > textComment :: Scraper String Comment--- > textComment = do--- > author <- text $ "span" @: [hasClass "author"]--- > commentText <- text $ "div" @: [hasClass "text"]--- > return $ TextComment author commentText--- >--- > imageComment :: Scraper String Comment--- > imageComment = do--- > author <- text $ "span" @: [hasClass "author"]--- > imageURL <- attr "src" $ "img" @: [hasClass "image"]--- > return $ ImageComment author imageURL+-- @+-- type Author = String --+-- data Comment+-- = TextComment Author String+-- | ImageComment Author URL+-- deriving (Show, Eq)+--+-- allComments :: IO (Maybe [Comment])+-- allComments = 'scrapeURL' \"http:\/\/example.com/article.html\" comments+-- where+-- comments :: Scraper String [Comment]+-- comments = 'chroots' ("div" '@:' ['hasClass' "container"]) comment+--+-- comment :: Scraper String Comment+-- comment = textComment `<|>` imageComment+--+-- textComment :: Scraper String Comment+-- textComment = do+-- author <- 'text' $ "span" \@: [hasClass "author"]+-- commentText <- text $ "div" \@: [hasClass "text"]+-- return $ TextComment author commentText+--+-- imageComment :: Scraper String Comment+-- imageComment = do+-- author <- text $ "span" \@: [hasClass "author"]+-- imageURL <- 'attr' "src" $ "img" \@: [hasClass "image"]+-- return $ ImageComment author imageURL+-- @+-- -- Complete examples can be found in the -- <https://github.com/fimad/scalpel/tree/master/examples examples> folder in -- the scalpel git repository.@@ -104,10 +106,12 @@ , AttributeName (..) , TagName (..) , tagSelector+, textSelector -- ** Wildcards , anySelector -- ** Tag combinators , (//)+, atDepth -- ** Attribute predicates , (@:) , (@=)@@ -130,18 +134,29 @@ , chroot , chroots , position+, matches -- ** Executing scrapers , scrape , scrapeStringLike , URL , scrapeURL-, scrapeURLWithOpts , scrapeURLWithConfig , Config (..) , Decoder , defaultDecoder , utf8Decoder , iso88591Decoder++-- * Serial Scraping+, SerialScraper+, inSerial+-- ** Primitives+, stepNext+, stepBack+, seekNext+, seekBack+, untilNext+, untilBack ) where import Text.HTML.Scalpel.Core
src/Text/HTML/Scalpel/Internal/Scrape/URL.hs view
@@ -1,3 +1,4 @@+{-# LANGUAGE OverloadedStrings #-} {-# OPTIONS_HADDOCK hide #-} module Text.HTML.Scalpel.Internal.Scrape.URL ( URL@@ -9,88 +10,68 @@ , iso88591Decoder , scrapeURL-, scrapeURLWithOpts , scrapeURLWithConfig ) where import Text.HTML.Scalpel.Core import Control.Applicative ((<$>))-import Data.Char (toLower)+import Data.CaseInsensitive () import Data.Default (def)-import Data.List (isInfixOf)-import Data.Maybe (listToMaybe)+import Data.Maybe (fromMaybe, listToMaybe) -import qualified Data.ByteString as BS+import qualified Data.ByteString.Lazy as LBS import qualified Data.Default as Default+import qualified Data.Text as Text import qualified Data.Text.Encoding as Text-import qualified Network.Curl as Curl+import qualified Network.HTTP.Client as HTTP+import qualified Network.HTTP.Client.TLS as HTTP import qualified Text.HTML.TagSoup as TagSoup import qualified Text.StringLike as TagSoup type URL = String -type CurlResponse = Curl.CurlResponse_ [(String, String)] BS.ByteString- -- | A method that takes a HTTP response as raw bytes and returns the body as a -- string type.-type Decoder str = Curl.CurlResponse_ [(String, String)] BS.ByteString -> str+type Decoder str = HTTP.Response LBS.ByteString -> str --- | A record type that determines how 'scrapeUrlWithConfig' interacts with the+-- | A record type that determines how 'scrapeURLWithConfig' interacts with the -- HTTP server and interprets the results. data Config str = Config {- curlOpts :: [Curl.CurlOption]-, decoder :: Decoder str+ decoder :: Decoder str+, manager :: Maybe HTTP.Manager } instance TagSoup.StringLike str => Default.Default (Config str) where def = Config {- curlOpts = [Curl.CurlFollowLocation True]- , decoder = defaultDecoder+ decoder = defaultDecoder+ , manager = Nothing } -- | The 'scrapeURL' function downloads the contents of the given URL and -- executes a 'Scraper' on it. ----- 'scrapeURL' makes use of curl to make HTTP requests. The dependency on curl--- may be too heavyweight for some use cases. In which case users who do not--- require inbuilt networking support can depend on--- <https://hackage.haskell.org/package/scalpel-core scalpel-core> for a--- lightweight subset of this library that does not depend on curl.-scrapeURL :: (Ord str, TagSoup.StringLike str)+-- The default behavior is to use the global manager provided by+-- http-client-tls (via 'HTTP.getGlobalManager'). Any exceptions thrown by+-- http-client are not caught and are bubbled up to the caller.+scrapeURL :: (TagSoup.StringLike str) => URL -> Scraper str a -> IO (Maybe a)-scrapeURL = scrapeURLWithOpts [Curl.CurlFollowLocation True]---- | The 'scrapeURLWithOpts' function take a list of curl options and downloads--- the contents of the given URL and executes a 'Scraper' on it.-scrapeURLWithOpts :: (Ord str, TagSoup.StringLike str)- => [Curl.CurlOption] -> URL -> Scraper str a -> IO (Maybe a)-scrapeURLWithOpts options = scrapeURLWithConfig (def {curlOpts = options})+scrapeURL = scrapeURLWithConfig def -- | The 'scrapeURLWithConfig' function takes a 'Config' record type and -- downloads the contents of the given URL and executes a 'Scraper' on it.-scrapeURLWithConfig :: (Ord str, TagSoup.StringLike str)+scrapeURLWithConfig :: (TagSoup.StringLike str) => Config str -> URL -> Scraper str a -> IO (Maybe a) scrapeURLWithConfig config url scraper = do- maybeTags <- downloadAsTags (decoder config) url- return (maybeTags >>= scrape scraper)+ manager <- fromMaybe HTTP.getGlobalManager (return <$> manager config)+ tags <- downloadAsTags (decoder config) manager url+ return (scrape scraper tags) where- downloadAsTags decoder url = do- maybeBytes <- openURIWithOpts url (curlOpts config)- return $ TagSoup.parseTags . decoder <$> maybeBytes--openURIWithOpts :: URL -> [Curl.CurlOption] -> IO (Maybe CurlResponse)-openURIWithOpts url opts = do- resp <- curlGetResponse_ url opts- return $ if Curl.respCurlCode resp /= Curl.CurlOK- then Nothing- else Just resp--curlGetResponse_ :: URL- -> [Curl.CurlOption]- -> IO (Curl.CurlResponse_ [(String, String)] BS.ByteString)-curlGetResponse_ = Curl.curlGetResponse_+ downloadAsTags decoder manager url = do+ request <- HTTP.parseRequest url+ response <- HTTP.httpLbs request manager+ return $ TagSoup.parseTags $ decoder response -- | The default response decoder. This decoder attempts to infer the character -- set of the HTTP response body from the `Content-Type` header. If this header@@ -99,24 +80,24 @@ defaultDecoder response = TagSoup.castString $ choosenDecoder body where- body = Curl.respBody response- headers = Curl.respHeaders response+ body = HTTP.responseBody response+ headers = HTTP.responseHeaders response contentType = listToMaybe- $ map (map toLower . snd)+ $ map (Text.decodeLatin1 . snd) $ take 1- $ dropWhile ((/= "content-type") . map toLower . fst)+ $ dropWhile ((/= "content-type") . fst) headers - isType t | Just ct <- contentType = ("charset=" ++ t) `isInfixOf` ct+ isType t | Just ct <- contentType = ("charset=" `Text.append` t) `Text.isInfixOf` ct | otherwise = False - choosenDecoder | isType "utf-8" = Text.decodeUtf8- | otherwise = Text.decodeLatin1+ choosenDecoder | isType "utf-8" = Text.decodeUtf8 . LBS.toStrict+ | otherwise = Text.decodeLatin1 . LBS.toStrict -- | A decoder that will always decode using `UTF-8`. utf8Decoder :: TagSoup.StringLike str => Decoder str-utf8Decoder = TagSoup.castString . Text.decodeUtf8 . Curl.respBody+utf8Decoder = TagSoup.castString . Text.decodeUtf8 . LBS.toStrict . HTTP.responseBody -- | A decoder that will always decode using `ISO-8859-1`. iso88591Decoder :: TagSoup.StringLike str => Decoder str-iso88591Decoder = TagSoup.castString . Text.decodeLatin1 . Curl.respBody+iso88591Decoder = TagSoup.castString . Text.decodeLatin1 . LBS.toStrict . HTTP.responseBody