rdf (empty) → 0.1.0.0
raw patch · 14 files changed
+1197/−0 lines, 14 filesdep +attoparsecdep +basedep +bytestringsetup-changed
Dependencies added: attoparsec, base, bytestring, criterion, deepseq, dlist, fgl, rdf, text, transformers
Files
- ChangeLog.md +3/−0
- LICENSE +21/−0
- README.md +5/−0
- Setup.hs +2/−0
- bench/Main.hs +48/−0
- rdf.cabal +59/−0
- src/Data/RDF/Encoder/Common.hs +104/−0
- src/Data/RDF/Encoder/NQuads.hs +63/−0
- src/Data/RDF/Graph.hs +112/−0
- src/Data/RDF/Internal.hs +475/−0
- src/Data/RDF/Parser/Common.hs +31/−0
- src/Data/RDF/Parser/NQuads.hs +112/−0
- src/Data/RDF/ToRDF.hs +128/−0
- src/Data/RDF/Types.hs +34/−0
+ ChangeLog.md view
@@ -0,0 +1,3 @@+* 0.1.0.0++Initial release. RDF generation DSL needs more instances.
+ LICENSE view
@@ -0,0 +1,21 @@+The MIT License (MIT)++Copyright (c) 2016 Travis Whitaker++Permission is hereby granted, free of charge, to any person obtaining a copy+of this software and associated documentation files (the "Software"), to deal+in the Software without restriction, including without limitation the rights+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell+copies of the Software, and to permit persons to whom the Software is+furnished to do so, subject to the following conditions:++The above copyright notice and this permission notice shall be included in all+copies or substantial portions of the Software.++THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE+SOFTWARE.
+ README.md view
@@ -0,0 +1,5 @@+# rdf++Data structures, parsers, and encoders for RDF data sets based on the RDF 1.1+abstract syntax and RFC 3987. The interface is intended to support incremental+graph processing in constant space.
+ Setup.hs view
@@ -0,0 +1,2 @@+import Distribution.Simple+main = defaultMain
+ bench/Main.hs view
@@ -0,0 +1,48 @@+{-# LANGUAGE OverloadedStrings #-}++module Main where++import Criterion.Main++import qualified Data.ByteString.Builder as B++import Data.RDF.Types+import Data.RDF.Encode.NQuads+import Data.RDF.Parser.NQuads++import Data.String++import qualified Data.Text as T++import qualified Data.Text.Lazy as TL+import qualified Data.Text.Lazy.Encoding as TL++-- | This function is inefficient. Run it outside of the benchmarks.+exampleGraph :: Int -> RDFGraph+exampleGraph n = RDFGraph (Just "http://user@benchmark.graph:8888/?graph") (take n (trips 0))+ where trips i = Triple (sub i) (pred i) (obj i) : trips (i+1)+ sub = fromString . ("_:" ++) . show+ pred = fromString . ("<http://user@benchmark.graph/succ#" ++) . (++ ">") . show+ obj = fromString . show++exampleDoc :: Int -> TL.Text+exampleDoc = TL.decodeUtf8 . B.toLazyByteString . encodeRDFGraph . exampleGraph++mkBenchEncodeGraph :: String -> Int -> Benchmark+mkBenchEncodeGraph p n = env (return (exampleGraph n))+ (\ ~g -> bench (p ++ "/" ++ show n)+ (nf (B.toLazyByteString . encodeRDFGraph) g))++mkBenchDecodeGraph :: String -> Int -> Benchmark+mkBenchDecodeGraph p n = env (return (exampleDoc n))+ (\ ~d -> bench (p ++ "/" ++ show n)+ (nf parseNQuads d))++main :: IO ()+main = defaultMain [ bgroup "fine" [ bgroup "encodeGraphFine" $ map (mkBenchEncodeGraph "encodeRDFGraph") [500,1000..100000]+ , bgroup "parseGraphsFine" $ map (mkBenchDecodeGraph "parseRDFGraph") [500,1000..100000]+ ]+ , bgroup "coarse" [ bgroup "encodeGraphCoarse" $ map (mkBenchEncodeGraph "encodeRDFGraph") [100000,200000..1000000]+ , bgroup "parseGraphsCoarse" $ map (mkBenchDecodeGraph "parseRDFGraph") [100000,200000..1000000]+ ]+ ]
+ rdf.cabal view
@@ -0,0 +1,59 @@+name: rdf+version: 0.1.0.0+synopsis: Representation and Incremental Processing of RDF Data+description:+ Data structures, parsers, and encoders for RDF data sets based on the+ RDF 1.1 abstract syntax and RFC 3987. The interface is intended to support+ incremental graph processing in constant space.++homepage: https://github.com/traviswhitaker/rdf+bug-reports: https://github.com/traviswhitaker/rdf/issues+license: MIT+license-file: LICENSE+author: Travis Whitaker+maintainer: pi.boy.travis@gmail.com+copyright: Travis Whitaker 2016+category: Data+build-type: Simple+extra-source-files: ChangeLog.md, README.md+cabal-version: >=1.10++library+ exposed-modules: Data.RDF.Types+ , Data.RDF.Encoder.Common+ , Data.RDF.Encoder.NQuads+ , Data.RDF.Graph+ , Data.RDF.Parser.Common+ , Data.RDF.Parser.NQuads+ , Data.RDF.ToRDF+ other-modules: Data.RDF.Internal+ build-depends: base >=4.8 && < 5.0+ , attoparsec >=0.13 && <0.14+ , bytestring >=0.10 && <0.11+ , deepseq >=1.4 && <1.5+ , dlist+ , fgl >=5.5 && <5.6+ , text >=1.2 && <1.3+ , transformers >=0.4 && <0.6+ hs-source-dirs: src+ ghc-options: -Wall+ -fwarn-identities+ -fwarn-missing-signatures+ default-language: Haskell2010++benchmark bench-rdf+ type: exitcode-stdio-1.0+ hs-source-dirs: bench+ main-is: Main.hs+ build-depends: base+ , bytestring+ , criterion+ , deepseq+ , rdf+ , text+ ghc-options: -rtsopts+ default-language: Haskell2010++source-repository head+ type: git+ location: https://github.com/TravisWhitaker/rdf.git
+ src/Data/RDF/Encoder/Common.hs view
@@ -0,0 +1,104 @@+{-|+Module : Data.RDF.Encode.Common+Description : Representation and Incremental Processing of RDF Data+Copyright : Travis Whitaker 2016+License : MIT+Maintainer : pi.boy.travis@gmail.com+Stability : Provisional+Portability : Portable++This module provides encoders for the primitive terms in the RDF abstract syntax+as described in RDF 1.1 Concepts and Abstract Syntax. These should be useful for+all RDF host languages.+-}++{-# LANGUAGE OverloadedStrings #-}++module Data.RDF.Encoder.Common (+ -- * Triple Components+ encodeSubject+ , encodePredicate+ , encodeObject+ -- * Terms+ , encodeBlankNode+ , encodeLiteral+ -- ** IRIs+ , encodeIRI+ , encodeEscapedIRI+ -- * Utilities+ , quoteString+ , maybeBuilder+ ) where++import qualified Data.ByteString.Builder as B++import Data.Monoid++import Data.RDF.Types++import qualified Data.Text as T+import qualified Data.Text.Encoding as T++-- | Escape the double quotes in a quoted string literal.+quoteString :: T.Text -> T.Text+quoteString = T.replace "\"" "\\\""++-- | Maps 'Nothing' to 'mempty'.+maybeBuilder :: Maybe B.Builder -> B.Builder+maybeBuilder Nothing = mempty+maybeBuilder (Just b) = b++-- | Encode an escaped 'IRI', i.e. between angle brackets.+encodeEscapedIRI :: IRI -> B.Builder+encodeEscapedIRI i = B.byteString "<" <> encodeIRI i <> B.byteString ">"++-- | Encode an 'IRI'.+encodeIRI :: IRI -> B.Builder+encodeIRI (IRI s a p q f) = T.encodeUtf8Builder s+ <> B.byteString ":"+ <> maybeBuilder (encodeIRIAuth <$> a)+ <> B.byteString "/"+ <> T.encodeUtf8Builder p+ <> maybeBuilder (((B.byteString "?" <>) . T.encodeUtf8Builder) <$> q)+ <> maybeBuilder (((B.byteString "#" <>) . T.encodeUtf8Builder) <$> f)++-- | Encode an 'IRIAuth'.+encodeIRIAuth :: IRIAuth -> B.Builder+encodeIRIAuth (IRIAuth u h p) = B.byteString "//"+ <> maybeBuilder (((<> B.byteString "@") . T.encodeUtf8Builder) <$> u)+ <> T.encodeUtf8Builder h+ <> maybeBuilder (((B.byteString ":" <>) . T.encodeUtf8Builder) <$> p)++-- | Encode a 'Literal', including the 'LiteralType'.+encodeLiteral :: Literal -> B.Builder+encodeLiteral (Literal v t) = B.byteString "\""+ <> T.encodeUtf8Builder (quoteString v)+ <> B.byteString "\""+ <> encodeLiteralType t++-- | Encode a 'LiteralType'.+encodeLiteralType :: LiteralType -> B.Builder+encodeLiteralType (LiteralIRIType i) = B.byteString "^^"+ <> encodeEscapedIRI i+encodeLiteralType (LiteralLangType l) = B.byteString "@"+ <> T.encodeUtf8Builder l+encodeLiteralType LiteralUntyped = mempty++-- | Encode a 'BlankNode'.+encodeBlankNode :: BlankNode -> B.Builder+encodeBlankNode (BlankNode l) = B.byteString "_:" <> T.encodeUtf8Builder l++-- | Encode a 'Subject'.+encodeSubject :: Subject -> B.Builder+encodeSubject (IRISubject i) = encodeEscapedIRI i+encodeSubject (BlankSubject b) = encodeBlankNode b++-- | Encode a 'Predicate'.+encodePredicate :: Predicate -> B.Builder+encodePredicate (Predicate i) = encodeEscapedIRI i++-- | Encode a 'Object'.+encodeObject :: Object -> B.Builder+encodeObject (IRIObject i) = encodeEscapedIRI i+encodeObject (LiteralObject l) = encodeLiteral l+encodeObject (BlankObject b) = encodeBlankNode b
+ src/Data/RDF/Encoder/NQuads.hs view
@@ -0,0 +1,63 @@+{-|+Module : Data.RDF.Encode.NQuads+Description : Representation and Incremental Processing of RDF Data+Copyright : Travis Whitaker 2016+License : MIT+Maintainer : pi.boy.travis@gmail.com+Stability : Provisional+Portability : Portable++An encoder for+<https://www.w3.org/TR/2014/REC-n-quads-20140225/ RDF 1.1 N-Quads>.+'B.Builder's are used to support efficient incremental output.+-}++{-# LANGUAGE OverloadedStrings #-}++module Data.RDF.Encoder.NQuads (+ -- * Graph Encoding+ encodeRDFGraph+ , encodeRDFGraphs+ , encodeTriple+ , encodeQuad+ ) where++import qualified Data.ByteString.Builder as B++import Data.Monoid++import Data.RDF.Types+import Data.RDF.Encoder.Common++-- | Encodes a 'Triple' as a single line, i.e. with no graph label. Includes the+-- terminating period and newline.+encodeTriple :: Triple -> B.Builder+encodeTriple (Triple s p o) = encodeSubject s+ <> B.byteString " "+ <> encodePredicate p+ <> B.byteString " "+ <> encodeObject o+ <> B.byteString " .\n"++-- | Encodes a 'Quad' as a single line. Includes the terminating period and+-- newline.+encodeQuad :: Quad -> B.Builder+encodeQuad (Quad t Nothing) = encodeTriple t+encodeQuad (Quad (Triple s p o) (Just g)) = encodeSubject s+ <> B.byteString " "+ <> encodePredicate p+ <> B.byteString " "+ <> encodeObject o+ <> B.byteString " "+ <> encodeEscapedIRI g+ <> B.byteString " .\n"++-- | Encode a single 'RDFGraph' as a 'B.Builder'.+encodeRDFGraph :: RDFGraph -> B.Builder+encodeRDFGraph (RDFGraph Nothing ts) = mconcat $ map encodeTriple ts+encodeRDFGraph (RDFGraph (Just g) ts) = let qs = map (\t -> Quad t (Just g)) ts+ in mconcat $ map encodeQuad qs++-- | Encode multiple 'RDFGraph's as a 'B.Builder'.+encodeRDFGraphs :: Foldable f => f RDFGraph -> B.Builder+encodeRDFGraphs = foldMap encodeRDFGraph
+ src/Data/RDF/Graph.hs view
@@ -0,0 +1,112 @@+{-|+Module : Data.RDF.Graph+Description : Representation and Incremental Processing of RDF Data+Copyright : Travis Whitaker 2016+License : MIT+Maintainer : pi.boy.travis@gmail.com+Stability : Provisional+Portability : Portable++This module provides conversion between RDF triples and @fgl@ graphs. Naturally+these functions will force the entire graph into memory.+-}++{-# LANGUAGE DeriveGeneric+ , DeriveAnyClass+ #-}++module Data.RDF.Graph (+ -- FGL Supporting Types+ GNode(..)+ , GEdge+ -- * Conversion to FGL Graphs+ , rdfGraph+ , triplesGraph+ -- * Conversion from FGL Graphs+ , graphRDF+ , graphTriples+ ) where++import Control.DeepSeq++import qualified Data.Graph.Inductive.Graph as G+import qualified Data.Graph.Inductive.NodeMap as G++import Data.Maybe++import Data.RDF.Types++import GHC.Generics++-- | An RDF 'Subject' or 'Object' as a 'G.Graph' node. This common+-- representation is necessary because the 'Object' of one 'Triple' might be+-- the 'Subject' of another.+data GNode = IRIGNode !IRI+ | BlankGNode !BlankNode+ | LiteralGNode !Literal+ deriving ( Eq+ , Ord+ , Read+ , Show+ , Generic+ , NFData+ )++-- | A 'G.Graph' edge is an RDF 'Predicate'.+type GEdge = Predicate++-- | Convert a 'Subject' to a 'GNode'.+subjectNode :: Subject -> GNode+subjectNode (IRISubject i) = IRIGNode i+subjectNode (BlankSubject b) = BlankGNode b++-- | Convert an 'Object' to a 'GNode'.+objectNode :: Object -> GNode+objectNode (IRIObject i) = IRIGNode i+objectNode (BlankObject b) = BlankGNode b+objectNode (LiteralObject l) = LiteralGNode l++-- | Convert a 'GNode' to a 'Subject'. This will fail if the 'GNode' contains a+-- 'Literal'.+nodeSubject :: GNode -> Either String Subject+nodeSubject (IRIGNode i) = Right (IRISubject i)+nodeSubject (BlankGNode b) = Right (BlankSubject b)+nodeSubject _ = Left "nodeSubject: subject must IRI or blank node."++-- | Convert a 'GNode' to an 'Object'.+nodeObject :: GNode -> Object+nodeObject (IRIGNode i) = IRIObject i+nodeObject (BlankGNode b) = BlankObject b+nodeObject (LiteralGNode l) = LiteralObject l++-- | Convert an 'RDFGraph' into a 'G.DynGraph' and 'G.NodeMap'. The 'graphLabel'+-- is discarded.+rdfGraph :: G.DynGraph g => RDFGraph -> (g GNode GEdge, G.NodeMap GNode)+rdfGraph (RDFGraph _ ts) = triplesGraph ts++-- | Convert a list of 'Triple's into a 'G.DynGraph' and a 'G.NodeMap'.+triplesGraph :: G.DynGraph g => [Triple] -> (g GNode GEdge, G.NodeMap GNode)+triplesGraph triples = G.mkMapGraph nodes edges+ where (nodes, edges) = go ([],[]) triples+ go (ns, es) [] = (ns, es)+ go (ns, es) (Triple s p o:ts) = let s' = subjectNode s+ o' = objectNode o+ in go (s':o':ns, (s', o', p):es) ts++-- | Convert a 'G.Graph' into an 'RDFGraph'. This will fail if the graph+-- contains any 'LiteralGNode's with an outward degree greater than zero,+-- since such a graph is illegal in RDF.+graphRDF :: G.Graph g => (Maybe IRI) -> g GNode GEdge -> Either String RDFGraph+graphRDF l = (RDFGraph l <$>) . graphTriples++-- | Convert a 'G.Graph' into a list of 'Triple's. This will fail if the graph+-- contains any 'LiteralGNode's with an outward degree greater than zero,+-- since such a graph is illegal in RDF.+graphTriples :: G.Graph g => g GNode GEdge -> Either String [Triple]+graphTriples g = go (G.labEdges g)+ -- The use of fromJust is safe here, since labEdges will never return+ -- an edge to a node not present in the graph.+ where go [] = Right []+ go ((si, oi, p):ts) = let s = nodeSubject (fromJust (G.lab g si))+ o = nodeObject (fromJust (G.lab g oi))+ in ((\s' -> (Triple s' p o:)) <$> s) <*> go ts
+ src/Data/RDF/Internal.hs view
@@ -0,0 +1,475 @@+{-|+Module : Data.RDF.Internal+Description : Representation and Incremental Processing of RDF Data+Copyright : Travis Whitaker 2016+License : MIT+Maintainer : pi.boy.travis@gmail.com+Stability : Provisional+Portability : Portable++Internal module.+-}++{-# LANGUAGE DeriveGeneric+ , DeriveAnyClass+ , OverloadedStrings+ #-}++module Data.RDF.Internal where++import Control.Applicative++import Control.DeepSeq++import qualified Data.Attoparsec.Combinator as A+import qualified Data.Attoparsec.Text as A++import Data.Char++import Data.String++import GHC.Generics++import qualified Data.Text as T++-- | A contiguous RDF graph with optional label. Note that a contiguous graph+-- within an RDF data set will not appear as a single contiguous graph to this+-- library if the graph's constituent triples are not contiguous in the+-- original data set. This strategy allows for incremental processing of RDF+-- data in constant space.+data RDFGraph = RDFGraph {+ -- | A named RDF graph includes an 'IRI'.+ rdfLabel :: !(Maybe IRI)+ -- | The constituent triples. A proper graph is a strict set of triples+ -- (i.e. no duplicate nodes or edges), but this guarantee cannot be made+ -- if the triples are to be processed incrementally in constant space.+ -- Programs using this type for interpreting RDF graphs should ignore any+ -- supernumerary triples in this list.+ , rdfTriples :: [Triple]+ } deriving ( Eq+ , Ord+ , Read+ , Show+ , Generic+ , NFData+ )++-- | An RDF quad, i.e. a triple belonging to a named graph.+data Quad = Quad {+ quadTriple :: !Triple+ , quadGraph :: !(Maybe IRI)+ } deriving ( Eq+ , Ord+ , Read+ , Show+ , Generic+ , NFData+ )++-- | An RDF triple.+data Triple = Triple !Subject !Predicate !Object+ deriving ( Eq+ , Ord+ , Read+ , Show+ , Generic+ , NFData+ )++-- | An RDF subject, i.e. either an 'IRI' or a 'BlankNode'.+--+-- This type has an 'IsString' instance, allowing string literals to be+-- interpreted as 'Subject's with @-XOverloadedStrings@, like so:+--+-- >>> "<http://example.com> :: Subject+-- IRISubject (IRI (...))+-- >>> "_:some-node" :: Subject+-- BlankSubject (BlankNode {unBlankNode = "some-node"})+data Subject = IRISubject !IRI+ | BlankSubject !BlankNode+ deriving ( Eq+ , Ord+ , Read+ , Show+ , Generic+ , NFData+ )++-- | An RDF predicate.+--+-- This type has an 'IsString' instance, allowing string literals to be+-- interpreted as 'Predicate's with @-XOverloadedStrings@, like so:+--+-- >>> "<http://example.com>" :: Predicate+-- Predicate {unPredicate = IRI (...)}+newtype Predicate = Predicate { unPredicate :: IRI }+ deriving ( Eq+ , Ord+ , Read+ , Show+ , Generic+ , NFData+ )++-- | An RDF object, i.e. either an 'IRI', a 'Literal', or a 'BlankNode'.+--+-- This type has an 'IsString' instance, allowing string literals to be+-- interpreted as 'Object's with @-XOverloadedStrings@, like so:+--+-- >>> "<http://example.com>" :: Object+-- IRIObject (IRI (...))+-- >>> "_:some-node" :: Object+-- BlankObject (BlankNode {unBlankNode = "some-node"})+-- >>> "computer" :: Object+-- LiteralObject (Literal {litString = "computer", litType = LiteralUntyped})+--+-- The precedence for literal interpretation is IRI > BlankNode > Literal. To+-- force a literal that is also a valid blank node label or IRI to be+-- interpreted as a 'LiteralObject', wrap it in an extra set of double quotes:+--+-- >>> "\"_:some-node\"" :: Object+-- LiteralObject (Literal {litString = "_:some-node", litType = LiteralUntyped})+data Object = IRIObject !IRI+ | BlankObject !BlankNode+ | LiteralObject !Literal+ deriving ( Eq+ , Ord+ , Read+ , Show+ , Generic+ , NFData+ )++-- | A blank node with its local label, without the preceeding "_:". Other+-- programs processing RDF are permitted to discard these node labels, i.e.+-- all blank node labels are local to a specific representation of an RDF data+-- set.+--+-- This type has an 'IsString' instance, allowing string literals to be+-- interpreted as 'BlankNode's with @-XOverloadedStrings@, like so:+--+-- >>> "_:some-node" :: BlankNode+-- BlankNode {unBlankNode = "some-node"}+newtype BlankNode = BlankNode { unBlankNode :: T.Text }+ deriving ( Eq+ , Ord+ , Read+ , Show+ , Generic+ , NFData+ )++-- | An RDF literal. As stipulated by the RDF standard, the 'litType' is merely+-- metadata; all RDF processing programs must try to handle literals that are+-- ill-typed.+--+-- This type has an 'IsString' instance, allowing string literals to be+-- interpreted as 'Literal's with @-XOverloadedStrings@, like so:+--+-- >>> "computer" :: Literal+-- Literal {litString = "computer", litType = LiteralUntyped}+--+-- For untyped literals the extra double quotes are not required. They are+-- required for typed literals:+--+-- >>> "\"computer\"@en" :: Literal+-- Literal {litString = "computer", litType = LiteralLangType "en"}+--+-- >>> "\"computer\"^^<http://computer.machine/machine>" :: Literal+-- Literal { litString = "computer", litType = LiteralIRIType (...)}+data Literal = Literal {+ litString :: !T.Text+ , litType :: !LiteralType+ } deriving ( Eq+ , Ord+ , Read+ , Show+ , Generic+ , NFData+ )++-- | An RDF literal type. As stipulated by the RDF standard, this is merely+-- metadata; all RDF processing programs must try to handle literals that are+-- ill-typed.+data LiteralType = LiteralIRIType !IRI+ | LiteralLangType !T.Text+ | LiteralUntyped+ deriving ( Eq+ , Ord+ , Read+ , Show+ , Generic+ , NFData+ )++-- | An Internationalized Resource Identifier. This library preferentially+-- follows RFC 3987 over the RDF 1.1 specification, as the two standards+-- disagree about precisely what constitutes an IRI. A notable exception is+-- the handling of IRI fragments; this library follows the RDF 1.1+-- specification, allowing IRI fragments to occur in absolute IRIs, even+-- though this is expressly prohibited by RFC 3987.+--+-- Unlike the @network-uri@ package's behavior with URI fields, this library+-- does not include the sentinel tokens in the parsed fields. For example,+-- when parsing @http://example.com@, @network-uri@ will provide the string+-- @http:@ as the scheme, while this library will provide @http@ as the+-- scheme.+--+-- This type has an 'IsString' instnace, allowing string literals to be+-- interpreted as 'IRI's with @-XOverloadedStrings@, like so:+--+-- >>> "http://example.com" :: IRI+-- IRI { iriScheme = "http"+-- , iriAuth = Just (IRIAuth { iriUser = Nothing+-- , iriHost = "example.com"+-- , iriPort = Nothing+-- })+-- , iriPath = ""+-- , iriQuery = Nothing+-- , iriFragment = Nothing+-- }+data IRI = IRI {+ -- | The IRI scheme, e.g. @http@+ iriScheme :: !T.Text+ -- | The IRI authority, e.g. @example.com@+ , iriAuth :: !(Maybe IRIAuth)+ -- | The IRI path, e.g. @/posts//index.html@+ , iriPath :: !T.Text+ -- | The IRI query, i.e. the component after the @?@ if present.+ , iriQuery :: !(Maybe T.Text)+ -- | The IRI fragment, i.e. the component after the @#@ if present.+ , iriFragment :: !(Maybe T.Text)+ } deriving ( Eq+ , Ord+ , Read+ , Show+ , Generic+ , NFData+ )++-- | An IRI Authority, as described by RFC 3987.+data IRIAuth = IRIAuth {+ -- | The IRI user, i.e. the component before the @\@@ if present.+ iriUser :: !(Maybe T.Text)+ -- | The IRI host, e.g. @example.com@.+ , iriHost :: T.Text+ -- | The IRI port, i.e. the numeral after the @:@ if present.+ , iriPort :: !(Maybe T.Text)+ } deriving ( Eq+ , Ord+ , Read+ , Show+ , Generic+ , NFData+ )++-- | Predicate on 'Char's for acceptability for inclusion in an 'IRI'.+isIRI :: Char -> Bool+isIRI c = (c /= '<')+ && (c /= '>')+ && (c /= '"')+ && (c /= '{')+ && (c /= '}')+ && (c /= '|')+ && (c /= '^')+ && (c /= '`')+ && (c /= '\\')++-- | 'IRI' parser.+parseIRI :: A.Parser IRI+parseIRI = IRI <$> (parseScheme <* A.char ':')+ <*> parseAuth+ <*> parsePath+ <*> parseQuery+ <*> parseFragment++-- | 'IRI' scheme parser.+parseScheme :: A.Parser T.Text+parseScheme = A.takeWhile1 isScheme >>= check+ where check t+ | isAlpha (T.head t) = pure t+ | otherwise = fail "parseScheme: must start with letter."+ isScheme c = isAlphaNum c+ || (c == '+')+ || (c == '-')+ || (c == '.')++-- | 'IRIAuth' parser.+parseAuth :: A.Parser (Maybe IRIAuth)+parseAuth = A.option Nothing (A.string "//" *> (Just <$> parseIRIAuth))+ where parseIRIAuth = IRIAuth <$> parseUser+ <*> parseHost+ <*> parsePort++-- | 'IRIAuth' user parser.+parseUser :: A.Parser (Maybe T.Text)+parseUser = A.option Nothing (Just <$> (A.takeWhile1 isUser <* A.char '@'))+ where isUser c = isIRI c && (c /= '@')++-- | 'IRIAuth' host parser.+parseHost :: A.Parser T.Text+parseHost = A.takeWhile1 isHost+ where isHost c = isIRI c && (c /= '/') && (c /= ':')++-- | 'IRIAuth' port parser.+parsePort :: A.Parser (Maybe T.Text)+parsePort = A.option Nothing (Just <$> (A.char ':' *> A.takeWhile1 isDigit))++-- | 'IRI' path parser.+parsePath :: A.Parser T.Text+parsePath = A.option "" (A.char '/' *> A.takeWhile1 isPath)+ where isPath c = isIRI c && (c /= '?') && (c /= '#')++-- | 'IRI' query parser.+parseQuery :: A.Parser (Maybe T.Text)+parseQuery = A.option Nothing (Just <$> (A.char '?' *> A.takeWhile1 isQuery))+ where isQuery c = isIRI c && (c/= '#')++-- | 'IRI' fragment parser.+parseFragment :: A.Parser (Maybe T.Text)+parseFragment = A.option Nothing (Just <$> (A.char '#' *> A.takeWhile1 isIRI))++-- | Parser for graph labels, i.e. either an escaped 'IRI' or the empty string.+parseGraphLabel :: A.Parser (Maybe IRI)+parseGraphLabel = A.option Nothing (Just <$> parseEscapedIRI)++-- | 'Subject' parser.+parseSubject :: A.Parser Subject+parseSubject = do+ c <- A.anyChar+ case c of '<' -> IRISubject <$> (parseIRI <* A.char '>')+ '_' -> BlankSubject <$> (A.char ':' *> parseBlankNodeLabel)+ _ -> fail "parseSubject: must be blank node or IRI."++-- | 'Predicate' parser.+parsePredicate :: A.Parser Predicate+parsePredicate = Predicate <$> parseEscapedIRI++-- | 'Object' parser.+parseObject :: A.Parser Object+parseObject = do+ c <- A.anyChar+ case c of '<' -> IRIObject <$> (parseIRI <* A.char '>')+ '_' -> BlankObject <$> (A.char ':' *> parseBlankNodeLabel)+ _ -> LiteralObject <$> parseLiteralBody++-- | Parse an escaped 'IRI', i.e. an IRI enclosed in angle brackets.+parseEscapedIRI :: A.Parser IRI+parseEscapedIRI = A.char '<' *> parseIRI <* A.char '>'++-- | Parse a blank node label.+parseBlankNodeLabel :: A.Parser BlankNode+parseBlankNodeLabel = BlankNode <$> (A.takeWhile1 isLabel >>= check)+ where check t+ | isHead (T.head t) && isTail (T.last t) = pure t+ | otherwise = fail "parseBlankNode"+ isLabel = not . isSpace+ isHead c = isLabel c+ && (c /= '-')+ && (c /= '.')+ isTail c = isLabel c+ && (c /= '.')++-- | Parse a blank node label, with the preceeding @_:@.+parseBlankNode :: A.Parser BlankNode+parseBlankNode = A.string "_:" *> parseBlankNodeLabel++-- | Like 'parseLiteral', but without the leading double quote.+parseLiteralBody :: A.Parser Literal+parseLiteralBody = Literal <$> escString <*> valType+ where valType = valIRIType <|> valLangType <|> pure LiteralUntyped+ valIRIType = LiteralIRIType <$> (A.string "^^" *> parseEscapedIRI)+ valLangType = LiteralLangType <$> (A.char '@' *> A.takeWhile1 isLang)+ isLang c = isAlphaNum c || (c == '-')+ escString = unescapeAll <$> A.scan False machine+ machine False '\\' = Just True+ machine False '"' = Nothing+ machine False _ = Just False+ machine True _ = Just False+ unescapeAll = T.concat . unescapeFrag . T.splitOn "\\"+ unescapeFrag [] = []+ unescapeFrag (f:fs) = case T.uncons f of+ Nothing -> f : unescapeFrag fs+ (Just (e, f')) -> T.singleton (unescape e) : f' : unescapeFrag fs+ unescape 't' = '\t'+ unescape 'b' = '\b'+ unescape 'n' = '\n'+ unescape 'r' = '\r'+ unescape 'f' = '\f'+ unescape c = c++-- | Parse an RDF 'Literal', including the 'LiteralType' if present.+parseLiteral :: A.Parser Literal+parseLiteral = A.char '"' *> parseLiteralBody++-- | Parse an unescaped untyped RDF 'Literal'.+parseUnescapedLiteral :: A.Parser Literal+parseUnescapedLiteral = Literal <$> A.takeText <*> pure LiteralUntyped++-- | Make implementations for 'fromString' from a 'A.Parser'.+fromStringParser :: A.Parser a -- ^ The literal parser.+ -> String -- ^ The literal type name for error messages.+ -> (String -> a) -- ^ The 'fromString' implementation.+fromStringParser p n s = let t = T.pack s+ r = A.parseOnly p t+ in case r of (Left e) -> error $ mconcat+ [ "Invalid "+ , n+ , " literal ("+ , s+ , ") "+ , e+ ]+ (Right x) -> x++-- | This instance uses 'parseIRI' and calls 'error' if the literal is invalid.+-- It is not clear exactly when 'fromString' is evaluated so this error is+-- difficult to explictly catch. This can be solved by ensuring that your+-- 'IRI' literals are eagerly evaluated so any malformed literals can be+-- caught immediately. It would be nicer if this happened at compile time.+instance IsString IRI where+ fromString = fromStringParser parseIRI "IRI"++-- | This instance uses 'parseLiteral' and calls 'error' if the literal is+-- invalid. It is not clear exactly when 'fromString' is evaluated so this+-- error is difficult to explictly catch. This can be solved by ensuring that+-- your 'Literal' literals are eagerly evaluated so any malformed literals can+-- be caught immediately. It would be nicer if this happened at compile time.+instance IsString Literal where+ fromString = fromStringParser p "Literal"+ where p = parseLiteral <|> parseUnescapedLiteral++-- | This instance uses 'parseBlankNode' and calls 'error' if the literal is+-- invalid. It is not clear exactly when 'fromString' is evaluated so this+-- error is difficult to explictly catch. This can be solved by ensuring that+-- your 'BlankNode' literals are eagerly evaluated so any malformed literals+-- can be caught immediately. It would be nicer if this happened at compile+-- time.+instance IsString BlankNode where+ fromString = fromStringParser parseBlankNode "BlankNode"++-- | This instance uses 'parseSubject' and calls 'error' if the literal+-- is invalid. It is not clear exactly when 'fromString' is evaluated so this+-- error is difficult to explictly catch. This can be solved by ensuring that+-- your 'Subject' literals are eagerly evaluated so any malformed literals can+-- be caught immediately. It would be nicer if this happened at compile time.+instance IsString Subject where+ fromString = fromStringParser parseSubject "Subject"++-- | This instance uses 'parsePredicate' and calls 'error' if the literal is+-- invalid. It is not clear exactly when 'fromString' is evaluated so this+-- error is difficult to explictly catch. This can be solved by ensuring that+-- your 'Predicate' literals are eagerly evaluated so any malformed literals+-- can be caught immediately. It would be nicer if this happened at compile+-- time.+instance IsString Predicate where+ fromString = fromStringParser parsePredicate "Predicate"++-- | This instance uses 'parseObject' and calls 'error' if the literal is+-- invalid. It is not clear exactly when 'fromString' is evaluated so this+-- error is difficult to explictly catch. This can be solved by ensuring that+-- your 'Object' literals are eagerly evaluated so any malformed literals can+-- be caught immediately. It would be nicer if this happened at compile time.+instance IsString Object where+ fromString = fromStringParser p "Object"+ where p = parseObject <|> (LiteralObject <$> parseUnescapedLiteral)
+ src/Data/RDF/Parser/Common.hs view
@@ -0,0 +1,31 @@+{-|+Module : Data.RDF.Parser.Common+Description : Representation and Incremental Processing of RDF Data+Copyright : Travis Whitaker 2016+License : MIT+Maintainer : pi.boy.travis@gmail.com+Stability : Provisional+Portability : Portable++This module provides parsers for the primitive terms in the RDF abstract syntax+as described in RDF 1.1 Concepts and Abstract Syntax. These should be useful for+all RDF host languages.+-}++{-# LANGUAGE OverloadedStrings #-}++module Data.RDF.Parser.Common (+ -- * Triple Components+ parseSubject+ , parsePredicate+ , parseObject+ , parseGraphLabel+ -- * Terms+ , parseBlankNode+ , parseLiteral+ -- ** IRIs+ , parseIRI+ , parseEscapedIRI+ ) where++import Data.RDF.Internal
+ src/Data/RDF/Parser/NQuads.hs view
@@ -0,0 +1,112 @@+{-|+Module : Data.RDF.Parser.NQuads+Description : Representation and Incremental Processing of RDF Data+Copyright : Travis Whitaker 2016+License : MIT+Maintainer : pi.boy.travis@gmail.com+Stability : Provisional+Portability : Portable++A parser for <https://www.w3.org/TR/2014/REC-n-quads-20140225/ RDF 1.1 N-Quads>.+-}++{-# LANGUAGE OverloadedStrings #-}++module Data.RDF.Parser.NQuads (+ Result+ , parseNQuads+ , parseTriple+ , parseQuad+ , parseQuadLine+ , foldGraphs+ , foldResults+ ) where++import qualified Data.Attoparsec.Text as A+import qualified Data.Attoparsec.Text.Lazy as AL++import Data.RDF.Types+import Data.RDF.Parser.Common++import qualified Data.Text.Lazy as TL++-- | Either an 'RDFGraph' or a parse error.+type Result = Either String RDFGraph++-- | A parser for+-- <https://www.w3.org/TR/2014/REC-n-quads-20140225/ RDF 1.1 N-Quads>. This+-- parser works incrementally by first lazily splitting the input into lines,+-- then parsing each line of the N-Quads document individually. This allows+-- for incremental processing in constant space, as well as extracting any+-- valid data from an N-Quads document that contains some invalid quads.+-- 'TL.Text' is used because the RDF 1.1 specification stipulates that RDF+-- should always be encoded with Unicode.+--+-- Due to its incremental nature, this parser will accept some N-Quads+-- documents that are not legal according to the RDF 1.1 specification.+-- Specifically, this parser will provide duplicate 'Triple's if they exist in+-- the input N-Quads document; a proper graph consists of true sets of nodes+-- and edges, i.e. no duplicate nodes or edges. Any downstream program+-- incrementally consuming this parser's output should take care to ignore any+-- supernumerary triples.+--+-- Likewise, if a graph's constituent triples are not contiguous in the input+-- N-Quads document, then they will not be folded into contiguous 'RDFGraph's+-- in this parser's output. Any downstream program incrementally consuming+-- this parser's output and performing graph processing that discriminates+-- based on graph labels will not necessarily be presented each contiguous+-- labeled graph as a single 'RDFGraph' record. For example, something like+-- this could be used to lazily find all 'RDFGraph' records containing a named+-- graph's 'Triple's. Downstream processing must then be able to handle a+-- single named graph spanning multiple 'RDFGraph' records.+--+-- > filterGraph :: (Maybe IRI) -> [RDFGraph] -> [RDFGraph]+-- > filterGraph gl = filter (\g -> (graphLabel g) == gl)+parseNQuads :: TL.Text -> [Result]+parseNQuads = foldResults+ . map (AL.eitherResult . AL.parse parseQuad)+ . TL.lines++-- | Fold a list of 'Quad's into a list of 'RDFGraph's, where adjacent 'Quad's+-- in the input are included in the same 'RDFGraph'.+foldGraphs :: [Quad] -> [RDFGraph]+foldGraphs [] = []+foldGraphs (quad:quads) = go (RDFGraph (quadGraph quad) [quadTriple quad]) quads+ where go g [] = [g]+ go g@(RDFGraph gl ts) (q:qs)+ | gl == quadGraph q = go (RDFGraph gl (quadTriple q:ts)) qs+ | otherwise = g : go (RDFGraph (quadGraph q)+ [quadTriple q]) qs++-- | Fold a list of parsed 'Quad's into a list of parsed 'RDFGraph's, where+-- adjacent 'Quad's in the input are included in the same 'RDFGraph'.+foldResults :: [Either String Quad] -> [Result]+foldResults [] = []+foldResults (Left e:quads) = Left e : foldResults quads+foldResults (Right quad:quads) = go (RDFGraph (quadGraph quad)+ [quadTriple quad])+ quads+ where go g [] = [Right g]+ go g (Left e:qs) = Right g : Left e : foldResults qs+ go g@(RDFGraph gl ts) (Right q:qs)+ | gl == quadGraph q = go (RDFGraph gl (quadTriple q:ts)) qs+ | otherwise = Right g : go (RDFGraph (quadGraph q)+ [quadTriple q]) qs++-- | Parse a single N-Quads 'Triple'.+parseTriple :: A.Parser Triple+parseTriple = Triple <$> (parseSubject <* A.skipSpace)+ <*> (parsePredicate <* A.skipSpace)+ <*> parseObject++-- | Parse a single N-Quads 'Quad'.+parseQuad :: A.Parser Quad+parseQuad = Quad <$> parseTriple+ <*> ((A.skipSpace *> parseGraphLabel) <*+ (A.skipSpace *> A.char '.'))++-- | Parse a single N-Quads 'Quad' on its own line. This parser is suitable for+-- using Attoparsec's incremental input mechanism 'parse'/'feed' instead of a+-- lazy 'T.Text'.+parseQuadLine :: A.Parser Quad+parseQuadLine = parseQuad <* A.char '\n'
+ src/Data/RDF/ToRDF.hs view
@@ -0,0 +1,128 @@+{-|+Module : Data.RDF.ToRDF+Description : DSL for Mapping Haskell Data to RDF Graphs+Copyright : Travis Whitaker 2016+License : MIT+Maintainer : pi.boy.travis@gmail.com+Stability : Provisional+Portability : Portable++This module provides a simple DSL for mapping Haskell data to RDF graphs.+-}++{-# LANGUAGE BangPatterns+ , DefaultSignatures+ , FlexibleContexts+ , FlexibleInstances+ , TupleSections+ #-}++module Data.RDF.ToRDF (+ ToRDF(..)+ , ToObject(..)+ , toTriples+ , Triples+ , RDFGen+ , runRDFGen+ , appBaseIRI+ , newBlankNode+ ) where++import Control.Monad.Trans.Reader+import Control.Monad.Trans.State.Lazy++import qualified Data.DList as DL++import Data.Int++import Data.Monoid++import Data.RDF.Types++import qualified Data.Text as T+import qualified Data.Text.Lazy as TL+import qualified Data.Text.Lazy.Builder as TL+import qualified Data.Text.Lazy.Builder.Int as TL+import qualified Data.Text.Lazy.Builder.RealFloat as TL++import Data.Word++type Triples = DL.DList Triple++-- | RDF generator monad. Provides 'ReaderT' for the base 'IRI', and 'StateT'+-- for a monotonically increasing blank node identifier.+type RDFGen a = ReaderT IRI (State Word64) a++runRDFGen :: RDFGen a -> IRI -> a+runRDFGen m i = evalState (runReaderT m i) 0++class ToRDF a where+ triples :: a -> RDFGen Triples++class ToObject a where+ object :: a -> RDFGen Object++instance ToObject Int where+ object = pure . toLObject . TL.decimal++instance ToObject Integer where+ object = pure . toLObject . TL.decimal++instance ToObject Int8 where+ object = pure . toLObject . TL.decimal++instance ToObject Int16 where+ object = pure . toLObject . TL.decimal++instance ToObject Int32 where+ object = pure . toLObject . TL.decimal++instance ToObject Int64 where+ object = pure . toLObject . TL.decimal++instance ToObject Word where+ object = pure . toLObject . TL.decimal++instance ToObject Word8 where+ object = pure . toLObject . TL.decimal++instance ToObject Word16 where+ object = pure . toLObject . TL.decimal++instance ToObject Word32 where+ object = pure . toLObject . TL.decimal++instance ToObject Word64 where+ object = pure . toLObject . TL.decimal++instance ToObject String where+ object s = pure $ LiteralObject (Literal (T.pack s) LiteralUntyped)++instance ToObject T.Text where+ object t = pure $ LiteralObject (Literal t LiteralUntyped)++-- | Forces the lazy 'TL.Text'.+instance ToObject TL.Text where+ object t = pure $ LiteralObject (Literal (TL.toStrict t) LiteralUntyped)++instance ToObject Float where+ object = pure . toLObject . TL.realFloat++instance ToObject Double where+ object = pure . toLObject . TL.realFloat++toTriples :: ToRDF a => IRI -> a -> [Triple]+toTriples i x = DL.toList (runRDFGen (triples x) i)++toText :: TL.Builder -> T.Text+toText = TL.toStrict . TL.toLazyText++toLObject :: TL.Builder -> Object+toLObject b = LiteralObject (Literal (toText b) LiteralUntyped)++appBaseIRI :: Endo IRI -> RDFGen IRI+appBaseIRI = asks . appEndo++newBlankNode :: RDFGen BlankNode+newBlankNode = ReaderT (const ((BlankNode . toText . TL.decimal)+ <$> get <* modify' (+1)))
+ src/Data/RDF/Types.hs view
@@ -0,0 +1,34 @@+{-|+Module : Data.RDF.Types+Description : Representation and Incremental Processing of RDF Data+Copyright : Travis Whitaker 2016+License : MIT+Maintainer : pi.boy.travis@gmail.com+Stability : Provisional+Portability : Portable++This module provides types for representing RDF data based on the abstract+syntax described in RDF 1.1 Concepts and Abstract Syntax.+-}++{-# LANGUAGE DeriveAnyClass #-}++module Data.RDF.Types (+ -- * Graphs+ RDFGraph(..)+ , Quad(..)+ , Triple(..)+ -- * Triple Components+ , Subject(..)+ , Predicate(..)+ , Object(..)+ -- * Terms+ , BlankNode(..)+ , Literal(..)+ , LiteralType(..)+ -- ** IRIs+ , IRI(..)+ , IRIAuth(..)+ ) where++import Data.RDF.Internal