chatter 0.5.1.0 → 0.5.2.0
raw patch · 11 files changed
+248/−150 lines, 11 filesdep −criteriondep −splitdep ~filepathdep ~textdep ~tokenizenew-component:exe:trainNERPVP ok
version bump matches the API change (PVP)
Dependencies removed: criterion, split
Dependency ranges changed: filepath, text, tokenize
API changes (from Hackage documentation)
+ NLP.Corpora.Conll: LOC :: NERTag
+ NLP.Corpora.Conll: MISC :: NERTag
+ NLP.Corpora.Conll: ORG :: NERTag
+ NLP.Corpora.Conll: PER :: NERTag
+ NLP.Corpora.Conll: data NERTag
+ NLP.Corpora.Conll: instance Arbitrary NERTag
+ NLP.Corpora.Conll: instance Bounded NERTag
+ NLP.Corpora.Conll: instance Constructor C1_0NERTag
+ NLP.Corpora.Conll: instance Constructor C1_1NERTag
+ NLP.Corpora.Conll: instance Constructor C1_2NERTag
+ NLP.Corpora.Conll: instance Constructor C1_3NERTag
+ NLP.Corpora.Conll: instance Datatype D1NERTag
+ NLP.Corpora.Conll: instance Enum NERTag
+ NLP.Corpora.Conll: instance Eq NERTag
+ NLP.Corpora.Conll: instance Generic NERTag
+ NLP.Corpora.Conll: instance NERTag NERTag
+ NLP.Corpora.Conll: instance Ord NERTag
+ NLP.Corpora.Conll: instance Read NERTag
+ NLP.Corpora.Conll: instance Serialize NERTag
+ NLP.Corpora.Conll: instance Show NERTag
+ NLP.Corpora.WikiNer: C_O :: Chunk
+ NLP.Corpora.WikiNer: LOC :: Chunk
+ NLP.Corpora.WikiNer: MISC :: Chunk
+ NLP.Corpora.WikiNer: ORG :: Chunk
+ NLP.Corpora.WikiNer: PER :: Chunk
+ NLP.Corpora.WikiNer: data Chunk
+ NLP.Corpora.WikiNer: instance Arbitrary Chunk
+ NLP.Corpora.WikiNer: instance Bounded Chunk
+ NLP.Corpora.WikiNer: instance ChunkTag Chunk
+ NLP.Corpora.WikiNer: instance Constructor C1_0Chunk
+ NLP.Corpora.WikiNer: instance Constructor C1_1Chunk
+ NLP.Corpora.WikiNer: instance Constructor C1_2Chunk
+ NLP.Corpora.WikiNer: instance Constructor C1_3Chunk
+ NLP.Corpora.WikiNer: instance Constructor C1_4Chunk
+ NLP.Corpora.WikiNer: instance Datatype D1Chunk
+ NLP.Corpora.WikiNer: instance Enum Chunk
+ NLP.Corpora.WikiNer: instance Eq Chunk
+ NLP.Corpora.WikiNer: instance Generic Chunk
+ NLP.Corpora.WikiNer: instance Ord Chunk
+ NLP.Corpora.WikiNer: instance Read Chunk
+ NLP.Corpora.WikiNer: instance Serialize Chunk
+ NLP.Corpora.WikiNer: instance Show Chunk
+ NLP.Corpora.WikiNer: parseWikiNer :: Text -> Either Error [[IOBChunk Chunk Tag]]
+ NLP.Corpora.WikiNer: trainChunker :: [FilePath] -> IO (Chunker Chunk Tag)
+ NLP.Corpora.WikiNer: wikiNerChunker :: IO (Chunker Chunk Tag)
+ NLP.Types.Tags: class (Ord a, Eq a, Read a, Show a, Generic a, Serialize a) => NERTag a where fromNERTag = pack . show parseNERTag txt = toEitherErr $ readEither $ unpack txt
+ NLP.Types.Tags: fromNERTag :: NERTag a => a -> Text
+ NLP.Types.Tags: parseNERTag :: NERTag a => Text -> Either Error a
Files
- appsrc/NERTrainer.hs +16/−0
- changelog.md +8/−0
- chatter.cabal +36/−17
- data/models/README +6/−0
- data/models/wikiner.ner.model.gz too large to diff
- src/NLP/Corpora/Conll.hs +15/−1
- src/NLP/Corpora/WikiNer.hs +136/−0
- src/NLP/Types/IOB.hs +19/−4
- src/NLP/Types/Tags.hs +12/−1
- tests/src/Bench.hs +0/−62
- tests/src/NLP/Similarity/VectorSimBench.hs +0/−65
+ appsrc/NERTrainer.hs view
@@ -0,0 +1,16 @@+{-# LANGUAGE OverloadedStrings #-}+module NERTrainer where++import System.Environment (getArgs)+import NLP.Chunk (saveChunker)+import qualified NLP.Corpora.WikiNer as W++main :: IO ()+main = do+ args <- getArgs+ let output = last args+ corpora = init args++ chunker <- W.trainChunker corpora+ saveChunker chunker output+
changelog.md view
@@ -1,3 +1,11 @@+= 0.5.2.0 =++ - Unceremoniously forced the Chunking model to do basic Named+ Entity Recognition, and added a simple NER model based on Nothman+ et al's WikiNER data set -- which is provided under a very+ permissive CC license.+ (http://schwa.org/projects/resources/wiki/Wikiner)+ = 0.5.1.0 = - Moved to Tasty from test-framework. I'm treating this as a
chatter.cabal view
@@ -1,5 +1,5 @@ name: chatter-version: 0.5.1.0+version: 0.5.2.0 synopsis: A library of simple NLP algorithms. description: chatter is a collection of simple Natural Language Processing algorithms.@@ -37,6 +37,7 @@ ./data/models/brown.pos.model.gz ./data/models/conll2000.pos.model.gz ./data/models/conll2000.chunk.model.gz+ ./data/models/wikiner.ner.model.gz source-repository head type: git@@ -65,6 +66,7 @@ NLP.Corpora.Email NLP.Corpora.Brown NLP.Corpora.Conll+ NLP.Corpora.WikiNer NLP.Similarity.VectorSim NLP.Extraction.Parsec NLP.Extraction.Examples.ParsecExamples@@ -141,9 +143,9 @@ ghc-options: -Wall -main-is ChunkTrainer -rtsopts -Executable eval+Executable trainNER default-language: Haskell2010- Main-Is: Evaluate.hs+ Main-Is: NERTrainer.hs hs-source-dirs: appsrc Build-depends: chatter,@@ -154,27 +156,44 @@ cereal >= 0.4.0.1, containers >= 0.5.0.0 - ghc-options: -Wall -main-is Evaluate -rtsopts+ ghc-options: -Wall -main-is NERTrainer -rtsopts -benchmark bench- type: exitcode-stdio-1.0- default-language: Haskell2010- Main-Is: Bench.hs- hs-source-dirs: tests/src - Other-modules: NLP.Similarity.VectorSimBench- Corpora+Executable eval+ default-language: Haskell2010+ Main-Is: Evaluate.hs+ hs-source-dirs: appsrc Build-depends: chatter,- criterion >= 0.8.0.1, filepath >= 1.3.0.1, text >= 0.11.3.0,- base >= 4.6 && <= 6,- deepseq,- split >= 0.1.2.3,- tokenize >= 0.2.0+ base >= 4.6 && <= 6,+ bytestring >= 0.10.0.0,+ cereal >= 0.4.0.1,+ containers >= 0.5.0.0 - ghc-options: -Wall -main-is Bench+ ghc-options: -Wall -main-is Evaluate -rtsopts++-- disabled because we aren't using it, and it's causing issues with `stack ghci`+-- benchmark bench+-- type: exitcode-stdio-1.0+-- default-language: Haskell2010+-- Main-Is: Bench.hs+-- hs-source-dirs: tests/src++-- Other-modules: NLP.Similarity.VectorSimBench+-- Corpora++-- Build-depends: chatter,+-- criterion >= 0.8.0.1,+-- filepath >= 1.3.0.1,+-- text >= 0.11.3.0,+-- base >= 4.6 && <= 6,+-- deepseq,+-- split >= 0.1.2.3,+-- tokenize >= 0.2.0++-- ghc-options: -Wall -main-is Bench test-suite tests
data/models/README view
@@ -40,3 +40,9 @@ real 0m20.730s user 0m20.298s sys 0m0.290s++---------------------------------------------------------------------+wikiner.ner.model.gz+---------------------------------------------------------------------++A Chunker that chunks to NER tags; trained on aij-wikiner-en-wp2
+ data/models/wikiner.ner.model.gz view
file too large to diff
src/NLP/Corpora/Conll.hs view
@@ -16,6 +16,20 @@ import qualified NLP.Types.Tags as T import NLP.Types.General +-- | Named entity categories defined for the Conll 2003 task.+data NERTag = PER+ | ORG+ | LOC+ | MISC+ deriving (Read, Show, Ord, Eq, Generic, Enum, Bounded)++instance Arbitrary NERTag where+ arbitrary = elements [minBound..]++instance Serialize NERTag+instance T.NERTag NERTag++-- | Phrase chunk tags defined for the Conll task. data Chunk = ADJP | ADVP | CONJP@@ -35,7 +49,6 @@ instance Serialize Chunk -instance Serialize Tag instance T.Tag Tag where fromTag = showTag@@ -56,6 +69,7 @@ instance Arbitrary Tag where arbitrary = elements [minBound ..]+instance Serialize Tag readTag :: Text -> Either Error Tag readTag "#" = Right Hash
+ src/NLP/Corpora/WikiNer.hs view
@@ -0,0 +1,136 @@+{-# LANGUAGE OverloadedStrings #-}+{-# LANGUAGE DeriveGeneric #-}+-- | A parser for the Wiki NER work presented in:+--+-- @Article{nothman2012:artint:wikiner,+-- author = {Joel Nothman and Nicky Ringland and Will Radford and Tara Murphy and James R. Curran},+-- title = {Learning multilingual named entity recognition from {Wikipedia}},+-- journal = {Artificial Intelligence},+-- publisher = {Elsevier},+-- volume = {194},+-- pages = {151--175},+-- year = {2012},+-- doi = {10.1016/j.artint.2012.03.006},+-- url = {http://dx.doi.org/10.1016/j.artint.2012.03.006}+-- }+--+-- And provided here: http://schwa.org/projects/resources/wiki/Wikiner+--+-- The format does not appear to be documented, but it looks like:+--+-- * One sentence per line.+--+-- * Tagged tokens are separated by spaces+--+-- * Items in a tagged token are separated by vertical bars ('|')+--+-- * Each line of `n` text tokens contains 3*n items, starting with a+-- text token, a POS tag, then a IOB tag with one of the NER classes+--+-- For example, the sentence:+-- The Oxford Companion to Philosophy says, "there is no single defining position that all anarchists hold, and those considered anarchists at best sharae a certain family resemblance."+--+-- Is rendered as:+-- The|DT|I-MISC Oxford|NNP|I-MISC Companion|NNP|I-MISC to|TO|I-MISC Philosophy|NNP|I-MISC says|VBZ|O ,|,|O "|LQU|O there|EX|O is|VBZ|O no|DT|O single|JJ|O defining|VBG|O position|NN|O that|IN|O all|DT|O anarchists|NNS|O hold|VBP|O ,|,|O and|CC|O those|DT|O considered|VBN|O anarchists|NNS|O at|IN|O best|JJS|O share|NN|O a|DT|O certain|JJ|O family|NN|O resemblance|NN|O .|.|O "|RQU|O+--+--+-- This module also provides a trained model for NER via the averaged+-- perceptron chunker. This actually kindof works, which is a bit+-- amazing. For example:+--+-- > import NLP.Corpora.WikiNer+-- > import NLP.POS+-- > import NLP.Chunk+-- > tgr <- defaultTagger+-- > chk <- wikiNerChunker+-- > chunkText tgr chk "Real World Haskell is a book created by Don Stewart, Bryan O'Sullivan, and Jon Goerzen."+-- > "[ORG Real/NNP] [MISC World/NNP] [PER Haskell/NNP] is/VBZ a/DT book/NN created/VBN by/IN [PER Don/NNP Stewart/NNP] ,/, [PER Bryan/NNP O'Sullivan/NNP] ,/, and/CC [PER Jon/NNP Goerzen/NNP] ./."+--+--+module NLP.Corpora.WikiNer+ ( parseWikiNer+ , trainChunker+ , wikiNerChunker+ , Chunk(..)+ )+where++import Data.Text (Text)+import qualified Data.Text as T+import qualified Data.Text.IO as T+import Data.Serialize (Serialize)+import GHC.Generics+import System.FilePath ((</>))+import Text.Read (readEither)+import Test.QuickCheck.Arbitrary (Arbitrary(..))+import Test.QuickCheck.Gen (elements)+++import NLP.Chunk (train, loadChunker)+import NLP.Chunk.AvgPerceptronChunker (Chunker(..), mkChunker)+import qualified NLP.Corpora.Conll as Conll+import NLP.ML.AvgPerceptron ( emptyPerceptron )+import NLP.Types.IOB hiding (parseIOB)+import NLP.Types.General (Error, toEitherErr)+import NLP.Types.Tags++import Paths_chatter++parseWikiNer :: Text -> Either Error [[IOBChunk Chunk Conll.Tag]]+parseWikiNer = parseIOB++-- | Convert wikiNer format to basic IOB (one token perline, space+-- separated tags, and a blank line between each sentence)+parseIOB :: (ChunkTag chunk, Tag tag) => Text -> Either Error [[IOBChunk chunk tag]]+parseIOB input = sequence $ map (parseSentence . toIOBLines) (filter (/="") $ T.lines input)++-- | Different classes of Named Entity used in the WikiNER data set.+data Chunk = LOC+ | MISC+ | ORG+ | PER+ | C_O -- ^ "out" not a chunk.+ deriving (Read, Show, Ord, Eq, Generic, Enum, Bounded)+++instance Arbitrary Chunk where+ arbitrary = elements [minBound ..]++instance Serialize Chunk++instance ChunkTag Chunk where+ fromChunk = T.pack . show+ parseChunk txt = toEitherErr $ readEither (T.unpack txt)+ notChunk = C_O++wikiNerChunker :: IO (Chunker Chunk Conll.Tag)+wikiNerChunker = do+ dir <- getDataDir+ loadChunker (dir </> "data" </> "models" </> "wikiner.ner.model.gz")++-- | Tranlsate a WikiNER sentence into a list of IOB-lines, for+-- parsing with `parseIOBLine`+toIOBLines :: Text -> [Text]+toIOBLines sent = map (T.replace "|" " ") (T.words sent)++-- | Train a chunker on a provided corpus.+trainChunker :: [FilePath] -> IO (Chunker Chunk Conll.Tag)+trainChunker corpora = do+ content <- mapM T.readFile corpora++ let trainingText = T.intercalate "\n" content++ eiobs = parseWikiNer trainingText++ chunker :: Chunker Chunk Conll.Tag+ chunker = mkChunker emptyPerceptron++ case eiobs of+ Left err -> do+ T.putStrLn err+ error (T.unpack err)+ Right iobs -> do+ print (take 1 iobs)+ let chunkSents = map toChunkTree iobs+ train chunker chunkSents+
src/NLP/Types/IOB.hs view
@@ -4,6 +4,7 @@ import Prelude hiding (print) import Control.Applicative ((<$>), (<*>)) import Data.Maybe (mapMaybe)+import Data.Monoid ((<>)) import Data.Text (Text) import qualified Data.Text as T @@ -14,6 +15,18 @@ import NLP.Types.Tree import NLP.Types.General (Error) ++-- TODO: This module needs to be rewritten to parse IOB represented+-- data as a tree, then once that tree is created, establish the+-- semantic types at the proper levels.+--+-- I think the levels should look something like this:+--+-- 0: Tokens+-- 1: POS Tags+-- 2/3: Chunks+-- 3/2: NER tags+ -- | Data type to indicate IOB tags for chunking data IOBChunk chunk tag = BChunk (POS tag) chunk -- ^ Beging marker. | IChunk (POS tag) chunk -- ^ In chunk tag@@ -46,10 +59,12 @@ -- parseIOBLine :: (ChunkTag chunk, Tag tag) => Text -> Either Error (IOBChunk chunk tag) parseIOBLine txt =- let (tokTxt:tagTxt:iobTxt:_) = T.words txt- token = Token tokTxt- tag = POS (parseTag tagTxt) token- in iobBuilder iobTxt tag+ case T.words txt of+ (tokTxt:tagTxt:iobTxt:_) ->+ let token = Token tokTxt+ tag = POS (parseTag tagTxt) token+ in iobBuilder iobTxt tag+ _ -> Left ("not enough words in IOB line: \"" <> txt <> "\"") iobBuilder :: (ChunkTag c, Tag t) => Text -> (POS t -> Either Error (IOBChunk c t)) iobBuilder iobTxt | "I-" `T.isPrefixOf` iobTxt = \tag -> (IChunk tag) <$> chunk
src/NLP/Types/Tags.hs view
@@ -9,11 +9,22 @@ import qualified Data.Text as T import Data.Text.Encoding (encodeUtf8, decodeUtf8) import GHC.Generics+import Text.Read (readEither) + import Test.QuickCheck (Arbitrary(..), NonEmptyList(..)) import Test.QuickCheck.Instances () -import NLP.Types.General (Error)+import NLP.Types.General (Error, toEitherErr)++-- | The class of named entity sets. This typeclass can be defined+-- entirely in terms of the required class constraints.+class (Ord a, Eq a, Read a, Show a, Generic a, Serialize a) => NERTag a where+ fromNERTag :: a -> Text+ fromNERTag = T.pack . show++ parseNERTag :: Text -> Either Error a+ parseNERTag txt = toEitherErr $ readEither $ T.unpack txt -- | The class of things that can be regarded as 'chunks'; Chunk tags -- are much like POS tags, but should not be confused. Generally,
− tests/src/Bench.hs
@@ -1,62 +0,0 @@-{-# LANGUAGE PackageImports #-}-{-# LANGUAGE OverloadedStrings #-}-module Bench where--import Data.Text (Text)-import qualified Data.Text as T-import qualified Data.Text.IO as T--import Control.DeepSeq-import Criterion.Main-import Criterion.Config (defaultConfig, Config(..), ljust)-import Criterion (bench, bgroup, Benchmark)--import NLP.POS (tagText)-import NLP.POS.AvgPerceptronTagger (trainNew, mkTagger)-import Corpora-import NLP.Corpora.Email--import qualified NLP.Similarity.VectorSimBench as VS--myConfig :: Config-myConfig = defaultConfig {- -- Always GC between runs.- cfgPerformGC = ljust True- }--main :: IO ()-main = do--- postagBench <- posTagging- muc3_1 <- VS.muc3_01- muc3_2 <- VS.muc3_02- muc3_3 <- VS.muc3_03- pTxt <- plugArchiveText- let len = length pTxt- plugTxt = take (len `div` 4) pTxt- plugStr = map T.unpack plugTxt- deepseq plugStr $ defaultMainWith myConfig (return ())- [ - -- bgroup "POS Tagging" [] -- postagBench- -- , bgroup "Similarity" $ VS.benchmarks (muc3_1++muc3_2) muc3_3- ]---- posTagging :: IO [Benchmark]--- posTagging = do--- ca01 <- T.readFile brownCA01--- ca02 <- T.readFile (brownCAFiles!!1)--- let ca1_2 = T.unlines [ca01, ca02]--- return [ bench "Train Brown ca01" $ trainNew ca01--- , bench "Train & test Brown ca01" $ trainAndTag ca01 "the dog jumped"---- , bench "Train Brown ca02" $ trainNew ca02--- , bench "Train & test Brown ca02" $ trainAndTag ca02 "the dog jumped"---- , bench "Train Brown ca01-02" $ trainNew ca1_2--- , bench "Train & test Brown ca01-02" $ trainAndTag ca1_2 "the dog jumped"--- ]--trainAndTag :: Text -> Text -> IO Text-trainAndTag corpus input = do- tagger <- trainNew corpus- return $ tagText (mkTagger tagger Nothing) input-
− tests/src/NLP/Similarity/VectorSimBench.hs
@@ -1,65 +0,0 @@-{-# LANGUAGE PackageImports #-}-{-# LANGUAGE OverloadedStrings #-}-module NLP.Similarity.VectorSimBench where--import Data.List.Split (splitWhen)-import Data.Text (Text)-import qualified Data.Text as T-import qualified Data.Text.IO as T-import Criterion (bench, whnf, Benchmark)--import NLP.Tokenize.Text (tokenize)-import NLP.Similarity.VectorSim-import NLP.Types (mkCorpus, Corpus)--benchmarks :: [[Text]] -> [[Text]] -> [Benchmark]-benchmarks docs testDocs = let- corpus = mkCorpus docs- in [ bench "Doc 1-2 vs 3-4" $ whnf (similarity corpus (concat $ take 2 testDocs))- ((testDocs!!2) ++ (testDocs!!3))- , bench "Doc 1-5 vs 6-10" $ whnf (similarity corpus (concat $ take 5 testDocs))- (concat $ take 5 $ drop 5 testDocs)- , bench "all pairs of 1-5" $ whnf (docsRunAllPairs corpus) (take 5 testDocs)-- , bench "TV all pairs of 1-5" $ whnf (tvDocsRunAllPairs corpus) (take 5 testDocs)- ]--docsRunAllPairs :: Corpus -> [[Text]] -> Double-docsRunAllPairs _ [] = 0-docsRunAllPairs corpus (d:ds) = let- firstRow = foldl (\v doc -> v + similarity corpus d doc) 0 ds- in firstRow + (docsRunAllPairs corpus ds)--tvDocsRunAllPairs :: Corpus -> [[Text]] -> Double-tvDocsRunAllPairs corpus docs = runVectors (map (mkVector corpus) docs)- where- runVectors :: [TermVector] -> Double- runVectors [] = 0- runVectors (d:ds) = let- firstRow = foldl (\v doc -> v + tvSim d doc) 0 ds- in firstRow + (runVectors ds)---readMucCorpus :: String -> IO [[Text]]-readMucCorpus file = do- content <- T.readFile ("./tests/resources/corpora/muc3_4/"++file)- let- docMarker :: Text -> Bool- docMarker txt = "DEV-MUC3-" `T.isPrefixOf` txt-- docLines :: [[Text]]- docLines = splitWhen docMarker $ T.lines content-- documents :: [Text]- documents = map T.unlines docLines-- return $ map tokenize documents--muc3_01 :: IO [[Text]]-muc3_01 = readMucCorpus "dev-muc3-0001-0100"--muc3_02 :: IO [[Text]]-muc3_02 = readMucCorpus "dev-muc3-0101-0200"--muc3_03 :: IO [[Text]]-muc3_03 = readMucCorpus "dev-muc3-0201-0300"