packages feed

BiobaseInfernal 0.0.2.1 → 0.5.0.0

raw patch · 12 files changed

+438/−665 lines, 12 filesdep +attoparsecdep +attoparsec-enumeratordep +bytestringdep −Biobasedep −arraydep −parsecdep ~basePVP ok

version bump matches the API change (PVP)

Dependencies added: attoparsec, attoparsec-enumerator, bytestring, enumerator, transformers, tuple

Dependencies removed: Biobase, array, parsec

Dependency ranges changed: base

API changes (from Hackage documentation)

- Biobase.Infernal.CM: B :: StateType
- Biobase.Infernal.CM: BEGL :: NodeType
- Biobase.Infernal.CM: BEGR :: NodeType
- Biobase.Infernal.CM: BIF :: NodeType
- Biobase.Infernal.CM: Branch :: Int -> Transition
- Biobase.Infernal.CM: CM :: Array Int (Node n) -> Array Int (State s) -> [(String, String)] -> Array Int Double -> Array Int Double -> CMType -> Array Nucleotide Double -> CM n s
- Biobase.Infernal.CM: CMProb :: CMType
- Biobase.Infernal.CM: CMScore :: CMType
- Biobase.Infernal.CM: D :: StateType
- Biobase.Infernal.CM: E :: StateType
- Biobase.Infernal.CM: END :: NodeType
- Biobase.Infernal.CM: EmitP :: Nucleotide -> Nucleotide -> Double -> Emission
- Biobase.Infernal.CM: EmitS :: Nucleotide -> Double -> Emission
- Biobase.Infernal.CM: IL :: StateType
- Biobase.Infernal.CM: IR :: StateType
- Biobase.Infernal.CM: MATL :: NodeType
- Biobase.Infernal.CM: MATP :: NodeType
- Biobase.Infernal.CM: MATR :: NodeType
- Biobase.Infernal.CM: ML :: StateType
- Biobase.Infernal.CM: MP :: StateType
- Biobase.Infernal.CM: MR :: StateType
- Biobase.Infernal.CM: Node :: Int -> NodeType -> [Int] -> [Int] -> [Int] -> n -> Node n
- Biobase.Infernal.CM: ROOT :: NodeType
- Biobase.Infernal.CM: S :: StateType
- Biobase.Infernal.CM: State :: Int -> StateType -> Int -> [Int] -> [Transition] -> [Emission] -> s -> State s
- Biobase.Infernal.CM: Transition :: Int -> Double -> Transition
- Biobase.Infernal.CM: cmMakeLocal :: Double -> Double -> CM n s -> CM n s
- Biobase.Infernal.CM: cmMakeLocalBegin :: Double -> CM n s -> CM n s
- Biobase.Infernal.CM: cmMakeLocalEnd :: Double -> CM n s -> CM n s
- Biobase.Infernal.CM: cmNormalizeProbabilities :: CM n s -> CM n s
- Biobase.Infernal.CM: cmProb2Score :: CM n s -> CM n s
- Biobase.Infernal.CM: cmScore2Prob :: CM n s -> CM n s
- Biobase.Infernal.CM: cmType :: CM n s -> CMType
- Biobase.Infernal.CM: data CM n s
- Biobase.Infernal.CM: data CMType
- Biobase.Infernal.CM: data Emission
- Biobase.Infernal.CM: data Node n
- Biobase.Infernal.CM: data NodeType
- Biobase.Infernal.CM: data State s
- Biobase.Infernal.CM: data StateType
- Biobase.Infernal.CM: data Transition
- Biobase.Infernal.CM: eNuc :: Emission -> Nucleotide
- Biobase.Infernal.CM: eNucL :: Emission -> Nucleotide
- Biobase.Infernal.CM: eNucR :: Emission -> Nucleotide
- Biobase.Infernal.CM: escore :: Emission -> Double
- Biobase.Infernal.CM: header :: CM n s -> [(String, String)]
- Biobase.Infernal.CM: instance (Show n, Show s) => Show (CM n s)
- Biobase.Infernal.CM: instance Bounded NodeType
- Biobase.Infernal.CM: instance Bounded StateType
- Biobase.Infernal.CM: instance Enum NodeType
- Biobase.Infernal.CM: instance Enum StateType
- Biobase.Infernal.CM: instance Eq CMType
- Biobase.Infernal.CM: instance Eq NodeType
- Biobase.Infernal.CM: instance Eq StateType
- Biobase.Infernal.CM: instance Ord NodeType
- Biobase.Infernal.CM: instance Ord StateType
- Biobase.Infernal.CM: instance Read NodeType
- Biobase.Infernal.CM: instance Read StateType
- Biobase.Infernal.CM: instance Show CMType
- Biobase.Infernal.CM: instance Show Emission
- Biobase.Infernal.CM: instance Show NodeType
- Biobase.Infernal.CM: instance Show StateType
- Biobase.Infernal.CM: instance Show Transition
- Biobase.Infernal.CM: instance Show n => Show (Node n)
- Biobase.Infernal.CM: instance Show s => Show (State s)
- Biobase.Infernal.CM: localBegin :: CM n s -> Array Int Double
- Biobase.Infernal.CM: localBeginPossible :: CM n s -> Node n -> Bool
- Biobase.Infernal.CM: localBeginProb2Score :: Array Int Double -> Array Int Double
- Biobase.Infernal.CM: localBeginScore2Prob :: Array Int Double -> Array Int Double
- Biobase.Infernal.CM: localEnd :: CM n s -> Array Int Double
- Biobase.Infernal.CM: localEndPossible :: CM n s -> Node n -> Bool
- Biobase.Infernal.CM: localEndProb2Score :: Array Int Double -> Array Int Double
- Biobase.Infernal.CM: localEndScore2Prob :: Array Int Double -> Array Int Double
- Biobase.Infernal.CM: nchildren :: Node n -> [Int]
- Biobase.Infernal.CM: nid :: Node n -> Int
- Biobase.Infernal.CM: nodeMainState :: CM n s -> Node n -> State s
- Biobase.Infernal.CM: nodeMainStateAssocs :: [(NodeType, StateType)]
- Biobase.Infernal.CM: nodes :: CM n s -> Array Int (Node n)
- Biobase.Infernal.CM: nparents :: Node n -> [Int]
- Biobase.Infernal.CM: nstates :: Node n -> [Int]
- Biobase.Infernal.CM: ntag :: Node n -> n
- Biobase.Infernal.CM: ntype :: Node n -> NodeType
- Biobase.Infernal.CM: nullModel :: CM n s -> Array Nucleotide Double
- Biobase.Infernal.CM: schildren :: State s -> [Transition]
- Biobase.Infernal.CM: semission :: State s -> [Emission]
- Biobase.Infernal.CM: sid :: State s -> Int
- Biobase.Infernal.CM: snode :: State s -> Int
- Biobase.Infernal.CM: sparents :: State s -> [Int]
- Biobase.Infernal.CM: stag :: State s -> s
- Biobase.Infernal.CM: stateProb2Score :: CM n s -> State s -> State s
- Biobase.Infernal.CM: stateScore2Prob :: CM n s -> State s -> State s
- Biobase.Infernal.CM: states :: CM n s -> Array Int (State s)
- Biobase.Infernal.CM: statesProb2Score :: CM n s -> Array Int (State s) -> Array Int (State s)
- Biobase.Infernal.CM: statesScore2Prob :: CM n s -> Array Int (State s) -> Array Int (State s)
- Biobase.Infernal.CM: stype :: State s -> StateType
- Biobase.Infernal.CM: tchild :: Transition -> Int
- Biobase.Infernal.CM: transitionTargets :: [Transition] -> [Int]
- Biobase.Infernal.CM: tscore :: Transition -> Double
- Biobase.Infernal.CM.Export: nodeToString :: CM n s -> Node n -> [String]
- Biobase.Infernal.CM.Export: stateToString :: CM n s -> State s -> String
- Biobase.Infernal.CM.Export: toString :: CM n s -> String
- Biobase.Infernal.CM.Import: fromFile :: String -> IO (Either ParseError [CM () ()])
- Biobase.Infernal.CM.Import: fromString :: [Char] -> Either ParseError [CM () ()]
- Biobase.Infernal.Stockholm: Stockholm :: [(String, String)] -> [(String, String)] -> [(String, String)] -> [String] -> Stockholm
- Biobase.Infernal.Stockholm: colAnnotations :: Stockholm -> [(String, String)]
- Biobase.Infernal.Stockholm: data Stockholm
- Biobase.Infernal.Stockholm: exAnnotations :: Stockholm -> [(String, String)]
- Biobase.Infernal.Stockholm: instance Show Stockholm
- Biobase.Infernal.Stockholm: sequences :: Stockholm -> [(String, String)]
- Biobase.Infernal.Stockholm: unknown :: Stockholm -> [String]
- Biobase.Infernal.Stockholm.Import: fromFile :: String -> IO Stockholm
+ Biobase.Infernal.Taxonomy: Species :: ByteString -> [ByteString] -> Int -> Species
+ Biobase.Infernal.Taxonomy: classification :: Species -> [ByteString]
+ Biobase.Infernal.Taxonomy: data Species
+ Biobase.Infernal.Taxonomy: instance Show Species
+ Biobase.Infernal.Taxonomy: name :: Species -> ByteString
+ Biobase.Infernal.Taxonomy: shortenName :: ByteString -> ByteString
+ Biobase.Infernal.Taxonomy: taxid :: Species -> Int
+ Biobase.Infernal.Taxonomy.Import: eeImport :: Monad m => Enumeratee ByteString Species m b
+ Biobase.Infernal.Taxonomy.Import: iSpeciesMap :: Monad m => Iteratee Species m (Map ByteString Species)
+ Biobase.Infernal.Taxonomy.Import: iTaxIdMap :: Monad m => Iteratee Species m (Map Int Species)
+ Biobase.Infernal.VerboseHit: VerboseHit :: (Int, Int) -> (Int, Int) -> ByteString -> Strand -> Double -> Double -> Double -> Int -> ByteString -> ByteString -> ByteString -> ByteString -> ByteString -> VerboseHit
+ Biobase.Infernal.VerboseHit: data VerboseHit
+ Biobase.Infernal.VerboseHit: instance Read VerboseHit
+ Biobase.Infernal.VerboseHit: instance Show VerboseHit
+ Biobase.Infernal.VerboseHit: type Strand = Char
+ Biobase.Infernal.VerboseHit: vhCM :: VerboseHit -> ByteString
+ Biobase.Infernal.VerboseHit: vhConsensus :: VerboseHit -> ByteString
+ Biobase.Infernal.VerboseHit: vhEvalue :: VerboseHit -> Double
+ Biobase.Infernal.VerboseHit: vhGC :: VerboseHit -> Int
+ Biobase.Infernal.VerboseHit: vhPvalue :: VerboseHit -> Double
+ Biobase.Infernal.VerboseHit: vhQuery :: VerboseHit -> (Int, Int)
+ Biobase.Infernal.VerboseHit: vhScaffold :: VerboseHit -> ByteString
+ Biobase.Infernal.VerboseHit: vhScore :: VerboseHit -> Double
+ Biobase.Infernal.VerboseHit: vhScoring :: VerboseHit -> ByteString
+ Biobase.Infernal.VerboseHit: vhSequence :: VerboseHit -> ByteString
+ Biobase.Infernal.VerboseHit: vhStrand :: VerboseHit -> Strand
+ Biobase.Infernal.VerboseHit: vhTarget :: VerboseHit -> (Int, Int)
+ Biobase.Infernal.VerboseHit: vhWuss :: VerboseHit -> ByteString
+ Biobase.Infernal.VerboseHit.Export: StreamCM :: ByteString -> HitStream a
+ Biobase.Infernal.VerboseHit.Export: StreamInsertion :: a -> HitStream a
+ Biobase.Infernal.VerboseHit.Export: StreamScaffold :: ByteString -> HitStream a
+ Biobase.Infernal.VerboseHit.Export: StreamStrand :: Char -> HitStream a
+ Biobase.Infernal.VerboseHit.Export: StreamVerboseHit :: VerboseHit -> HitStream a
+ Biobase.Infernal.VerboseHit.Export: data HitStream a
+ Biobase.Infernal.VerboseHit.Export: eeFlattenStream :: Monad m => Enumeratee [HitStream z] (HitStream z) m a
+ Biobase.Infernal.VerboseHit.Export: eeHitToStream :: Monad m => Enumeratee VerboseHit [HitStream z] m a
+ Biobase.Infernal.VerboseHit.Export: eeStreamToByteString :: (Monad m, Show z) => StreamToByteString m z
+ Biobase.Infernal.VerboseHit.Export: eeStreamToByteString' :: Monad m => StreamToByteString m ()
+ Biobase.Infernal.VerboseHit.Export: instance Show a => Show (HitStream a)
+ Biobase.Infernal.VerboseHit.Export: showCM :: ByteString -> ByteString
+ Biobase.Infernal.VerboseHit.Export: showScaffold :: ByteString -> ByteString
+ Biobase.Infernal.VerboseHit.Export: showStrand :: Char -> ByteString
+ Biobase.Infernal.VerboseHit.Export: showVerboseHit :: VerboseHit -> ByteString
+ Biobase.Infernal.VerboseHit.Export: streamCM :: HitStream a -> ByteString
+ Biobase.Infernal.VerboseHit.Export: streamInsertion :: HitStream a -> a
+ Biobase.Infernal.VerboseHit.Export: streamScaffold :: HitStream a -> ByteString
+ Biobase.Infernal.VerboseHit.Export: streamStrand :: HitStream a -> Char
+ Biobase.Infernal.VerboseHit.Export: streamVerboseHit :: HitStream a -> VerboseHit
+ Biobase.Infernal.VerboseHit.Export: type StreamToByteString m z = forall a. Enumeratee (HitStream z) ByteString m a
+ Biobase.Infernal.VerboseHit.Import: eeToVerboseHit :: MonadIO m => Enumeratee ByteString VerboseHit m b
+ Biobase.Infernal.VerboseHit.Internal: AliGo :: ByteString -> ByteString -> Char -> AliGo
+ Biobase.Infernal.VerboseHit.Internal: aliCM :: AliGo -> ByteString
+ Biobase.Infernal.VerboseHit.Internal: aliScaffold :: AliGo -> ByteString
+ Biobase.Infernal.VerboseHit.Internal: aliStrand :: AliGo -> Char
+ Biobase.Infernal.VerboseHit.Internal: data AliGo
+ Biobase.Infernal.VerboseHit.Internal: instance Show AliGo

Files

− Biobase/Infernal/CM.hs
@@ -1,326 +0,0 @@------- Stores Infernal Covariance Model data. Built for ease of use, not speed but--- should work reasonably well.------- TODO Data.Vector?------ TODO use generics?------ TODO functions to change to probabilities!------ TODO cmCanonize function!------ TODO put functions into their own modules, a bit of cleanup------ TODO add functions to insert a new node between two already existing nodes; think about how to handle BIF------ TODO add ability to remove node; think about how to handle BIF------ NOTE maybe BIF should not be insertable/removable right now?--module Biobase.Infernal.CM where---import Data.Array.IArray-import Data.List (genericLength)--import Biobase.RNA hiding (nucE) -- "E type" nucleotides do not happen in CMs!------ * Data types for Covariance Models---- {{{ Data types---- | A complete covariance model. Each node and each state can be tagged with--- additional data. Typically, say after parsing, the tag will be ().--data CM n s = CM-  { nodes      :: Array Int (Node n)-  , states     :: Array Int (State s)-  , header     :: [(String,String)] -- keeps the list of header entries sorted!-  , localBegin :: Array Int Double-  , localEnd   :: Array Int Double-  , cmType     :: CMType-  , nullModel  :: Array Nucleotide Double-  } deriving (Show)---- | Describes one node--data Node n = Node-  { nid :: Int-  , ntype :: NodeType-  , nparents :: [Int] -- TODO can there be more than one?-  , nchildren :: [Int]-  , nstates :: [Int]-  , ntag :: n-  } deriving (Show)---- | One state--data State s = State-  { sid :: Int-  , stype :: StateType-  , snode :: Int-  , sparents :: [Int]-  , schildren :: [Transition]-  , semission :: [Emission]-  , stag :: s-  } deriving (Show)---- | CMType is important if we want to set localBegin / localEnd!--data CMType = CMProb | CMScore-  deriving (Show,Eq)---- | can emit either one nucleotide or a pair--data Emission-  = EmitS {eNuc :: Nucleotide, escore :: Double}-  | EmitP { eNucL :: Nucleotide, eNucR :: Nucleotide, escore :: Double}-  deriving (Show)---- | branches are transition without attached probability becaue both branches are always taken--data Transition-  = Branch {tchild :: Int}-  | Transition {tchild :: Int, tscore :: Double}-  deriving (Show)---- | the different node types--data NodeType = MATP | MATL | MATR | BIF | ROOT | BEGL | BEGR | END-  deriving (Read,Show,Eq,Ord,Enum,Bounded)---- | the different state types--data StateType = MP | IL | IR | D | ML | MR | B | S | E-  deriving (Read,Show,Eq,Ord,Enum,Bounded)---- }}}----- * make a local model out of a global one---- | generate a local model with local begin prob and local end prob--cmMakeLocal :: Double -> Double -> CM n s -> CM n s-cmMakeLocal pbegin pend cm = cmMakeLocalBegin pbegin $ cmMakeLocalEnd pend cm----cmMakeLocalBegin :: Double -> CM n s -> CM n s-cmMakeLocalBegin pbegin cm = cm{localBegin = localBegin cm // changes} where-  changes = rootS : (start : intern)-  rootS = (0, prob2Score 0 1.0) -- root disabled!-  start = (head . nstates $ nodes cm ! 1, prob2Score (1-pbegin) 1.0) -- the first state after "root 0"-  intern = map (\k -> (sid $ nodeMainState cm k,prob2Score (pbegin / l) 1.0)) nds-  nds = filter (localBeginPossible cm) . elems $ nodes cm-  l = genericLength nds------ TODO have to change the transition score, too!--cmMakeLocalEnd :: Double -> CM n s -> CM n s-cmMakeLocalEnd pend cm = cm{localEnd = localEnd cm // changes} where-  changes = map (\k -> (sid $ nodeMainState cm k,prob2Score (pend / l) 1.0)) nds-  nds = filter (localBeginPossible cm) . elems $ nodes cm-  l = genericLength nds------ * Transform between score and probability mode---- | given a CM in score mode, change it to probability mode--cmScore2Prob :: CM n s -> CM n s-cmScore2Prob cm' = if cmType cm' == CMProb then cm' else CM-  (nodes  cm)-  (statesScore2Prob cm $ states cm)-  (header cm)-  (localBeginScore2Prob $ localBegin cm)-  (localEndScore2Prob   $ localEnd   cm)-  CMProb-  nm-  where-    nm = amap (flip score2Prob 0.25) $ nullModel cm'-    cm = cm' {nullModel = nm}----- | Given a CM in prob mode, change to score mode--cmProb2Score :: CM n s -> CM n s-cmProb2Score cm' = if cmType cm' == CMScore then cm' else CM-  (nodes  cm)-  (statesProb2Score cm $ states cm)-  (header cm)-  (localBeginProb2Score $ localBegin cm)-  (localEndProb2Score   $ localEnd   cm)-  CMScore-  nm-  where-    nm = amap (flip prob2Score 0.25) $ nullModel cm'-    cm = cm' {nullModel = nm}------ | normalize all PROBabilities in a CM--cmNormalizeProbabilities :: CM n s -> CM n s-cmNormalizeProbabilities cm-  | cmType cm == CMScore = error "cannot normalize score-type CM"-  | otherwise            = cm -- TODO have to map normalization over all scores!------ {{{ CM score/prob conversion helpers--statesScore2Prob :: CM n s -> Array Int (State s) -> Array Int (State s)-statesScore2Prob cm sA = amap f sA where-  f s = s {schildren = map fT $ schildren s, semission = map fE $ semission s}-  fT b@(Branch _) = b-  fT (Transition k v) = Transition k (score2Prob v 1.0)-  fE (EmitS k v) = EmitS k (score2Prob v $ nullModel cm ! k)-  fE (EmitP k1 k2 v) = EmitP k1 k2 (score2Prob v $ (nullModel cm ! k1) * (nullModel cm ! k2))----localBeginScore2Prob :: Array Int Double -> Array Int Double-localBeginScore2Prob sA = amap f sA where-  f s = score2Prob s 1.0----localEndScore2Prob :: Array Int Double -> Array Int Double-localEndScore2Prob sA = amap f sA where-  f s = score2Prob s 1.0----statesProb2Score :: CM n s -> Array Int (State s) -> Array Int (State s)-statesProb2Score cm sA = amap f sA where-  f s = s {schildren = map fT $ schildren s, semission = map fE $ semission s}-  fT b@(Branch _) = b-  fT (Transition k v) = Transition k (prob2Score v 1.0)-  fE (EmitS k v) = EmitS k (prob2Score v $ nullModel cm ! k)-  fE (EmitP k1 k2 v) = EmitP k1 k2 (prob2Score v $ (nullModel cm ! k1) * (nullModel cm ! k2))----localBeginProb2Score :: Array Int Double -> Array Int Double-localBeginProb2Score sA = amap f sA where-  f s = prob2Score s 1.0----localEndProb2Score :: Array Int Double -> Array Int Double-localEndProb2Score sA = amap f sA where-  f s = score2Prob s 1.0---- }}}------- * Helper Functions---- {{{ helper functions---- | extract the main state for each node (eg MP state for MATP node)---- TODO shouldn't this just be "head $ nstates n"?--nodeMainState :: CM n s -> Node n -> State s-nodeMainState cm n = head $ filter ((==st) . stype) ss where-  (Just st) = (ntype n) `lookup` nodeMainStateAssocs-  ss = map (states cm !) $ nstates n----- | Checks for each node, if it can be target of a local begin.--localBeginPossible :: CM n s -> Node n -> Bool-localBeginPossible cm n =-  if ntype n `elem` okNodes-  && (not . any (==0) $ nparents n) -- nodes reachable from "root" (that is: node 1) have handled specially-    then True-    else False-  where-    okNodes = [MATP,MATL,MATR,BIF]------ | Checks for each node if it can lead to a local end.--localEndPossible :: CM n s -> Node n -> Bool-localEndPossible cm n =-  if ntype n `elem` okNodes-  && (END /= (ntype $ nodes cm ! (nid n +1)))-    then True-    else False-  where-    okNodes = [MATP,MATL,MATR,BEGL,BEGR]------ | transform scores into probabilities, given a nullmodel for x---- TODO quickcheck!--score2Prob x null-  | x == (-1/0) = 0-  | otherwise   = exp (x * log 2) * null------ | back into scores--prob2Score x null-  | x == 0    = (-1/0)-  | otherwise = log (x / null) / log 2------ | Transform a state, setting probabilities instead of scores. Requires CM--- knowledge for background model.---- TODO actually use the background model--stateScore2Prob :: CM n s -> State s -> State s-stateScore2Prob cm s = error "implement me"------ | Transform a state, setting scores instead of probabilities.--stateProb2Score :: CM n s -> State s -> State s-stateProb2Score cm s = error "implement me"-transitionTargets :: [Transition] -> [Int]-transitionTargets xs = map f xs where-  f (Branch x)       = x-  f (Transition x _) = x---nodeMainStateAssocs :: [(NodeType,StateType)]-nodeMainStateAssocs =-  [ (MATP, MP)-  , (MATL, ML)-  , (MATR, MR)-  , (BIF,  B)-  , (ROOT, S)-  , (BEGL, S)-  , (BEGR, S)-  , (END,  E)-  ]---- }}}------ TODO score -> prob : exp (x / log 2) -- CHECK THIS!!!--- score -> prob : exp (x * log 2)--- prob -> score : (log x) / log 2--- TODO mkGlobal (needed?)--- TODO use logsum : log (exp x + exp y) = x + log (1 + exp (y-x)), where x>y
− Biobase/Infernal/CM/Export.hs
@@ -1,76 +0,0 @@-{-# LANGUAGE RecordWildCards #-}---- | write an Infernal model. We aim for version 1.0.2 of Infernal but right--- now, there is no explicit version management! We just return header--- information as we read it. So we are compatible to all versions that can be--- parsed but have no safeguards in place.------ NOTE: exported model can apparently be calibrated.------ TODO make sure that (Export.toString . Import.fromString == id); the other--- way around is not absolutely necessary (except if the Infernal parser should--- require this at some point)------ TODO put functions like stringS into another module?--module Biobase.Infernal.CM.Export where--import qualified Data.Array.IArray as A-import Text.Printf--import Biobase.Infernal.CM--import Biobase.Infernal.CM.Import -- TODO for testing only!------ | Main export function for CMs. Creates a string that is accepted by--- Infernal--toString :: CM n s -> String-toString cm@CM{..} = unlines $ hdr ++ ["MODEL:"] ++ nds ++ ["//"]-  where-    hdr = map (\(k,v) -> k ++ " " ++ v) header-    nds = concatMap (nodeToString cm) $ A.elems nodes------ | export a specific node, used by 'toString'--nodeToString :: CM n s -> Node n -> [String]-nodeToString cm Node{..} = printf "\t\t\t\t[ %s %d ]" (show ntype) nid : map (stateToString cm . (states cm A.!)) nstates------ | export a specific states, used by 'nodeToString'--stateToString :: CM n s -> State s -> String-stateToString cm State{..} = printf "    %2s %6d %6d %2d %6d %6d %-55s %s" (show stype) sid maxP numP minC numC tscores escores-  where-    maxP-      | null sparents = -1-      | otherwise     = maximum sparents-    numP = length sparents-    minC-      | null schildren = -1-      | otherwise      = minimum $ map tchild schildren-    numC-      | stype == B = maximum $ map tchild schildren-      | otherwise  = length schildren-    tscores :: String-    tscores-      | stype == B = ""-      | otherwise  = concatMap (printf "%s   " . stringS . tscore) schildren-    escores :: String-    escores-      | null semission = ""-      | otherwise      = concatMap (printf "%.3f " . escore) semission------ | export scores for transitions with the Infernal-specific "-infinity" value--- of "*"--stringS x-  | x == (-1/0) = "      *"-  | otherwise   = printf "%.3f" x
− Biobase/Infernal/CM/Import.hs
@@ -1,194 +0,0 @@------- A parser for Infernal's Covariance Models (CM). Should work with version 1.0--- of Infernal.------- TODO use generics?------ TODO importing the cmcalibrate calibration data is currently broken! (data is parsed as separate lines)--module Biobase.Infernal.CM.Import (fromFile,fromString) where--import Data.Maybe (fromJust)-import qualified Data.Array.IArray as A-import qualified Data.List as L-import Text.ParserCombinators.Parsec hiding (State)--import Biobase.Infernal.CM-import Biobase.RNA hiding (nucE)------ {{{ Parsec--myChar = alphaNum <|> char '-' <|> char '_' <|> char '.'-headerChar = myChar <|> char '[' <|> char ']'-floating = many1 $ digit <|> char '.' <|> char 'e' <|> char '-'--headerLine = do-  key <- many1 headerChar-  --spaces-  many1 $ char ' '-  val <- many1 (headerChar <|> char ' ' <|> char '/' <|> char ':')-  newline-  return (key,val)--unknownNumbers = do-  many1 $ char ' '-  ns <- floating `sepEndBy` (many1 $ char ' ')-  newline-  return ("",concat ns)--theHeader = do-  hs <- many $ (try headerLine <|> try unknownNumbers)-  string "MODEL:"-  newline-  return hs---- }}}---- {{{ Node--node = do-  spaces-  string "[ "-  name <- many letter-  spaces-  num <- many digit >>= (return . read)-  spaces-  string "]"-  newline-  states <- many $ try (state num)-  return $ (Node-             { ntype = read name-             , nid = num-             , nstates = map sid states-             , nparents = []-             , nchildren = []-             , ntag = ()-             }-            , states)---- }}}---- {{{ stuff---- TODO put some of this into HsTools/Parsec stuff--aNum = many1 $ char '-' <|> digit-addneginf :: String -> Double-addneginf "*" = (-1)/0 --"-1000000000.0"-addneginf x = read x---- }}}---- {{{ State--state nodeID = do-  spaces-  name <- many1 myChar-  spaces-  sid <- aNum >>= (return . read)-  spaces-  plast <- aNum >>= (return . read)-  spaces-  pnum <- aNum >>= (return . read)-  spaces-  cfirst <- aNum >>= (return . read)-  spaces-  cnum <- aNum >>= (return . read)-  probs <- ( (many $ myChar <|> char '*') `sepBy` (many1 $ char ' ') >>=-             return . map addneginf . filter ((/=) "") )-  newline-  let s = case name of-            "B" -> State-                    { stype = read name-                    , sid = sid-                    , snode = nodeID-                    , sparents = [plast-pnum+1 .. plast]-                    , schildren = [Branch cfirst, Branch cnum]-                    , semission = []-                    , stag = ()-                    }-            _   -> State-                    { stype = read name-                    , sid = sid-                    , snode = nodeID-                    , sparents = [plast-pnum+1 .. plast]-                    , schildren = zipWith Transition [cfirst .. cfirst+cnum-1] probs-                    , semission = let keep = drop cnum probs in-                        case length keep of-                          0  -> []-                          4  -> zipWith EmitS acgu keep-                          16 -> zipWith (\(k1,k2) v -> EmitP k1 k2 v) acguPairs keep-                          _  -> error $ "strange number of probabilities" ++ show (keep)-                    , stag = ()-                    }-  return s---- }}}---- {{{ Model--models = do-  ms <- many model-  eof-  return ms--model = do-  h <- theHeader-  ns <- many node-  let states = concatMap snd ns-  let nodes = map (addPCinfo states . fst) ns-  -- just add all the node parent / child info-  string "//"-  newline-  -- eof -- removed, we want to be able to read concatenated models!-  return $ CM-            { nodes      = A.array (0, length nodes -1)  $ zip (map nid nodes)  nodes-            , states     = A.array (0, length states -1) $ zip (map sid states) states-            , header     = h-            , localBegin = A.array (0, length states -1) $ zip [0 .. length states -1] (0.0 : repeat (-1/0))-            , localEnd   = A.array (0, length states -1) $ zip [0 .. length states -1] (repeat (-1/0))-            , cmType     = CMScore-            , nullModel  = A.array (nucA,nucU) $ zip acgu (map read . words . fromJust $ "NULL" `lookup` h) -- TODO circumvents the whole parsing stuff!-            }---- }}}---- {{{ Stuff--addPCinfo states n =-  let-    s = nstates n-    sp = L.nub $ L.sort $ concatMap (sparents . (states !!)) s-    sc = L.nub $ L.sort $ concatMap (transitionTargets . schildren . (states !!)) s-    np = (L.nub $ L.sort $ map (snode . (states !!)) sp) L.\\ [nid n]-    nc = (L.nub $ L.sort $ map (snode . (states !!)) sc) L.\\ [nid n]-  in-    n {nparents = np, nchildren = nc}------ | Two types of parsing, once using a file and once by parsing a string.--fromFile f = parseFromFile models f-fromString s = parse models "(stdin)" s---- }}}---- | Helper function to remove impossible state transitions (those that have--- -infty score).---- TODO move to InfernalCM.hs and have it for Prob and Score both--{--canonize cm = cm {states = A.amap f $ states cm} where-  f s = s {schildren = filter g $ schildren s}-  g (Branch _)    = True-  g (Transition _ v) = -  h (_,s)-    | s == (-1)/0 = False-    | otherwise   = True--}
− Biobase/Infernal/Stockholm.hs
@@ -1,16 +0,0 @@---- | The beginnings of a Stockholm File parser.--module Biobase.Infernal.Stockholm where---- | Stockholm format data. We have a set of sequences with sequence data, a--- set of column annotations and unknown data (actually: known, but we do not--- care).--data Stockholm = Stockholm-  { sequences :: [(String,String)]-  , colAnnotations :: [(String,String)]   -- #=GC tag-  , exAnnotations :: [(String,String)]    -- NOTE this is not in the Stockholm format!-  , unknown :: [String]-  }-  deriving (Show)
− Biobase/Infernal/Stockholm/Import.hs
@@ -1,35 +0,0 @@---- | A simple parser for Stockholm data. As we do not interpret most stuff--- right now, Parsec is not required.--module Biobase.Infernal.Stockholm.Import where--import Biobase.Infernal.Stockholm-import Data.List--fromFile :: String -> IO Stockholm-fromFile fname = do-  ls <- readFile fname >>= return . filter (not . null) . lines-  let (colanno, rest1) = partition (\x -> length x >= 4 && (and $ zipWith (==) "#=GC" x)) ls-  let (seqdata, rest2) = partition (\(x:_) -> x/='#' && x/='/') rest1-  let (exdata , unk)   = partition (\x -> length x >= 4 && (and $ zipWith (==) "#=EX" x)) rest2-  {--  print colanno-  print rest1-  print seqdata-  print rest2-  print exdata-  print unk-  -}-  return Stockholm-    { sequences      = map (mkPair . words) seqdata-    , colAnnotations = map (mkPair . drop 1 . words) colanno-    , exAnnotations  = map (mkPair . drop 1 . words) exdata-    , unknown        = unk-    }----mkPair []     = error "empty list"-mkPair [x]    = error $ "just one element: " ++ show x-mkPair (x:xs) = (x,unwords xs)
+ Biobase/Infernal/Taxonomy.hs view
@@ -0,0 +1,30 @@+{-# LANGUAGE PatternGuards #-}++-- | Infernal contains a taxonomy database. This is a simple module reflecting+-- said database.++module Biobase.Infernal.Taxonomy where++import qualified Data.ByteString.Char8 as BS+import Data.Char (toLower)++++-- | For each species, we store the name and a classification list from most+-- general (head) to most specific (last). The database comes with the NCBI+-- taxon identifier (taxid).++data Species = Species+  { name :: BS.ByteString+  , classification :: [BS.ByteString]+  , taxid :: Int+  } deriving (Show)++-- | Given a name such as "Drosophila Melanogaster", returns "d.melanogaster".++shortenName :: BS.ByteString -> BS.ByteString+shortenName xs+  | null ws   = xs+  | [w] <- ws = w+  | otherwise = BS.map toLower $ BS.take 1 (ws!!0) `BS.append` (BS.cons '.' $ ws!!1)+  where ws = BS.words xs
+ Biobase/Infernal/Taxonomy/Import.hs view
@@ -0,0 +1,68 @@++module Biobase.Infernal.Taxonomy.Import+  ( eeImport+  , iSpeciesMap+  , iTaxIdMap+  ) where++import qualified Data.Enumerator as E+import qualified Data.Enumerator.List as EL+import qualified Data.ByteString.Char8 as BS+import qualified Data.Attoparsec as A+import Control.Applicative+import qualified Data.Attoparsec.Char8 as A8+import qualified Data.Attoparsec.Enumerator as EAP+import qualified Data.Map as M++import Biobase.Infernal.Taxonomy++import qualified Data.Enumerator.Binary as EB++++-- | Provide name-based lookup as the most-common usage scenario.+--+-- TODO there are 9 duplicates in the names, let's find them and see what is+-- going on++iSpeciesMap :: Monad m => E.Iteratee Species m (M.Map BS.ByteString Species)+iSpeciesMap = EL.fold (\m x -> M.insert (name x) x m) M.empty++-- | And a map based on taxon id++iTaxIdMap :: Monad m => E.Iteratee Species m (M.Map Int Species)+iTaxIdMap = EL.fold (\m x -> M.insert (taxid x) x m) M.empty++-- | Imports taxonomy data.+--+-- NOTE The taxonomy format is, for each species, a line consisting of: taxid -+-- tab - species name - tab - semicolon separated list of classification names+-- - dot - end of line.++eeImport :: Monad m => E.Enumeratee BS.ByteString Species m b+eeImport = E.sequence $ EAP.iterParser mkSpecies++-- | Given a 'ByteString', create a species entry.++mkSpecies :: A.Parser Species+mkSpecies = f <$> ptaxid <* tab <*> pname <* tab <*> A8.takeWhile (/='\n') <* A8.endOfLine where+  f k n xs = let+               cs = map (BS.copy . BS.dropWhile (==' ')) . BS.split ';' . BS.init $ xs+             in Species (BS.copy n) cs k+  ptaxid   = A8.decimal+  pname    = A8.takeWhile (/='\t')+  tab      = A8.char '\t'++++-- * Testing++test :: IO ()+test = do+  m1 <- E.run_ $ (EB.enumFile "./Tests/Infernal/taxonomy" E.$$ (eeImport E.=$ iSpeciesMap))+  m2 <- E.run_ $ (EB.enumFile "./Tests/Infernal/taxonomy" E.$$ (eeImport E.=$ iTaxIdMap))+  print $ M.size m1+  print $ m1 M.! BS.pack "Cenarchaeum symbiosum B"+  -- print $ M.size $ M.mapKeys shortenName m1+  -- print $ M.size m2+  return ()
+ Biobase/Infernal/VerboseHit.hs view
@@ -0,0 +1,31 @@+{-# LANGUAGE RecordWildCards #-}++-- | Provides a datatype for cmsearch verbose output.++module Biobase.Infernal.VerboseHit where++import qualified Data.ByteString.Char8 as BS+import Text.Printf++++-- | Captures a complete alignment++data VerboseHit = VerboseHit+  { vhTarget  :: (Int,Int)  -- ^ part of target sequence (start counting at 1)+  , vhQuery   :: (Int,Int)  -- ^ which part of the CM/stk do we align to+  , vhCM      :: BS.ByteString  -- ^ the CM for this alignment+  , vhStrand  :: Strand     -- ^ should be either '+' or '-'+  , vhScore   :: Double     -- ^ bit score+  , vhEvalue  :: Double     -- ^ number of hits we expect to find with 'score' or higher for 'targetSequence' length+  , vhPvalue  :: Double     -- ^ ?+  , vhGC      :: Int        -- ^ ?+  , vhScaffold :: BS.ByteString -- ^ scaffold, chromosome, ... (the name of the sequence, not the sequence data!)+  , vhWuss :: BS.ByteString -- ^ fancy secondary structure annotation using wuss notation+  , vhConsensus :: BS.ByteString -- ^ query consensus (upper: highly, lower: weak/no)+  , vhScoring   :: BS.ByteString -- ^ represents where positive and negative scores come from+  , vhSequence :: BS.ByteString -- ^ the target sequence which aligns to the model+  } deriving (Show,Read)++type Strand = Char+
+ Biobase/Infernal/VerboseHit/Export.hs view
@@ -0,0 +1,170 @@+{-# LANGUAGE RankNTypes #-}+{-# LANGUAGE RecordWildCards #-}+{-# LANGUAGE OverloadedStrings #-}++-- | Exports VerboseHit results back into text. As a likely scenario is a+-- pipeline where hits are to be filtered out, this provides enumeratee's that+-- handle additional annotations as required by the file format for CMs,+-- scaffolds, and strand information. If you just need a way to show the data,+-- use printVerboseHit.++module Biobase.Infernal.VerboseHit.Export where+{-+  ( eeFromVerboseHit+  ) where+-}++import Control.Monad.Trans.Class (lift)+import qualified Data.ByteString.Char8 as BS+import qualified Data.Enumerator as E+import qualified Data.Enumerator.List as EL+import Text.Printf++import Biobase.Infernal.VerboseHit+import Biobase.Infernal.VerboseHit.Internal++++-- | Takes a list of 'VerboseHit's and produces a list of bytestrings. Unlining+-- those bytestrings produces a file that is \in essence\ an Infernal+-- verbose-hit output file and should be parse-able by ours and other+-- importers.+--+-- TODO block length (for the alignment of query/sequenc) ?!+--+-- TODO is there a more elegant treatment of the eof condition than asking at+-- every verbose hit that is created?++{-+eeFromVerboseHit :: Monad m => E.Enumeratee VerboseHit BS.ByteString m a+eeFromVerboseHit = goS (AliGo "" "" '?') where+  goS s (E.Continue k) = EL.head >>= go s where+    go s Nothing = return $ E.Continue k+    go s@AliGo{..} (Just l@VerboseHit{..}) = do+      eof <- E.isEOF+      let res =  [ "\n//\n" | vhCM /= aliCM && aliCM /= "" ]+              ++ [ "\nCM: " `BS.append` vhCM `BS.append` "\n" | vhCM /= aliCM]+              ++ [ "\n>" `BS.append` vhSeqName `BS.append` "\n" | vhSeqName /= aliScaffold]+              ++ [ strand vhStrand | vhStrand /= aliStrand]+              ++ [ BS.pack $ printf " Query = %d - %d, Target = %d - %d"+                              (fst vhQuery) (snd vhQuery) (fst vhTarget) (snd vhTarget)+                 , BS.pack $ printf " Score = %.2f, E = %f, P = %.4e, GC = %d"+                              vhScore vhEvalue vhPvalue vhGC+                 , ""+                 , ws11 `BS.append` vhWuss+                 , (BS.pack $ printf "%10d " (fst vhQuery))+                      `BS.append` vhConsensus+                      `BS.append` (BS.pack $ printf " %d" (snd vhQuery))+                 , ws11 `BS.append` vhScoring+                 , (BS.pack $ printf "%10d " (fst vhTarget))+                      `BS.append` vhSequence+                      `BS.append` (BS.pack $ printf " %d" (snd vhTarget))+                 ]+              ++ [ "\n//" | eof]+      newStep <- lift $ E.runIteratee $ k $ E.Chunks res+      goS (AliGo vhCM vhSequence vhStrand) newStep+    strand '+' = "  Plus strand results:\n"+    strand '-' = "  Minus strand results:\n"+    strand _   = "  Unknown strand results:\n"+    ws11 = BS.pack $ replicate 11 ' '+  goS _ step = return step+-}++-- | Convert a 'VerboseHit' to a string, ready for printing as in the input+-- file.++showVerboseHit :: VerboseHit -> BS.ByteString+showVerboseHit VerboseHit{..} = BS.unlines+  [ BS.pack $ printf " Query = %d - %d, Target = %d - %d"+                (fst vhQuery) (snd vhQuery) (fst vhTarget) (snd vhTarget)+  , BS.pack $ printf " Score = %.2f, E = %f, P = %.4e, GC = %d"+                vhScore vhEvalue vhPvalue vhGC+  , ""+  , ws11 `BS.append` vhWuss+  , (BS.pack $ printf "%10d " (fst vhQuery))+    `BS.append` vhConsensus+    `BS.append` (BS.pack $ printf " %d" (snd vhQuery))+  , ws11 `BS.append` vhScoring+  , (BS.pack $ printf "%10d " (fst vhTarget))+    `BS.append` vhSequence+    `BS.append` (BS.pack $ printf " %d" (snd vhTarget))+  ] where+    ws11 = BS.pack $ replicate 11 ' '++-- | CM information, ready for printing.++showCM :: BS.ByteString -> BS.ByteString+showCM cm = "CM: " `BS.append` cm++-- | Scaffold information++showScaffold :: BS.ByteString -> BS.ByteString+showScaffold sc = ">" `BS.append` sc++-- | Strand information++showStrand :: Char -> BS.ByteString+showStrand = f where+  f '+' = "  Plus strand results:"+  f '-' = "  Minus strand results:"+  f _   = "  Unknown strand results:"++-- | Turning a list of 'VerboseHit's back into lines of characters is, in+-- principle, not too hard. But just before we actually stream out, we might+-- want to inject arbitrary data into the stream. This is done via+-- 'StreamInsertion'. The other constructors merely wrap certain data.+--+-- One way to, say, tag verbose hits is like this (note the output type of+-- 'eeHitToStream'):+--+-- > tag [s@(StreamVerboseHit _)] = [StreamInsertion (), s]+-- > tag xs = xs++data HitStream a+  = StreamVerboseHit {streamVerboseHit :: VerboseHit}+  | StreamCM {streamCM :: BS.ByteString}+  | StreamScaffold {streamScaffold :: BS.ByteString}+  | StreamStrand {streamStrand :: Char}+  | StreamInsertion {streamInsertion :: a}+  deriving (Show)++-- | This enumeratee turns 'VerboseHit's into a 'HitStream'. Each VerboseHit+-- can emit one or more elements, depending on if the CM, scaffold, or strand+-- changes.+--+-- TODO try to rewrite use Control.Monad.State++eeHitToStream :: Monad m => E.Enumeratee VerboseHit [HitStream z] m a+eeHitToStream = EL.mapAccum go (AliGo "" "" '?') where+  go AliGo{..} vh@VerboseHit{..} = (AliGo vhCM vhScaffold vhStrand,+    [StreamCM vhCM | aliCM /= vhCM] +++    [StreamScaffold vhScaffold | aliCM /= vhCM || aliScaffold /= vhScaffold] +++    [StreamStrand vhStrand | aliCM /= vhCM || aliScaffold /= vhScaffold || aliStrand /= vhStrand] +++    [StreamVerboseHit vh]+    )++-- | Flattens a stream from a list of lists to a single list. After this point,+-- you probably want to insert elements into the stream, then flatten again.++eeFlattenStream :: Monad m => E.Enumeratee [HitStream z] (HitStream z) m a+eeFlattenStream = EL.concatMap id++-- | If the 'HitStream' contains 'StreamInsertion's that are an instance of+-- 'Show', this provides a default method to turn the stream into a bytestring.+--+-- TODO add some newline characters for good measure+--+-- TODO on end-of-stream, we should print out "//"++eeStreamToByteString :: (Monad m, Show z) => StreamToByteString m z+eeStreamToByteString = EL.map f where+  f StreamVerboseHit{..} = showVerboseHit streamVerboseHit `BS.snoc` '\n'+  f StreamCM{..} = showCM streamCM `BS.append` "\n\n"+  f StreamScaffold{..} = showScaffold streamScaffold `BS.append` "\n\n"+  f StreamStrand{..} = showStrand streamStrand `BS.append` "\n\n"+  f StreamInsertion{..} = BS.pack . show $ streamInsertion++eeStreamToByteString' :: (Monad m) => StreamToByteString m ()+eeStreamToByteString' = eeStreamToByteString++type StreamToByteString m z = forall a . E.Enumeratee (HitStream z) BS.ByteString m a
+ Biobase/Infernal/VerboseHit/Import.hs view
@@ -0,0 +1,96 @@+{-# LANGUAGE DoAndIfThenElse #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE OverloadedStrings #-}+{-# LANGUAGE RecordWildCards #-}++-- | Enumeratee that transforms a stream of 'ByteString's into a stream of+-- 'VerboseHit's.++module Biobase.Infernal.VerboseHit.Import +  ( eeToVerboseHit+  ) where++import Control.Applicative+import Control.Monad.IO.Class (MonadIO)+import Control.Monad.Trans.Class (lift)+import Data.Char (isSpace)+import Data.Tuple.Select+import qualified Data.Attoparsec as A+import qualified Data.Attoparsec.Char8 as A8+import qualified Data.Attoparsec.Enumerator as EAP+import qualified Data.ByteString.Char8 as BS+import qualified Data.Enumerator as E+import qualified Data.Enumerator.List as EL++import Biobase.Infernal.VerboseHit+import Biobase.Infernal.VerboseHit.Internal++++-- | Stateful reading of 'VerboseHit's. This pipe groups certain lines and+-- condenses them to one hit. As multiple hits share certain state (cm,+-- scaffold, and strand), we thread some state through using "goS" internally.+--+-- TODO is MonadIO really necessary?++eeToVerboseHit :: MonadIO m => E.Enumeratee BS.ByteString VerboseHit m b+eeToVerboseHit = goS (AliGo "" "" '?') where+  goS s (E.Continue k) = EL.dropWhile BS.null >> E.peek >>= go s where+    go s Nothing = return $ E.Continue k -- ???+    go s (Just (l :: BS.ByteString))+      | "CM: "    `BS.isInfixOf` l = drops >> E.peek >>= go s{aliCM = BS.copy $ BS.drop 4 l}+      | ">"       `BS.isInfixOf` l = drops >> E.peek >>= go s{aliScaffold = BS.copy $ BS.drop 1 l}+      | "  Plus"  `BS.isInfixOf` l = drops >> E.peek >>= go s{aliStrand = '+'}+      | "  Minus" `BS.isInfixOf` l = drops >> E.peek >>= go s{aliStrand = '-'}+      | " Query"  `BS.isInfixOf` l = do+          x <- qs (aliCM s) (aliScaffold s) (aliStrand s)+          newStep <- lift $ E.runIteratee $ k $ E.Chunks [x]+          goS s newStep+    go s l = drops >> E.peek >>= go s+    drops = EL.drop 1 >> EL.dropWhile BS.null+  goS _ step = return step++-- | Parses one "Alignment". This is build by Query/Score lines and multiple+-- 4-tuples or strings with the actual alignment.++qs :: Monad m => BS.ByteString -> BS.ByteString -> Char -> E.Iteratee BS.ByteString m VerboseHit+qs cm scaf pm = do+  q <- EAP.iterParser qt+  s <- EAP.iterParser sepg+  l <- fourLines $ sel4 q+  return $ VerboseHit+    { vhScaffold = scaf+    , vhCM = cm+    , vhStrand = pm+    , vhQuery = (sel1 q, sel2 q)+    , vhTarget = (sel3 q, sel4 q)+    , vhScore = sel1 s+    , vhEvalue = sel2 s+    , vhPvalue = sel3 s+    , vhGC = sel4 s+    , vhWuss = cpy $ l!!0+    , vhConsensus = cpy $ l!!1+    , vhScoring = cpy $ l!!2+    , vhSequence = cpy $ l!!3+    }+  where+    cpy = BS.copy . BS.concat+    qt = (,,,) <$ A.string " Query = "   <*> A8.decimal <* A.string " - " <*> A8.decimal+               <* A.string ", Target = " <*> A8.decimal <* A.string " - " <*> A8.decimal+    sepg = (,,,) <$ A.string " Score = " <*> A8.double+                 <* A.string ", E = "    <*> A8.double+                 <* A.string ", P = "    <*> A8.double+                 <* A.string ", GC = " <* A8.skipSpace <*> A8.decimal++-- | Parse 4-tuples of cmsearch alignment output until we have hit the last, or+-- "to", 4-tuple. Assumes that "to" is correct++fourLines to = do+  EL.dropWhile BS.null+  ls <- EL.take 4+  let ws = BS.length . BS.takeWhile isSpace . head $ ls+  let cs = BS.length . BS.dropWhile isSpace . head $ ls+  let xs = map (BS.take cs . BS.drop ws) ls+  if (to == (read . BS.unpack . last . BS.words . last $ ls))+  then return . map (:[]) $ xs+  else fourLines to >>= return . (zipWith (:) xs)
+ Biobase/Infernal/VerboseHit/Internal.hs view
@@ -0,0 +1,17 @@++-- | Shared, internal stuff.++module Biobase.Infernal.VerboseHit.Internal where++import qualified Data.ByteString.Char8 as BS++++-- | State for import and export functions++data AliGo = AliGo+  { aliCM :: BS.ByteString+  , aliScaffold :: BS.ByteString+  , aliStrand :: Char+  } deriving (Show)+
BiobaseInfernal.cabal view
@@ -1,38 +1,46 @@ name:           BiobaseInfernal-version:        0.0.2.1+version:        0.5.0.0 author:         Christian Hoener zu Siederdissen maintainer:     choener@tbi.univie.ac.at-copyright:      Christian Hoener zu Siederdissen, 2010-homepage:       http://www.tbi.univie.ac.at/~choener/Haskell/+homepage:       http://www.tbi.univie.ac.at/~choener/+copyright:      Christian Hoener zu Siederdissen, 2011 category:       Bioinformatics-synopsis:       (deprecated) Infernal CM manipulation+synopsis:       Infernal data structures and tools license:        GPL-3 license-file:   LICENSE build-type:     Simple stability:      experimental cabal-version:  >= 1.4.0 description:-                This library provides functions to load Infernal covariance-                models. Models can be converted from score mode into-                probability mode. They, too, can be switched into local mode.-                In additon, Stockholm files can be loaded.+                Provides import and export facilities for Infernal/Rfam data+                formats. We include Stockholm, CM, verbose Infernal results,+                and tabulated Infernal results. Some small tools are included.+                .+                With the BioHaskell library split, this package is officially+                back! And the package is not feature-complete yet, take the+                above as a will-include-soon list. +extra-source-files:  library   build-depends:-    base >= 4 && < 5,-    parsec >= 3 && < 4,+    base >3 && <5,+    attoparsec,+    attoparsec-enumerator,+    bytestring,     containers,-    array,--    Biobase >= 0.0.2 && < 0.0.3+    enumerator,+    transformers,+    tuple    exposed-modules:-    Biobase.Infernal.CM-    Biobase.Infernal.CM.Import-    Biobase.Infernal.CM.Export-    Biobase.Infernal.Stockholm-    Biobase.Infernal.Stockholm.Import+    Biobase.Infernal.Taxonomy+    Biobase.Infernal.Taxonomy.Import+    Biobase.Infernal.VerboseHit+    Biobase.Infernal.VerboseHit.Export+    Biobase.Infernal.VerboseHit.Import+    Biobase.Infernal.VerboseHit.Internal    ghc-options:     -O2+