fasta 0.10.1.0 → 0.10.2.0
raw patch · 11 files changed
+244/−143 lines, 11 filesPVP: major bump suggested
API removals or changes: PVP suggests a major version bump
API changes (from Hackage documentation)
+ Data.Fasta.ByteString.Lazy.Translation: customCodon2aa :: [(Codon, Char)] -> Codon -> Either ByteString AA
+ Data.Fasta.ByteString.Lazy.Translation: customTranslate :: [(Codon, AA)] -> Int64 -> FastaSequence -> Either ByteString FastaSequence
+ Data.Fasta.ByteString.Lazy.Types: type AA = Char
+ Data.Fasta.ByteString.Translation: customCodon2aa :: [(Codon, Char)] -> Codon -> Either ByteString AA
+ Data.Fasta.ByteString.Translation: customTranslate :: [(Codon, AA)] -> Int -> FastaSequence -> Either ByteString FastaSequence
+ Data.Fasta.ByteString.Types: type AA = Char
+ Data.Fasta.String.Translation: customCodon2aa :: [(Codon, Char)] -> Codon -> Either String AA
+ Data.Fasta.String.Translation: customTranslate :: [(Codon, AA)] -> Int -> FastaSequence -> Either String FastaSequence
+ Data.Fasta.String.Types: type AA = Char
+ Data.Fasta.Text.Lazy.Translation: customCodon2aa :: [(Codon, Char)] -> Codon -> Either Text AA
+ Data.Fasta.Text.Lazy.Translation: customTranslate :: [(Codon, AA)] -> Int64 -> FastaSequence -> Either Text FastaSequence
+ Data.Fasta.Text.Lazy.Types: type AA = Char
+ Data.Fasta.Text.Translation: customCodon2aa :: [(Codon, Char)] -> Codon -> Either Text AA
+ Data.Fasta.Text.Translation: customTranslate :: [(Codon, AA)] -> Int -> FastaSequence -> Either Text FastaSequence
+ Data.Fasta.Text.Types: type AA = Char
- Data.Fasta.ByteString.Lazy.Translation: codon2aa :: Codon -> Either ByteString ByteString
+ Data.Fasta.ByteString.Lazy.Translation: codon2aa :: Codon -> Either ByteString AA
- Data.Fasta.ByteString.Translation: codon2aa :: Codon -> Either ByteString ByteString
+ Data.Fasta.ByteString.Translation: codon2aa :: Codon -> Either ByteString AA
- Data.Fasta.String.Translation: codon2aa :: Codon -> Either String Char
+ Data.Fasta.String.Translation: codon2aa :: Codon -> Either String AA
- Data.Fasta.Text.Lazy.Translation: codon2aa :: Codon -> Either Text Text
+ Data.Fasta.Text.Lazy.Translation: codon2aa :: Codon -> Either Text Char
- Data.Fasta.Text.Translation: codon2aa :: Codon -> Either Text Text
+ Data.Fasta.Text.Translation: codon2aa :: Codon -> Either Text AA
Files
- fasta.cabal +1/−1
- src/Data/Fasta/ByteString/Lazy/Translation.hs +52/−33
- src/Data/Fasta/ByteString/Lazy/Types.hs +2/−1
- src/Data/Fasta/ByteString/Translation.hs +52/−33
- src/Data/Fasta/ByteString/Types.hs +2/−1
- src/Data/Fasta/String/Translation.hs +26/−6
- src/Data/Fasta/String/Types.hs +1/−0
- src/Data/Fasta/Text/Lazy/Translation.hs +52/−33
- src/Data/Fasta/Text/Lazy/Types.hs +2/−1
- src/Data/Fasta/Text/Translation.hs +52/−33
- src/Data/Fasta/Text/Types.hs +2/−1
fasta.cabal view
@@ -10,7 +10,7 @@ -- PVP summary: +-+------- breaking API changes -- | | +----- non-breaking API additions -- | | | +--- code changes with no API change-version: 0.10.1.0+version: 0.10.2.0 -- A short (one-line) description of the package. synopsis: A simple, mindless parser for fasta files.
src/Data/Fasta/ByteString/Lazy/Translation.hs view
@@ -9,7 +9,10 @@ module Data.Fasta.ByteString.Lazy.Translation ( chunksOf , codon2aa- , translate ) where+ , customCodon2aa+ , translate+ , customTranslate+ ) where -- Built in import Data.Char@@ -31,50 +34,60 @@ -- | Converts a codon to an amino acid -- Remember, if there is an "N" in that DNA sequence, then it is translated -- as an X, an unknown amino acid.-codon2aa :: Codon -> Either BL.ByteString BL.ByteString+codon2aa :: Codon -> Either BL.ByteString AA codon2aa x- | codon `elem` ["GCT", "GCC", "GCA", "GCG"] = Right "A"- | codon `elem` ["CGT", "CGC", "CGA", "CGG", "AGA", "AGG"] = Right "R"- | codon `elem` ["AAT", "AAC"] = Right "N"- | codon `elem` ["GAT", "GAC"] = Right "D"- | codon `elem` ["TGT", "TGC"] = Right "C"- | codon `elem` ["CAA", "CAG"] = Right "Q"- | codon `elem` ["GAA", "GAG"] = Right "E"- | codon `elem` ["GGT", "GGC", "GGA", "GGG"] = Right "G"- | codon `elem` ["CAT", "CAC"] = Right "H"- | codon `elem` ["ATT", "ATC", "ATA"] = Right "I"- | codon `elem` ["ATG"] = Right "M"- | codon `elem` ["TTA", "TTG", "CTT", "CTC", "CTA", "CTG"] = Right "L"- | codon `elem` ["AAA", "AAG"] = Right "K"- | codon `elem` ["TTT", "TTC"] = Right "F"- | codon `elem` ["CCT", "CCC", "CCA", "CCG"] = Right "P"- | codon `elem` ["TCT", "TCC", "TCA", "TCG", "AGT", "AGC"] = Right "S"- | codon `elem` ["ACT", "ACC", "ACA", "ACG"] = Right "T"- | codon `elem` ["TGG"] = Right "W"- | codon `elem` ["TAT", "TAC"] = Right "Y"- | codon `elem` ["GTT", "GTC", "GTA", "GTG"] = Right "V"- | codon `elem` ["TAA", "TGA", "TAG"] = Right "*"- | codon `elem` ["---", "..."] = Right "-"- | codon == "~~~" = Right "-"- | 'N' `BL.elem` codon = Right "X"- | '-' `BL.elem` codon = Right "-"- | '.' `BL.elem` codon = Right "-"+ | codon `elem` ["GCT", "GCC", "GCA", "GCG"] = Right 'A'+ | codon `elem` ["CGT", "CGC", "CGA", "CGG", "AGA", "AGG"] = Right 'R'+ | codon `elem` ["AAT", "AAC"] = Right 'N'+ | codon `elem` ["GAT", "GAC"] = Right 'D'+ | codon `elem` ["TGT", "TGC"] = Right 'C'+ | codon `elem` ["CAA", "CAG"] = Right 'Q'+ | codon `elem` ["GAA", "GAG"] = Right 'E'+ | codon `elem` ["GGT", "GGC", "GGA", "GGG"] = Right 'G'+ | codon `elem` ["CAT", "CAC"] = Right 'H'+ | codon `elem` ["ATT", "ATC", "ATA"] = Right 'I'+ | codon `elem` ["ATG"] = Right 'M'+ | codon `elem` ["TTA", "TTG", "CTT", "CTC", "CTA", "CTG"] = Right 'L'+ | codon `elem` ["AAA", "AAG"] = Right 'K'+ | codon `elem` ["TTT", "TTC"] = Right 'F'+ | codon `elem` ["CCT", "CCC", "CCA", "CCG"] = Right 'P'+ | codon `elem` ["TCT", "TCC", "TCA", "TCG", "AGT", "AGC"] = Right 'S'+ | codon `elem` ["ACT", "ACC", "ACA", "ACG"] = Right 'T'+ | codon `elem` ["TGG"] = Right 'W'+ | codon `elem` ["TAT", "TAC"] = Right 'Y'+ | codon `elem` ["GTT", "GTC", "GTA", "GTG"] = Right 'V'+ | codon `elem` ["TAA", "TGA", "TAG"] = Right '*'+ | codon `elem` ["---", "..."] = Right '-'+ | codon == "~~~" = Right '-'+ | 'N' `BL.elem` codon = Right 'X'+ | '-' `BL.elem` codon = Right '-'+ | '.' `BL.elem` codon = Right '-' | otherwise = Left errorMsg where codon = BL.map toUpper x errorMsg = BL.append "Unidentified codon: " codon +-- | Translate a codon using a custom table+customCodon2aa :: [(Codon, Char)] -> Codon -> Either BL.ByteString AA+customCodon2aa table codon = case lookup codon table of+ (Just x) -> Right x+ Nothing -> codon2aa codon+ -- | Translates a bytestring of nucleotides given a reading frame (1, 2, or -- 3) -- drops the first 0, 1, or 2 nucleotides respectively. Returns--- a bytestring with the error if the codon is invalid.-translate :: Int64 -> FastaSequence -> Either BL.ByteString FastaSequence-translate pos x+-- a bytestring with the error if the codon is invalid. Also has customized+-- codon translations as well overriding the defaults.+customTranslate :: [(Codon, AA)]+ -> Int64+ -> FastaSequence+ -> Either BL.ByteString FastaSequence+customTranslate table pos x | any isLeft' translation = Left $ head . lefts $ translation- | otherwise = Right $ x { fastaSeq = BL.concat+ | otherwise = Right $ x { fastaSeq = BL.pack . rights $ translation } where- translation = map codon2aa+ translation = map (customCodon2aa table) . filter ((== 3) . BL.length) . chunksOf 3 . BL.drop (pos - 1)@@ -82,3 +95,9 @@ $ x isLeft' (Left _) = True isLeft' _ = False++-- | Translates a bytestring of nucleotides given a reading frame (1, 2, or+-- 3) -- drops the first 0, 1, or 2 nucleotides respectively. Returns+-- a bytestring with the error if the codon is invalid.+translate :: Int64 -> FastaSequence -> Either BL.ByteString FastaSequence+translate = customTranslate []
src/Data/Fasta/ByteString/Lazy/Types.hs view
@@ -18,9 +18,10 @@ } deriving (Eq, Ord, Show) -- Basic+type Codon = BL.ByteString+type AA = Char type Clone = FastaSequence type Germline = FastaSequence-type Codon = BL.ByteString -- Advanced -- | A clone is a collection of sequences derived from a germline with
src/Data/Fasta/ByteString/Translation.hs view
@@ -9,7 +9,10 @@ module Data.Fasta.ByteString.Translation ( chunksOf , codon2aa- , translate ) where+ , customCodon2aa+ , translate+ , customTranslate+ ) where -- Built in import Data.Char@@ -30,50 +33,60 @@ -- | Converts a codon to an amino acid -- Remember, if there is an "N" in that DNA sequence, then it is translated -- as an X, an unknown amino acid.-codon2aa :: Codon -> Either B.ByteString B.ByteString+codon2aa :: Codon -> Either B.ByteString AA codon2aa x- | codon `elem` ["GCT", "GCC", "GCA", "GCG"] = Right "A"- | codon `elem` ["CGT", "CGC", "CGA", "CGG", "AGA", "AGG"] = Right "R"- | codon `elem` ["AAT", "AAC"] = Right "N"- | codon `elem` ["GAT", "GAC"] = Right "D"- | codon `elem` ["TGT", "TGC"] = Right "C"- | codon `elem` ["CAA", "CAG"] = Right "Q"- | codon `elem` ["GAA", "GAG"] = Right "E"- | codon `elem` ["GGT", "GGC", "GGA", "GGG"] = Right "G"- | codon `elem` ["CAT", "CAC"] = Right "H"- | codon `elem` ["ATT", "ATC", "ATA"] = Right "I"- | codon `elem` ["ATG"] = Right "M"- | codon `elem` ["TTA", "TTG", "CTT", "CTC", "CTA", "CTG"] = Right "L"- | codon `elem` ["AAA", "AAG"] = Right "K"- | codon `elem` ["TTT", "TTC"] = Right "F"- | codon `elem` ["CCT", "CCC", "CCA", "CCG"] = Right "P"- | codon `elem` ["TCT", "TCC", "TCA", "TCG", "AGT", "AGC"] = Right "S"- | codon `elem` ["ACT", "ACC", "ACA", "ACG"] = Right "T"- | codon `elem` ["TGG"] = Right "W"- | codon `elem` ["TAT", "TAC"] = Right "Y"- | codon `elem` ["GTT", "GTC", "GTA", "GTG"] = Right "V"- | codon `elem` ["TAA", "TGA", "TAG"] = Right "*"- | codon `elem` ["---", "..."] = Right "-"- | codon == "~~~" = Right "-"- | "N" `B.isInfixOf` codon = Right "X"- | "-" `B.isInfixOf` codon = Right "-"- | "." `B.isInfixOf` codon = Right "-"+ | codon `elem` ["GCT", "GCC", "GCA", "GCG"] = Right 'A'+ | codon `elem` ["CGT", "CGC", "CGA", "CGG", "AGA", "AGG"] = Right 'R'+ | codon `elem` ["AAT", "AAC"] = Right 'N'+ | codon `elem` ["GAT", "GAC"] = Right 'D'+ | codon `elem` ["TGT", "TGC"] = Right 'C'+ | codon `elem` ["CAA", "CAG"] = Right 'Q'+ | codon `elem` ["GAA", "GAG"] = Right 'E'+ | codon `elem` ["GGT", "GGC", "GGA", "GGG"] = Right 'G'+ | codon `elem` ["CAT", "CAC"] = Right 'H'+ | codon `elem` ["ATT", "ATC", "ATA"] = Right 'I'+ | codon `elem` ["ATG"] = Right 'M'+ | codon `elem` ["TTA", "TTG", "CTT", "CTC", "CTA", "CTG"] = Right 'L'+ | codon `elem` ["AAA", "AAG"] = Right 'K'+ | codon `elem` ["TTT", "TTC"] = Right 'F'+ | codon `elem` ["CCT", "CCC", "CCA", "CCG"] = Right 'P'+ | codon `elem` ["TCT", "TCC", "TCA", "TCG", "AGT", "AGC"] = Right 'S'+ | codon `elem` ["ACT", "ACC", "ACA", "ACG"] = Right 'T'+ | codon `elem` ["TGG"] = Right 'W'+ | codon `elem` ["TAT", "TAC"] = Right 'Y'+ | codon `elem` ["GTT", "GTC", "GTA", "GTG"] = Right 'V'+ | codon `elem` ["TAA", "TGA", "TAG"] = Right '*'+ | codon `elem` ["---", "..."] = Right '-'+ | codon == "~~~" = Right '-'+ | "N" `B.isInfixOf` codon = Right 'X'+ | "-" `B.isInfixOf` codon = Right '-'+ | "." `B.isInfixOf` codon = Right '-' | otherwise = Left errorMsg where codon = B.map toUpper x errorMsg = B.append "Unidentified codon: " codon +-- | Translate a codon using a custom table+customCodon2aa :: [(Codon, Char)] -> Codon -> Either B.ByteString AA+customCodon2aa table codon = case lookup codon table of+ (Just x) -> Right x+ Nothing -> codon2aa codon+ -- | Translates a bytestring of nucleotides given a reading frame (1, 2, or -- 3) -- drops the first 0, 1, or 2 nucleotides respectively. Returns--- a bytestring with the error if the codon is invalid.-translate :: Int -> FastaSequence -> Either B.ByteString FastaSequence-translate pos x+-- a bytestring with the error if the codon is invalid. Also has customized+-- codon translations as well overriding the defaults.+customTranslate :: [(Codon, AA)]+ -> Int+ -> FastaSequence+ -> Either B.ByteString FastaSequence+customTranslate table pos x | any isLeft' translation = Left $ head . lefts $ translation- | otherwise = Right $ x { fastaSeq = B.concat+ | otherwise = Right $ x { fastaSeq = B.pack . rights $ translation } where- translation = map codon2aa+ translation = map (customCodon2aa table) . filter ((== 3) . B.length) . chunksOf 3 . B.drop (pos - 1)@@ -81,3 +94,9 @@ $ x isLeft' (Left _) = True isLeft' _ = False++-- | Translates a bytestring of nucleotides given a reading frame (1, 2, or+-- 3) -- drops the first 0, 1, or 2 nucleotides respectively. Returns+-- a bytestring with the error if the codon is invalid.+translate :: Int -> FastaSequence -> Either B.ByteString FastaSequence+translate = customTranslate []
src/Data/Fasta/ByteString/Types.hs view
@@ -18,9 +18,10 @@ } deriving (Eq, Ord, Show) -- Basic+type Codon = B.ByteString+type AA = Char type Clone = FastaSequence type Germline = FastaSequence-type Codon = B.ByteString -- Advanced -- | A clone is a collection of sequences derived from a germline with
src/Data/Fasta/String/Translation.hs view
@@ -5,7 +5,11 @@ amino acids for strings. -} -module Data.Fasta.String.Translation where+module Data.Fasta.String.Translation ( codon2aa+ , customCodon2aa+ , translate+ , customTranslate+ ) where -- Built in import Data.Either@@ -20,7 +24,7 @@ -- | Converts a codon to an amino acid -- Remember, if there is an "N" in that DNA sequence, then it is translated -- as an X, an unknown amino acid.-codon2aa :: Codon -> Either String Char+codon2aa :: Codon -> Either String AA codon2aa x | codon `elem` ["GCT", "GCC", "GCA", "GCG"] = Right 'A' | codon `elem` ["CGT", "CGC", "CGA", "CGG", "AGA", "AGG"] = Right 'R'@@ -53,15 +57,25 @@ codon = map toUpper x errorMsg = "Unidentified codon: " ++ codon +-- | Translate a codon using a custom table+customCodon2aa :: [(Codon, Char)] -> Codon -> Either String AA+customCodon2aa table codon = case lookup codon table of+ (Just x) -> Right x+ Nothing -> codon2aa codon+ -- | Translates a string of nucleotides given a reading frame (1, 2, or -- 3) -- drops the first 0, 1, or 2 nucleotides respectively. Returns--- a string with the error if the codon is invalid.-translate :: Int -> FastaSequence -> Either String FastaSequence-translate pos x+-- a string with the error if the codon is invalid. Also has customized+-- codon translations as well overriding the defaults.+customTranslate :: [(Codon, AA)]+ -> Int+ -> FastaSequence+ -> Either String FastaSequence+customTranslate table pos x | any isLeft' translation = Left $ head . lefts $ translation | otherwise = Right $ x { fastaSeq = rights translation } where- translation = map codon2aa+ translation = map (customCodon2aa table) . filter ((== 3) . length) . Split.chunksOf 3 . drop (pos - 1)@@ -69,3 +83,9 @@ $ x isLeft' (Left _) = True isLeft' _ = False++-- | Translates a string of nucleotides given a reading frame (1, 2, or+-- 3) -- drops the first 0, 1, or 2 nucleotides respectively. Returns+-- a string with the error if the codon is invalid.+translate :: Int -> FastaSequence -> Either String FastaSequence+translate = customTranslate []
src/Data/Fasta/String/Types.hs view
@@ -16,6 +16,7 @@ -- Basic type Codon = String+type AA = Char type Clone = FastaSequence type Germline = FastaSequence
src/Data/Fasta/Text/Lazy/Translation.hs view
@@ -8,7 +8,10 @@ {-# LANGUAGE OverloadedStrings #-} module Data.Fasta.Text.Lazy.Translation ( codon2aa- , translate ) where+ , customCodon2aa+ , translate+ , customTranslate+ ) where -- Built in import Data.Either@@ -21,50 +24,60 @@ -- | Converts a codon to an amino acid -- Remember, if there is an "N" in that DNA sequence, then it is translated -- as an X, an unknown amino acid.-codon2aa :: Codon -> Either T.Text T.Text+codon2aa :: Codon -> Either T.Text Char codon2aa x- | codon `elem` ["GCT", "GCC", "GCA", "GCG"] = Right "A"- | codon `elem` ["CGT", "CGC", "CGA", "CGG", "AGA", "AGG"] = Right "R"- | codon `elem` ["AAT", "AAC"] = Right "N"- | codon `elem` ["GAT", "GAC"] = Right "D"- | codon `elem` ["TGT", "TGC"] = Right "C"- | codon `elem` ["CAA", "CAG"] = Right "Q"- | codon `elem` ["GAA", "GAG"] = Right "E"- | codon `elem` ["GGT", "GGC", "GGA", "GGG"] = Right "G"- | codon `elem` ["CAT", "CAC"] = Right "H"- | codon `elem` ["ATT", "ATC", "ATA"] = Right "I"- | codon `elem` ["ATG"] = Right "M"- | codon `elem` ["TTA", "TTG", "CTT", "CTC", "CTA", "CTG"] = Right "L"- | codon `elem` ["AAA", "AAG"] = Right "K"- | codon `elem` ["TTT", "TTC"] = Right "F"- | codon `elem` ["CCT", "CCC", "CCA", "CCG"] = Right "P"- | codon `elem` ["TCT", "TCC", "TCA", "TCG", "AGT", "AGC"] = Right "S"- | codon `elem` ["ACT", "ACC", "ACA", "ACG"] = Right "T"- | codon `elem` ["TGG"] = Right "W"- | codon `elem` ["TAT", "TAC"] = Right "Y"- | codon `elem` ["GTT", "GTC", "GTA", "GTG"] = Right "V"- | codon `elem` ["TAA", "TGA", "TAG"] = Right "*"- | codon `elem` ["---", "..."] = Right "-"- | codon == "~~~" = Right "-"- | "N" `T.isInfixOf` codon = Right "X"- | "-" `T.isInfixOf` codon = Right "-"- | "." `T.isInfixOf` codon = Right "-"+ | codon `elem` ["GCT", "GCC", "GCA", "GCG"] = Right 'A'+ | codon `elem` ["CGT", "CGC", "CGA", "CGG", "AGA", "AGG"] = Right 'R'+ | codon `elem` ["AAT", "AAC"] = Right 'N'+ | codon `elem` ["GAT", "GAC"] = Right 'D'+ | codon `elem` ["TGT", "TGC"] = Right 'C'+ | codon `elem` ["CAA", "CAG"] = Right 'Q'+ | codon `elem` ["GAA", "GAG"] = Right 'E'+ | codon `elem` ["GGT", "GGC", "GGA", "GGG"] = Right 'G'+ | codon `elem` ["CAT", "CAC"] = Right 'H'+ | codon `elem` ["ATT", "ATC", "ATA"] = Right 'I'+ | codon `elem` ["ATG"] = Right 'M'+ | codon `elem` ["TTA", "TTG", "CTT", "CTC", "CTA", "CTG"] = Right 'L'+ | codon `elem` ["AAA", "AAG"] = Right 'K'+ | codon `elem` ["TTT", "TTC"] = Right 'F'+ | codon `elem` ["CCT", "CCC", "CCA", "CCG"] = Right 'P'+ | codon `elem` ["TCT", "TCC", "TCA", "TCG", "AGT", "AGC"] = Right 'S'+ | codon `elem` ["ACT", "ACC", "ACA", "ACG"] = Right 'T'+ | codon `elem` ["TGG"] = Right 'W'+ | codon `elem` ["TAT", "TAC"] = Right 'Y'+ | codon `elem` ["GTT", "GTC", "GTA", "GTG"] = Right 'V'+ | codon `elem` ["TAA", "TGA", "TAG"] = Right '*'+ | codon `elem` ["---", "..."] = Right '-'+ | codon == "~~~" = Right '-'+ | "N" `T.isInfixOf` codon = Right 'X'+ | "-" `T.isInfixOf` codon = Right '-'+ | "." `T.isInfixOf` codon = Right '-' | otherwise = Left errorMsg where codon = T.toUpper x errorMsg = T.append "Unidentified codon: " codon +-- | Translate a codon using a custom table+customCodon2aa :: [(Codon, Char)] -> Codon -> Either T.Text AA+customCodon2aa table codon = case lookup codon table of+ (Just x) -> Right x+ Nothing -> codon2aa codon+ -- | Translates a text of nucleotides given a reading frame (1, 2, or -- 3) -- drops the first 0, 1, or 2 nucleotides respectively. Returns--- a text with the error if the codon is invalid.-translate :: Int64 -> FastaSequence -> Either T.Text FastaSequence-translate pos x+-- a text with the error if the codon is invalid. Also has customized codon+-- translations as well overriding the defaults.+customTranslate :: [(Codon, AA)]+ -> Int64+ -> FastaSequence+ -> Either T.Text FastaSequence+customTranslate table pos x | any isLeft' translation = Left $ head . lefts $ translation- | otherwise = Right $ x { fastaSeq = T.concat+ | otherwise = Right $ x { fastaSeq = T.pack . rights $ translation } where- translation = map codon2aa+ translation = map (customCodon2aa table) . filter ((== 3) . T.length) . T.chunksOf 3 . T.drop (pos - 1)@@ -72,3 +85,9 @@ $ x isLeft' (Left _) = True isLeft' _ = False++-- | Translates a text of nucleotides given a reading frame (1, 2, or+-- 3) -- drops the first 0, 1, or 2 nucleotides respectively. Returns+-- a text with the error if the codon is invalid.+translate :: Int64 -> FastaSequence -> Either T.Text FastaSequence+translate = customTranslate []
src/Data/Fasta/Text/Lazy/Types.hs view
@@ -18,9 +18,10 @@ } deriving (Eq, Ord, Show) -- Basic+type Codon = T.Text+type AA = Char type Clone = FastaSequence type Germline = FastaSequence-type Codon = T.Text -- Advanced -- | A clone is a collection of sequences derived from a germline with
src/Data/Fasta/Text/Translation.hs view
@@ -8,7 +8,10 @@ {-# LANGUAGE OverloadedStrings #-} module Data.Fasta.Text.Translation ( codon2aa- , translate ) where+ , customCodon2aa+ , translate+ , customTranslate+ ) where -- Built in import Data.Either@@ -20,50 +23,60 @@ -- | Converts a codon to an amino acid -- Remember, if there is an "N" in that DNA sequence, then it is translated -- as an X, an unknown amino acid.-codon2aa :: Codon -> Either T.Text T.Text+codon2aa :: Codon -> Either T.Text AA codon2aa x- | codon `elem` ["GCT", "GCC", "GCA", "GCG"] = Right "A"- | codon `elem` ["CGT", "CGC", "CGA", "CGG", "AGA", "AGG"] = Right "R"- | codon `elem` ["AAT", "AAC"] = Right "N"- | codon `elem` ["GAT", "GAC"] = Right "D"- | codon `elem` ["TGT", "TGC"] = Right "C"- | codon `elem` ["CAA", "CAG"] = Right "Q"- | codon `elem` ["GAA", "GAG"] = Right "E"- | codon `elem` ["GGT", "GGC", "GGA", "GGG"] = Right "G"- | codon `elem` ["CAT", "CAC"] = Right "H"- | codon `elem` ["ATT", "ATC", "ATA"] = Right "I"- | codon `elem` ["ATG"] = Right "M"- | codon `elem` ["TTA", "TTG", "CTT", "CTC", "CTA", "CTG"] = Right "L"- | codon `elem` ["AAA", "AAG"] = Right "K"- | codon `elem` ["TTT", "TTC"] = Right "F"- | codon `elem` ["CCT", "CCC", "CCA", "CCG"] = Right "P"- | codon `elem` ["TCT", "TCC", "TCA", "TCG", "AGT", "AGC"] = Right "S"- | codon `elem` ["ACT", "ACC", "ACA", "ACG"] = Right "T"- | codon `elem` ["TGG"] = Right "W"- | codon `elem` ["TAT", "TAC"] = Right "Y"- | codon `elem` ["GTT", "GTC", "GTA", "GTG"] = Right "V"- | codon `elem` ["TAA", "TGA", "TAG"] = Right "*"- | codon `elem` ["---", "..."] = Right "-"- | codon == "~~~" = Right "-"- | "N" `T.isInfixOf` codon = Right "X"- | "-" `T.isInfixOf` codon = Right "-"- | "." `T.isInfixOf` codon = Right "-"+ | codon `elem` ["GCT", "GCC", "GCA", "GCG"] = Right 'A'+ | codon `elem` ["CGT", "CGC", "CGA", "CGG", "AGA", "AGG"] = Right 'R'+ | codon `elem` ["AAT", "AAC"] = Right 'N'+ | codon `elem` ["GAT", "GAC"] = Right 'D'+ | codon `elem` ["TGT", "TGC"] = Right 'C'+ | codon `elem` ["CAA", "CAG"] = Right 'Q'+ | codon `elem` ["GAA", "GAG"] = Right 'E'+ | codon `elem` ["GGT", "GGC", "GGA", "GGG"] = Right 'G'+ | codon `elem` ["CAT", "CAC"] = Right 'H'+ | codon `elem` ["ATT", "ATC", "ATA"] = Right 'I'+ | codon `elem` ["ATG"] = Right 'M'+ | codon `elem` ["TTA", "TTG", "CTT", "CTC", "CTA", "CTG"] = Right 'L'+ | codon `elem` ["AAA", "AAG"] = Right 'K'+ | codon `elem` ["TTT", "TTC"] = Right 'F'+ | codon `elem` ["CCT", "CCC", "CCA", "CCG"] = Right 'P'+ | codon `elem` ["TCT", "TCC", "TCA", "TCG", "AGT", "AGC"] = Right 'S'+ | codon `elem` ["ACT", "ACC", "ACA", "ACG"] = Right 'T'+ | codon `elem` ["TGG"] = Right 'W'+ | codon `elem` ["TAT", "TAC"] = Right 'Y'+ | codon `elem` ["GTT", "GTC", "GTA", "GTG"] = Right 'V'+ | codon `elem` ["TAA", "TGA", "TAG"] = Right '*'+ | codon `elem` ["---", "..."] = Right '-'+ | codon == "~~~" = Right '-'+ | "N" `T.isInfixOf` codon = Right 'X'+ | "-" `T.isInfixOf` codon = Right '-'+ | "." `T.isInfixOf` codon = Right '-' | otherwise = Left errorMsg where codon = T.toUpper x errorMsg = T.append "Unidentified codon: " codon +-- | Translate a codon using a custom table+customCodon2aa :: [(Codon, Char)] -> Codon -> Either T.Text AA+customCodon2aa table codon = case lookup codon table of+ (Just x) -> Right x+ Nothing -> codon2aa codon+ -- | Translates a text of nucleotides given a reading frame (1, 2, or -- 3) -- drops the first 0, 1, or 2 nucleotides respectively. Returns--- a text with the error if the codon is invalid.-translate :: Int -> FastaSequence -> Either T.Text FastaSequence-translate pos x+-- a text with the error if the codon is invalid. Also has customized codon+-- translations as well overriding the defaults.+customTranslate :: [(Codon, AA)]+ -> Int+ -> FastaSequence+ -> Either T.Text FastaSequence+customTranslate table pos x | any isLeft' translation = Left $ head . lefts $ translation- | otherwise = Right $ x { fastaSeq = T.concat+ | otherwise = Right $ x { fastaSeq = T.pack . rights $ translation } where- translation = map codon2aa+ translation = map (customCodon2aa table) . filter ((== 3) . T.length) . T.chunksOf 3 . T.drop (pos - 1)@@ -71,3 +84,9 @@ $ x isLeft' (Left _) = True isLeft' _ = False++-- | Translates a text of nucleotides given a reading frame (1, 2, or+-- 3) -- drops the first 0, 1, or 2 nucleotides respectively. Returns+-- a text with the error if the codon is invalid.+translate :: Int -> FastaSequence -> Either T.Text FastaSequence+translate = customTranslate []
src/Data/Fasta/Text/Types.hs view
@@ -18,9 +18,10 @@ } deriving (Eq, Ord, Show) -- Basic+type Codon = T.Text+type AA = Char type Clone = FastaSequence type Germline = FastaSequence-type Codon = T.Text -- Advanced -- | A clone is a collection of sequences derived from a germline with