dcatoolkit 0.1.5__tar.gz → 0.1.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dcatoolkit-0.1.5 → dcatoolkit-0.1.7}/LICENSE +21 -21
- {dcatoolkit-0.1.5/src/dcatoolkit.egg-info → dcatoolkit-0.1.7}/PKG-INFO +72 -57
- dcatoolkit-0.1.7/README.md +17 -0
- {dcatoolkit-0.1.5 → dcatoolkit-0.1.7}/pyproject.toml +53 -53
- {dcatoolkit-0.1.5 → dcatoolkit-0.1.7}/setup.cfg +4 -4
- {dcatoolkit-0.1.5 → dcatoolkit-0.1.7}/src/dcatoolkit/__init__.py +3 -3
- {dcatoolkit-0.1.5 → dcatoolkit-0.1.7}/src/dcatoolkit/analytics.py +159 -144
- {dcatoolkit-0.1.5 → dcatoolkit-0.1.7}/src/dcatoolkit/representation.py +978 -928
- {dcatoolkit-0.1.5 → dcatoolkit-0.1.7/src/dcatoolkit.egg-info}/PKG-INFO +72 -57
- dcatoolkit-0.1.5/README.md +0 -2
- {dcatoolkit-0.1.5 → dcatoolkit-0.1.7}/src/dcatoolkit.egg-info/SOURCES.txt +0 -0
- {dcatoolkit-0.1.5 → dcatoolkit-0.1.7}/src/dcatoolkit.egg-info/dependency_links.txt +0 -0
- {dcatoolkit-0.1.5 → dcatoolkit-0.1.7}/src/dcatoolkit.egg-info/requires.txt +0 -0
- {dcatoolkit-0.1.5 → dcatoolkit-0.1.7}/src/dcatoolkit.egg-info/top_level.txt +0 -0
|
@@ -1,21 +1,21 @@
|
|
|
1
|
-
MIT License
|
|
2
|
-
|
|
3
|
-
Copyright (c) 2024 Raheel Syed Ahmed
|
|
4
|
-
|
|
5
|
-
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
-
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
-
in the Software without restriction, including without limitation the rights
|
|
8
|
-
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
-
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
-
furnished to do so, subject to the following conditions:
|
|
11
|
-
|
|
12
|
-
The above copyright notice and this permission notice shall be included in all
|
|
13
|
-
copies or substantial portions of the Software.
|
|
14
|
-
|
|
15
|
-
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
-
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
-
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
-
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
-
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
-
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
-
SOFTWARE.
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2024 Raheel Syed Ahmed
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -1,57 +1,72 @@
|
|
|
1
|
-
Metadata-Version: 2.1
|
|
2
|
-
Name: dcatoolkit
|
|
3
|
-
Version: 0.1.
|
|
4
|
-
Summary: Collection of useful modules and representations for managing DCA output data.
|
|
5
|
-
Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
6
|
-
Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
7
|
-
License: MIT License
|
|
8
|
-
|
|
9
|
-
Copyright (c) 2024 Raheel Syed Ahmed
|
|
10
|
-
|
|
11
|
-
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
12
|
-
of this software and associated documentation files (the "Software"), to deal
|
|
13
|
-
in the Software without restriction, including without limitation the rights
|
|
14
|
-
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
15
|
-
copies of the Software, and to permit persons to whom the Software is
|
|
16
|
-
furnished to do so, subject to the following conditions:
|
|
17
|
-
|
|
18
|
-
The above copyright notice and this permission notice shall be included in all
|
|
19
|
-
copies or substantial portions of the Software.
|
|
20
|
-
|
|
21
|
-
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
22
|
-
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
23
|
-
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
24
|
-
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
25
|
-
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
26
|
-
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
27
|
-
SOFTWARE.
|
|
28
|
-
|
|
29
|
-
Keywords: dca,toolkit,DI,coevolution
|
|
30
|
-
Classifier: Development Status :: 4 - Beta
|
|
31
|
-
Classifier: Intended Audience :: Science/Research
|
|
32
|
-
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
33
|
-
Classifier: License :: OSI Approved :: MIT License
|
|
34
|
-
Classifier: Programming Language :: Python :: 3
|
|
35
|
-
Classifier: Programming Language :: Python :: 3.10
|
|
36
|
-
Classifier: Programming Language :: Python :: 3.11
|
|
37
|
-
Classifier: Programming Language :: Python :: 3.12
|
|
38
|
-
Requires-Python: >=3.10
|
|
39
|
-
Description-Content-Type: text/markdown
|
|
40
|
-
License-File: LICENSE
|
|
41
|
-
Requires-Dist: biotite
|
|
42
|
-
Requires-Dist: matplotlib>=3.8.0
|
|
43
|
-
Requires-Dist: numpy>=1.26.0
|
|
44
|
-
Requires-Dist: pandas>=2.1.0
|
|
45
|
-
Requires-Dist: scikit-learn>=1.3
|
|
46
|
-
Requires-Dist: scipy>=1.11.0
|
|
47
|
-
Provides-Extra: tests
|
|
48
|
-
Requires-Dist: pytest; extra == "tests"
|
|
49
|
-
Provides-Extra: docs
|
|
50
|
-
Requires-Dist: sphinx; extra == "docs"
|
|
51
|
-
Requires-Dist: pdoc; extra == "docs"
|
|
52
|
-
Requires-Dist: numpydoc; extra == "docs"
|
|
53
|
-
Provides-Extra: lint
|
|
54
|
-
Requires-Dist: ruffle; extra == "lint"
|
|
55
|
-
|
|
56
|
-
# dcatoolkit
|
|
57
|
-
Collection of useful modules and representations for managing DCA output data.
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: dcatoolkit
|
|
3
|
+
Version: 0.1.7
|
|
4
|
+
Summary: Collection of useful modules and representations for managing DCA output data.
|
|
5
|
+
Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
6
|
+
Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
7
|
+
License: MIT License
|
|
8
|
+
|
|
9
|
+
Copyright (c) 2024 Raheel Syed Ahmed
|
|
10
|
+
|
|
11
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
12
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
13
|
+
in the Software without restriction, including without limitation the rights
|
|
14
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
15
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
16
|
+
furnished to do so, subject to the following conditions:
|
|
17
|
+
|
|
18
|
+
The above copyright notice and this permission notice shall be included in all
|
|
19
|
+
copies or substantial portions of the Software.
|
|
20
|
+
|
|
21
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
22
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
23
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
24
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
25
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
26
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
27
|
+
SOFTWARE.
|
|
28
|
+
|
|
29
|
+
Keywords: dca,toolkit,DI,coevolution
|
|
30
|
+
Classifier: Development Status :: 4 - Beta
|
|
31
|
+
Classifier: Intended Audience :: Science/Research
|
|
32
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
33
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
34
|
+
Classifier: Programming Language :: Python :: 3
|
|
35
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
36
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
37
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
38
|
+
Requires-Python: >=3.10
|
|
39
|
+
Description-Content-Type: text/markdown
|
|
40
|
+
License-File: LICENSE
|
|
41
|
+
Requires-Dist: biotite
|
|
42
|
+
Requires-Dist: matplotlib>=3.8.0
|
|
43
|
+
Requires-Dist: numpy>=1.26.0
|
|
44
|
+
Requires-Dist: pandas>=2.1.0
|
|
45
|
+
Requires-Dist: scikit-learn>=1.3
|
|
46
|
+
Requires-Dist: scipy>=1.11.0
|
|
47
|
+
Provides-Extra: tests
|
|
48
|
+
Requires-Dist: pytest; extra == "tests"
|
|
49
|
+
Provides-Extra: docs
|
|
50
|
+
Requires-Dist: sphinx; extra == "docs"
|
|
51
|
+
Requires-Dist: pdoc; extra == "docs"
|
|
52
|
+
Requires-Dist: numpydoc; extra == "docs"
|
|
53
|
+
Provides-Extra: lint
|
|
54
|
+
Requires-Dist: ruffle; extra == "lint"
|
|
55
|
+
|
|
56
|
+
# dcatoolkit
|
|
57
|
+
Collection of useful modules and representations for managing DCA output data.
|
|
58
|
+
|
|
59
|
+
## Major Sections
|
|
60
|
+
### Representations
|
|
61
|
+
* Use Pairs to load lists, tuples, sets, and ndarrays with the correct orientation of elements. This will allow you to yield integer pairs that can be mirrored (where y becomes x and vice versa) and to subset various pairs.
|
|
62
|
+
* Use DirectInformationData to create 3-column structured ndarrays that can be sorted by "DI", mapped to a protein with a ResidueAlignment, and used to generate output for other programs (including UCSF Chimera)
|
|
63
|
+
* Use ResidueAlignment to generate a reference map. Indices of one sequence of characters can be linked to their corresponding indices of the other sequence of characters. The dictionaries produced, domain-to-protein and protein-to-domain, allow for forward mapping and backmapping.
|
|
64
|
+
* Use StructureInformation to find contacts in a protein structure and find atomic information related to specific pairs of interest.
|
|
65
|
+
### Analytics
|
|
66
|
+
* Use MSATools to load in Multiple Sequence Alignment (MSA) data and provide functionality including generating frequency statistics on "gappiness" in the MSA and filtering and cleaning MSAs.
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
## Diagram of Hidden Markov Machine & Direct Coupling Analysis Pipeline
|
|
70
|
+
<p align="center">
|
|
71
|
+
<img src="https://github.com/user-attachments/assets/4768e08f-d513-4dbf-abc5-c80c1b3d42aa"/>
|
|
72
|
+
</p>
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
# dcatoolkit
|
|
2
|
+
Collection of useful modules and representations for managing DCA output data.
|
|
3
|
+
|
|
4
|
+
## Major Sections
|
|
5
|
+
### Representations
|
|
6
|
+
* Use Pairs to load lists, tuples, sets, and ndarrays with the correct orientation of elements. This will allow you to yield integer pairs that can be mirrored (where y becomes x and vice versa) and to subset various pairs.
|
|
7
|
+
* Use DirectInformationData to create 3-column structured ndarrays that can be sorted by "DI", mapped to a protein with a ResidueAlignment, and used to generate output for other programs (including UCSF Chimera)
|
|
8
|
+
* Use ResidueAlignment to generate a reference map. Indices of one sequence of characters can be linked to their corresponding indices of the other sequence of characters. The dictionaries produced, domain-to-protein and protein-to-domain, allow for forward mapping and backmapping.
|
|
9
|
+
* Use StructureInformation to find contacts in a protein structure and find atomic information related to specific pairs of interest.
|
|
10
|
+
### Analytics
|
|
11
|
+
* Use MSATools to load in Multiple Sequence Alignment (MSA) data and provide functionality including generating frequency statistics on "gappiness" in the MSA and filtering and cleaning MSAs.
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
## Diagram of Hidden Markov Machine & Direct Coupling Analysis Pipeline
|
|
15
|
+
<p align="center">
|
|
16
|
+
<img src="https://github.com/user-attachments/assets/4768e08f-d513-4dbf-abc5-c80c1b3d42aa"/>
|
|
17
|
+
</p>
|
|
@@ -1,54 +1,54 @@
|
|
|
1
|
-
[build-system]
|
|
2
|
-
requires = ["setuptools >= 61.0"]
|
|
3
|
-
build-backend = "setuptools.build_meta"
|
|
4
|
-
|
|
5
|
-
[project]
|
|
6
|
-
name = "dcatoolkit"
|
|
7
|
-
version = "0.1.
|
|
8
|
-
description = "Collection of useful modules and representations for managing DCA output data."
|
|
9
|
-
keywords = ["dca", "toolkit", "DI", "coevolution"]
|
|
10
|
-
|
|
11
|
-
readme = "README.md"
|
|
12
|
-
license = {file = "LICENSE"}
|
|
13
|
-
|
|
14
|
-
requires-python = ">=3.10"
|
|
15
|
-
|
|
16
|
-
authors = [
|
|
17
|
-
{name = "Raheel Syed Ahmed", email = "raheelsyedahmed@gmail.com"}
|
|
18
|
-
]
|
|
19
|
-
maintainers = [
|
|
20
|
-
{name = "Raheel Syed Ahmed", email = "raheelsyedahmed@gmail.com"}
|
|
21
|
-
]
|
|
22
|
-
|
|
23
|
-
dependencies = [
|
|
24
|
-
"biotite",
|
|
25
|
-
"matplotlib>=3.8.0",
|
|
26
|
-
"numpy>=1.26.0",
|
|
27
|
-
"pandas>=2.1.0",
|
|
28
|
-
"scikit-learn>=1.3",
|
|
29
|
-
"scipy>=1.11.0",
|
|
30
|
-
]
|
|
31
|
-
|
|
32
|
-
classifiers = [
|
|
33
|
-
"Development Status :: 4 - Beta",
|
|
34
|
-
"Intended Audience :: Science/Research",
|
|
35
|
-
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
36
|
-
"License :: OSI Approved :: MIT License",
|
|
37
|
-
"Programming Language :: Python :: 3",
|
|
38
|
-
"Programming Language :: Python :: 3.10",
|
|
39
|
-
"Programming Language :: Python :: 3.11",
|
|
40
|
-
"Programming Language :: Python :: 3.12",
|
|
41
|
-
]
|
|
42
|
-
|
|
43
|
-
[project.optional-dependencies]
|
|
44
|
-
tests = [
|
|
45
|
-
"pytest",
|
|
46
|
-
]
|
|
47
|
-
docs = [
|
|
48
|
-
"sphinx",
|
|
49
|
-
"pdoc",
|
|
50
|
-
"numpydoc"
|
|
51
|
-
]
|
|
52
|
-
lint = [
|
|
53
|
-
"ruffle",
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools >= 61.0"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "dcatoolkit"
|
|
7
|
+
version = "0.1.7"
|
|
8
|
+
description = "Collection of useful modules and representations for managing DCA output data."
|
|
9
|
+
keywords = ["dca", "toolkit", "DI", "coevolution"]
|
|
10
|
+
|
|
11
|
+
readme = "README.md"
|
|
12
|
+
license = {file = "LICENSE"}
|
|
13
|
+
|
|
14
|
+
requires-python = ">=3.10"
|
|
15
|
+
|
|
16
|
+
authors = [
|
|
17
|
+
{name = "Raheel Syed Ahmed", email = "raheelsyedahmed@gmail.com"}
|
|
18
|
+
]
|
|
19
|
+
maintainers = [
|
|
20
|
+
{name = "Raheel Syed Ahmed", email = "raheelsyedahmed@gmail.com"}
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
dependencies = [
|
|
24
|
+
"biotite",
|
|
25
|
+
"matplotlib>=3.8.0",
|
|
26
|
+
"numpy>=1.26.0",
|
|
27
|
+
"pandas>=2.1.0",
|
|
28
|
+
"scikit-learn>=1.3",
|
|
29
|
+
"scipy>=1.11.0",
|
|
30
|
+
]
|
|
31
|
+
|
|
32
|
+
classifiers = [
|
|
33
|
+
"Development Status :: 4 - Beta",
|
|
34
|
+
"Intended Audience :: Science/Research",
|
|
35
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
36
|
+
"License :: OSI Approved :: MIT License",
|
|
37
|
+
"Programming Language :: Python :: 3",
|
|
38
|
+
"Programming Language :: Python :: 3.10",
|
|
39
|
+
"Programming Language :: Python :: 3.11",
|
|
40
|
+
"Programming Language :: Python :: 3.12",
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
[project.optional-dependencies]
|
|
44
|
+
tests = [
|
|
45
|
+
"pytest",
|
|
46
|
+
]
|
|
47
|
+
docs = [
|
|
48
|
+
"sphinx",
|
|
49
|
+
"pdoc",
|
|
50
|
+
"numpydoc"
|
|
51
|
+
]
|
|
52
|
+
lint = [
|
|
53
|
+
"ruffle",
|
|
54
54
|
]
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
[egg_info]
|
|
2
|
-
tag_build =
|
|
3
|
-
tag_date = 0
|
|
4
|
-
|
|
1
|
+
[egg_info]
|
|
2
|
+
tag_build =
|
|
3
|
+
tag_date = 0
|
|
4
|
+
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
|
|
2
|
-
__version__ = "0.1.
|
|
3
|
-
from .representation import Pairs, DirectInformationData, StructureInformation, ResidueAlignment
|
|
1
|
+
|
|
2
|
+
__version__ = "0.1.7"
|
|
3
|
+
from .representation import Pairs, DirectInformationData, StructureInformation, ResidueAlignment
|
|
4
4
|
from .analytics import MSATools
|
|
@@ -1,145 +1,160 @@
|
|
|
1
|
-
import re
|
|
2
|
-
from collections import Counter
|
|
3
|
-
from typing import Optional
|
|
4
|
-
import string
|
|
5
|
-
|
|
6
|
-
class MSATools:
|
|
7
|
-
"""
|
|
8
|
-
Tools and interface for encapsulating MSA data and providing functionality for filtering and analysis.
|
|
9
|
-
|
|
10
|
-
Parameters
|
|
11
|
-
----------
|
|
12
|
-
MSA : list of tuple of str, str
|
|
13
|
-
Loaded MSA that is a list of tuples where the first element is the header and the second element is its corresponding sequence.
|
|
14
|
-
"""
|
|
15
|
-
def __init__(self, MSA: list[tuple[str, str]]):
|
|
16
|
-
self.MSA = MSA
|
|
17
|
-
|
|
18
|
-
@staticmethod
|
|
19
|
-
def load_from_file(
|
|
20
|
-
"""
|
|
21
|
-
Generates MSATools object from an MSA file in ".afa" format.
|
|
22
|
-
|
|
23
|
-
Parameters
|
|
24
|
-
----------
|
|
25
|
-
|
|
26
|
-
Filepath of the MSA in ".afa" format that is provided.
|
|
27
|
-
|
|
28
|
-
Returns
|
|
29
|
-
-------
|
|
30
|
-
MSATools
|
|
31
|
-
An MSATools instance with the appropriate list of (header, sequence) tuples where sequences are simplified and converted to single line format.
|
|
32
|
-
"""
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
""
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
1
|
+
import re
|
|
2
|
+
from collections import Counter
|
|
3
|
+
from typing import Optional, Union
|
|
4
|
+
import string, io
|
|
5
|
+
|
|
6
|
+
class MSATools:
|
|
7
|
+
"""
|
|
8
|
+
Tools and interface for encapsulating MSA data and providing functionality for filtering and analysis.
|
|
9
|
+
|
|
10
|
+
Parameters
|
|
11
|
+
----------
|
|
12
|
+
MSA : list of tuple of str, str
|
|
13
|
+
Loaded MSA that is a list of tuples where the first element is the header and the second element is its corresponding sequence.
|
|
14
|
+
"""
|
|
15
|
+
def __init__(self, MSA: list[tuple[str, str]]):
|
|
16
|
+
self.MSA = MSA
|
|
17
|
+
|
|
18
|
+
@staticmethod
|
|
19
|
+
def load_from_file(msa_source: Union[str, io.IOBase]) -> 'MSATools':
|
|
20
|
+
"""
|
|
21
|
+
Generates MSATools object from an MSA file in ".afa" format.
|
|
22
|
+
|
|
23
|
+
Parameters
|
|
24
|
+
----------
|
|
25
|
+
msa_source : str or io.IOBase
|
|
26
|
+
Filepath or IOBase of the MSA in ".afa" format that is provided.
|
|
27
|
+
|
|
28
|
+
Returns
|
|
29
|
+
-------
|
|
30
|
+
MSATools
|
|
31
|
+
An MSATools instance with the appropriate list of (header, sequence) tuples where sequences are simplified and converted to single line format.
|
|
32
|
+
"""
|
|
33
|
+
data = ""
|
|
34
|
+
msa_entries: list[tuple[str, str]] = []
|
|
35
|
+
if isinstance(msa_source, str):
|
|
36
|
+
with open(msa_source, 'r') as fs:
|
|
37
|
+
data = fs.read()
|
|
38
|
+
elif isinstance(msa_source, io.BytesIO):
|
|
39
|
+
data = msa_source.getvalue().decode()
|
|
40
|
+
elif isinstance(msa_source, io.StringIO):
|
|
41
|
+
data = msa_source.getvalue()
|
|
42
|
+
else:
|
|
43
|
+
raise Exception("msa_file is not bytesIO, stringIO, or a filepath.")
|
|
44
|
+
split_data = data.split(">")[1:]
|
|
45
|
+
for entry in split_data:
|
|
46
|
+
line_split_entry = entry.split("\n")
|
|
47
|
+
header = line_split_entry[0]
|
|
48
|
+
sequence = "".join(line_split_entry[1:])
|
|
49
|
+
msa_entries.append((">"+header, sequence))
|
|
50
|
+
return MSATools(msa_entries)
|
|
51
|
+
|
|
52
|
+
@staticmethod
|
|
53
|
+
def get_sequence_max_cont_gaps(sequence: str) -> int:
|
|
54
|
+
"""
|
|
55
|
+
Find maximum number of continuous gaps in a specific sequence.
|
|
56
|
+
|
|
57
|
+
Parameters
|
|
58
|
+
----------
|
|
59
|
+
sequence : str
|
|
60
|
+
Sequence of characters, potentially containing multiple of '-', a gap character.
|
|
61
|
+
|
|
62
|
+
Returns
|
|
63
|
+
-------
|
|
64
|
+
int
|
|
65
|
+
The maximum number of continuous gaps in a sequence.
|
|
66
|
+
"""
|
|
67
|
+
dash_match = re.findall(r"-+", sequence)
|
|
68
|
+
gap_counts = [len(match) for match in dash_match]
|
|
69
|
+
if len(gap_counts) > 0:
|
|
70
|
+
return max(gap_counts)
|
|
71
|
+
else:
|
|
72
|
+
return 0
|
|
73
|
+
|
|
74
|
+
def gap_frequency(self) -> tuple[dict[int, int], dict[int, float]]:
|
|
75
|
+
"""
|
|
76
|
+
Calculates the frequency of maximum continuous gaps throughout the MSA where the key corresponds to the number of continous gaps and the value corresponds to the number of sequences or the cumulative percentage of their sequences.
|
|
77
|
+
|
|
78
|
+
Returns
|
|
79
|
+
-------
|
|
80
|
+
tuple of dict of int, int and dict of int, int
|
|
81
|
+
Two element tuple where first element is a frequency count dictionary and the second element is a cumulative percentage of sequences with a specific maximum number of continous gaps.
|
|
82
|
+
"""
|
|
83
|
+
max_gap_counts = []
|
|
84
|
+
for header, sequence in self.MSA:
|
|
85
|
+
max_gap_counts.append(MSATools.get_sequence_max_cont_gaps(sequence))
|
|
86
|
+
frequency_count_dict = dict(Counter(max_gap_counts))
|
|
87
|
+
cumul_perc_dict = {}
|
|
88
|
+
cumul_count = 0
|
|
89
|
+
for key in sorted(frequency_count_dict.keys()):
|
|
90
|
+
value = frequency_count_dict[key]
|
|
91
|
+
cumul_count += value
|
|
92
|
+
cumul_perc_dict[key] = cumul_count / len(self.MSA)
|
|
93
|
+
return (frequency_count_dict, cumul_perc_dict)
|
|
94
|
+
|
|
95
|
+
def filter_by_continuous_gaps(self, max_gaps: Optional[int]=None) -> list[tuple[str, str]]:
|
|
96
|
+
"""
|
|
97
|
+
Filter out entries in your MSA by the number of maximum continuous gaps specified unless None is provided. Also, removes .s and lowercase letters from the sequence.
|
|
98
|
+
|
|
99
|
+
Parameters
|
|
100
|
+
----------
|
|
101
|
+
max_gaps : int
|
|
102
|
+
The maximum allowed number of continuous gaps in a sequence
|
|
103
|
+
|
|
104
|
+
Returns
|
|
105
|
+
-------
|
|
106
|
+
list of tuple of str, str
|
|
107
|
+
List of entries that are valid in that their sequences' number of maximum continuous gaps is within the threshold supplied as `max_gaps`.
|
|
108
|
+
"""
|
|
109
|
+
table = str.maketrans('', '', string.ascii_lowercase+".")
|
|
110
|
+
if max_gaps == None:
|
|
111
|
+
kept_entries = []
|
|
112
|
+
for header, sequence in self.MSA:
|
|
113
|
+
sequence = sequence.translate(table)
|
|
114
|
+
kept_entries.append((header, sequence))
|
|
115
|
+
return kept_entries
|
|
116
|
+
else:
|
|
117
|
+
kept_entries = []
|
|
118
|
+
for header, sequence in self.MSA:
|
|
119
|
+
sequence = sequence.translate(table)
|
|
120
|
+
if MSATools.get_sequence_max_cont_gaps(sequence) <= max_gaps:
|
|
121
|
+
kept_entries.append((header, sequence))
|
|
122
|
+
return kept_entries
|
|
123
|
+
|
|
124
|
+
def write(self, destination: Union[str, io.IOBase]) -> None:
|
|
125
|
+
"""
|
|
126
|
+
Writes this MSA's headers and sequences to the destination specified.
|
|
127
|
+
|
|
128
|
+
Parameters
|
|
129
|
+
----------
|
|
130
|
+
destination : str or io.IOBase
|
|
131
|
+
Filepath or IO to write the MSA supplied to.
|
|
132
|
+
|
|
133
|
+
Returns
|
|
134
|
+
-------
|
|
135
|
+
None
|
|
136
|
+
"""
|
|
137
|
+
if isinstance(destination, str):
|
|
138
|
+
with open(destination, 'w') as fs:
|
|
139
|
+
for header, sequence in self.MSA:
|
|
140
|
+
fs.write(header)
|
|
141
|
+
fs.write("\n")
|
|
142
|
+
fs.write(sequence)
|
|
143
|
+
fs.write("\n")
|
|
144
|
+
elif isinstance(destination, io.IOBase):
|
|
145
|
+
for header, sequence in self.MSA:
|
|
146
|
+
destination.write(header)
|
|
147
|
+
destination.write("\n")
|
|
148
|
+
destination.write(sequence)
|
|
149
|
+
destination.write("\n")
|
|
150
|
+
|
|
151
|
+
def __len__(self):
|
|
152
|
+
"""
|
|
153
|
+
Returns the number of sequences, and equivalently, the number of headers in the MSA.
|
|
154
|
+
|
|
155
|
+
Returns
|
|
156
|
+
-------
|
|
157
|
+
int
|
|
158
|
+
length of the MSA list of header, sequence tuples.
|
|
159
|
+
"""
|
|
145
160
|
return len(self.MSA)
|