dcatoolkit 0.2.2__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dcatoolkit-0.2.2 → dcatoolkit-0.3.0}/LICENSE +21 -21
- {dcatoolkit-0.2.2/src/dcatoolkit.egg-info → dcatoolkit-0.3.0}/PKG-INFO +122 -72
- dcatoolkit-0.3.0/README.md +63 -0
- dcatoolkit-0.3.0/pyproject.toml +86 -0
- {dcatoolkit-0.2.2 → dcatoolkit-0.3.0}/setup.cfg +4 -4
- dcatoolkit-0.3.0/src/dcatoolkit/__init__.py +15 -0
- dcatoolkit-0.3.0/src/dcatoolkit/analytics.py +230 -0
- dcatoolkit-0.3.0/src/dcatoolkit/representation/__init__.py +6 -0
- dcatoolkit-0.3.0/src/dcatoolkit/representation/alignment.py +225 -0
- dcatoolkit-0.3.0/src/dcatoolkit/representation/di.py +326 -0
- dcatoolkit-0.3.0/src/dcatoolkit/representation/pairs.py +236 -0
- dcatoolkit-0.3.0/src/dcatoolkit/representation/structure.py +808 -0
- {dcatoolkit-0.2.2 → dcatoolkit-0.3.0/src/dcatoolkit.egg-info}/PKG-INFO +122 -72
- dcatoolkit-0.3.0/src/dcatoolkit.egg-info/SOURCES.txt +21 -0
- {dcatoolkit-0.2.2 → dcatoolkit-0.3.0}/src/dcatoolkit.egg-info/requires.txt +4 -3
- dcatoolkit-0.3.0/tests/test_alignments.py +170 -0
- dcatoolkit-0.3.0/tests/test_analytics.py +36 -0
- {dcatoolkit-0.2.2 → dcatoolkit-0.3.0}/tests/test_contacts.py +95 -94
- dcatoolkit-0.3.0/tests/test_di.py +131 -0
- dcatoolkit-0.3.0/tests/test_pairs.py +195 -0
- dcatoolkit-0.3.0/tests/test_structure.py +37 -0
- dcatoolkit-0.2.2/README.md +0 -17
- dcatoolkit-0.2.2/pyproject.toml +0 -54
- dcatoolkit-0.2.2/src/dcatoolkit/__init__.py +0 -6
- dcatoolkit-0.2.2/src/dcatoolkit/analytics.py +0 -161
- dcatoolkit-0.2.2/src/dcatoolkit/representation.py +0 -1271
- dcatoolkit-0.2.2/src/dcatoolkit.egg-info/SOURCES.txt +0 -12
- {dcatoolkit-0.2.2 → dcatoolkit-0.3.0}/src/dcatoolkit.egg-info/dependency_links.txt +0 -0
- {dcatoolkit-0.2.2 → dcatoolkit-0.3.0}/src/dcatoolkit.egg-info/top_level.txt +0 -0
|
@@ -1,21 +1,21 @@
|
|
|
1
|
-
MIT License
|
|
2
|
-
|
|
3
|
-
Copyright (c) 2024 Raheel Syed Ahmed
|
|
4
|
-
|
|
5
|
-
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
-
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
-
in the Software without restriction, including without limitation the rights
|
|
8
|
-
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
-
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
-
furnished to do so, subject to the following conditions:
|
|
11
|
-
|
|
12
|
-
The above copyright notice and this permission notice shall be included in all
|
|
13
|
-
copies or substantial portions of the Software.
|
|
14
|
-
|
|
15
|
-
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
-
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
-
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
-
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
-
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
-
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
-
SOFTWARE.
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2024 Raheel Syed Ahmed
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -1,72 +1,122 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
2
|
-
Name: dcatoolkit
|
|
3
|
-
Version: 0.
|
|
4
|
-
Summary: Collection of useful modules and representations for managing DCA output data.
|
|
5
|
-
Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
6
|
-
Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
7
|
-
License: MIT License
|
|
8
|
-
|
|
9
|
-
Copyright (c) 2024 Raheel Syed Ahmed
|
|
10
|
-
|
|
11
|
-
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
12
|
-
of this software and associated documentation files (the "Software"), to deal
|
|
13
|
-
in the Software without restriction, including without limitation the rights
|
|
14
|
-
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
15
|
-
copies of the Software, and to permit persons to whom the Software is
|
|
16
|
-
furnished to do so, subject to the following conditions:
|
|
17
|
-
|
|
18
|
-
The above copyright notice and this permission notice shall be included in all
|
|
19
|
-
copies or substantial portions of the Software.
|
|
20
|
-
|
|
21
|
-
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
22
|
-
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
23
|
-
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
24
|
-
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
25
|
-
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
26
|
-
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
27
|
-
SOFTWARE.
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
Classifier:
|
|
34
|
-
Classifier:
|
|
35
|
-
Classifier:
|
|
36
|
-
Classifier:
|
|
37
|
-
Classifier: Programming Language :: Python :: 3
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
Requires-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
Requires-Dist:
|
|
45
|
-
Requires-Dist:
|
|
46
|
-
Requires-Dist:
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
Requires-Dist:
|
|
52
|
-
Requires-Dist:
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: dcatoolkit
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Collection of useful modules and representations for managing DCA output data.
|
|
5
|
+
Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
6
|
+
Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
7
|
+
License: MIT License
|
|
8
|
+
|
|
9
|
+
Copyright (c) 2024 Raheel Syed Ahmed
|
|
10
|
+
|
|
11
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
12
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
13
|
+
in the Software without restriction, including without limitation the rights
|
|
14
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
15
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
16
|
+
furnished to do so, subject to the following conditions:
|
|
17
|
+
|
|
18
|
+
The above copyright notice and this permission notice shall be included in all
|
|
19
|
+
copies or substantial portions of the Software.
|
|
20
|
+
|
|
21
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
22
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
23
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
24
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
25
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
26
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
27
|
+
SOFTWARE.
|
|
28
|
+
|
|
29
|
+
Project-URL: Homepage, https://github.com/RaheelSyedAhmed/dcatoolkit
|
|
30
|
+
Project-URL: Changelog, https://github.com/RaheelSyedAhmed/dcatoolkit/blob/main/CHANGELOG.md
|
|
31
|
+
Project-URL: Issues, https://github.com/RaheelSyedAhmed/dcatoolkit/issues
|
|
32
|
+
Keywords: dca,toolkit,DI,coevolution
|
|
33
|
+
Classifier: Development Status :: 4 - Beta
|
|
34
|
+
Classifier: Intended Audience :: Science/Research
|
|
35
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
36
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
37
|
+
Classifier: Programming Language :: Python :: 3
|
|
38
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
39
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
40
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
41
|
+
Requires-Python: >=3.10
|
|
42
|
+
Description-Content-Type: text/markdown
|
|
43
|
+
License-File: LICENSE
|
|
44
|
+
Requires-Dist: biotite>=1.0.1
|
|
45
|
+
Requires-Dist: numpy>=1.26.0
|
|
46
|
+
Requires-Dist: pandas>=2.1.0
|
|
47
|
+
Requires-Dist: scipy>=1.11.0
|
|
48
|
+
Provides-Extra: tests
|
|
49
|
+
Requires-Dist: pytest; extra == "tests"
|
|
50
|
+
Provides-Extra: docs
|
|
51
|
+
Requires-Dist: sphinx; extra == "docs"
|
|
52
|
+
Requires-Dist: pdoc; extra == "docs"
|
|
53
|
+
Requires-Dist: numpydoc; extra == "docs"
|
|
54
|
+
Provides-Extra: lint
|
|
55
|
+
Requires-Dist: ruff; extra == "lint"
|
|
56
|
+
Provides-Extra: plot
|
|
57
|
+
Requires-Dist: matplotlib>=3.10.9; extra == "plot"
|
|
58
|
+
Dynamic: license-file
|
|
59
|
+
|
|
60
|
+
# dcatoolkit
|
|
61
|
+
Collection of useful modules and representations for managing DCA output data.
|
|
62
|
+
|
|
63
|
+
## Installation
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
pip install dcatoolkit
|
|
67
|
+
|
|
68
|
+
# optional: adds matplotlib for the plotting example"
|
|
69
|
+
pip install "dcatoolkit[plot]"
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
Requires Python 3.10+.
|
|
73
|
+
Upgrading from 0.2.x? See the [changelog](https://github.com/RaheelSyedAhmed/dcatoolkit/blob/main/CHANGELOG.md).
|
|
74
|
+
|
|
75
|
+
## Major Sections
|
|
76
|
+
### Representations
|
|
77
|
+
* Use Pairs to load lists, tuples, sets, and ndarrays with the correct orientation of elements. This will allow you to store integer pairs, in the form of structured arrays with fields `residue1` and `residue2` that can be mirrored (where y becomes x and vice versa) and subset.
|
|
78
|
+
* Use DirectInformationData to create structured ndarrays with `residue1`, `residue2`, and `DI` fields that can be sorted by `DI`, mapped to a protein with a ResidueAlignment, and used to generate output for other programs (including UCSF Chimera)
|
|
79
|
+
* Use ResidueAlignment to generate a reference map. Indices of one sequence of characters can be linked to their corresponding indices of the other sequence of characters. The dictionaries produced, domain-to-protein and protein-to-domain, allow for forward mapping and backmapping.
|
|
80
|
+
* Use StructureInformation to find contacts in a protein structure and find atomic information related to specific pairs of interest. It can read in PDBx/mmCIF and PDB files or fetch them from RCSB and find contacts between residues.
|
|
81
|
+
### Analytics
|
|
82
|
+
* Use MSATools to load in Multiple Sequence Alignment (MSA) data and provide functionality including generating frequency statistics on "gappiness" in the MSA and filtering and cleaning MSAs.
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
## Quick Start
|
|
86
|
+
```python
|
|
87
|
+
from dcatoolkit import DirectInformationData, ResidueAlignment, StructureInformation
|
|
88
|
+
|
|
89
|
+
# Rank DI pairs and map them from MSA (domain) numbering onto the protein sequence.
|
|
90
|
+
di = DirectInformationData.load_from_DI_file("my_dca_output.DI")
|
|
91
|
+
alignment = ResidueAlignment.load_from_align_file("my_domain.align")
|
|
92
|
+
top_pairs = di.get_ranked_mapped_pairs(alignment, alignment, number=50)
|
|
93
|
+
|
|
94
|
+
# Compare the top pairs with residue contacts in a structure.
|
|
95
|
+
structure = StructureInformation.fetch_pdb("2KLL")
|
|
96
|
+
contacts = structure.get_contacts(ca_only=False, threshold=8, chain1="A", chain2="A")
|
|
97
|
+
hits = [pair for pair in top_pairs.tolist() if pair in contacts]
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
## Diagram of Hidden Markov Model & Direct Coupling Analysis Pipeline
|
|
101
|
+
<p align="center">
|
|
102
|
+
<img src="https://github.com/user-attachments/assets/4768e08f-d513-4dbf-abc5-c80c1b3d42aa"/>
|
|
103
|
+
</p>
|
|
104
|
+
|
|
105
|
+
## Development
|
|
106
|
+
|
|
107
|
+
This project uses [uv](https://docs.astral.sh/uv/) for dependency management.
|
|
108
|
+
|
|
109
|
+
```bash
|
|
110
|
+
git clone https://github.com/RaheelSyedAhmed/dcatoolkit.git
|
|
111
|
+
cd dcatoolkit
|
|
112
|
+
uv sync --all-extras
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
Run the test suite (requires internet to fetch from RCSB):
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
uv run pytest
|
|
119
|
+
|
|
120
|
+
uv sync --all-extras # Installs packages from tests, docs, lint, and plot.
|
|
121
|
+
uv run ruff check # linting
|
|
122
|
+
```
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
# dcatoolkit
|
|
2
|
+
Collection of useful modules and representations for managing DCA output data.
|
|
3
|
+
|
|
4
|
+
## Installation
|
|
5
|
+
|
|
6
|
+
```bash
|
|
7
|
+
pip install dcatoolkit
|
|
8
|
+
|
|
9
|
+
# optional: adds matplotlib for the plotting example"
|
|
10
|
+
pip install "dcatoolkit[plot]"
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
Requires Python 3.10+.
|
|
14
|
+
Upgrading from 0.2.x? See the [changelog](https://github.com/RaheelSyedAhmed/dcatoolkit/blob/main/CHANGELOG.md).
|
|
15
|
+
|
|
16
|
+
## Major Sections
|
|
17
|
+
### Representations
|
|
18
|
+
* Use Pairs to load lists, tuples, sets, and ndarrays with the correct orientation of elements. This will allow you to store integer pairs, in the form of structured arrays with fields `residue1` and `residue2` that can be mirrored (where y becomes x and vice versa) and subset.
|
|
19
|
+
* Use DirectInformationData to create structured ndarrays with `residue1`, `residue2`, and `DI` fields that can be sorted by `DI`, mapped to a protein with a ResidueAlignment, and used to generate output for other programs (including UCSF Chimera)
|
|
20
|
+
* Use ResidueAlignment to generate a reference map. Indices of one sequence of characters can be linked to their corresponding indices of the other sequence of characters. The dictionaries produced, domain-to-protein and protein-to-domain, allow for forward mapping and backmapping.
|
|
21
|
+
* Use StructureInformation to find contacts in a protein structure and find atomic information related to specific pairs of interest. It can read in PDBx/mmCIF and PDB files or fetch them from RCSB and find contacts between residues.
|
|
22
|
+
### Analytics
|
|
23
|
+
* Use MSATools to load in Multiple Sequence Alignment (MSA) data and provide functionality including generating frequency statistics on "gappiness" in the MSA and filtering and cleaning MSAs.
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
## Quick Start
|
|
27
|
+
```python
|
|
28
|
+
from dcatoolkit import DirectInformationData, ResidueAlignment, StructureInformation
|
|
29
|
+
|
|
30
|
+
# Rank DI pairs and map them from MSA (domain) numbering onto the protein sequence.
|
|
31
|
+
di = DirectInformationData.load_from_DI_file("my_dca_output.DI")
|
|
32
|
+
alignment = ResidueAlignment.load_from_align_file("my_domain.align")
|
|
33
|
+
top_pairs = di.get_ranked_mapped_pairs(alignment, alignment, number=50)
|
|
34
|
+
|
|
35
|
+
# Compare the top pairs with residue contacts in a structure.
|
|
36
|
+
structure = StructureInformation.fetch_pdb("2KLL")
|
|
37
|
+
contacts = structure.get_contacts(ca_only=False, threshold=8, chain1="A", chain2="A")
|
|
38
|
+
hits = [pair for pair in top_pairs.tolist() if pair in contacts]
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## Diagram of Hidden Markov Model & Direct Coupling Analysis Pipeline
|
|
42
|
+
<p align="center">
|
|
43
|
+
<img src="https://github.com/user-attachments/assets/4768e08f-d513-4dbf-abc5-c80c1b3d42aa"/>
|
|
44
|
+
</p>
|
|
45
|
+
|
|
46
|
+
## Development
|
|
47
|
+
|
|
48
|
+
This project uses [uv](https://docs.astral.sh/uv/) for dependency management.
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
git clone https://github.com/RaheelSyedAhmed/dcatoolkit.git
|
|
52
|
+
cd dcatoolkit
|
|
53
|
+
uv sync --all-extras
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
Run the test suite (requires internet to fetch from RCSB):
|
|
57
|
+
|
|
58
|
+
```bash
|
|
59
|
+
uv run pytest
|
|
60
|
+
|
|
61
|
+
uv sync --all-extras # Installs packages from tests, docs, lint, and plot.
|
|
62
|
+
uv run ruff check # linting
|
|
63
|
+
```
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools >= 61.0"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "dcatoolkit"
|
|
7
|
+
version = "0.3.0"
|
|
8
|
+
description = "Collection of useful modules and representations for managing DCA output data."
|
|
9
|
+
keywords = ["dca", "toolkit", "DI", "coevolution"]
|
|
10
|
+
|
|
11
|
+
readme = "README.md"
|
|
12
|
+
license = {file = "LICENSE"}
|
|
13
|
+
|
|
14
|
+
requires-python = ">=3.10"
|
|
15
|
+
|
|
16
|
+
authors = [
|
|
17
|
+
{name = "Raheel Syed Ahmed", email = "raheelsyedahmed@gmail.com"}
|
|
18
|
+
]
|
|
19
|
+
maintainers = [
|
|
20
|
+
{name = "Raheel Syed Ahmed", email = "raheelsyedahmed@gmail.com"}
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
dependencies = [
|
|
24
|
+
"biotite>=1.0.1",
|
|
25
|
+
"numpy>=1.26.0",
|
|
26
|
+
"pandas>=2.1.0",
|
|
27
|
+
"scipy>=1.11.0",
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
classifiers = [
|
|
31
|
+
"Development Status :: 4 - Beta",
|
|
32
|
+
"Intended Audience :: Science/Research",
|
|
33
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
34
|
+
"License :: OSI Approved :: MIT License",
|
|
35
|
+
"Programming Language :: Python :: 3",
|
|
36
|
+
"Programming Language :: Python :: 3.10",
|
|
37
|
+
"Programming Language :: Python :: 3.11",
|
|
38
|
+
"Programming Language :: Python :: 3.12",
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
[project.urls]
|
|
42
|
+
Homepage = "https://github.com/RaheelSyedAhmed/dcatoolkit"
|
|
43
|
+
Changelog = "https://github.com/RaheelSyedAhmed/dcatoolkit/blob/main/CHANGELOG.md"
|
|
44
|
+
Issues = "https://github.com/RaheelSyedAhmed/dcatoolkit/issues"
|
|
45
|
+
|
|
46
|
+
[project.optional-dependencies]
|
|
47
|
+
tests = [
|
|
48
|
+
"pytest",
|
|
49
|
+
]
|
|
50
|
+
docs = [
|
|
51
|
+
"sphinx",
|
|
52
|
+
"pdoc",
|
|
53
|
+
"numpydoc"
|
|
54
|
+
]
|
|
55
|
+
lint = [
|
|
56
|
+
"ruff",
|
|
57
|
+
]
|
|
58
|
+
plot = [
|
|
59
|
+
"matplotlib>=3.10.9",
|
|
60
|
+
]
|
|
61
|
+
[tool.ruff]
|
|
62
|
+
# The target Python version is inferred from requires-python above.
|
|
63
|
+
# Lint the package and its tests; examples are standalone scripts.
|
|
64
|
+
extend-exclude = [
|
|
65
|
+
"examples",
|
|
66
|
+
# Python 2 scripts that generated the reference contact files in tests/pdb_info; not valid Python 3.
|
|
67
|
+
"tests/pdb_info/interface_contacts_allatom_args.py",
|
|
68
|
+
"tests/pdb_info/interface_contacts_calpha_args.py",
|
|
69
|
+
]
|
|
70
|
+
|
|
71
|
+
[tool.ruff.lint]
|
|
72
|
+
# Explicit rule selection, so results don't change with ruff's defaults between versions.
|
|
73
|
+
select = [
|
|
74
|
+
"E4", "E7", "E9", # pycodestyle errors: imports, statements, syntax
|
|
75
|
+
"F", # Pyflakes: unused imports/variables, undefined names
|
|
76
|
+
"B", # flake8-bugbear: likely bugs
|
|
77
|
+
"I", # isort: import ordering
|
|
78
|
+
"UP", # pyupgrade: modern syntax for the target Python version
|
|
79
|
+
"C4", # flake8-comprehensions
|
|
80
|
+
"SIM", # flake8-simplify
|
|
81
|
+
"PIE", # flake8-pie: unnecessary code
|
|
82
|
+
"RUF", # Ruff-specific rules
|
|
83
|
+
]
|
|
84
|
+
ignore = [
|
|
85
|
+
"B905", # zip() without strict=: most zips are equal-length by construction, and ResidueAlignment deliberately truncates mismatched align texts.
|
|
86
|
+
]
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
[egg_info]
|
|
2
|
-
tag_build =
|
|
3
|
-
tag_date = 0
|
|
4
|
-
|
|
1
|
+
[egg_info]
|
|
2
|
+
tag_build =
|
|
3
|
+
tag_date = 0
|
|
4
|
+
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
|
|
2
|
+
from importlib.metadata import version as _version
|
|
3
|
+
|
|
4
|
+
__version__ = _version("dcatoolkit")
|
|
5
|
+
from .analytics import MSATools
|
|
6
|
+
from .representation import (
|
|
7
|
+
DirectInformationData,
|
|
8
|
+
MMCIFInformation,
|
|
9
|
+
Pairs,
|
|
10
|
+
PDBInformation,
|
|
11
|
+
ResidueAlignment,
|
|
12
|
+
StructureInformation,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
__all__ = ['DirectInformationData', 'MMCIFInformation', 'MSATools', 'PDBInformation', 'Pairs', 'ResidueAlignment', 'StructureInformation']
|
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
import io
|
|
2
|
+
import re
|
|
3
|
+
import string
|
|
4
|
+
from collections import Counter
|
|
5
|
+
from collections.abc import Callable
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Literal
|
|
8
|
+
|
|
9
|
+
import numpy as np
|
|
10
|
+
import numpy.typing as npt
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class MSATools:
|
|
14
|
+
"""
|
|
15
|
+
Tools and interface for encapsulating MSA data and providing functionality for filtering and analysis.
|
|
16
|
+
|
|
17
|
+
Parameters
|
|
18
|
+
----------
|
|
19
|
+
MSA : list of tuple of (str, str)
|
|
20
|
+
Loaded MSA that is a list of tuples where the first element is the header (including its leading ``">"``) and the second element is its corresponding sequence.
|
|
21
|
+
|
|
22
|
+
Attributes
|
|
23
|
+
----------
|
|
24
|
+
MSA : list of tuple of (str, str)
|
|
25
|
+
The `MSA` supplied.
|
|
26
|
+
"""
|
|
27
|
+
_GAP_CLEANUP_TABLE = str.maketrans('', '', string.ascii_lowercase + ".")
|
|
28
|
+
|
|
29
|
+
def __init__(self, MSA: list[tuple[str, str]]):
|
|
30
|
+
self.MSA = MSA
|
|
31
|
+
|
|
32
|
+
@staticmethod
|
|
33
|
+
def load_from_file(msa_source: str | io.IOBase | Path) -> 'MSATools':
|
|
34
|
+
"""
|
|
35
|
+
Generates MSATools object from an MSA file in aligned FASTA (``.afa``) format.
|
|
36
|
+
|
|
37
|
+
Parameters
|
|
38
|
+
----------
|
|
39
|
+
msa_source : str or pathlib.Path or io.BytesIO or io.TextIOBase
|
|
40
|
+
Filepath, or in-memory bytes or text stream, of the MSA in ``.afa`` format. Streams are read from the beginning.
|
|
41
|
+
|
|
42
|
+
Returns
|
|
43
|
+
-------
|
|
44
|
+
MSATools
|
|
45
|
+
An MSATools instance with the appropriate list of ``(header, sequence)`` tuples where sequences are simplified and converted to single line format.
|
|
46
|
+
|
|
47
|
+
Raises
|
|
48
|
+
------
|
|
49
|
+
TypeError
|
|
50
|
+
If `msa_source` is not a filepath, ``io.BytesIO``, or ``io.TextIOBase``.
|
|
51
|
+
|
|
52
|
+
Notes
|
|
53
|
+
-----
|
|
54
|
+
Windows (``\\r\\n``) and old Mac (``\\r``) line endings are normalized to ``\\n`` before parsing. Headers keep their leading ``">"``.
|
|
55
|
+
"""
|
|
56
|
+
data = ""
|
|
57
|
+
msa_entries: list[tuple[str, str]] = []
|
|
58
|
+
if isinstance(msa_source, (str, Path)):
|
|
59
|
+
with open(msa_source) as fs:
|
|
60
|
+
data = fs.read()
|
|
61
|
+
elif isinstance(msa_source, io.BytesIO):
|
|
62
|
+
data = msa_source.getvalue().decode()
|
|
63
|
+
elif isinstance(msa_source, io.TextIOBase):
|
|
64
|
+
msa_source.seek(0)
|
|
65
|
+
data = msa_source.read()
|
|
66
|
+
else:
|
|
67
|
+
raise TypeError("msa_file is not bytesIO, a TextIO, or a filepath.")
|
|
68
|
+
|
|
69
|
+
data = data.replace("\r\n", "\n").replace("\r", "\n")
|
|
70
|
+
split_data = data.split(">")[1:]
|
|
71
|
+
for entry in split_data:
|
|
72
|
+
header, _, rest = entry.partition("\n")
|
|
73
|
+
sequence = rest.replace("\n", "")
|
|
74
|
+
msa_entries.append((">"+header, sequence))
|
|
75
|
+
return MSATools(msa_entries)
|
|
76
|
+
|
|
77
|
+
@staticmethod
|
|
78
|
+
def get_sequence_max_cont_gaps(sequence: str) -> int:
|
|
79
|
+
"""
|
|
80
|
+
Find maximum number of continuous gaps in a specific sequence.
|
|
81
|
+
|
|
82
|
+
Parameters
|
|
83
|
+
----------
|
|
84
|
+
sequence : str
|
|
85
|
+
Sequence of characters, potentially containing multiple of ``"-"``, a gap character.
|
|
86
|
+
|
|
87
|
+
Returns
|
|
88
|
+
-------
|
|
89
|
+
int
|
|
90
|
+
The maximum number of continuous gaps (``"-"``) in a sequence, or 0 if there are none. Other characters, such as ``"."``, break a run of gaps.
|
|
91
|
+
"""
|
|
92
|
+
return max((m.end() - m.start() for m in re.finditer(r"-+", sequence)), default=0)
|
|
93
|
+
|
|
94
|
+
def gap_frequency(self) -> tuple[dict[int, int], dict[int, float]]:
|
|
95
|
+
"""
|
|
96
|
+
Calculates the frequency of maximum continuous gaps throughout the MSA where the key corresponds to the number of continuous gaps and the value corresponds to the number of sequences or the cumulative fraction of sequences.
|
|
97
|
+
|
|
98
|
+
Returns
|
|
99
|
+
-------
|
|
100
|
+
tuple of (dict of {int : int}, dict of {int : float})
|
|
101
|
+
Two element tuple where the first element maps each maximum continuous gap length to the number of sequences with it, and the second element maps it to the cumulative fraction (0 to 1) of sequences with that maximum or fewer.
|
|
102
|
+
|
|
103
|
+
See Also
|
|
104
|
+
--------
|
|
105
|
+
get_sequence_max_cont_gaps : Computes each sequence's maximum continuous gap length.
|
|
106
|
+
filter_by_continuous_gaps : Filters sequences by that length.
|
|
107
|
+
|
|
108
|
+
Notes
|
|
109
|
+
-----
|
|
110
|
+
Gaps are counted on the sequences as stored. To match the counts `filter_by_continuous_gaps()` uses, run this on an already filtered MSA, e.g. ``MSATools(msa.filter_by_continuous_gaps()).gap_frequency()``.
|
|
111
|
+
"""
|
|
112
|
+
max_gap_counts = []
|
|
113
|
+
for _header, sequence in self.MSA:
|
|
114
|
+
max_gap_counts.append(MSATools.get_sequence_max_cont_gaps(sequence))
|
|
115
|
+
frequency_count_dict = dict(Counter(max_gap_counts))
|
|
116
|
+
cumul_perc_dict = {}
|
|
117
|
+
cumul_count = 0
|
|
118
|
+
for key in sorted(frequency_count_dict.keys()):
|
|
119
|
+
value = frequency_count_dict[key]
|
|
120
|
+
cumul_count += value
|
|
121
|
+
cumul_perc_dict[key] = cumul_count / len(self.MSA)
|
|
122
|
+
return (frequency_count_dict, cumul_perc_dict)
|
|
123
|
+
|
|
124
|
+
def filter_by_continuous_gaps(self, max_gaps: int | None=None) -> list[tuple[str, str]]:
|
|
125
|
+
"""
|
|
126
|
+
Filter out entries in your MSA by the number of maximum continuous gaps specified unless None is provided. Also, removes ``"."`` characters and lowercase letters (insert positions) from the sequence.
|
|
127
|
+
|
|
128
|
+
Parameters
|
|
129
|
+
----------
|
|
130
|
+
max_gaps : int, optional
|
|
131
|
+
The maximum allowed number of continuous gaps in a sequence. If None, no entries are removed, but sequences are still cleaned.
|
|
132
|
+
|
|
133
|
+
Returns
|
|
134
|
+
-------
|
|
135
|
+
list of tuple of (str, str)
|
|
136
|
+
List of entries that are valid in that their sequences' number of maximum continuous gaps is within the threshold supplied as `max_gaps`. Wrap it in ``MSATools(...)`` to keep working with it as an MSA.
|
|
137
|
+
|
|
138
|
+
Notes
|
|
139
|
+
-----
|
|
140
|
+
Insert characters are removed before gaps are counted, so gap runs that were separated only by ``"."`` or lowercase letters count as one run.
|
|
141
|
+
"""
|
|
142
|
+
kept_entries = []
|
|
143
|
+
for header, sequence in self.MSA:
|
|
144
|
+
sequence = sequence.translate(MSATools._GAP_CLEANUP_TABLE)
|
|
145
|
+
if max_gaps is None or MSATools.get_sequence_max_cont_gaps(sequence) <= max_gaps:
|
|
146
|
+
kept_entries.append((header, sequence))
|
|
147
|
+
return kept_entries
|
|
148
|
+
|
|
149
|
+
def gap_proportion(self, agg_func: Callable[..., float | int]=np.mean, axis: Literal[0, 1] = 0) -> float | int:
|
|
150
|
+
"""
|
|
151
|
+
Evaluates the gap proportion per alignment position (column) or per sequence (row) in the MSA, then aggregates it into one value.
|
|
152
|
+
|
|
153
|
+
Parameters
|
|
154
|
+
----------
|
|
155
|
+
agg_func : callable, default numpy.mean
|
|
156
|
+
The aggregation function applied to get the expected result, usually a mean, max, or min value, of the gap proportions.
|
|
157
|
+
axis : {0, 1}, default 0
|
|
158
|
+
When `axis` is 0, gap proportions are computed per column, over every sequence in that column. When `axis` is 1, they are computed per row, over every column in that sequence.
|
|
159
|
+
|
|
160
|
+
Returns
|
|
161
|
+
-------
|
|
162
|
+
float or int
|
|
163
|
+
A numerical value determined by `agg_func` over the gap proportions.
|
|
164
|
+
|
|
165
|
+
Notes
|
|
166
|
+
-----
|
|
167
|
+
Any character that is not an ASCII letter (e.g. ``"-"`` or ``"."``) counts as a gap. All sequences must have the same length, as in an unfiltered or a filtered MSA.
|
|
168
|
+
"""
|
|
169
|
+
num_rows = len(self.MSA)
|
|
170
|
+
num_cols = len(self.MSA[0][1])
|
|
171
|
+
flat = np.frombuffer("".join(seq for _, seq in self.MSA).encode("ascii"), dtype=np.uint8).reshape(num_rows, num_cols)
|
|
172
|
+
is_alpha = ((flat >= 65) & (flat <= 90)) | ((flat >= 97) & (flat <= 122))
|
|
173
|
+
non_alpha_counts = np.sum(~is_alpha, axis) / flat.shape[axis]
|
|
174
|
+
return agg_func(non_alpha_counts)
|
|
175
|
+
|
|
176
|
+
def write(self, destination: str | Path | io.TextIOBase) -> None:
|
|
177
|
+
"""
|
|
178
|
+
Writes this MSA's headers and sequences to the destination specified, each header followed by its sequence on a single line.
|
|
179
|
+
|
|
180
|
+
Parameters
|
|
181
|
+
----------
|
|
182
|
+
destination : str or pathlib.Path or io.TextIOBase
|
|
183
|
+
Filepath or writable text stream to write the MSA to.
|
|
184
|
+
|
|
185
|
+
Raises
|
|
186
|
+
------
|
|
187
|
+
TypeError
|
|
188
|
+
If `destination` is not a filepath or a writable ``io.TextIOBase``.
|
|
189
|
+
"""
|
|
190
|
+
lines = (f"{header}\n{sequence}\n" for header, sequence in self.MSA)
|
|
191
|
+
if isinstance(destination, (str, Path)):
|
|
192
|
+
with open(destination, 'w') as fs:
|
|
193
|
+
fs.writelines(lines)
|
|
194
|
+
elif isinstance(destination, io.TextIOBase) and destination.writable():
|
|
195
|
+
destination.writelines(lines)
|
|
196
|
+
else:
|
|
197
|
+
raise TypeError(f"Destination supplied is either not a filepath or is not a writeable TextIO object (got {type(destination).__name__}).")
|
|
198
|
+
|
|
199
|
+
def as_matrix(self) -> npt.NDArray:
|
|
200
|
+
"""
|
|
201
|
+
Represents the MSA as a numpy matrix of sequences.
|
|
202
|
+
|
|
203
|
+
Returns
|
|
204
|
+
-------
|
|
205
|
+
numpy.ndarray
|
|
206
|
+
A ``<U1`` (one-character string) matrix with one row per sequence and one column per alignment position. Each cell is the sequence character for that sequence at that position.
|
|
207
|
+
"""
|
|
208
|
+
return np.array([list(seq) for _, seq in self.MSA])
|
|
209
|
+
|
|
210
|
+
def __str__(self) -> str:
|
|
211
|
+
"""
|
|
212
|
+
Returns the sequences present in the loaded MSA in string format.
|
|
213
|
+
|
|
214
|
+
Returns
|
|
215
|
+
-------
|
|
216
|
+
str
|
|
217
|
+
Each sequence in the instance separated with newline characters.
|
|
218
|
+
"""
|
|
219
|
+
return "\n".join([seq for _, seq in self.MSA])
|
|
220
|
+
|
|
221
|
+
def __len__(self):
|
|
222
|
+
"""
|
|
223
|
+
Returns the number of sequences, and equivalently, the number of headers in the MSA.
|
|
224
|
+
|
|
225
|
+
Returns
|
|
226
|
+
-------
|
|
227
|
+
int
|
|
228
|
+
Length of the MSA list of ``(header, sequence)`` tuples.
|
|
229
|
+
"""
|
|
230
|
+
return len(self.MSA)
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
from .alignment import ResidueAlignment
|
|
2
|
+
from .di import DirectInformationData
|
|
3
|
+
from .pairs import Pairs
|
|
4
|
+
from .structure import MMCIFInformation, PDBInformation, StructureInformation
|
|
5
|
+
|
|
6
|
+
__all__ = ['DirectInformationData', 'MMCIFInformation', 'PDBInformation', 'Pairs', 'ResidueAlignment', 'StructureInformation']
|