dcatoolkit 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dcatoolkit-0.1.0/LICENSE +21 -0
- dcatoolkit-0.1.0/PKG-INFO +57 -0
- dcatoolkit-0.1.0/README.md +2 -0
- dcatoolkit-0.1.0/pyproject.toml +54 -0
- dcatoolkit-0.1.0/setup.cfg +4 -0
- dcatoolkit-0.1.0/src/dcatoolkit/__init__.py +3 -0
- dcatoolkit-0.1.0/src/dcatoolkit/representation.py +741 -0
- dcatoolkit-0.1.0/src/dcatoolkit.egg-info/PKG-INFO +57 -0
- dcatoolkit-0.1.0/src/dcatoolkit.egg-info/SOURCES.txt +10 -0
- dcatoolkit-0.1.0/src/dcatoolkit.egg-info/dependency_links.txt +1 -0
- dcatoolkit-0.1.0/src/dcatoolkit.egg-info/requires.txt +17 -0
- dcatoolkit-0.1.0/src/dcatoolkit.egg-info/top_level.txt +1 -0
dcatoolkit-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2024 Raheel Syed Ahmed
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: dcatoolkit
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Collection of useful modules and representations for managing DCA output data.
|
|
5
|
+
Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
6
|
+
Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
7
|
+
License: MIT License
|
|
8
|
+
|
|
9
|
+
Copyright (c) 2024 Raheel Syed Ahmed
|
|
10
|
+
|
|
11
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
12
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
13
|
+
in the Software without restriction, including without limitation the rights
|
|
14
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
15
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
16
|
+
furnished to do so, subject to the following conditions:
|
|
17
|
+
|
|
18
|
+
The above copyright notice and this permission notice shall be included in all
|
|
19
|
+
copies or substantial portions of the Software.
|
|
20
|
+
|
|
21
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
22
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
23
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
24
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
25
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
26
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
27
|
+
SOFTWARE.
|
|
28
|
+
|
|
29
|
+
Keywords: dca,toolkit,DI,coevolution
|
|
30
|
+
Classifier: Development Status :: 4 - Beta
|
|
31
|
+
Classifier: Intended Audience :: Science/Research
|
|
32
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
33
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
34
|
+
Classifier: Programming Language :: Python :: 3
|
|
35
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
36
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
37
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
38
|
+
Requires-Python: >=3.10
|
|
39
|
+
Description-Content-Type: text/markdown
|
|
40
|
+
License-File: LICENSE
|
|
41
|
+
Requires-Dist: biotite
|
|
42
|
+
Requires-Dist: matplotlib>=3.8.0
|
|
43
|
+
Requires-Dist: numpy>=1.26.0
|
|
44
|
+
Requires-Dist: pandas>=2.1.0
|
|
45
|
+
Requires-Dist: pyhmmer>=0.10.14
|
|
46
|
+
Requires-Dist: scikit-learn>=1.3
|
|
47
|
+
Requires-Dist: scipy>=1.11.0
|
|
48
|
+
Provides-Extra: tests
|
|
49
|
+
Requires-Dist: pytest; extra == "tests"
|
|
50
|
+
Provides-Extra: docs
|
|
51
|
+
Requires-Dist: sphinx; extra == "docs"
|
|
52
|
+
Requires-Dist: numpydoc; extra == "docs"
|
|
53
|
+
Provides-Extra: lint
|
|
54
|
+
Requires-Dist: ruffle; extra == "lint"
|
|
55
|
+
|
|
56
|
+
# dcatoolkit
|
|
57
|
+
Collection of useful modules and representations for managing DCA output data.
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools >= 61.0"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "dcatoolkit"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Collection of useful modules and representations for managing DCA output data."
|
|
9
|
+
keywords = ["dca", "toolkit", "DI", "coevolution"]
|
|
10
|
+
|
|
11
|
+
readme = "README.md"
|
|
12
|
+
license = {file = "LICENSE"}
|
|
13
|
+
|
|
14
|
+
requires-python = ">=3.10"
|
|
15
|
+
|
|
16
|
+
authors = [
|
|
17
|
+
{name = "Raheel Syed Ahmed", email = "raheelsyedahmed@gmail.com"}
|
|
18
|
+
]
|
|
19
|
+
maintainers = [
|
|
20
|
+
{name = "Raheel Syed Ahmed", email = "raheelsyedahmed@gmail.com"}
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
dependencies = [
|
|
24
|
+
"biotite",
|
|
25
|
+
"matplotlib>=3.8.0",
|
|
26
|
+
"numpy>=1.26.0",
|
|
27
|
+
"pandas>=2.1.0",
|
|
28
|
+
"pyhmmer>=0.10.14",
|
|
29
|
+
"scikit-learn>=1.3",
|
|
30
|
+
"scipy>=1.11.0",
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
classifiers = [
|
|
34
|
+
"Development Status :: 4 - Beta",
|
|
35
|
+
"Intended Audience :: Science/Research",
|
|
36
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
37
|
+
"License :: OSI Approved :: MIT License",
|
|
38
|
+
"Programming Language :: Python :: 3",
|
|
39
|
+
"Programming Language :: Python :: 3.10",
|
|
40
|
+
"Programming Language :: Python :: 3.11",
|
|
41
|
+
"Programming Language :: Python :: 3.12",
|
|
42
|
+
]
|
|
43
|
+
|
|
44
|
+
[project.optional-dependencies]
|
|
45
|
+
tests = [
|
|
46
|
+
"pytest",
|
|
47
|
+
]
|
|
48
|
+
docs = [
|
|
49
|
+
"sphinx",
|
|
50
|
+
"numpydoc"
|
|
51
|
+
]
|
|
52
|
+
lint = [
|
|
53
|
+
"ruffle",
|
|
54
|
+
]
|
|
@@ -0,0 +1,741 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
import pandas as pd
|
|
3
|
+
from scipy.spatial.distance import cdist
|
|
4
|
+
|
|
5
|
+
import biotite.structure.io.pdbx as pdbx
|
|
6
|
+
import biotite.database.rcsb as rcsb
|
|
7
|
+
|
|
8
|
+
from typing import Optional
|
|
9
|
+
import numpy.typing as npt
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class Pairs:
|
|
13
|
+
"""
|
|
14
|
+
Object that contains a representation (as an ndarray) of pairs of entities that are related. This may extend to Direct Information Pairs or Structural contacts, where each residue is one component of the pair.
|
|
15
|
+
|
|
16
|
+
Note
|
|
17
|
+
----
|
|
18
|
+
Either a filepath or a ndarr has to be specified in order to produce a Pairs representation.
|
|
19
|
+
|
|
20
|
+
Parameters
|
|
21
|
+
----------
|
|
22
|
+
filepath : str, optional
|
|
23
|
+
Filepath of the pairs in tabular representation, separated by whitespace between the pair components and newlines between each pair.
|
|
24
|
+
ndarr : numpy.ndarray, optional
|
|
25
|
+
Populated ndarray that contains pair information.
|
|
26
|
+
delimiter : str, optional
|
|
27
|
+
String used to specify separator between two pairs. See numpy.loadtxt() for details.
|
|
28
|
+
|
|
29
|
+
Attributes
|
|
30
|
+
----------
|
|
31
|
+
pairs : numpy.ndarray
|
|
32
|
+
Ndarray representation of pairs supplied by the user. This is produced via the np.loadtxt() function.
|
|
33
|
+
"""
|
|
34
|
+
def __init__(self, filepath: Optional[str]=None, ndarr: Optional[npt.NDArray]=None, delimiter: Optional[str]=None) -> None:
|
|
35
|
+
if (filepath is not None and ndarr is not None) or (filepath is None and ndarr is None):
|
|
36
|
+
raise Exception("Please specify either a filepath or a NumPy array to populate your pairs.")
|
|
37
|
+
elif filepath is not None:
|
|
38
|
+
if delimiter:
|
|
39
|
+
self.pairs = np.loadtxt(filepath, dtype=int, delimiter=delimiter)
|
|
40
|
+
else:
|
|
41
|
+
self.pairs = np.loadtxt(filepath, dtype=int)
|
|
42
|
+
elif ndarr is not None:
|
|
43
|
+
self.pairs = ndarr
|
|
44
|
+
|
|
45
|
+
@staticmethod
|
|
46
|
+
def mirror_diagonal(pairs: npt.NDArray) -> npt.NDArray:
|
|
47
|
+
"""
|
|
48
|
+
Flip 2D ndarray with 2 columns columnwise. Flips pair positions for diagonal-mirrored representation.
|
|
49
|
+
|
|
50
|
+
Parameters
|
|
51
|
+
----------
|
|
52
|
+
pairs : numpy.ndarray
|
|
53
|
+
2d array with (n, 2) shape.
|
|
54
|
+
|
|
55
|
+
Returns
|
|
56
|
+
-------
|
|
57
|
+
numpy.ndarray
|
|
58
|
+
Values flipped along the column axis.
|
|
59
|
+
"""
|
|
60
|
+
return np.flip(pairs, axis=1)
|
|
61
|
+
|
|
62
|
+
@staticmethod
|
|
63
|
+
def subset_pairs(pairs: npt.NDArray, number : Optional[int]=None) -> npt.NDArray:
|
|
64
|
+
"""
|
|
65
|
+
Picks out a subset of 'number' pairs if a number is supplied. Otherwise, returns all pairs.
|
|
66
|
+
|
|
67
|
+
Parameters
|
|
68
|
+
----------
|
|
69
|
+
pairs : numpy.ndarray
|
|
70
|
+
Ndarray to select number of rows from.
|
|
71
|
+
number : int, optional
|
|
72
|
+
Specific number of rows of pairs to subset.
|
|
73
|
+
|
|
74
|
+
Returns
|
|
75
|
+
-------
|
|
76
|
+
numpy.ndarray
|
|
77
|
+
Subset of pairs from rows 0 to number.
|
|
78
|
+
pairs : numpy.ndarray
|
|
79
|
+
All pairs specified from the parameters section.
|
|
80
|
+
"""
|
|
81
|
+
if number is not None:
|
|
82
|
+
return pairs[:number, ]
|
|
83
|
+
else:
|
|
84
|
+
return pairs
|
|
85
|
+
|
|
86
|
+
@staticmethod
|
|
87
|
+
def mirror_pairs(pairs: npt.NDArray, mirror: bool=False) -> npt.NDArray:
|
|
88
|
+
"""
|
|
89
|
+
Produces combined array of pairs and potentially their mirrored representation.
|
|
90
|
+
|
|
91
|
+
Parameters
|
|
92
|
+
----------
|
|
93
|
+
pairs : numpy.ndarray
|
|
94
|
+
Ndarray to mirror and vertically append if mirror is set to True.
|
|
95
|
+
mirror : bool
|
|
96
|
+
Whether or not to append mirrored representation of pairs to the original pairs ndarray.
|
|
97
|
+
|
|
98
|
+
Returns
|
|
99
|
+
-------
|
|
100
|
+
mirrored_ndarray : numpy.ndarray
|
|
101
|
+
combined ndarray of pairs and mirrored pairs.
|
|
102
|
+
pairs : numpy.ndarray
|
|
103
|
+
The original pairs specified from the parameters section.
|
|
104
|
+
"""
|
|
105
|
+
if mirror:
|
|
106
|
+
return np.vstack((pairs, Pairs.mirror_diagonal(pairs)))
|
|
107
|
+
else:
|
|
108
|
+
return pairs
|
|
109
|
+
|
|
110
|
+
@staticmethod
|
|
111
|
+
def get_pairs(pairs: npt.NDArray, mirror: bool=False, number: Optional[int]=None) -> npt.NDArray:
|
|
112
|
+
"""
|
|
113
|
+
Returns pairs based on user specification, offering options to produce mirrored representation of pairs and to select a specific number of pairs.
|
|
114
|
+
|
|
115
|
+
Parameters
|
|
116
|
+
----------
|
|
117
|
+
pairs : numpy.ndarray
|
|
118
|
+
ndarray of pairs to select from or to mirror.
|
|
119
|
+
mirror : bool
|
|
120
|
+
Whether or not to append mirrored representation of pairs to the original pairs ndarray.
|
|
121
|
+
number : int
|
|
122
|
+
Specific number of rows of pairs to subset.
|
|
123
|
+
|
|
124
|
+
Returns
|
|
125
|
+
-------
|
|
126
|
+
numpy.ndarray
|
|
127
|
+
mirrored, subset version of pairs produced via subset_pairs() and mirror_pairs() on pairs.
|
|
128
|
+
"""
|
|
129
|
+
# Check to see if user requested mirrored pairs, if so, add in pairs that are mirrored across diagonal
|
|
130
|
+
pairs = Pairs.subset_pairs(pairs, number)
|
|
131
|
+
if mirror:
|
|
132
|
+
pairs = Pairs.mirror_pairs(pairs, mirror)
|
|
133
|
+
return pairs
|
|
134
|
+
|
|
135
|
+
class ResidueAlignment:
|
|
136
|
+
"""
|
|
137
|
+
A representation of a residue alignment, often from a query HMM to a protein structure target sequence.
|
|
138
|
+
|
|
139
|
+
Parameters
|
|
140
|
+
----------
|
|
141
|
+
domain_name : str
|
|
142
|
+
The name of the query HMM.
|
|
143
|
+
protein_name : str
|
|
144
|
+
The name of the target protein sequence.
|
|
145
|
+
domain_start : int
|
|
146
|
+
The starting index of the domain alignment in the query HMM.
|
|
147
|
+
protein_start : int
|
|
148
|
+
The starting index of the domain alignment in the protein target sequence.
|
|
149
|
+
domain_text : str
|
|
150
|
+
The sequence of the domain in the query HMM corresponding to this alignment.
|
|
151
|
+
protein_text : str
|
|
152
|
+
The sequence of the protein target sequence corresponding to this alignment.
|
|
153
|
+
|
|
154
|
+
Attributes
|
|
155
|
+
----------
|
|
156
|
+
reference_mapping : pandas.DataFrame
|
|
157
|
+
The representation of the mapping where a row constitutes a residue pair and its indices in the format: 'domain_index', 'domain_residue', 'protein_residue', 'protein_index'.
|
|
158
|
+
domain_to_protein : dict[int, int]
|
|
159
|
+
A dictionary allowing for mapping from indices corresponding to the query HMM and Multiple Sequence Alignment to the protein target sequence.
|
|
160
|
+
protein_to_domain : dict[int, int]
|
|
161
|
+
A dictionary allowing for mapping from indices corresponding to the protein target sequence to the query HMM and Multiple Sequence Alignment.
|
|
162
|
+
"""
|
|
163
|
+
def __init__(self, domain_name: str, protein_name: str, domain_start: int, protein_start: int, domain_text: str, protein_text: str) -> None:
|
|
164
|
+
self.domain_name = domain_name
|
|
165
|
+
self.protein_name = protein_name
|
|
166
|
+
self.set_reference_mapping(domain_start, protein_start, domain_text, protein_text)
|
|
167
|
+
|
|
168
|
+
def set_reference_mapping(self, domain_start: int, protein_start: int, domain_text: str, protein_text: str) -> None:
|
|
169
|
+
"""
|
|
170
|
+
Set values for reference_mapping and mapping dictionaries, domain_to_protein and protein_to_domain.
|
|
171
|
+
|
|
172
|
+
Note
|
|
173
|
+
----
|
|
174
|
+
For details on `domain_start`, `protein_start`, `domain_text`, `protein_text`, please refer to the `ResidueAlignment` docstring.
|
|
175
|
+
|
|
176
|
+
Returns
|
|
177
|
+
-------
|
|
178
|
+
None
|
|
179
|
+
"""
|
|
180
|
+
# Convert text to list variant for iteration
|
|
181
|
+
domain_sequence = list(domain_text)
|
|
182
|
+
protein_sequence = list(protein_text)
|
|
183
|
+
|
|
184
|
+
mapping_entries = []
|
|
185
|
+
|
|
186
|
+
for i in range(len(domain_sequence)):
|
|
187
|
+
mapping_entry = []
|
|
188
|
+
# Check to see if domain residue is valid, if so, we can assign the proper index.
|
|
189
|
+
if domain_sequence[i] != '.':
|
|
190
|
+
mapping_entry.append(domain_start)
|
|
191
|
+
domain_start += 1
|
|
192
|
+
else:
|
|
193
|
+
mapping_entry.append(pd.NA)
|
|
194
|
+
# Assign the values of the residues mapped together.
|
|
195
|
+
mapping_entry.append(domain_sequence[i])
|
|
196
|
+
mapping_entry.append(protein_sequence[i])
|
|
197
|
+
# Check to see if protein residue is valid, if so, we can assign the proper index.
|
|
198
|
+
if protein_sequence[i] != '-':
|
|
199
|
+
mapping_entry.append(protein_start)
|
|
200
|
+
protein_start += 1
|
|
201
|
+
else:
|
|
202
|
+
mapping_entry.append(pd.NA)
|
|
203
|
+
# Store the resulting mapping in the reference map.
|
|
204
|
+
mapping_entries.append(mapping_entry)
|
|
205
|
+
self.reference_mapping = pd.DataFrame(mapping_entries, columns=['domain_index', 'domain_residue', 'protein_residue', 'protein_index'])
|
|
206
|
+
self.reference_mapping = self.reference_mapping.astype({'domain_index': pd.Int32Dtype(), 'protein_index': pd.Int32Dtype(), 'domain_residue': pd.StringDtype(), 'protein_residue': pd.StringDtype()})
|
|
207
|
+
reference_mapping_notna = self.reference_mapping.dropna()
|
|
208
|
+
|
|
209
|
+
self.domain_to_protein = dict(zip(reference_mapping_notna.domain_index, reference_mapping_notna.protein_index))
|
|
210
|
+
self.protein_to_domain = dict(zip(reference_mapping_notna.protein_index, reference_mapping_notna.domain_index))
|
|
211
|
+
|
|
212
|
+
@staticmethod
|
|
213
|
+
def load_from_align_file(align_filepath: str) -> 'ResidueAlignment':
|
|
214
|
+
"""
|
|
215
|
+
Generate ResidueAlignment from a standard align file generated from HMM scan.
|
|
216
|
+
|
|
217
|
+
Parameters
|
|
218
|
+
----------
|
|
219
|
+
align_filepath : str
|
|
220
|
+
Filepath of the align file generated from a scan file produced via hmmscan.
|
|
221
|
+
|
|
222
|
+
Returns
|
|
223
|
+
-------
|
|
224
|
+
ResidueAlignment
|
|
225
|
+
ResidueAlignment with domain and protein starting indices and corresponding sequence texts.
|
|
226
|
+
|
|
227
|
+
File Format
|
|
228
|
+
-----------
|
|
229
|
+
Domain_name
|
|
230
|
+
1
|
|
231
|
+
XXXXXXXXXXXXXXXXXXXX
|
|
232
|
+
20
|
|
233
|
+
|
|
234
|
+
Protein_name
|
|
235
|
+
70
|
|
236
|
+
XXXXXXXXXXXXXXXXXXXX
|
|
237
|
+
89
|
|
238
|
+
"""
|
|
239
|
+
# Read the alignment file and parse the important information from each alignment entry.
|
|
240
|
+
alignment_entries = ResidueAlignment.read_align_file(align_filepath)
|
|
241
|
+
hmm_entry, protein_entry = alignment_entries
|
|
242
|
+
domain_name, domain_start, domain_text, _ = hmm_entry
|
|
243
|
+
protein_name, protein_start, protein_text, _ = protein_entry
|
|
244
|
+
|
|
245
|
+
# Convert to ints for iteration
|
|
246
|
+
domain_start = int(domain_start)
|
|
247
|
+
protein_start = int(protein_start)
|
|
248
|
+
|
|
249
|
+
return ResidueAlignment(domain_name, protein_name, domain_start, protein_start, domain_text, protein_text)
|
|
250
|
+
|
|
251
|
+
@staticmethod
|
|
252
|
+
def read_align_file(align_filepath: str) -> list[list[str]]:
|
|
253
|
+
"""
|
|
254
|
+
Reads standard align file, where a scan file is selected for a particular domain and processed into an align file format. Details are present in produce_align_from_scan().
|
|
255
|
+
|
|
256
|
+
Parameters
|
|
257
|
+
----------
|
|
258
|
+
align_filepath : str
|
|
259
|
+
Filepath and filename of alignment file that contains information on the domain / protein of interest and its mapping to a protein's structural sequence
|
|
260
|
+
|
|
261
|
+
Returns
|
|
262
|
+
-------
|
|
263
|
+
alignment_entries : list of list of strings
|
|
264
|
+
list of associated lines (one which corresponds to the HMM produced sequence and its indices and one that corresponds to the protein's seqeuence and its indices), which are also contained in a list.
|
|
265
|
+
"""
|
|
266
|
+
with open(align_filepath, 'r') as fs:
|
|
267
|
+
alignment_entries: list[list[str]] = []
|
|
268
|
+
current_entry: list[str] = []
|
|
269
|
+
line_count = 0
|
|
270
|
+
for line in fs:
|
|
271
|
+
line = line.strip()
|
|
272
|
+
if line != '':
|
|
273
|
+
line_count += 1
|
|
274
|
+
current_entry.append(line)
|
|
275
|
+
if line_count == 4:
|
|
276
|
+
line_count = 0
|
|
277
|
+
alignment_entries.append(current_entry)
|
|
278
|
+
current_entry: list[str] = []
|
|
279
|
+
return alignment_entries
|
|
280
|
+
|
|
281
|
+
def __str__(self) -> str:
|
|
282
|
+
"""
|
|
283
|
+
Returns string representation of the ResidueAlignment pandas DataFrame in tab-separated value (tsv) format.
|
|
284
|
+
|
|
285
|
+
Returns
|
|
286
|
+
-------
|
|
287
|
+
str
|
|
288
|
+
reference_mapping pandas DataFrame exported to TSV format via the to_csv(sep="\t") function from pandas.
|
|
289
|
+
"""
|
|
290
|
+
return self.reference_mapping.to_csv(sep="\t")
|
|
291
|
+
|
|
292
|
+
class DirectInformationData:
|
|
293
|
+
"""
|
|
294
|
+
Representation and interface for Direct Information data including residue indices for a pair and its corresponding DI value represented as a 3-column ndarray.
|
|
295
|
+
|
|
296
|
+
Parameters
|
|
297
|
+
----------
|
|
298
|
+
structured_ndarray : numpy.ndarray
|
|
299
|
+
Ndarray with the shape (n,3) with dtype={'names': ('residue1', 'residue2', 'DI'), 'formats': (int, int, float, float)}
|
|
300
|
+
|
|
301
|
+
Attributes
|
|
302
|
+
----------
|
|
303
|
+
DI_data : numpy.ndarray
|
|
304
|
+
The structured_ndarray in the parameters section where column 1 corresponds to a pair's first residue, column 2 corresponds to the pair's second residue, and column 3 corresponds to the Direct Information of the pair.
|
|
305
|
+
"""
|
|
306
|
+
def __init__(self, structured_ndarray: npt.NDArray) -> None:
|
|
307
|
+
self.DI_data = structured_ndarray
|
|
308
|
+
|
|
309
|
+
@staticmethod
|
|
310
|
+
def load_from_dca_output(dca_filepath: str) -> 'DirectInformationData':
|
|
311
|
+
"""
|
|
312
|
+
Function to generate a DirectInformationData object from the direct output of the MATLab dca function.
|
|
313
|
+
|
|
314
|
+
Parameters
|
|
315
|
+
----------
|
|
316
|
+
dca_filepath : str
|
|
317
|
+
Filepath of the DCA output to be read and compiled into a structured ndarray. DCA output is a 4 column text file with the following columns: (residue 1, residue 2, Mutual Information, Direct Information).
|
|
318
|
+
|
|
319
|
+
Returns
|
|
320
|
+
-------
|
|
321
|
+
DirectInformationData
|
|
322
|
+
DirectInformationData object with named structured array containing residue indices and the DI value of the pair.
|
|
323
|
+
"""
|
|
324
|
+
file_data = np.loadtxt(dca_filepath, dtype={'names': ('residue1', 'residue2', 'MI', 'DI'), 'formats': (int, int, float, float)})
|
|
325
|
+
return DirectInformationData(file_data[['residue1', 'residue2', 'DI']])
|
|
326
|
+
|
|
327
|
+
@staticmethod
|
|
328
|
+
def load_from_DI_file(DI_filepath: str) -> 'DirectInformationData':
|
|
329
|
+
"""
|
|
330
|
+
Function to generate a DirectInformationData object from the modified DI-only version of the DCA output generated via the MATLab dca function.
|
|
331
|
+
|
|
332
|
+
Parameters
|
|
333
|
+
----------
|
|
334
|
+
DI_filepath : str
|
|
335
|
+
Filepath of the DI file to be read and compile into a structured ndarray. DI file is a 3 column text file with the following columns: (residue 1, residue 2, Direct Information).
|
|
336
|
+
|
|
337
|
+
Returns
|
|
338
|
+
-------
|
|
339
|
+
DirectInformationData
|
|
340
|
+
DirectInformationData object with named structured array containing residue indices and the DI value of the pair.
|
|
341
|
+
"""
|
|
342
|
+
return DirectInformationData(np.loadtxt(DI_filepath, dtype={'names': ('residue1', 'residue2', 'DI'), 'formats': (int, int, float)}))
|
|
343
|
+
|
|
344
|
+
@staticmethod
|
|
345
|
+
def load_as_ndarray(ndarray: npt.NDArray) -> 'DirectInformationData':
|
|
346
|
+
"""
|
|
347
|
+
Function to generate a Direct Information object from a ndarray.
|
|
348
|
+
|
|
349
|
+
Parameters
|
|
350
|
+
----------
|
|
351
|
+
ndarray : numpy.ndarray
|
|
352
|
+
An ndarray of shape (n,3) where its columns are (residue 1, residue 2, and Direct Information)
|
|
353
|
+
|
|
354
|
+
Returns
|
|
355
|
+
-------
|
|
356
|
+
DirectInformationData
|
|
357
|
+
DirectInformationData object with named structured array containing residue indices and the DI value of the pair.
|
|
358
|
+
"""
|
|
359
|
+
if ndarray.shape[1] != 3:
|
|
360
|
+
raise Exception("Dimensions of numpy array supplied are different from what is expected. Please supply residue1, residue2, and DI column in int, int, float format and with shape of (n, 3).")
|
|
361
|
+
# Structured ndarrays require list of tuples for conversion.
|
|
362
|
+
DI_data = np.array([tuple(x) for x in ndarray], dtype={'names': ('residue1', 'residue2', 'DI'), 'formats': (int, int, float)})
|
|
363
|
+
|
|
364
|
+
return DirectInformationData(DI_data)
|
|
365
|
+
|
|
366
|
+
def get_ranked_mapped_pairs(self, RA1: ResidueAlignment, RA2: ResidueAlignment, pairs_only: bool=True, mirror: bool=False, number: Optional[int]=None) -> npt.NDArray:
|
|
367
|
+
"""
|
|
368
|
+
Uses DirectInformationData and Pairs interface methods to obtain ranked, mapped residues. See rank_pairs() function and map_DIs() function for details on rank and mapping. Residue Alignments can be the same for intra-domain / intra-protein mapping.
|
|
369
|
+
|
|
370
|
+
Parameters
|
|
371
|
+
----------
|
|
372
|
+
RA1 : ResidueAlignment
|
|
373
|
+
The ResidueAlignment used for mapping the first column of residues to the appropriate target sequence.
|
|
374
|
+
RA2 : ResidueAlignment
|
|
375
|
+
The ResidueAlignment used for mapping the second column of residues to the appropriate target sequence.
|
|
376
|
+
pairs_only : bool
|
|
377
|
+
True if the final ndarray should contain only columns 1 and 2, corresponding to the residues that constitute the pair. This would drop the DI column.
|
|
378
|
+
mirror : bool
|
|
379
|
+
See Pairs.mirror_pairs() or get_pairs() for details. NOTE: This option is overriden entirely if pairs_only is False. If true, this will produce an ndarray that has the original residue indices and repeated residue indices but with residue 1 and residue 2 switched. This is useful for plotting across the upper diagonal of a contact map.
|
|
380
|
+
number : int, None
|
|
381
|
+
Number of ranked, mapped pairs to return.
|
|
382
|
+
|
|
383
|
+
Returns
|
|
384
|
+
-------
|
|
385
|
+
numpy.ndarray
|
|
386
|
+
Structured ndarray with columns residue 1, residue 2 and optionally DI. Only has specified number of pairs if `number` is specified and mirrored pairs if `mirror` is True and pairs_only is False.
|
|
387
|
+
|
|
388
|
+
Notes
|
|
389
|
+
-----
|
|
390
|
+
ResidueAlignments contain dictionaries like domain_to_protein to map residues produced via Direct Coupling Analysis (DCA) on an MSA generated in context to an HMM. The residues are mapped to a protein structure via alignment of the HMM hit / domain to the protein sequence.
|
|
391
|
+
"""
|
|
392
|
+
ranked_pairs = DirectInformationData.rank_pairs(DirectInformationData.nonlocal_pairs(self.DI_data))
|
|
393
|
+
ranked_mapped_pairs = DirectInformationData.map_DIs(ranked_pairs, RA1, RA2)
|
|
394
|
+
if pairs_only:
|
|
395
|
+
return Pairs.get_pairs(ranked_mapped_pairs[['residue1', 'residue2']], mirror=mirror, number=number)
|
|
396
|
+
else:
|
|
397
|
+
return Pairs.get_pairs(ranked_mapped_pairs, mirror=False, number=number)
|
|
398
|
+
|
|
399
|
+
@staticmethod
|
|
400
|
+
def map_DIs(DI_data : npt.NDArray, RA1: ResidueAlignment, RA2: ResidueAlignment) -> npt.NDArray:
|
|
401
|
+
"""
|
|
402
|
+
Uses domain-to-protein mappings present in the Residue Alignments provided to generate mapped representations of the residues from the DI_data structured ndarray provided.
|
|
403
|
+
|
|
404
|
+
Parameters
|
|
405
|
+
----------
|
|
406
|
+
DI_data : numpy.ndarray
|
|
407
|
+
Structured ndarray that contains columns "residue1" and "residue2".
|
|
408
|
+
RA1 : ResidueAlignment
|
|
409
|
+
The ResidueAlignment used for mapping the first column of residues to the appropriate target sequence.
|
|
410
|
+
RA2 : ResidueAlignment
|
|
411
|
+
The ResidueAlignment used for mapping the second column of residues to the appropriate target sequence.
|
|
412
|
+
|
|
413
|
+
Returns
|
|
414
|
+
-------
|
|
415
|
+
mappable_DI_data : numpy.ndarray
|
|
416
|
+
DI_data that has been mapped to the target sequence specified in the generation of the corresponding ResidueAlignments.
|
|
417
|
+
|
|
418
|
+
Note:
|
|
419
|
+
Residues that do not map to the target sequence of the ResidueAlignment are dropped.
|
|
420
|
+
"""
|
|
421
|
+
# Alternative approach is to just use get function instead of [x] and default to np.nan and drop nans row-wise.
|
|
422
|
+
mapping_key_mask = (np.isin(DI_data['residue1'], list(RA1.domain_to_protein.keys()))) & (np.isin(DI_data['residue2'], list(RA2.domain_to_protein.keys())))
|
|
423
|
+
mappable_DI_data = DI_data[mapping_key_mask]
|
|
424
|
+
mappable_DI_data['residue1'] = np.vectorize(lambda x: RA1.domain_to_protein[x])(mappable_DI_data['residue1'])
|
|
425
|
+
mappable_DI_data['residue2'] = np.vectorize(lambda x: RA2.domain_to_protein[x])(mappable_DI_data['residue2'])
|
|
426
|
+
return mappable_DI_data
|
|
427
|
+
|
|
428
|
+
@staticmethod
|
|
429
|
+
def rank_pairs(DI_data: npt.NDArray) -> npt.NDArray:
|
|
430
|
+
"""
|
|
431
|
+
Sorts a structured ndarray of pairs information to order the ndarray by Direct Information (DI) score.
|
|
432
|
+
|
|
433
|
+
Parameters
|
|
434
|
+
----------
|
|
435
|
+
DI_data : numpy.ndarray
|
|
436
|
+
Structured ndarray that contains columns with names "residue1", "residue2", and "DI" (Direct Information)
|
|
437
|
+
|
|
438
|
+
Returns
|
|
439
|
+
-------
|
|
440
|
+
numpy.ndarray
|
|
441
|
+
Structured ndarray sorted upon column that is named DI in descending order.
|
|
442
|
+
"""
|
|
443
|
+
# [::-1] reverses the order from ascending DI Score to descending DI score.
|
|
444
|
+
return np.sort(DI_data, order='DI')[::-1]
|
|
445
|
+
|
|
446
|
+
@staticmethod
|
|
447
|
+
def nonlocal_pairs(DI_data: npt.NDArray) -> npt.NDArray:
|
|
448
|
+
"""
|
|
449
|
+
Subsets a structured ndarray of pairs information to find nonlocal pairs, where residue interactions are likely not involved in secondary structure formation i.e. helices and sheet interactions. Nonlocal pairs must be at least 4 residues apart.
|
|
450
|
+
|
|
451
|
+
Parameters
|
|
452
|
+
----------
|
|
453
|
+
DI_data : numpy.ndarray
|
|
454
|
+
Structured ndarray that contains (at least) the first and second columns with the names "residue1" and "residue2" respectively.
|
|
455
|
+
|
|
456
|
+
Returns
|
|
457
|
+
-------
|
|
458
|
+
numpy.ndarray
|
|
459
|
+
Structured ndarray of DI pairs where residue 1 and residue 2 are at least 4 residues apart.
|
|
460
|
+
"""
|
|
461
|
+
return DI_data[abs(DI_data['residue1'] - DI_data['residue2']) > 4]
|
|
462
|
+
|
|
463
|
+
@staticmethod
|
|
464
|
+
def get_dist_commands(model1: str | int, model2: str | int, chain1: str, chain2: str, pairs: npt.NDArray, ca_only: bool=True, auth_res_ids=False) -> list[str]:
|
|
465
|
+
"""
|
|
466
|
+
Get UCSF Chimera commands for displaying distance commands for usage in displaying distances between residue pairs. Options are present for alpha-carbon to alpha-carbon distance or for specified atom to specified atom distance.
|
|
467
|
+
|
|
468
|
+
Parameters
|
|
469
|
+
----------
|
|
470
|
+
model1 : str, int
|
|
471
|
+
Number of model corresponding to the structure containing the first column of residues.
|
|
472
|
+
model2 : str, int
|
|
473
|
+
Number of model corresponding to the structure containing the second column of residues.
|
|
474
|
+
chain1 : str
|
|
475
|
+
The chain present in the structure in model1 containing the first column of residues.
|
|
476
|
+
chain2 : str
|
|
477
|
+
The chain present in the structure in model2 containing the second column of residues.
|
|
478
|
+
pairs : numpy.ndarray
|
|
479
|
+
Structured ndarray that contains (at least) the first and second columns with the names "residue1" and "residue2" respectively. `ca_only` should be set to false and an additional "atom_name1" and "atom_name2" column should be added if atoms are specified per pair.
|
|
480
|
+
ca_only : bool
|
|
481
|
+
True if distance commands are for displaying distances between the two alpha-carbons of the residue pair. If false, specific_atom_names is used in lieu of "CA" as an atom identifier.
|
|
482
|
+
auth_res_ids : bool
|
|
483
|
+
If true, use the "auth_residue1" and "auth_residue2" columns instead of "residue1" and "residue2" columns. These residue ids correspond to the auth protein residue ids.
|
|
484
|
+
|
|
485
|
+
Returns
|
|
486
|
+
-------
|
|
487
|
+
distance_commands : list of str
|
|
488
|
+
List of distance commands generated between two residues with model and chain information needed, either between two alpha-carbons or the specified atoms.
|
|
489
|
+
|
|
490
|
+
Note
|
|
491
|
+
----
|
|
492
|
+
model1 and model2 can be equivalent if both columns involve residues referenced by the same model. The same would apply for chains if the residues are present on the same chain.
|
|
493
|
+
"""
|
|
494
|
+
distance_commands: list[str] = []
|
|
495
|
+
for i in range(np.shape(pairs)[0]):
|
|
496
|
+
if auth_res_ids:
|
|
497
|
+
residue1 = pairs[i]['auth_residue1']
|
|
498
|
+
residue2 = pairs[i]['auth_residue2']
|
|
499
|
+
else:
|
|
500
|
+
residue1 = pairs[i]['residue1']
|
|
501
|
+
residue2 = pairs[i]['residue2']
|
|
502
|
+
if ca_only:
|
|
503
|
+
distance_commands.append(f"distance #{model1}:{residue1}.{chain1}@CA #{model2}:{residue2}.{chain2}@CA;")
|
|
504
|
+
else:
|
|
505
|
+
atom1 = pairs[i]['atom_name1']
|
|
506
|
+
atom2 = pairs[i]['atom_name2']
|
|
507
|
+
distance_commands.append(f"distance #{model1}:{residue1}.{chain1}@{atom1} #{model2}:{residue2}.{chain2}@{atom2};")
|
|
508
|
+
return distance_commands
|
|
509
|
+
|
|
510
|
+
@staticmethod
|
|
511
|
+
def write_DI_data(filepath: str, pairs: npt.NDArray, delimiter: str="\t", fmt: tuple[str, str, str] | tuple[str, str]=('%d', '%d', '%.3f')) -> None:
|
|
512
|
+
"""
|
|
513
|
+
Writes pairs ndarray to file with specified delimiter between the pairs' row elements, i.e. residue 1, residue 2, and DI score.
|
|
514
|
+
|
|
515
|
+
Parameters
|
|
516
|
+
----------
|
|
517
|
+
filepath : str
|
|
518
|
+
Path of the file to write DirectInformation data to.
|
|
519
|
+
pairs : numpy.ndarray
|
|
520
|
+
Ndarray of at-least pairs information (residue 1, residue 2) and optionally Direct Information to write to a file via numpy.savetxt().
|
|
521
|
+
delimiter : str
|
|
522
|
+
Delimiter to separate columns of the pairs ndarray when writing to a file.
|
|
523
|
+
fmt : tuple of str, default=('%d', '%d', '%.3f)
|
|
524
|
+
format passed as an argument to numpy.savetxt() to define type of column and output format. Set to ('%d', '%d') if only two columns are present in the ndarray.
|
|
525
|
+
|
|
526
|
+
Returns
|
|
527
|
+
-------
|
|
528
|
+
None
|
|
529
|
+
"""
|
|
530
|
+
if len(pairs[0]) == 2:
|
|
531
|
+
fmt = ('%d', '%d')
|
|
532
|
+
np.savetxt(filepath, pairs, delimiter=delimiter, fmt=fmt)
|
|
533
|
+
|
|
534
|
+
class StructureInformation:
|
|
535
|
+
"""
|
|
536
|
+
Information regarding a protein structure, obtained from a protein structure file.
|
|
537
|
+
|
|
538
|
+
Parameters
|
|
539
|
+
----------
|
|
540
|
+
structure : biotite.structure
|
|
541
|
+
Structure obtained from an RCSB entry with a provided pdbx/mmcif file with a specified model number.
|
|
542
|
+
pdbx_file : biotite.io.pdbx.CIFFile
|
|
543
|
+
mmCIF file that contains generic information and atomic information of the protein structure categorized into mmCIF blocks.
|
|
544
|
+
|
|
545
|
+
Attributes
|
|
546
|
+
----------
|
|
547
|
+
struct_ref_seq : zip
|
|
548
|
+
zipped version of parallel arrays that contain chain ids, beginning align indices of the sequence, and beginning align indices of the auth sequence.
|
|
549
|
+
"""
|
|
550
|
+
def __init__(self, structure, pdbx_file: pdbx.CIFFile):
|
|
551
|
+
self.structure = structure
|
|
552
|
+
self.pdbx_file = pdbx_file
|
|
553
|
+
strand_ids = []
|
|
554
|
+
align_beg = []
|
|
555
|
+
auth_align_beg = []
|
|
556
|
+
# Obtain struct ref seq information (chain names, beginning residue, and the corresponding protein beginning residue)
|
|
557
|
+
for col_name, col in self.pdbx_file[list(self.pdbx_file.keys())[0]]['struct_ref_seq'].items():
|
|
558
|
+
if col_name == "pdbx_strand_id":
|
|
559
|
+
strand_ids = col.data.array
|
|
560
|
+
if col_name == "seq_align_beg":
|
|
561
|
+
align_beg = col.data.array
|
|
562
|
+
if col_name == "pdbx_auth_seq_align_beg":
|
|
563
|
+
auth_align_beg = col.data.array
|
|
564
|
+
# Put all three lists together
|
|
565
|
+
self.struct_ref_seq = zip(strand_ids, align_beg, auth_align_beg)
|
|
566
|
+
|
|
567
|
+
@staticmethod
|
|
568
|
+
def fetch_pdb(pdb_id: str, model_num: int=1, struc_format: str="mmcif") -> 'StructureInformation':
|
|
569
|
+
"""
|
|
570
|
+
Fetches PDB as mmCIF file from RCSB and compiles the information into a StructureInformation instance.
|
|
571
|
+
|
|
572
|
+
Parameters
|
|
573
|
+
----------
|
|
574
|
+
pdb_id : str
|
|
575
|
+
PDB ID to be fetched from the RCSB database.
|
|
576
|
+
model_num : int
|
|
577
|
+
The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
|
|
578
|
+
struc_format : str
|
|
579
|
+
The format of the file to pull from the RCSB database.
|
|
580
|
+
|
|
581
|
+
Returns
|
|
582
|
+
-------
|
|
583
|
+
StructureInformation
|
|
584
|
+
StructureInformation generated from pdbx.get_structure() function using the pdbx file fetched from RCSB. The pdbx file is also supplied as an argument.
|
|
585
|
+
|
|
586
|
+
Raises
|
|
587
|
+
------
|
|
588
|
+
TypeError
|
|
589
|
+
Fetched data was not found and returned None instead.
|
|
590
|
+
"""
|
|
591
|
+
fetched_data = rcsb.fetch(pdb_id, struc_format)
|
|
592
|
+
if fetched_data is None:
|
|
593
|
+
raise TypeError("RCSB fetch failed. Try fetch again.")
|
|
594
|
+
pdbx_file = pdbx.CIFFile.read(fetched_data)
|
|
595
|
+
return StructureInformation(pdbx.get_structure(pdbx_file=pdbx_file, model=model_num, use_author_fields=False), pdbx_file)
|
|
596
|
+
|
|
597
|
+
@staticmethod
|
|
598
|
+
def read_pdb_mmCIF(pdb_filepath: str, model_num: int=1) -> 'StructureInformation':
|
|
599
|
+
"""
|
|
600
|
+
Reads PDB mmCIF file from filepath and compiles the information into a StructureInformation instance.
|
|
601
|
+
|
|
602
|
+
Parameters
|
|
603
|
+
----------
|
|
604
|
+
pdb_filepath : str
|
|
605
|
+
Filepath of the PDB mmCIF file to be read.
|
|
606
|
+
model_num : int
|
|
607
|
+
The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
|
|
608
|
+
|
|
609
|
+
Returns
|
|
610
|
+
-------
|
|
611
|
+
StructureInformation
|
|
612
|
+
StructureInformation generated from pdbx.get_structure() function using the pdbx file fetched from RCSB. The pdbx file is also supplied as an argument.
|
|
613
|
+
"""
|
|
614
|
+
pdbx_file = pdbx.CIFFile.read(pdb_filepath)
|
|
615
|
+
return StructureInformation(pdbx.get_structure(pdbx_file, model=model_num, use_author_fields=False), pdbx_file)
|
|
616
|
+
|
|
617
|
+
def get_chain_specific_structure(self, ca_only: bool, chain1: str, chain2: str, remove_hetero=True) -> tuple:
|
|
618
|
+
"""
|
|
619
|
+
Subsets structure attribute to select for chain specific portions of the structure.
|
|
620
|
+
|
|
621
|
+
Parameters
|
|
622
|
+
----------
|
|
623
|
+
ca_only : bool
|
|
624
|
+
If true, the structure will also be subsetted for atom entries where the atom_name annotation is "CA" (referring to alpha-carbons)
|
|
625
|
+
chain1 : str
|
|
626
|
+
Chain id corresponding to the first column of residues in the structure.
|
|
627
|
+
chain2 : str
|
|
628
|
+
Chain id corresponding to the second column of residues in the structure.
|
|
629
|
+
remove_hetero : bool, default=True
|
|
630
|
+
If true, the structure will also be subsetted for atom entries where the hetero annotation is False, thus removing heteroatoms.
|
|
631
|
+
|
|
632
|
+
Returns
|
|
633
|
+
-------
|
|
634
|
+
tuple of biotite.structure.AtomArray, biotite.structure.AtomArray
|
|
635
|
+
Two AtomArrays that refer to atoms in the first chain and second chain, respectively without accounting for the presence of heteroatoms if `remove_hetero` is True.
|
|
636
|
+
"""
|
|
637
|
+
selected_structure = self.structure
|
|
638
|
+
if remove_hetero:
|
|
639
|
+
# Remove hetero atoms via hetero column of structure ndarray
|
|
640
|
+
selected_structure = self.structure[self.structure.hetero == False]
|
|
641
|
+
if ca_only:
|
|
642
|
+
# Consider selection of alpha-carbon atoms only
|
|
643
|
+
selected_structure = selected_structure[selected_structure.atom_name == "CA"]
|
|
644
|
+
chain1_structure = selected_structure[selected_structure.chain_id == chain1]
|
|
645
|
+
chain2_structure = selected_structure[selected_structure.chain_id == chain2]
|
|
646
|
+
return (chain1_structure, chain2_structure)
|
|
647
|
+
|
|
648
|
+
def generate_dist_matrix(self, ca_only: bool, chain1: str, chain2: str):
|
|
649
|
+
"""
|
|
650
|
+
Generates distance matrix between two chains in the structure attribute.
|
|
651
|
+
|
|
652
|
+
Parameters
|
|
653
|
+
----------
|
|
654
|
+
ca_only : bool
|
|
655
|
+
If True, only atoms that have the name "CA" are selected in the chains the distance matrix is calculated between.
|
|
656
|
+
chain1 : str
|
|
657
|
+
Chain id corresponding to the first column of residues in the structure.
|
|
658
|
+
chain2 : str
|
|
659
|
+
Chain id corresponding to the first column of residues in the structure.
|
|
660
|
+
|
|
661
|
+
Returns
|
|
662
|
+
-------
|
|
663
|
+
tuple of biotite.structure.AtomArray, biotite.structure.AtomArray, numpy.ndarray
|
|
664
|
+
Tuple containing the chain 1 structure, the chain 2 structure, and the distance matrix of chain 1 and chain 2's pairwise distances.
|
|
665
|
+
"""
|
|
666
|
+
chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only, chain1, chain2, remove_hetero=True)
|
|
667
|
+
dist_matrix = cdist(chain1_structure.coord, chain2_structure.coord)
|
|
668
|
+
return (chain1_structure, chain2_structure, dist_matrix)
|
|
669
|
+
|
|
670
|
+
def get_min_dist_atom_info(self, pairs: npt.NDArray, chain1: str, chain2: str) -> npt.NDArray:
|
|
671
|
+
"""
|
|
672
|
+
Generate a ndarray of residue ids and their corresponding atom names such that the distance is the minimum between the initial residues provided.
|
|
673
|
+
|
|
674
|
+
Parameters
|
|
675
|
+
----------
|
|
676
|
+
pairs : numpy.ndarray
|
|
677
|
+
Pairs structured ndarray with "residue1" and "residue2" columns.
|
|
678
|
+
chain1 : str
|
|
679
|
+
Chain id corresponding to the first column of residues in the structure.
|
|
680
|
+
chain2 : str
|
|
681
|
+
Chain id corresponding to the second column of residues in the structure.
|
|
682
|
+
|
|
683
|
+
Returns
|
|
684
|
+
-------
|
|
685
|
+
min_dist_pairs_atoms_arr : numpy.ndarray
|
|
686
|
+
Structured ndarray that has residue indices, auth residue indices (corresponding to the protein numbering), and atomic names in the format {'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,str,str]}
|
|
687
|
+
"""
|
|
688
|
+
shift1 = 0
|
|
689
|
+
shift2 = 0
|
|
690
|
+
for row in list(self.struct_ref_seq):
|
|
691
|
+
ref_seq_chain, ref_seq_beg, auth_ref_seq_beg = row
|
|
692
|
+
if ref_seq_chain == chain1:
|
|
693
|
+
shift1 = int(auth_ref_seq_beg) - int(ref_seq_beg)
|
|
694
|
+
if ref_seq_chain == chain2:
|
|
695
|
+
shift2 = int(auth_ref_seq_beg) - int(ref_seq_beg)
|
|
696
|
+
|
|
697
|
+
chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only=False, chain1=chain1, chain2=chain2, remove_hetero=True)
|
|
698
|
+
min_dist_pairs_atoms = []
|
|
699
|
+
for row in pairs:
|
|
700
|
+
# Obtain structure information for chains 1 and 2
|
|
701
|
+
chain1_res1_structure = chain1_structure[chain1_structure.res_id == row['residue1']]
|
|
702
|
+
chain2_res2_structure = chain2_structure[chain2_structure.res_id == row['residue2']]
|
|
703
|
+
|
|
704
|
+
# Calculate a distance matrix and find the indices of the minimal value in the matrix
|
|
705
|
+
dist_matrix = cdist(chain1_res1_structure.coord, chain2_res2_structure.coord)
|
|
706
|
+
|
|
707
|
+
ind = np.unravel_index(np.argmin(dist_matrix), dist_matrix.shape)
|
|
708
|
+
# Use the indices to access the atom in the atom array and get the correct atom name.
|
|
709
|
+
# Generate the auth ids of the residues in the pairs ndarray
|
|
710
|
+
auth_res_id1 = row['residue1'] + shift1
|
|
711
|
+
auth_res_id2 = row['residue2'] + shift2
|
|
712
|
+
min_dist_pairs_atoms.append((row['residue1'], row['residue2'],auth_res_id1, auth_res_id2, chain1_res1_structure[ind[0]].atom_name, chain2_res2_structure[ind[1]].atom_name))
|
|
713
|
+
min_dist_pairs_atoms_arr = np.array(min_dist_pairs_atoms, dtype={'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,int,int,'<U10','<U10']})
|
|
714
|
+
return min_dist_pairs_atoms_arr
|
|
715
|
+
|
|
716
|
+
def get_contacts(self, ca_only: bool, threshold: float, chain1: str, chain2: str) -> set[tuple[int, int]]:
|
|
717
|
+
"""
|
|
718
|
+
Get contacts from the structure attribute where the distance between two residues is less than the threshold.
|
|
719
|
+
|
|
720
|
+
Parameters
|
|
721
|
+
----------
|
|
722
|
+
ca_only : bool
|
|
723
|
+
If true, only consider alpha-carbon to alpha-carbon distances.
|
|
724
|
+
threshold : float
|
|
725
|
+
Maximum distance to consider between two atoms.
|
|
726
|
+
chain1 : str
|
|
727
|
+
Chain id corresponding to the first column of residues in the structure.
|
|
728
|
+
chain2 : str
|
|
729
|
+
Chain id corresponding to the second column of residues in the structure.
|
|
730
|
+
|
|
731
|
+
Returns
|
|
732
|
+
-------
|
|
733
|
+
contacts_set : set of tuple of ints
|
|
734
|
+
Set of contacts, tuples with "residue1" and "residue2" from the structure that are within the distance threshold.
|
|
735
|
+
"""
|
|
736
|
+
chain1_structure, chain2_structure, dist_matrix = self.generate_dist_matrix(ca_only, chain1, chain2)
|
|
737
|
+
contacts = np.argwhere(dist_matrix <= threshold)
|
|
738
|
+
contacts_set = set()
|
|
739
|
+
for contact in contacts:
|
|
740
|
+
contacts_set.add((chain1_structure[contact[0]].res_id, chain2_structure[contact[1]].res_id))
|
|
741
|
+
return contacts_set
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: dcatoolkit
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Collection of useful modules and representations for managing DCA output data.
|
|
5
|
+
Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
6
|
+
Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
7
|
+
License: MIT License
|
|
8
|
+
|
|
9
|
+
Copyright (c) 2024 Raheel Syed Ahmed
|
|
10
|
+
|
|
11
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
12
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
13
|
+
in the Software without restriction, including without limitation the rights
|
|
14
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
15
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
16
|
+
furnished to do so, subject to the following conditions:
|
|
17
|
+
|
|
18
|
+
The above copyright notice and this permission notice shall be included in all
|
|
19
|
+
copies or substantial portions of the Software.
|
|
20
|
+
|
|
21
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
22
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
23
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
24
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
25
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
26
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
27
|
+
SOFTWARE.
|
|
28
|
+
|
|
29
|
+
Keywords: dca,toolkit,DI,coevolution
|
|
30
|
+
Classifier: Development Status :: 4 - Beta
|
|
31
|
+
Classifier: Intended Audience :: Science/Research
|
|
32
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
33
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
34
|
+
Classifier: Programming Language :: Python :: 3
|
|
35
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
36
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
37
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
38
|
+
Requires-Python: >=3.10
|
|
39
|
+
Description-Content-Type: text/markdown
|
|
40
|
+
License-File: LICENSE
|
|
41
|
+
Requires-Dist: biotite
|
|
42
|
+
Requires-Dist: matplotlib>=3.8.0
|
|
43
|
+
Requires-Dist: numpy>=1.26.0
|
|
44
|
+
Requires-Dist: pandas>=2.1.0
|
|
45
|
+
Requires-Dist: pyhmmer>=0.10.14
|
|
46
|
+
Requires-Dist: scikit-learn>=1.3
|
|
47
|
+
Requires-Dist: scipy>=1.11.0
|
|
48
|
+
Provides-Extra: tests
|
|
49
|
+
Requires-Dist: pytest; extra == "tests"
|
|
50
|
+
Provides-Extra: docs
|
|
51
|
+
Requires-Dist: sphinx; extra == "docs"
|
|
52
|
+
Requires-Dist: numpydoc; extra == "docs"
|
|
53
|
+
Provides-Extra: lint
|
|
54
|
+
Requires-Dist: ruffle; extra == "lint"
|
|
55
|
+
|
|
56
|
+
# dcatoolkit
|
|
57
|
+
Collection of useful modules and representations for managing DCA output data.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
src/dcatoolkit/__init__.py
|
|
5
|
+
src/dcatoolkit/representation.py
|
|
6
|
+
src/dcatoolkit.egg-info/PKG-INFO
|
|
7
|
+
src/dcatoolkit.egg-info/SOURCES.txt
|
|
8
|
+
src/dcatoolkit.egg-info/dependency_links.txt
|
|
9
|
+
src/dcatoolkit.egg-info/requires.txt
|
|
10
|
+
src/dcatoolkit.egg-info/top_level.txt
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
dcatoolkit
|