dcatoolkit 0.2.2__tar.gz → 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dcatoolkit-0.2.2/src/dcatoolkit.egg-info → dcatoolkit-0.2.3}/PKG-INFO +3 -2
- {dcatoolkit-0.2.2 → dcatoolkit-0.2.3}/pyproject.toml +1 -1
- {dcatoolkit-0.2.2 → dcatoolkit-0.2.3}/src/dcatoolkit/__init__.py +1 -1
- {dcatoolkit-0.2.2 → dcatoolkit-0.2.3}/src/dcatoolkit/analytics.py +47 -1
- {dcatoolkit-0.2.2 → dcatoolkit-0.2.3}/src/dcatoolkit/representation.py +220 -134
- {dcatoolkit-0.2.2 → dcatoolkit-0.2.3/src/dcatoolkit.egg-info}/PKG-INFO +3 -2
- {dcatoolkit-0.2.2 → dcatoolkit-0.2.3}/src/dcatoolkit.egg-info/SOURCES.txt +3 -1
- dcatoolkit-0.2.3/tests/test_alignments.py +88 -0
- {dcatoolkit-0.2.2 → dcatoolkit-0.2.3}/tests/test_contacts.py +7 -7
- dcatoolkit-0.2.3/tests/test_sequence_numbering.py +14 -0
- {dcatoolkit-0.2.2 → dcatoolkit-0.2.3}/LICENSE +0 -0
- {dcatoolkit-0.2.2 → dcatoolkit-0.2.3}/README.md +0 -0
- {dcatoolkit-0.2.2 → dcatoolkit-0.2.3}/setup.cfg +0 -0
- {dcatoolkit-0.2.2 → dcatoolkit-0.2.3}/src/dcatoolkit.egg-info/dependency_links.txt +0 -0
- {dcatoolkit-0.2.2 → dcatoolkit-0.2.3}/src/dcatoolkit.egg-info/requires.txt +0 -0
- {dcatoolkit-0.2.2 → dcatoolkit-0.2.3}/src/dcatoolkit.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: dcatoolkit
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: Collection of useful modules and representations for managing DCA output data.
|
|
5
5
|
Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
6
6
|
Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
@@ -52,6 +52,7 @@ Requires-Dist: pdoc; extra == "docs"
|
|
|
52
52
|
Requires-Dist: numpydoc; extra == "docs"
|
|
53
53
|
Provides-Extra: lint
|
|
54
54
|
Requires-Dist: ruffle; extra == "lint"
|
|
55
|
+
Dynamic: license-file
|
|
55
56
|
|
|
56
57
|
# dcatoolkit
|
|
57
58
|
Collection of useful modules and representations for managing DCA output data.
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "dcatoolkit"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.3"
|
|
8
8
|
description = "Collection of useful modules and representations for managing DCA output data."
|
|
9
9
|
keywords = ["dca", "toolkit", "DI", "coevolution"]
|
|
10
10
|
|
|
@@ -1,8 +1,11 @@
|
|
|
1
1
|
import re
|
|
2
2
|
from collections import Counter
|
|
3
|
-
from
|
|
3
|
+
from collections.abc import Callable
|
|
4
|
+
from typing import Optional, Union, Literal
|
|
4
5
|
import string
|
|
5
6
|
import io
|
|
7
|
+
import numpy as np
|
|
8
|
+
import numpy.typing as npt
|
|
6
9
|
|
|
7
10
|
class MSATools:
|
|
8
11
|
"""
|
|
@@ -122,6 +125,27 @@ class MSATools:
|
|
|
122
125
|
kept_entries.append((header, sequence))
|
|
123
126
|
return kept_entries
|
|
124
127
|
|
|
128
|
+
def gap_proportion(self, agg_func: Callable[..., float | int]=np.mean, axis: Literal[0, 1] = 0) -> float | int:
|
|
129
|
+
"""
|
|
130
|
+
Evaluates gap frequency per alignment position, or column, in the MSA.
|
|
131
|
+
|
|
132
|
+
Parameters
|
|
133
|
+
----------
|
|
134
|
+
agg_func: function = np.mean
|
|
135
|
+
The aggregation function applied to get the expected result, usually a mean, max, or min value, of the gap frequencies present per alignment position in the MSA.
|
|
136
|
+
axis: int or str
|
|
137
|
+
When axis is 0, the aggregation function is applied per column for entries from every row in that column. When axis is 1, the aggregation function is applied per row for entries from every column in that row.
|
|
138
|
+
|
|
139
|
+
Return
|
|
140
|
+
------
|
|
141
|
+
float
|
|
142
|
+
A numerical value determined by the aggregation function supplied over the gap frequencies of the alignment positions in the MSA.
|
|
143
|
+
"""
|
|
144
|
+
sequence_matrix = self.as_matrix()
|
|
145
|
+
non_alpha_counts = np.sum(~np.char.isalpha(sequence_matrix), axis)
|
|
146
|
+
non_alpha_counts = non_alpha_counts / sequence_matrix.shape[axis]
|
|
147
|
+
return agg_func(non_alpha_counts)
|
|
148
|
+
|
|
125
149
|
def write(self, destination: Union[str, io.IOBase]) -> None:
|
|
126
150
|
"""
|
|
127
151
|
Writes this MSA's headers and sequences to the destination specified.
|
|
@@ -149,6 +173,28 @@ class MSATools:
|
|
|
149
173
|
destination.write(sequence)
|
|
150
174
|
destination.write("\n")
|
|
151
175
|
|
|
176
|
+
def as_matrix(self) -> npt.NDArray:
|
|
177
|
+
"""
|
|
178
|
+
Represents the MSA as a numpy matrix of sequences.
|
|
179
|
+
|
|
180
|
+
Returns
|
|
181
|
+
-------
|
|
182
|
+
npt.NDArray
|
|
183
|
+
A matrix of "number of sequences" rows and "number of alignment positions" columns. Each cell is the sequence character for that sequence at that position.
|
|
184
|
+
"""
|
|
185
|
+
return np.array([list(seq) for _, seq in self.MSA])
|
|
186
|
+
|
|
187
|
+
def __str__(self) -> str:
|
|
188
|
+
"""
|
|
189
|
+
Returns the sequences present in the loaded MSA in string format.
|
|
190
|
+
|
|
191
|
+
Returns
|
|
192
|
+
-------
|
|
193
|
+
str
|
|
194
|
+
Each sequence in the instance separated with newline characters.
|
|
195
|
+
"""
|
|
196
|
+
return "\n".join([seq for _, seq in self.MSA])
|
|
197
|
+
|
|
152
198
|
def __len__(self):
|
|
153
199
|
"""
|
|
154
200
|
Returns the number of sequences, and equivalently, the number of headers in the MSA.
|
|
@@ -6,6 +6,7 @@ import biotite.structure as struc
|
|
|
6
6
|
import biotite.structure.io.pdbx as pdbx
|
|
7
7
|
import biotite.structure.io.pdb as pdb
|
|
8
8
|
import biotite.database.rcsb as rcsb
|
|
9
|
+
from biotite.sequence import ProteinSequence
|
|
9
10
|
|
|
10
11
|
from collections.abc import Iterable
|
|
11
12
|
from typing import Optional, Union, Literal, overload
|
|
@@ -195,6 +196,8 @@ class ResidueAlignment:
|
|
|
195
196
|
The sequence of the domain in the query HMM corresponding to this alignment.
|
|
196
197
|
protein_text : str
|
|
197
198
|
The sequence of the protein target sequence corresponding to this alignment.
|
|
199
|
+
valid_residues : list of tuple of int, str
|
|
200
|
+
A list of tuples that contain first residue index then residue name (e.g. [(1, 'A'), (2, 'W'), (3, 'C')] )
|
|
198
201
|
|
|
199
202
|
Attributes
|
|
200
203
|
----------
|
|
@@ -205,12 +208,73 @@ class ResidueAlignment:
|
|
|
205
208
|
protein_to_domain : dict[int, int]
|
|
206
209
|
A dictionary allowing for mapping from indices corresponding to the protein target sequence to the query HMM and Multiple Sequence Alignment.
|
|
207
210
|
"""
|
|
208
|
-
def __init__(self, domain_name: str, protein_name: str, domain_start: int, protein_start: int, domain_text: str, protein_text: str) -> None:
|
|
211
|
+
def __init__(self, domain_name: str, protein_name: str, domain_start: int, protein_start: int, domain_text: str, protein_text: str, valid_residues: Optional[list[tuple[int, str]]]=None) -> None:
|
|
209
212
|
self.domain_name = domain_name
|
|
210
213
|
self.protein_name = protein_name
|
|
211
|
-
self.
|
|
214
|
+
self.valid_residues = valid_residues
|
|
215
|
+
if valid_residues:
|
|
216
|
+
self._set_restricted_reference_mapping(domain_start, protein_start, domain_text, protein_text, valid_residues)
|
|
217
|
+
else:
|
|
218
|
+
self._set_reference_mapping(domain_start, protein_start, domain_text, protein_text)
|
|
212
219
|
|
|
213
|
-
def
|
|
220
|
+
def _set_restricted_reference_mapping(self, domain_start: int, protein_start: int, domain_text: str, protein_text: str, valid_residues: list[tuple[int, str]]) -> None:
|
|
221
|
+
"""
|
|
222
|
+
Set values for reference_mapping and mapping dictionaries, domain_to_protein and protein_to_domain.
|
|
223
|
+
|
|
224
|
+
Parameters
|
|
225
|
+
----------
|
|
226
|
+
valid_residues : list of tuple of int, str
|
|
227
|
+
List of valid residues, non-missing residues in a structure, in the format of (seq_id, residue_name). These are iteratively selected in the order of the sequence to map to.
|
|
228
|
+
|
|
229
|
+
Notes
|
|
230
|
+
-----
|
|
231
|
+
For details on `domain_start`, `protein_start`, `domain_text`, `protein_text`, please refer to the `ResidueAlignment` docstring.
|
|
232
|
+
|
|
233
|
+
Returns
|
|
234
|
+
-------
|
|
235
|
+
None
|
|
236
|
+
"""
|
|
237
|
+
invalid_chars = [".", "_", "-"]
|
|
238
|
+
# Convert text to list variant for iteration
|
|
239
|
+
domain_sequence = list(domain_text)
|
|
240
|
+
protein_sequence = list(protein_text)
|
|
241
|
+
# Store mapping values per iteration here.
|
|
242
|
+
mapping_entries = []
|
|
243
|
+
|
|
244
|
+
# Go through aligned sequences and append data concerning domain index, domain residue, protein residue, and protein index per valid aligned residues.
|
|
245
|
+
for domain_aa, protein_aa in zip(domain_sequence, protein_sequence):
|
|
246
|
+
mapping_entry = []
|
|
247
|
+
if domain_aa not in invalid_chars:
|
|
248
|
+
mapping_entry.append(domain_start)
|
|
249
|
+
domain_start += 1
|
|
250
|
+
else:
|
|
251
|
+
mapping_entry.append(pd.NA)
|
|
252
|
+
mapping_entry.append(domain_aa)
|
|
253
|
+
|
|
254
|
+
if protein_aa not in invalid_chars:
|
|
255
|
+
while len(valid_residues) >= protein_start:
|
|
256
|
+
prot_index, valid_residue = valid_residues.pop(protein_start-1)
|
|
257
|
+
if protein_aa.lower() == valid_residue.lower():
|
|
258
|
+
mapping_entry.append(protein_aa)
|
|
259
|
+
mapping_entry.append(prot_index)
|
|
260
|
+
break
|
|
261
|
+
else:
|
|
262
|
+
# Default case for if valid residues are missing towards the end.
|
|
263
|
+
mapping_entry.append(protein_aa)
|
|
264
|
+
mapping_entry.append(pd.NA)
|
|
265
|
+
else:
|
|
266
|
+
mapping_entry.append(protein_aa)
|
|
267
|
+
mapping_entry.append(pd.NA)
|
|
268
|
+
mapping_entries.append(mapping_entry)
|
|
269
|
+
|
|
270
|
+
self.reference_mapping = pd.DataFrame(mapping_entries, columns=['domain_index', 'domain_residue', 'protein_residue', 'protein_index'])
|
|
271
|
+
self.reference_mapping = self.reference_mapping.astype({'domain_index': pd.Int32Dtype(), 'protein_index': pd.Int32Dtype(), 'domain_residue': pd.StringDtype(), 'protein_residue': pd.StringDtype()})
|
|
272
|
+
reference_mapping_notna = self.reference_mapping.dropna()
|
|
273
|
+
|
|
274
|
+
self.domain_to_protein = dict(zip(reference_mapping_notna.domain_index, reference_mapping_notna.protein_index))
|
|
275
|
+
self.protein_to_domain = dict(zip(reference_mapping_notna.protein_index, reference_mapping_notna.domain_index))
|
|
276
|
+
|
|
277
|
+
def _set_reference_mapping(self, domain_start: int, protein_start: int, domain_text: str, protein_text: str) -> None:
|
|
214
278
|
"""
|
|
215
279
|
Set values for reference_mapping and mapping dictionaries, domain_to_protein and protein_to_domain.
|
|
216
280
|
|
|
@@ -222,25 +286,26 @@ class ResidueAlignment:
|
|
|
222
286
|
-------
|
|
223
287
|
None
|
|
224
288
|
"""
|
|
289
|
+
invalid_chars = [".", "_", "-"]
|
|
225
290
|
# Convert text to list variant for iteration
|
|
226
291
|
domain_sequence = list(domain_text)
|
|
227
292
|
protein_sequence = list(protein_text)
|
|
228
|
-
|
|
293
|
+
# Store mapping values per iteration here.
|
|
229
294
|
mapping_entries = []
|
|
230
295
|
|
|
231
|
-
for
|
|
296
|
+
for domain_aa, protein_aa in zip(domain_sequence, protein_sequence):
|
|
232
297
|
mapping_entry = []
|
|
233
298
|
# Check to see if domain residue is valid, if so, we can assign the proper index.
|
|
234
|
-
if
|
|
299
|
+
if domain_aa not in invalid_chars:
|
|
235
300
|
mapping_entry.append(domain_start)
|
|
236
301
|
domain_start += 1
|
|
237
302
|
else:
|
|
238
303
|
mapping_entry.append(pd.NA)
|
|
239
304
|
# Assign the values of the residues mapped together.
|
|
240
|
-
mapping_entry.append(
|
|
241
|
-
mapping_entry.append(
|
|
305
|
+
mapping_entry.append(domain_aa)
|
|
306
|
+
mapping_entry.append(protein_aa)
|
|
242
307
|
# Check to see if protein residue is valid, if so, we can assign the proper index.
|
|
243
|
-
if
|
|
308
|
+
if protein_aa not in invalid_chars:
|
|
244
309
|
mapping_entry.append(protein_start)
|
|
245
310
|
protein_start += 1
|
|
246
311
|
else:
|
|
@@ -282,7 +347,7 @@ class ResidueAlignment:
|
|
|
282
347
|
89
|
|
283
348
|
"""
|
|
284
349
|
# Read the alignment file and parse the important information from each alignment entry.
|
|
285
|
-
alignment_entries = ResidueAlignment.
|
|
350
|
+
alignment_entries = ResidueAlignment._read_align_file(align_filepath)
|
|
286
351
|
hmm_entry, protein_entry = alignment_entries
|
|
287
352
|
domain_name, domain_start, domain_text, _ = hmm_entry
|
|
288
353
|
protein_name, protein_start, protein_text, _ = protein_entry
|
|
@@ -294,7 +359,7 @@ class ResidueAlignment:
|
|
|
294
359
|
return ResidueAlignment(domain_name, protein_name, domain_start, protein_start, domain_text, protein_text)
|
|
295
360
|
|
|
296
361
|
@staticmethod
|
|
297
|
-
def
|
|
362
|
+
def _read_align_file(align_filepath: str) -> list[list[str]]:
|
|
298
363
|
"""
|
|
299
364
|
Reads standard align file, where a scan file is selected for a particular domain and processed into an align file format. Details are present in produce_align_from_scan().
|
|
300
365
|
|
|
@@ -770,9 +835,9 @@ class MMCIFInformation(StructureInformation):
|
|
|
770
835
|
self.full_sequences = pdbx.get_sequence(pdbx_file)
|
|
771
836
|
non_hetero_structure = self.structure[self.structure.hetero == False]
|
|
772
837
|
self.non_missing_sequences = {str(chain): str(sequence) for (chain, sequence) in list(zip(struc.get_chains(non_hetero_structure), struc.to_sequence(non_hetero_structure)[0]))}
|
|
773
|
-
self.
|
|
838
|
+
self._generate_auth_info()
|
|
774
839
|
|
|
775
|
-
def
|
|
840
|
+
def _generate_auth_info(self) -> None:
|
|
776
841
|
"""
|
|
777
842
|
Ran as part of constructor function. Generates information needed to access auth information including auth_seq_id and auth_asym_id, which correspond to alternative chain ids and alternative residue indices.
|
|
778
843
|
|
|
@@ -786,28 +851,25 @@ class MMCIFInformation(StructureInformation):
|
|
|
786
851
|
"""
|
|
787
852
|
if len(self.pdbx_file.keys()) > 0:
|
|
788
853
|
self.first_block = list(self.pdbx_file)[0]
|
|
789
|
-
|
|
854
|
+
atom_site_category = self.pdbx_file[self.first_block].get('atom_site')
|
|
790
855
|
self.chain_auth_dict: dict[str, str] = {}
|
|
791
856
|
self.auth_chain_dict: dict[str, str] = {}
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
if self.atom_site_category:
|
|
857
|
+
if atom_site_category:
|
|
795
858
|
categories = ['group_PDB', 'label_seq_id', 'label_asym_id', 'auth_seq_id', 'auth_asym_id', 'pdbx_PDB_model_num']
|
|
796
|
-
atom_site_data = np.column_stack([
|
|
859
|
+
atom_site_data = np.column_stack([atom_site_category[category].as_array() for category in categories])
|
|
797
860
|
_, idx = np.unique(atom_site_data, axis=0, return_index=True)
|
|
798
861
|
atom_site_data = atom_site_data[np.sort(idx)]
|
|
799
|
-
|
|
800
|
-
self.
|
|
801
|
-
self.unique_chains = np.unique(self.atom_data[:,2])
|
|
862
|
+
atom_data = atom_site_data[atom_site_data[:,0] == "ATOM"]
|
|
863
|
+
self.unique_chains = np.unique(atom_data[:,2])
|
|
802
864
|
for unique_chain in self.unique_chains:
|
|
803
|
-
unique_entry =
|
|
865
|
+
unique_entry = atom_data[atom_data[:,2] == unique_chain][0]
|
|
804
866
|
self.chain_auth_dict[unique_entry[2]] = unique_entry[4]
|
|
805
867
|
self.auth_chain_dict[unique_entry[4]] = unique_entry[2]
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
def get_start_res_id(self, chain_id: str,
|
|
868
|
+
self.atom_site_df = pd.DataFrame(np.column_stack([atom_site_category[category].as_array() for category in atom_site_category.keys()]), columns=atom_site_category.keys())
|
|
869
|
+
type_conversion_dict = {'label_seq_id': 'int64', 'auth_seq_id': 'int64', 'id': 'int64', 'Cartn_x': 'float', 'Cartn_y': 'float','Cartn_z': 'float', 'B_iso_or_equiv': 'float'}
|
|
870
|
+
self.atom_df = self.atom_site_df[self.atom_site_df['group_PDB'] == 'ATOM'].astype(type_conversion_dict)
|
|
871
|
+
|
|
872
|
+
def get_start_res_id(self, chain_id: str, get_auth_res_ids: bool=False, auth_chain_id_supplied: bool=False) -> int:
|
|
811
873
|
"""
|
|
812
874
|
Gets starting residue id of the specified chain excluding heteroatom group entries.
|
|
813
875
|
|
|
@@ -815,7 +877,7 @@ class MMCIFInformation(StructureInformation):
|
|
|
815
877
|
----------
|
|
816
878
|
chain_id : str
|
|
817
879
|
The chain id supplied and selected for from the structure.
|
|
818
|
-
|
|
880
|
+
get_auth_res_ids : bool
|
|
819
881
|
True if you want alt_ids for residues indices, False if cif residue indexing is needed.
|
|
820
882
|
auth_chain_id_supplied : bool
|
|
821
883
|
If True, the chain_id supplied is the auth chain id found on the RCSB website.
|
|
@@ -826,11 +888,13 @@ class MMCIFInformation(StructureInformation):
|
|
|
826
888
|
The residue id of the first atom in the chain provided.
|
|
827
889
|
"""
|
|
828
890
|
if auth_chain_id_supplied:
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
891
|
+
chain_df = self.atom_df[self.atom_df['auth_asym_id'] == chain_id]
|
|
892
|
+
else:
|
|
893
|
+
chain_df = self.atom_df[self.atom_df['label_asym_id'] == chain_id]
|
|
894
|
+
if get_auth_res_ids:
|
|
895
|
+
return chain_df['auth_seq_id'][0]
|
|
832
896
|
else:
|
|
833
|
-
return
|
|
897
|
+
return chain_df['label_seq_id'][0]
|
|
834
898
|
|
|
835
899
|
def get_full_sequence(self, chain_id: str, auth_chain_id_supplied: bool=False) -> str:
|
|
836
900
|
"""
|
|
@@ -875,7 +939,7 @@ class MMCIFInformation(StructureInformation):
|
|
|
875
939
|
else:
|
|
876
940
|
return self.non_missing_sequences[chain_id]
|
|
877
941
|
|
|
878
|
-
def get_chain_specific_structure(self, ca_only: bool,
|
|
942
|
+
def get_chain_specific_structure(self, ca_only: bool, chain_id: str, remove_hetero=True, auth_chain_id_supplied: bool=False):
|
|
879
943
|
"""
|
|
880
944
|
Subsets structure attribute to select for chain specific portions of the structure.
|
|
881
945
|
|
|
@@ -883,10 +947,8 @@ class MMCIFInformation(StructureInformation):
|
|
|
883
947
|
----------
|
|
884
948
|
ca_only : bool
|
|
885
949
|
If true, the structure will also be subsetted for atom entries where the atom_name annotation is "CA" (referring to alpha-carbons)
|
|
886
|
-
|
|
887
|
-
|
|
888
|
-
chain2 : str
|
|
889
|
-
Chain id corresponding to the second column of residues in the structure.
|
|
950
|
+
chain_id : str
|
|
951
|
+
The name of the chain to be selected for within the structure.
|
|
890
952
|
remove_hetero : bool, default=True
|
|
891
953
|
If true, the structure will also be subsetted for atom entries where the hetero annotation is False, thus removing heteroatoms.
|
|
892
954
|
auth_chain_id_supplied : bool
|
|
@@ -898,9 +960,7 @@ class MMCIFInformation(StructureInformation):
|
|
|
898
960
|
Two AtomArrays that refer to atoms in the first chain and second chain, respectively without accounting for the presence of heteroatoms if `remove_hetero` is True.
|
|
899
961
|
"""
|
|
900
962
|
if auth_chain_id_supplied:
|
|
901
|
-
|
|
902
|
-
chain2 = self.auth_chain_dict[chain2]
|
|
903
|
-
|
|
963
|
+
chain_id = self.auth_chain_dict[chain_id]
|
|
904
964
|
selected_structure = self.structure
|
|
905
965
|
if remove_hetero:
|
|
906
966
|
# Remove hetero atoms via hetero column of structure ndarray
|
|
@@ -908,64 +968,110 @@ class MMCIFInformation(StructureInformation):
|
|
|
908
968
|
if ca_only:
|
|
909
969
|
# Consider selection of alpha-carbon atoms only
|
|
910
970
|
selected_structure = selected_structure[selected_structure.atom_name == "CA"]
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
def generate_dist_matrix(self, ca_only: bool, chain1: str, chain2: str, auth_chain_id_supplied: bool=False):
|
|
971
|
+
chain_structure = selected_structure[selected_structure.chain_id == chain_id]
|
|
972
|
+
return chain_structure
|
|
973
|
+
|
|
974
|
+
def get_chain_site_data(self, ca_only: bool, chain_id: str, remove_hetero=True, auth_chain_id_supplied: bool=False):
|
|
916
975
|
"""
|
|
917
|
-
|
|
976
|
+
Subsets the atom_site dataframe to get atom information where the conditions are met.
|
|
918
977
|
|
|
919
978
|
Parameters
|
|
920
979
|
----------
|
|
921
980
|
ca_only : bool
|
|
922
|
-
If
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
|
|
926
|
-
|
|
981
|
+
If true, the dataframe will also be subsetted for atom entries where the label_atom_id annotation is "CA" (referring to alpha-carbons)
|
|
982
|
+
chain_id : str
|
|
983
|
+
The name of the chain to be selected for within the dataframe.
|
|
984
|
+
remove_hetero : bool, default=True
|
|
985
|
+
If true, the dataframe will also be subsetted for atom entries where the group_PDB annotation is ATOM rather than HETATM, thus removing heteroatoms.
|
|
986
|
+
auth_chain_id_supplied : bool
|
|
987
|
+
If True, the chain_id supplied is the auth chain id found on the RCSB website.
|
|
988
|
+
"""
|
|
989
|
+
atom_df = self.atom_df.copy()
|
|
990
|
+
if ca_only:
|
|
991
|
+
atom_df = atom_df[atom_df['label_atom_id'] == 'CA']
|
|
992
|
+
if remove_hetero:
|
|
993
|
+
atom_df = atom_df[atom_df['group_PDB'] == 'ATOM']
|
|
994
|
+
if auth_chain_id_supplied:
|
|
995
|
+
return atom_df[atom_df['auth_asym_id'] == chain_id]
|
|
996
|
+
else:
|
|
997
|
+
return atom_df[atom_df['label_asym_id'] == chain_id]
|
|
998
|
+
|
|
999
|
+
def get_seq_id_mapping(self, chain_id: str, seq_to_auth: bool, auth_chain_id_supplied: bool=False) -> dict[int, int]:
|
|
1000
|
+
"""
|
|
1001
|
+
Gets mapping from auth seq ids to label seq ids or vice-versa.
|
|
1002
|
+
|
|
1003
|
+
Parameters
|
|
1004
|
+
----------
|
|
1005
|
+
chain_id : str
|
|
1006
|
+
Chain id of the chain addressed for determining residue index mappings.
|
|
1007
|
+
seq_to_auth : bool
|
|
1008
|
+
If True, this indicates the mapping uses the label_seq_id as a key and the auth_seq_id as a value. Otherwise, keys and values are switched.
|
|
927
1009
|
auth_chain_id_supplied : bool
|
|
928
1010
|
If True, the chain_id supplied is the auth chain id found on the RCSB website.
|
|
929
1011
|
|
|
930
1012
|
Returns
|
|
931
1013
|
-------
|
|
932
|
-
|
|
933
|
-
|
|
1014
|
+
dict of int, int
|
|
1015
|
+
Dictionary with either label seq id or auth seq id as a key and the other as a value. The directionality is dependent on seq_to_auth.
|
|
934
1016
|
"""
|
|
935
|
-
|
|
936
|
-
|
|
937
|
-
|
|
1017
|
+
chain_df = self.get_chain_site_data(ca_only=True, chain_id=chain_id, remove_hetero=True, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
1018
|
+
if seq_to_auth:
|
|
1019
|
+
return dict(zip(chain_df['label_seq_id'], chain_df['auth_seq_id']))
|
|
1020
|
+
else:
|
|
1021
|
+
return dict(zip(chain_df['auth_seq_id'], chain_df['label_seq_id']))
|
|
938
1022
|
|
|
939
|
-
def
|
|
1023
|
+
def get_valid_chain_residues(self, chain_id: str, auth_seq_id: bool=False, auth_chain_id_supplied: bool=False) -> list[tuple[int, str]]:
|
|
940
1024
|
"""
|
|
941
|
-
|
|
1025
|
+
Gets valid indexing for residues of a specified chain. This is directly analogous to get_non_missing_sequence, does not contain missing residues, and provides the corresponding indices as well.
|
|
1026
|
+
|
|
1027
|
+
Parameters
|
|
1028
|
+
----------
|
|
1029
|
+
chain_id : str
|
|
1030
|
+
Chain id of the chain to be selected from the structure. This chain's sequence and corresponding residue indices are what are exclusively selected for.
|
|
1031
|
+
auth_seq_id: bool
|
|
1032
|
+
If True, the seq_ids that are the first element of the tuples in the returned list are auth_seq_ids.
|
|
1033
|
+
auth_chain_id_supplied : bool
|
|
1034
|
+
If True, the chain_id supplied is the auth chain id found on the RCSB website.
|
|
942
1035
|
|
|
1036
|
+
Returns
|
|
1037
|
+
-------
|
|
1038
|
+
list of tuple of int, str
|
|
1039
|
+
A list of residue information in sequential order reflecting the structure. The list consists of tuple elements where each tuple is the residue index and its corresponding one-letter amino acid.
|
|
1040
|
+
"""
|
|
1041
|
+
chain_structure = self.get_chain_specific_structure(ca_only=True, chain_id=chain_id, remove_hetero=True, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
1042
|
+
res_ids = chain_structure.res_id.tolist()
|
|
1043
|
+
res_names = chain_structure.res_name
|
|
1044
|
+
if auth_seq_id:
|
|
1045
|
+
seq_id_mapping = self.get_seq_id_mapping(chain_id=chain_id, seq_to_auth=True, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
1046
|
+
auth_res_ids = [seq_id_mapping[res_id] for res_id in res_ids]
|
|
1047
|
+
return list(zip(auth_res_ids, map(lambda symbol: ProteinSequence.convert_letter_3to1(symbol), res_names)))
|
|
1048
|
+
else:
|
|
1049
|
+
return list(zip(res_ids, map(lambda symbol: ProteinSequence.convert_letter_3to1(symbol), res_names)))
|
|
1050
|
+
|
|
1051
|
+
def generate_dist_matrix(self, ca_only: bool, chain1: str, chain2: str, auth_chain_id_supplied: bool=False):
|
|
1052
|
+
"""
|
|
1053
|
+
Generates distance matrix between two chains in the structure attribute.
|
|
1054
|
+
|
|
943
1055
|
Parameters
|
|
944
1056
|
----------
|
|
1057
|
+
ca_only : bool
|
|
1058
|
+
If True, only atoms that have the name "CA" are selected in the chains the distance matrix is calculated between.
|
|
945
1059
|
chain1 : str
|
|
946
|
-
|
|
1060
|
+
Chain id corresponding to the first column of residues in the structure.
|
|
947
1061
|
chain2 : str
|
|
948
|
-
|
|
1062
|
+
Chain id corresponding to the first column of residues in the structure.
|
|
949
1063
|
auth_chain_id_supplied : bool
|
|
950
1064
|
If True, the chain_id supplied is the auth chain id found on the RCSB website.
|
|
951
1065
|
|
|
952
1066
|
Returns
|
|
953
1067
|
-------
|
|
954
|
-
|
|
955
|
-
Tuple containing
|
|
1068
|
+
tuple of biotite.structure.AtomArray, biotite.structure.AtomArray, numpy.ndarray
|
|
1069
|
+
Tuple containing the chain 1 structure, the chain 2 structure, and the distance matrix of chain 1 and chain 2's pairwise distances.
|
|
956
1070
|
"""
|
|
957
|
-
|
|
958
|
-
|
|
959
|
-
|
|
960
|
-
|
|
961
|
-
shift1 = 0
|
|
962
|
-
shift2 = 0
|
|
963
|
-
if self.atom_site_category:
|
|
964
|
-
shift1 = self.res_auth_dict[chain1][1] - self.res_auth_dict[chain1][0]
|
|
965
|
-
shift2 = self.res_auth_dict[chain2][1] - self.res_auth_dict[chain2][0]
|
|
966
|
-
return shift1, shift2
|
|
967
|
-
else:
|
|
968
|
-
return shift1, shift2
|
|
1071
|
+
chain1_structure = self.get_chain_specific_structure(ca_only=ca_only, chain_id=chain1, remove_hetero=True, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
1072
|
+
chain2_structure = self.get_chain_specific_structure(ca_only=ca_only, chain_id=chain2, remove_hetero=True, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
1073
|
+
dist_matrix = cdist(chain1_structure.coord, chain2_structure.coord)
|
|
1074
|
+
return (chain1_structure, chain2_structure, dist_matrix)
|
|
969
1075
|
|
|
970
1076
|
def get_min_dist_atom_info(self, pairs: npt.NDArray, chain1: str, chain2: str, auth_chain_id_supplied: bool=False) -> npt.NDArray:
|
|
971
1077
|
"""
|
|
@@ -987,8 +1093,8 @@ class MMCIFInformation(StructureInformation):
|
|
|
987
1093
|
min_dist_pairs_atoms_arr : numpy.ndarray
|
|
988
1094
|
Structured ndarray that has residue indices, auth residue indices (corresponding to the protein numbering), and atomic names in the format {'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,int,int,'<U10','<U10']}
|
|
989
1095
|
"""
|
|
990
|
-
|
|
991
|
-
|
|
1096
|
+
chain1_structure = self.get_chain_specific_structure(ca_only=False, chain_id=chain1, remove_hetero=True, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
1097
|
+
chain2_structure = self.get_chain_specific_structure(ca_only=False, chain_id=chain2, remove_hetero=True, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
992
1098
|
min_dist_pairs_atoms = []
|
|
993
1099
|
for row in pairs:
|
|
994
1100
|
# Obtain structure information for chains 1 and 2
|
|
@@ -1001,13 +1107,15 @@ class MMCIFInformation(StructureInformation):
|
|
|
1001
1107
|
ind = np.unravel_index(np.argmin(dist_matrix), dist_matrix.shape)
|
|
1002
1108
|
# Use the indices to access the atom in the atom array and get the correct atom name.
|
|
1003
1109
|
# Generate the auth ids of the residues in the pairs ndarray
|
|
1004
|
-
|
|
1005
|
-
|
|
1110
|
+
seq_mapping_chain1 = self.get_seq_id_mapping(chain_id=chain1, seq_to_auth=True, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
1111
|
+
seq_mapping_chain2 = self.get_seq_id_mapping(chain_id=chain2, seq_to_auth=True, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
1112
|
+
auth_res_id1 = seq_mapping_chain1[row['residue1']]
|
|
1113
|
+
auth_res_id2 = seq_mapping_chain2[row['residue2']]
|
|
1006
1114
|
min_dist_pairs_atoms.append((row['residue1'], row['residue2'], auth_res_id1, auth_res_id2, chain1_res1_structure[ind[0]].atom_name, chain2_res2_structure[ind[1]].atom_name))
|
|
1007
1115
|
min_dist_pairs_atoms_arr = np.array(min_dist_pairs_atoms, dtype={'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,int,int,'<U10','<U10']})
|
|
1008
1116
|
return min_dist_pairs_atoms_arr
|
|
1009
1117
|
|
|
1010
|
-
def get_contacts(self, ca_only: bool, threshold: float, chain1: str, chain2: str,
|
|
1118
|
+
def get_contacts(self, ca_only: bool, threshold: float, chain1: str, chain2: str, auth_seq_id: bool=False, auth_chain_id_supplied: bool=False) -> set[tuple[int, int]]:
|
|
1011
1119
|
"""
|
|
1012
1120
|
Get contacts from the structure attribute where the distance between two residues is less than the threshold.
|
|
1013
1121
|
|
|
@@ -1021,8 +1129,8 @@ class MMCIFInformation(StructureInformation):
|
|
|
1021
1129
|
Chain id corresponding to the first column of residues in the structure.
|
|
1022
1130
|
chain2 : str
|
|
1023
1131
|
Chain id corresponding to the second column of residues in the structure.
|
|
1024
|
-
|
|
1025
|
-
True if you want
|
|
1132
|
+
auth_seq_id : bool
|
|
1133
|
+
True if you want auth_seq_ids for residues indices, False if cif residue indexing is needed.
|
|
1026
1134
|
auth_chain_id_supplied : bool
|
|
1027
1135
|
If True, the chain_id supplied is the auth chain id found on the RCSB website.
|
|
1028
1136
|
|
|
@@ -1031,8 +1139,9 @@ class MMCIFInformation(StructureInformation):
|
|
|
1031
1139
|
contacts_set : set of tuple of ints
|
|
1032
1140
|
Set of contacts, tuples with "residue1" and "residue2" from the structure that are within the distance threshold.
|
|
1033
1141
|
"""
|
|
1034
|
-
|
|
1035
1142
|
chain1_structure, chain2_structure, dist_matrix = self.generate_dist_matrix(ca_only, chain1, chain2, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
1143
|
+
seq_mapping_chain1 = self.get_seq_id_mapping(chain_id=chain1, seq_to_auth=True, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
1144
|
+
seq_mapping_chain2 = self.get_seq_id_mapping(chain_id=chain2, seq_to_auth=True, auth_chain_id_supplied=auth_chain_id_supplied)
|
|
1036
1145
|
thresh_ind = np.argwhere(dist_matrix <= threshold)
|
|
1037
1146
|
contacts_set = set()
|
|
1038
1147
|
for indices in thresh_ind:
|
|
@@ -1041,9 +1150,8 @@ class MMCIFInformation(StructureInformation):
|
|
|
1041
1150
|
res1 = chain1_atom.res_id
|
|
1042
1151
|
res2 = chain2_atom.res_id
|
|
1043
1152
|
if not(chain1==chain2 and res1 >= res2):
|
|
1044
|
-
if
|
|
1045
|
-
|
|
1046
|
-
contacts_set.add((res1 + shift1, res2 + shift2))
|
|
1153
|
+
if auth_seq_id:
|
|
1154
|
+
contacts_set.add((seq_mapping_chain1[res1], seq_mapping_chain2[res2]))
|
|
1047
1155
|
else:
|
|
1048
1156
|
contacts_set.add((res1, res2))
|
|
1049
1157
|
return contacts_set
|
|
@@ -1111,7 +1219,7 @@ class PDBInformation(StructureInformation):
|
|
|
1111
1219
|
"""
|
|
1112
1220
|
return self.non_missing_sequences[chain_id]
|
|
1113
1221
|
|
|
1114
|
-
def get_chain_specific_structure(self, ca_only: bool,
|
|
1222
|
+
def get_chain_specific_structure(self, ca_only: bool, chain_id: str, remove_hetero=True):
|
|
1115
1223
|
"""
|
|
1116
1224
|
Subsets structure attribute to select for chain specific portions of the structure.
|
|
1117
1225
|
|
|
@@ -1131,7 +1239,6 @@ class PDBInformation(StructureInformation):
|
|
|
1131
1239
|
tuple of biotite.structure.AtomArray, biotite.structure.AtomArray
|
|
1132
1240
|
Two AtomArrays that refer to atoms in the first chain and second chain, respectively without accounting for the presence of heteroatoms if `remove_hetero` is True.
|
|
1133
1241
|
"""
|
|
1134
|
-
|
|
1135
1242
|
selected_structure = self.structure
|
|
1136
1243
|
if remove_hetero:
|
|
1137
1244
|
# Remove hetero atoms via hetero column of structure ndarray
|
|
@@ -1139,10 +1246,26 @@ class PDBInformation(StructureInformation):
|
|
|
1139
1246
|
if ca_only:
|
|
1140
1247
|
# Consider selection of alpha-carbon atoms only
|
|
1141
1248
|
selected_structure = selected_structure[selected_structure.atom_name == "CA"]
|
|
1142
|
-
|
|
1143
|
-
|
|
1144
|
-
return (chain1_structure, chain2_structure)
|
|
1249
|
+
chain_structure = selected_structure[selected_structure.chain_id == chain_id]
|
|
1250
|
+
return chain_structure
|
|
1145
1251
|
|
|
1252
|
+
def get_valid_chain_residues(self, chain_id: str) -> list[tuple[int, str]]:
|
|
1253
|
+
"""
|
|
1254
|
+
Gets valid indexing for residues of a specified chain. This is directly analogous to get_non_missing_sequence, does not contain missing residues, and provides the corresponding indices as well.
|
|
1255
|
+
|
|
1256
|
+
Parameters
|
|
1257
|
+
----------
|
|
1258
|
+
chain_id : str
|
|
1259
|
+
Chain id of the chain to be selected from the structure. This chain's sequence and corresponding residue indices are what are exclusively selected for.
|
|
1260
|
+
|
|
1261
|
+
Returns
|
|
1262
|
+
-------
|
|
1263
|
+
list of tuple of int, str
|
|
1264
|
+
A list of residue information in sequential order reflecting the structure. The list consists of tuple elements where each tuple is the residue index and its corresponding one-letter amino acid.
|
|
1265
|
+
"""
|
|
1266
|
+
chain_structure = self.get_chain_specific_structure(ca_only=True, chain_id=chain_id, remove_hetero=True)
|
|
1267
|
+
return list(zip(chain_structure.res_id.tolist(), map(lambda symbol: ProteinSequence.convert_letter_3to1(symbol), chain_structure.res_name)))
|
|
1268
|
+
|
|
1146
1269
|
def generate_dist_matrix(self, ca_only: bool, chain1: str, chain2: str):
|
|
1147
1270
|
"""
|
|
1148
1271
|
Generates distance matrix between two chains in the structure attribute.
|
|
@@ -1161,38 +1284,11 @@ class PDBInformation(StructureInformation):
|
|
|
1161
1284
|
tuple of biotite.structure.AtomArray, biotite.structure.AtomArray, numpy.ndarray
|
|
1162
1285
|
Tuple containing the chain 1 structure, the chain 2 structure, and the distance matrix of chain 1 and chain 2's pairwise distances.
|
|
1163
1286
|
"""
|
|
1164
|
-
chain1_structure
|
|
1287
|
+
chain1_structure = self.get_chain_specific_structure(ca_only=ca_only, chain_id=chain1, remove_hetero=True)
|
|
1288
|
+
chain2_structure = self.get_chain_specific_structure(ca_only=ca_only, chain_id=chain2, remove_hetero=True)
|
|
1165
1289
|
dist_matrix = cdist(chain1_structure.coord, chain2_structure.coord)
|
|
1166
1290
|
return (chain1_structure, chain2_structure, dist_matrix)
|
|
1167
1291
|
|
|
1168
|
-
def get_shift_values(self, chain1: str, chain2: str) -> tuple[int, int]:
|
|
1169
|
-
"""
|
|
1170
|
-
Get shift values needed for production of auth residue ids.
|
|
1171
|
-
|
|
1172
|
-
Parameters
|
|
1173
|
-
----------
|
|
1174
|
-
chain1 : str
|
|
1175
|
-
Name of the chain id present in the struct_ref_seq block of cif files referring to the second column of residues.
|
|
1176
|
-
chain2 : str
|
|
1177
|
-
Name of the chain id present in the struct_ref_seq block of cif files referring to the second column of residues.
|
|
1178
|
-
|
|
1179
|
-
Returns
|
|
1180
|
-
-------
|
|
1181
|
-
(shift1, shift2) : tuple of int, int
|
|
1182
|
-
Tuple containing both shift values, the difference between the auth_res_id and res_id.
|
|
1183
|
-
"""
|
|
1184
|
-
non_hetero_structure = self.structure[self.structure.hetero == False]
|
|
1185
|
-
shift1 = 0
|
|
1186
|
-
shift2 = 0
|
|
1187
|
-
if chain1 in self.unique_chains and chain2 in self.unique_chains:
|
|
1188
|
-
shift1 = non_hetero_structure[non_hetero_structure.chain_id == chain1][0].res_id - 1
|
|
1189
|
-
shift2 = non_hetero_structure[non_hetero_structure.chain_id == chain2][0].res_id - 1
|
|
1190
|
-
shift1 *= -1
|
|
1191
|
-
shift2 *= -1
|
|
1192
|
-
return shift1, shift2
|
|
1193
|
-
else:
|
|
1194
|
-
return shift1, shift2
|
|
1195
|
-
|
|
1196
1292
|
def get_min_dist_atom_info(self, pairs: npt.NDArray, chain1: str, chain2: str) -> npt.NDArray:
|
|
1197
1293
|
"""
|
|
1198
1294
|
Generate a ndarray of residue ids and their corresponding atom names such that the distance is the minimum between the initial residues provided.
|
|
@@ -1211,8 +1307,8 @@ class PDBInformation(StructureInformation):
|
|
|
1211
1307
|
min_dist_pairs_atoms_arr : numpy.ndarray
|
|
1212
1308
|
Structured ndarray that has residue indices, auth residue indices (corresponding to the protein numbering), and atomic names in the format {'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,int,int,'<U10','<U10']}
|
|
1213
1309
|
"""
|
|
1214
|
-
|
|
1215
|
-
|
|
1310
|
+
chain1_structure = self.get_chain_specific_structure(ca_only=False, chain_id=chain1, remove_hetero=True)
|
|
1311
|
+
chain2_structure = self.get_chain_specific_structure(ca_only=False, chain_id=chain2, remove_hetero=True)
|
|
1216
1312
|
min_dist_pairs_atoms = []
|
|
1217
1313
|
for row in pairs:
|
|
1218
1314
|
# Obtain structure information for chains 1 and 2
|
|
@@ -1223,15 +1319,11 @@ class PDBInformation(StructureInformation):
|
|
|
1223
1319
|
dist_matrix = cdist(chain1_res1_structure.coord, chain2_res2_structure.coord)
|
|
1224
1320
|
|
|
1225
1321
|
ind = np.unravel_index(np.argmin(dist_matrix), dist_matrix.shape)
|
|
1226
|
-
|
|
1227
|
-
# Generate the auth ids of the residues in the pairs ndarray
|
|
1228
|
-
orig_res_id1 = row['residue1'] + shift1
|
|
1229
|
-
orig_res_id2 = row['residue2'] + shift2
|
|
1230
|
-
min_dist_pairs_atoms.append((orig_res_id1, orig_res_id2, row['residue1'], row['residue2'], chain1_res1_structure[ind[0]].atom_name, chain2_res2_structure[ind[1]].atom_name))
|
|
1322
|
+
min_dist_pairs_atoms.append((row['residue1'], row['residue2'], row['residue1'], row['residue2'], chain1_res1_structure[ind[0]].atom_name, chain2_res2_structure[ind[1]].atom_name))
|
|
1231
1323
|
min_dist_pairs_atoms_arr = np.array(min_dist_pairs_atoms, dtype={'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,int,int,'<U10','<U10']})
|
|
1232
1324
|
return min_dist_pairs_atoms_arr
|
|
1233
1325
|
|
|
1234
|
-
def get_contacts(self, ca_only: bool, threshold: float, chain1: str, chain2: str
|
|
1326
|
+
def get_contacts(self, ca_only: bool, threshold: float, chain1: str, chain2: str) -> set[tuple[int, int]]:
|
|
1235
1327
|
"""
|
|
1236
1328
|
Get contacts from the structure attribute where the distance between two residues is less than the threshold.
|
|
1237
1329
|
|
|
@@ -1245,8 +1337,6 @@ class PDBInformation(StructureInformation):
|
|
|
1245
1337
|
Chain id corresponding to the first column of residues in the structure.
|
|
1246
1338
|
chain2 : str
|
|
1247
1339
|
Chain id corresponding to the second column of residues in the structure.
|
|
1248
|
-
auth_contacts : bool
|
|
1249
|
-
True if you want alt_ids for residues indices, False if cif residue indexing is needed.
|
|
1250
1340
|
|
|
1251
1341
|
Returns
|
|
1252
1342
|
-------
|
|
@@ -1263,9 +1353,5 @@ class PDBInformation(StructureInformation):
|
|
|
1263
1353
|
res1 = chain1_atom.res_id
|
|
1264
1354
|
res2 = chain2_atom.res_id
|
|
1265
1355
|
if not(chain1==chain2 and res1 >= res2):
|
|
1266
|
-
|
|
1267
|
-
shift1, shift2 = self.get_shift_values(chain1, chain2)
|
|
1268
|
-
contacts_set.add((res1 + shift1, res2 + shift2))
|
|
1269
|
-
else:
|
|
1270
|
-
contacts_set.add((res1, res2))
|
|
1356
|
+
contacts_set.add((res1, res2))
|
|
1271
1357
|
return contacts_set
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: dcatoolkit
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: Collection of useful modules and representations for managing DCA output data.
|
|
5
5
|
Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
6
6
|
Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
@@ -52,6 +52,7 @@ Requires-Dist: pdoc; extra == "docs"
|
|
|
52
52
|
Requires-Dist: numpydoc; extra == "docs"
|
|
53
53
|
Provides-Extra: lint
|
|
54
54
|
Requires-Dist: ruffle; extra == "lint"
|
|
55
|
+
Dynamic: license-file
|
|
55
56
|
|
|
56
57
|
# dcatoolkit
|
|
57
58
|
Collection of useful modules and representations for managing DCA output data.
|
|
@@ -9,4 +9,6 @@ src/dcatoolkit.egg-info/SOURCES.txt
|
|
|
9
9
|
src/dcatoolkit.egg-info/dependency_links.txt
|
|
10
10
|
src/dcatoolkit.egg-info/requires.txt
|
|
11
11
|
src/dcatoolkit.egg-info/top_level.txt
|
|
12
|
-
tests/
|
|
12
|
+
tests/test_alignments.py
|
|
13
|
+
tests/test_contacts.py
|
|
14
|
+
tests/test_sequence_numbering.py
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
from context import ResidueAlignment
|
|
2
|
+
import pandas as pd
|
|
3
|
+
|
|
4
|
+
test_cases = """
|
|
5
|
+
Test Case 1
|
|
6
|
+
First: MA.KLT
|
|
7
|
+
Second: MAAKLT
|
|
8
|
+
|
|
9
|
+
Test Case 2
|
|
10
|
+
First: Q..EWLP
|
|
11
|
+
Second: QATEWLP
|
|
12
|
+
|
|
13
|
+
Test Case 3
|
|
14
|
+
First: D.G.H.V
|
|
15
|
+
Second: DAGAHAV
|
|
16
|
+
|
|
17
|
+
Test Case 4
|
|
18
|
+
First: TAKAPF
|
|
19
|
+
Second: TAMAPF
|
|
20
|
+
|
|
21
|
+
Test Case 5
|
|
22
|
+
First: FACARAA
|
|
23
|
+
Second: FACAR--
|
|
24
|
+
|
|
25
|
+
Test Case 6
|
|
26
|
+
First: ALMAY
|
|
27
|
+
Second: ALMAY
|
|
28
|
+
|
|
29
|
+
Test Case 7
|
|
30
|
+
First: VAIDTSK
|
|
31
|
+
Second: VAID--K
|
|
32
|
+
|
|
33
|
+
Test Case 8
|
|
34
|
+
First: NG.TA
|
|
35
|
+
Second: NG-TA
|
|
36
|
+
|
|
37
|
+
Test Case 9
|
|
38
|
+
First: STWLPL
|
|
39
|
+
Second: SAWLPL
|
|
40
|
+
|
|
41
|
+
Test Case 10
|
|
42
|
+
First: HCSTCRAAC
|
|
43
|
+
Second: H--TAR--C
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
test_cases = [
|
|
48
|
+
(47, 15, 'MA.KLT', 'MAAKLT'),
|
|
49
|
+
(39, 3, 'Q..EWLP', 'QATEWLP'),
|
|
50
|
+
(8, 45, 'D.G.H.V', 'DAGAHAV'),
|
|
51
|
+
(26, 19, 'TAKAPF', 'TAMAPF'),
|
|
52
|
+
(33, 9, 'FACARAA', 'FACAR--')
|
|
53
|
+
]
|
|
54
|
+
test_cases_validation = [
|
|
55
|
+
(42, 28, 'ALMAY', 'ALMAY'),
|
|
56
|
+
(14, 11, 'VAIDTSK', 'VAID--K'),
|
|
57
|
+
(10, 5, 'NG.TA', 'NG-TA'),
|
|
58
|
+
(2, 44, 'STWLPL', 'SAWLPL'),
|
|
59
|
+
(17, 34, 'HCSTCRAAC', 'H--TAR--C')
|
|
60
|
+
]
|
|
61
|
+
|
|
62
|
+
test_answers = [
|
|
63
|
+
[(47, 48, pd.NA, 49, 50, 51), tuple('MA.KLT'), tuple('MAAKLT'), (15,16,17,18,19,20)],
|
|
64
|
+
[(39, pd.NA, pd.NA, 40, 41, 42, 43), tuple('Q..EWLP'), tuple('QATEWLP'), (3,4,5,6,7,8,9)],
|
|
65
|
+
[(8, pd.NA, 9, pd.NA, 10, pd.NA, 11), tuple('D.G.H.V'), tuple('DAGAHAV'), (45,46,47,48,49,50,51)],
|
|
66
|
+
[(26, 27, 28, 29, 30, 31), tuple('TAKAPF'), tuple('TAMAPF'), (19,20,21,22,23,24)],
|
|
67
|
+
[(33, 34, 35, 36, 37, 38, 39), tuple('FACARAA'), tuple('FACAR--'), (9,10,11,12,13,pd.NA,pd.NA)],
|
|
68
|
+
[(42, 43, 44, 45, 46), tuple('ALMAY'), tuple('ALMAY'), (28,29,30,31,32)],
|
|
69
|
+
[(14, 15, 16, 17, 18, 19, 20), tuple('VAIDTSK'), tuple('VAID--K'), (11, 12, 13, 14, pd.NA, pd.NA, 15)],
|
|
70
|
+
[(10,11,pd.NA,12,13), tuple('NG.TA'), tuple('NG-TA'), (5,6,pd.NA,7,8)],
|
|
71
|
+
[(2,3,4,5,6,7), tuple('STWLPL'), tuple('SAWLPL'), (44,45,46,47,48,49)],
|
|
72
|
+
[(17,18,19,20,21,22,23,24,25), tuple('HCSTCRAAC'), tuple('H--TAR--C'), (34, pd.NA, pd.NA, 35,36,37, pd.NA, pd.NA, 38)]
|
|
73
|
+
]
|
|
74
|
+
|
|
75
|
+
def test_residue_alignments():
|
|
76
|
+
for test_num, test_case in enumerate(test_cases):
|
|
77
|
+
domain_start, protein_start, first_seq, second_seq = test_case
|
|
78
|
+
module_result = list(ResidueAlignment(f"Test_{test_num}", f"Test {test_num}", domain_start, protein_start, first_seq, second_seq).reference_mapping.itertuples(index=False, name=None))
|
|
79
|
+
answer = list(zip(*test_answers[test_num]))
|
|
80
|
+
assert module_result == answer
|
|
81
|
+
for test_num, test_case in enumerate(test_cases_validation, start=test_num+1):
|
|
82
|
+
domain_start, protein_start, first_seq, second_seq = test_case
|
|
83
|
+
module_result = list(ResidueAlignment(f"Test_{test_num}", f"Test {test_num}", domain_start, protein_start, first_seq, second_seq).reference_mapping.itertuples(index=False, name=None))
|
|
84
|
+
answer = list(zip(*test_answers[test_num]))
|
|
85
|
+
assert module_result == answer
|
|
86
|
+
|
|
87
|
+
# Can handle excess residues, but not missing any ones that are supposed to be there.
|
|
88
|
+
print(ResidueAlignment('name1', 'name2', 1, 1, 'MAAFT', 'MAAFT', valid_residues=[(5, 'M'), (6, 'A'), (7, 'A'), (8, 'R'), (12, 'F')]))
|
|
@@ -59,12 +59,12 @@ def check_contacts(test_CA: bool, threshold: float):
|
|
|
59
59
|
|
|
60
60
|
cif_file_contacts = read_contacts(corresponding_file)
|
|
61
61
|
chain1, auth_chain1, chain2, auth_chain2 = pdb_id_chain_map[pdb_id]
|
|
62
|
-
fetch_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.fetch_pdb(pdb_id, 'mmcif').get_contacts(test_CA, threshold, chain1, chain2,
|
|
63
|
-
read_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.read_mmCIF_file(str(cif_file)).get_contacts(test_CA, threshold, chain1, chain2,
|
|
64
|
-
fetch_authchain_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.fetch_pdb(pdb_id, 'mmcif').get_contacts(test_CA, threshold, auth_chain1, auth_chain2,
|
|
65
|
-
read_authchain_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.read_mmCIF_file(str(cif_file)).get_contacts(test_CA, threshold, auth_chain1, auth_chain2,
|
|
66
|
-
fetch_pdb_contacts = {(int(x[0]), int(x[1])) for x in PDBInformation.fetch_pdb(pdb_id, struc_format="pdb").get_contacts(test_CA, threshold, auth_chain1, auth_chain2
|
|
67
|
-
read_pdb_contacts = {(int(x[0]), int(x[1])) for x in PDBInformation.read_pdb_file(f"tests/pdb_info/{pdb_id.lower()}.pdb").get_contacts(test_CA, threshold, auth_chain1, auth_chain2
|
|
62
|
+
fetch_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.fetch_pdb(pdb_id, 'mmcif').get_contacts(test_CA, threshold, chain1, chain2, auth_seq_id=True)}
|
|
63
|
+
read_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.read_mmCIF_file(str(cif_file)).get_contacts(test_CA, threshold, chain1, chain2, auth_seq_id=True)}
|
|
64
|
+
fetch_authchain_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.fetch_pdb(pdb_id, 'mmcif').get_contacts(test_CA, threshold, auth_chain1, auth_chain2, auth_seq_id=True, auth_chain_id_supplied=True)}
|
|
65
|
+
read_authchain_cif_contacts = {(int(x[0]), int(x[1])) for x in MMCIFInformation.read_mmCIF_file(str(cif_file)).get_contacts(test_CA, threshold, auth_chain1, auth_chain2, auth_seq_id=True, auth_chain_id_supplied=True)}
|
|
66
|
+
fetch_pdb_contacts = {(int(x[0]), int(x[1])) for x in PDBInformation.fetch_pdb(pdb_id, struc_format="pdb").get_contacts(test_CA, threshold, auth_chain1, auth_chain2)}
|
|
67
|
+
read_pdb_contacts = {(int(x[0]), int(x[1])) for x in PDBInformation.read_pdb_file(f"tests/pdb_info/{pdb_id.lower()}.pdb").get_contacts(test_CA, threshold, auth_chain1, auth_chain2)}
|
|
68
68
|
|
|
69
69
|
|
|
70
70
|
fetch_cif_contacts = drop_inord_res(fetch_cif_contacts)
|
|
@@ -86,7 +86,7 @@ def check_contacts(test_CA: bool, threshold: float):
|
|
|
86
86
|
else:
|
|
87
87
|
compare(fetch_cif_contacts, fetch_pdb_contacts, read_cif_contacts, read_pdb_contacts)
|
|
88
88
|
print(f"{pdb_id} has no difference between cif and pdb reading.")
|
|
89
|
-
|
|
89
|
+
|
|
90
90
|
compare(fetch_cif_contacts, fetch_authchain_cif_contacts, read_cif_contacts, read_authchain_cif_contacts)
|
|
91
91
|
print(f"{pdb_id} has no issue reading with auth chains and asym chains.")
|
|
92
92
|
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
from context import MMCIFInformation, PDBInformation
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
import biotite.structure.io.pdbx as pdbx
|
|
4
|
+
import biotite.database.rcsb as rcsb
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
pdb_ids = ['1pzs', '3ddv', '6avj', '3d7i', '4OO8']
|
|
8
|
+
|
|
9
|
+
for index, AA in MMCIFInformation.fetch_pdb("4OO8", "mmcif").get_valid_chain_residues("A"):
|
|
10
|
+
if index > 30:
|
|
11
|
+
break
|
|
12
|
+
else:
|
|
13
|
+
pass
|
|
14
|
+
print(index, AA)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|