dcatoolkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
dcatoolkit/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+
2
+ __version__ = "0.1.0"
3
+ from .representation import Pairs, DirectInformationData, StructureInformation
@@ -0,0 +1,741 @@
1
+ import numpy as np
2
+ import pandas as pd
3
+ from scipy.spatial.distance import cdist
4
+
5
+ import biotite.structure.io.pdbx as pdbx
6
+ import biotite.database.rcsb as rcsb
7
+
8
+ from typing import Optional
9
+ import numpy.typing as npt
10
+
11
+
12
+ class Pairs:
13
+ """
14
+ Object that contains a representation (as an ndarray) of pairs of entities that are related. This may extend to Direct Information Pairs or Structural contacts, where each residue is one component of the pair.
15
+
16
+ Note
17
+ ----
18
+ Either a filepath or a ndarr has to be specified in order to produce a Pairs representation.
19
+
20
+ Parameters
21
+ ----------
22
+ filepath : str, optional
23
+ Filepath of the pairs in tabular representation, separated by whitespace between the pair components and newlines between each pair.
24
+ ndarr : numpy.ndarray, optional
25
+ Populated ndarray that contains pair information.
26
+ delimiter : str, optional
27
+ String used to specify separator between two pairs. See numpy.loadtxt() for details.
28
+
29
+ Attributes
30
+ ----------
31
+ pairs : numpy.ndarray
32
+ Ndarray representation of pairs supplied by the user. This is produced via the np.loadtxt() function.
33
+ """
34
+ def __init__(self, filepath: Optional[str]=None, ndarr: Optional[npt.NDArray]=None, delimiter: Optional[str]=None) -> None:
35
+ if (filepath is not None and ndarr is not None) or (filepath is None and ndarr is None):
36
+ raise Exception("Please specify either a filepath or a NumPy array to populate your pairs.")
37
+ elif filepath is not None:
38
+ if delimiter:
39
+ self.pairs = np.loadtxt(filepath, dtype=int, delimiter=delimiter)
40
+ else:
41
+ self.pairs = np.loadtxt(filepath, dtype=int)
42
+ elif ndarr is not None:
43
+ self.pairs = ndarr
44
+
45
+ @staticmethod
46
+ def mirror_diagonal(pairs: npt.NDArray) -> npt.NDArray:
47
+ """
48
+ Flip 2D ndarray with 2 columns columnwise. Flips pair positions for diagonal-mirrored representation.
49
+
50
+ Parameters
51
+ ----------
52
+ pairs : numpy.ndarray
53
+ 2d array with (n, 2) shape.
54
+
55
+ Returns
56
+ -------
57
+ numpy.ndarray
58
+ Values flipped along the column axis.
59
+ """
60
+ return np.flip(pairs, axis=1)
61
+
62
+ @staticmethod
63
+ def subset_pairs(pairs: npt.NDArray, number : Optional[int]=None) -> npt.NDArray:
64
+ """
65
+ Picks out a subset of 'number' pairs if a number is supplied. Otherwise, returns all pairs.
66
+
67
+ Parameters
68
+ ----------
69
+ pairs : numpy.ndarray
70
+ Ndarray to select number of rows from.
71
+ number : int, optional
72
+ Specific number of rows of pairs to subset.
73
+
74
+ Returns
75
+ -------
76
+ numpy.ndarray
77
+ Subset of pairs from rows 0 to number.
78
+ pairs : numpy.ndarray
79
+ All pairs specified from the parameters section.
80
+ """
81
+ if number is not None:
82
+ return pairs[:number, ]
83
+ else:
84
+ return pairs
85
+
86
+ @staticmethod
87
+ def mirror_pairs(pairs: npt.NDArray, mirror: bool=False) -> npt.NDArray:
88
+ """
89
+ Produces combined array of pairs and potentially their mirrored representation.
90
+
91
+ Parameters
92
+ ----------
93
+ pairs : numpy.ndarray
94
+ Ndarray to mirror and vertically append if mirror is set to True.
95
+ mirror : bool
96
+ Whether or not to append mirrored representation of pairs to the original pairs ndarray.
97
+
98
+ Returns
99
+ -------
100
+ mirrored_ndarray : numpy.ndarray
101
+ combined ndarray of pairs and mirrored pairs.
102
+ pairs : numpy.ndarray
103
+ The original pairs specified from the parameters section.
104
+ """
105
+ if mirror:
106
+ return np.vstack((pairs, Pairs.mirror_diagonal(pairs)))
107
+ else:
108
+ return pairs
109
+
110
+ @staticmethod
111
+ def get_pairs(pairs: npt.NDArray, mirror: bool=False, number: Optional[int]=None) -> npt.NDArray:
112
+ """
113
+ Returns pairs based on user specification, offering options to produce mirrored representation of pairs and to select a specific number of pairs.
114
+
115
+ Parameters
116
+ ----------
117
+ pairs : numpy.ndarray
118
+ ndarray of pairs to select from or to mirror.
119
+ mirror : bool
120
+ Whether or not to append mirrored representation of pairs to the original pairs ndarray.
121
+ number : int
122
+ Specific number of rows of pairs to subset.
123
+
124
+ Returns
125
+ -------
126
+ numpy.ndarray
127
+ mirrored, subset version of pairs produced via subset_pairs() and mirror_pairs() on pairs.
128
+ """
129
+ # Check to see if user requested mirrored pairs, if so, add in pairs that are mirrored across diagonal
130
+ pairs = Pairs.subset_pairs(pairs, number)
131
+ if mirror:
132
+ pairs = Pairs.mirror_pairs(pairs, mirror)
133
+ return pairs
134
+
135
+ class ResidueAlignment:
136
+ """
137
+ A representation of a residue alignment, often from a query HMM to a protein structure target sequence.
138
+
139
+ Parameters
140
+ ----------
141
+ domain_name : str
142
+ The name of the query HMM.
143
+ protein_name : str
144
+ The name of the target protein sequence.
145
+ domain_start : int
146
+ The starting index of the domain alignment in the query HMM.
147
+ protein_start : int
148
+ The starting index of the domain alignment in the protein target sequence.
149
+ domain_text : str
150
+ The sequence of the domain in the query HMM corresponding to this alignment.
151
+ protein_text : str
152
+ The sequence of the protein target sequence corresponding to this alignment.
153
+
154
+ Attributes
155
+ ----------
156
+ reference_mapping : pandas.DataFrame
157
+ The representation of the mapping where a row constitutes a residue pair and its indices in the format: 'domain_index', 'domain_residue', 'protein_residue', 'protein_index'.
158
+ domain_to_protein : dict[int, int]
159
+ A dictionary allowing for mapping from indices corresponding to the query HMM and Multiple Sequence Alignment to the protein target sequence.
160
+ protein_to_domain : dict[int, int]
161
+ A dictionary allowing for mapping from indices corresponding to the protein target sequence to the query HMM and Multiple Sequence Alignment.
162
+ """
163
+ def __init__(self, domain_name: str, protein_name: str, domain_start: int, protein_start: int, domain_text: str, protein_text: str) -> None:
164
+ self.domain_name = domain_name
165
+ self.protein_name = protein_name
166
+ self.set_reference_mapping(domain_start, protein_start, domain_text, protein_text)
167
+
168
+ def set_reference_mapping(self, domain_start: int, protein_start: int, domain_text: str, protein_text: str) -> None:
169
+ """
170
+ Set values for reference_mapping and mapping dictionaries, domain_to_protein and protein_to_domain.
171
+
172
+ Note
173
+ ----
174
+ For details on `domain_start`, `protein_start`, `domain_text`, `protein_text`, please refer to the `ResidueAlignment` docstring.
175
+
176
+ Returns
177
+ -------
178
+ None
179
+ """
180
+ # Convert text to list variant for iteration
181
+ domain_sequence = list(domain_text)
182
+ protein_sequence = list(protein_text)
183
+
184
+ mapping_entries = []
185
+
186
+ for i in range(len(domain_sequence)):
187
+ mapping_entry = []
188
+ # Check to see if domain residue is valid, if so, we can assign the proper index.
189
+ if domain_sequence[i] != '.':
190
+ mapping_entry.append(domain_start)
191
+ domain_start += 1
192
+ else:
193
+ mapping_entry.append(pd.NA)
194
+ # Assign the values of the residues mapped together.
195
+ mapping_entry.append(domain_sequence[i])
196
+ mapping_entry.append(protein_sequence[i])
197
+ # Check to see if protein residue is valid, if so, we can assign the proper index.
198
+ if protein_sequence[i] != '-':
199
+ mapping_entry.append(protein_start)
200
+ protein_start += 1
201
+ else:
202
+ mapping_entry.append(pd.NA)
203
+ # Store the resulting mapping in the reference map.
204
+ mapping_entries.append(mapping_entry)
205
+ self.reference_mapping = pd.DataFrame(mapping_entries, columns=['domain_index', 'domain_residue', 'protein_residue', 'protein_index'])
206
+ self.reference_mapping = self.reference_mapping.astype({'domain_index': pd.Int32Dtype(), 'protein_index': pd.Int32Dtype(), 'domain_residue': pd.StringDtype(), 'protein_residue': pd.StringDtype()})
207
+ reference_mapping_notna = self.reference_mapping.dropna()
208
+
209
+ self.domain_to_protein = dict(zip(reference_mapping_notna.domain_index, reference_mapping_notna.protein_index))
210
+ self.protein_to_domain = dict(zip(reference_mapping_notna.protein_index, reference_mapping_notna.domain_index))
211
+
212
+ @staticmethod
213
+ def load_from_align_file(align_filepath: str) -> 'ResidueAlignment':
214
+ """
215
+ Generate ResidueAlignment from a standard align file generated from HMM scan.
216
+
217
+ Parameters
218
+ ----------
219
+ align_filepath : str
220
+ Filepath of the align file generated from a scan file produced via hmmscan.
221
+
222
+ Returns
223
+ -------
224
+ ResidueAlignment
225
+ ResidueAlignment with domain and protein starting indices and corresponding sequence texts.
226
+
227
+ File Format
228
+ -----------
229
+ Domain_name
230
+ 1
231
+ XXXXXXXXXXXXXXXXXXXX
232
+ 20
233
+
234
+ Protein_name
235
+ 70
236
+ XXXXXXXXXXXXXXXXXXXX
237
+ 89
238
+ """
239
+ # Read the alignment file and parse the important information from each alignment entry.
240
+ alignment_entries = ResidueAlignment.read_align_file(align_filepath)
241
+ hmm_entry, protein_entry = alignment_entries
242
+ domain_name, domain_start, domain_text, _ = hmm_entry
243
+ protein_name, protein_start, protein_text, _ = protein_entry
244
+
245
+ # Convert to ints for iteration
246
+ domain_start = int(domain_start)
247
+ protein_start = int(protein_start)
248
+
249
+ return ResidueAlignment(domain_name, protein_name, domain_start, protein_start, domain_text, protein_text)
250
+
251
+ @staticmethod
252
+ def read_align_file(align_filepath: str) -> list[list[str]]:
253
+ """
254
+ Reads standard align file, where a scan file is selected for a particular domain and processed into an align file format. Details are present in produce_align_from_scan().
255
+
256
+ Parameters
257
+ ----------
258
+ align_filepath : str
259
+ Filepath and filename of alignment file that contains information on the domain / protein of interest and its mapping to a protein's structural sequence
260
+
261
+ Returns
262
+ -------
263
+ alignment_entries : list of list of strings
264
+ list of associated lines (one which corresponds to the HMM produced sequence and its indices and one that corresponds to the protein's seqeuence and its indices), which are also contained in a list.
265
+ """
266
+ with open(align_filepath, 'r') as fs:
267
+ alignment_entries: list[list[str]] = []
268
+ current_entry: list[str] = []
269
+ line_count = 0
270
+ for line in fs:
271
+ line = line.strip()
272
+ if line != '':
273
+ line_count += 1
274
+ current_entry.append(line)
275
+ if line_count == 4:
276
+ line_count = 0
277
+ alignment_entries.append(current_entry)
278
+ current_entry: list[str] = []
279
+ return alignment_entries
280
+
281
+ def __str__(self) -> str:
282
+ """
283
+ Returns string representation of the ResidueAlignment pandas DataFrame in tab-separated value (tsv) format.
284
+
285
+ Returns
286
+ -------
287
+ str
288
+ reference_mapping pandas DataFrame exported to TSV format via the to_csv(sep="\t") function from pandas.
289
+ """
290
+ return self.reference_mapping.to_csv(sep="\t")
291
+
292
+ class DirectInformationData:
293
+ """
294
+ Representation and interface for Direct Information data including residue indices for a pair and its corresponding DI value represented as a 3-column ndarray.
295
+
296
+ Parameters
297
+ ----------
298
+ structured_ndarray : numpy.ndarray
299
+ Ndarray with the shape (n,3) with dtype={'names': ('residue1', 'residue2', 'DI'), 'formats': (int, int, float, float)}
300
+
301
+ Attributes
302
+ ----------
303
+ DI_data : numpy.ndarray
304
+ The structured_ndarray in the parameters section where column 1 corresponds to a pair's first residue, column 2 corresponds to the pair's second residue, and column 3 corresponds to the Direct Information of the pair.
305
+ """
306
+ def __init__(self, structured_ndarray: npt.NDArray) -> None:
307
+ self.DI_data = structured_ndarray
308
+
309
+ @staticmethod
310
+ def load_from_dca_output(dca_filepath: str) -> 'DirectInformationData':
311
+ """
312
+ Function to generate a DirectInformationData object from the direct output of the MATLab dca function.
313
+
314
+ Parameters
315
+ ----------
316
+ dca_filepath : str
317
+ Filepath of the DCA output to be read and compiled into a structured ndarray. DCA output is a 4 column text file with the following columns: (residue 1, residue 2, Mutual Information, Direct Information).
318
+
319
+ Returns
320
+ -------
321
+ DirectInformationData
322
+ DirectInformationData object with named structured array containing residue indices and the DI value of the pair.
323
+ """
324
+ file_data = np.loadtxt(dca_filepath, dtype={'names': ('residue1', 'residue2', 'MI', 'DI'), 'formats': (int, int, float, float)})
325
+ return DirectInformationData(file_data[['residue1', 'residue2', 'DI']])
326
+
327
+ @staticmethod
328
+ def load_from_DI_file(DI_filepath: str) -> 'DirectInformationData':
329
+ """
330
+ Function to generate a DirectInformationData object from the modified DI-only version of the DCA output generated via the MATLab dca function.
331
+
332
+ Parameters
333
+ ----------
334
+ DI_filepath : str
335
+ Filepath of the DI file to be read and compile into a structured ndarray. DI file is a 3 column text file with the following columns: (residue 1, residue 2, Direct Information).
336
+
337
+ Returns
338
+ -------
339
+ DirectInformationData
340
+ DirectInformationData object with named structured array containing residue indices and the DI value of the pair.
341
+ """
342
+ return DirectInformationData(np.loadtxt(DI_filepath, dtype={'names': ('residue1', 'residue2', 'DI'), 'formats': (int, int, float)}))
343
+
344
+ @staticmethod
345
+ def load_as_ndarray(ndarray: npt.NDArray) -> 'DirectInformationData':
346
+ """
347
+ Function to generate a Direct Information object from a ndarray.
348
+
349
+ Parameters
350
+ ----------
351
+ ndarray : numpy.ndarray
352
+ An ndarray of shape (n,3) where its columns are (residue 1, residue 2, and Direct Information)
353
+
354
+ Returns
355
+ -------
356
+ DirectInformationData
357
+ DirectInformationData object with named structured array containing residue indices and the DI value of the pair.
358
+ """
359
+ if ndarray.shape[1] != 3:
360
+ raise Exception("Dimensions of numpy array supplied are different from what is expected. Please supply residue1, residue2, and DI column in int, int, float format and with shape of (n, 3).")
361
+ # Structured ndarrays require list of tuples for conversion.
362
+ DI_data = np.array([tuple(x) for x in ndarray], dtype={'names': ('residue1', 'residue2', 'DI'), 'formats': (int, int, float)})
363
+
364
+ return DirectInformationData(DI_data)
365
+
366
+ def get_ranked_mapped_pairs(self, RA1: ResidueAlignment, RA2: ResidueAlignment, pairs_only: bool=True, mirror: bool=False, number: Optional[int]=None) -> npt.NDArray:
367
+ """
368
+ Uses DirectInformationData and Pairs interface methods to obtain ranked, mapped residues. See rank_pairs() function and map_DIs() function for details on rank and mapping. Residue Alignments can be the same for intra-domain / intra-protein mapping.
369
+
370
+ Parameters
371
+ ----------
372
+ RA1 : ResidueAlignment
373
+ The ResidueAlignment used for mapping the first column of residues to the appropriate target sequence.
374
+ RA2 : ResidueAlignment
375
+ The ResidueAlignment used for mapping the second column of residues to the appropriate target sequence.
376
+ pairs_only : bool
377
+ True if the final ndarray should contain only columns 1 and 2, corresponding to the residues that constitute the pair. This would drop the DI column.
378
+ mirror : bool
379
+ See Pairs.mirror_pairs() or get_pairs() for details. NOTE: This option is overriden entirely if pairs_only is False. If true, this will produce an ndarray that has the original residue indices and repeated residue indices but with residue 1 and residue 2 switched. This is useful for plotting across the upper diagonal of a contact map.
380
+ number : int, None
381
+ Number of ranked, mapped pairs to return.
382
+
383
+ Returns
384
+ -------
385
+ numpy.ndarray
386
+ Structured ndarray with columns residue 1, residue 2 and optionally DI. Only has specified number of pairs if `number` is specified and mirrored pairs if `mirror` is True and pairs_only is False.
387
+
388
+ Notes
389
+ -----
390
+ ResidueAlignments contain dictionaries like domain_to_protein to map residues produced via Direct Coupling Analysis (DCA) on an MSA generated in context to an HMM. The residues are mapped to a protein structure via alignment of the HMM hit / domain to the protein sequence.
391
+ """
392
+ ranked_pairs = DirectInformationData.rank_pairs(DirectInformationData.nonlocal_pairs(self.DI_data))
393
+ ranked_mapped_pairs = DirectInformationData.map_DIs(ranked_pairs, RA1, RA2)
394
+ if pairs_only:
395
+ return Pairs.get_pairs(ranked_mapped_pairs[['residue1', 'residue2']], mirror=mirror, number=number)
396
+ else:
397
+ return Pairs.get_pairs(ranked_mapped_pairs, mirror=False, number=number)
398
+
399
+ @staticmethod
400
+ def map_DIs(DI_data : npt.NDArray, RA1: ResidueAlignment, RA2: ResidueAlignment) -> npt.NDArray:
401
+ """
402
+ Uses domain-to-protein mappings present in the Residue Alignments provided to generate mapped representations of the residues from the DI_data structured ndarray provided.
403
+
404
+ Parameters
405
+ ----------
406
+ DI_data : numpy.ndarray
407
+ Structured ndarray that contains columns "residue1" and "residue2".
408
+ RA1 : ResidueAlignment
409
+ The ResidueAlignment used for mapping the first column of residues to the appropriate target sequence.
410
+ RA2 : ResidueAlignment
411
+ The ResidueAlignment used for mapping the second column of residues to the appropriate target sequence.
412
+
413
+ Returns
414
+ -------
415
+ mappable_DI_data : numpy.ndarray
416
+ DI_data that has been mapped to the target sequence specified in the generation of the corresponding ResidueAlignments.
417
+
418
+ Note:
419
+ Residues that do not map to the target sequence of the ResidueAlignment are dropped.
420
+ """
421
+ # Alternative approach is to just use get function instead of [x] and default to np.nan and drop nans row-wise.
422
+ mapping_key_mask = (np.isin(DI_data['residue1'], list(RA1.domain_to_protein.keys()))) & (np.isin(DI_data['residue2'], list(RA2.domain_to_protein.keys())))
423
+ mappable_DI_data = DI_data[mapping_key_mask]
424
+ mappable_DI_data['residue1'] = np.vectorize(lambda x: RA1.domain_to_protein[x])(mappable_DI_data['residue1'])
425
+ mappable_DI_data['residue2'] = np.vectorize(lambda x: RA2.domain_to_protein[x])(mappable_DI_data['residue2'])
426
+ return mappable_DI_data
427
+
428
+ @staticmethod
429
+ def rank_pairs(DI_data: npt.NDArray) -> npt.NDArray:
430
+ """
431
+ Sorts a structured ndarray of pairs information to order the ndarray by Direct Information (DI) score.
432
+
433
+ Parameters
434
+ ----------
435
+ DI_data : numpy.ndarray
436
+ Structured ndarray that contains columns with names "residue1", "residue2", and "DI" (Direct Information)
437
+
438
+ Returns
439
+ -------
440
+ numpy.ndarray
441
+ Structured ndarray sorted upon column that is named DI in descending order.
442
+ """
443
+ # [::-1] reverses the order from ascending DI Score to descending DI score.
444
+ return np.sort(DI_data, order='DI')[::-1]
445
+
446
+ @staticmethod
447
+ def nonlocal_pairs(DI_data: npt.NDArray) -> npt.NDArray:
448
+ """
449
+ Subsets a structured ndarray of pairs information to find nonlocal pairs, where residue interactions are likely not involved in secondary structure formation i.e. helices and sheet interactions. Nonlocal pairs must be at least 4 residues apart.
450
+
451
+ Parameters
452
+ ----------
453
+ DI_data : numpy.ndarray
454
+ Structured ndarray that contains (at least) the first and second columns with the names "residue1" and "residue2" respectively.
455
+
456
+ Returns
457
+ -------
458
+ numpy.ndarray
459
+ Structured ndarray of DI pairs where residue 1 and residue 2 are at least 4 residues apart.
460
+ """
461
+ return DI_data[abs(DI_data['residue1'] - DI_data['residue2']) > 4]
462
+
463
+ @staticmethod
464
+ def get_dist_commands(model1: str | int, model2: str | int, chain1: str, chain2: str, pairs: npt.NDArray, ca_only: bool=True, auth_res_ids=False) -> list[str]:
465
+ """
466
+ Get UCSF Chimera commands for displaying distance commands for usage in displaying distances between residue pairs. Options are present for alpha-carbon to alpha-carbon distance or for specified atom to specified atom distance.
467
+
468
+ Parameters
469
+ ----------
470
+ model1 : str, int
471
+ Number of model corresponding to the structure containing the first column of residues.
472
+ model2 : str, int
473
+ Number of model corresponding to the structure containing the second column of residues.
474
+ chain1 : str
475
+ The chain present in the structure in model1 containing the first column of residues.
476
+ chain2 : str
477
+ The chain present in the structure in model2 containing the second column of residues.
478
+ pairs : numpy.ndarray
479
+ Structured ndarray that contains (at least) the first and second columns with the names "residue1" and "residue2" respectively. `ca_only` should be set to false and an additional "atom_name1" and "atom_name2" column should be added if atoms are specified per pair.
480
+ ca_only : bool
481
+ True if distance commands are for displaying distances between the two alpha-carbons of the residue pair. If false, specific_atom_names is used in lieu of "CA" as an atom identifier.
482
+ auth_res_ids : bool
483
+ If true, use the "auth_residue1" and "auth_residue2" columns instead of "residue1" and "residue2" columns. These residue ids correspond to the auth protein residue ids.
484
+
485
+ Returns
486
+ -------
487
+ distance_commands : list of str
488
+ List of distance commands generated between two residues with model and chain information needed, either between two alpha-carbons or the specified atoms.
489
+
490
+ Note
491
+ ----
492
+ model1 and model2 can be equivalent if both columns involve residues referenced by the same model. The same would apply for chains if the residues are present on the same chain.
493
+ """
494
+ distance_commands: list[str] = []
495
+ for i in range(np.shape(pairs)[0]):
496
+ if auth_res_ids:
497
+ residue1 = pairs[i]['auth_residue1']
498
+ residue2 = pairs[i]['auth_residue2']
499
+ else:
500
+ residue1 = pairs[i]['residue1']
501
+ residue2 = pairs[i]['residue2']
502
+ if ca_only:
503
+ distance_commands.append(f"distance #{model1}:{residue1}.{chain1}@CA #{model2}:{residue2}.{chain2}@CA;")
504
+ else:
505
+ atom1 = pairs[i]['atom_name1']
506
+ atom2 = pairs[i]['atom_name2']
507
+ distance_commands.append(f"distance #{model1}:{residue1}.{chain1}@{atom1} #{model2}:{residue2}.{chain2}@{atom2};")
508
+ return distance_commands
509
+
510
+ @staticmethod
511
+ def write_DI_data(filepath: str, pairs: npt.NDArray, delimiter: str="\t", fmt: tuple[str, str, str] | tuple[str, str]=('%d', '%d', '%.3f')) -> None:
512
+ """
513
+ Writes pairs ndarray to file with specified delimiter between the pairs' row elements, i.e. residue 1, residue 2, and DI score.
514
+
515
+ Parameters
516
+ ----------
517
+ filepath : str
518
+ Path of the file to write DirectInformation data to.
519
+ pairs : numpy.ndarray
520
+ Ndarray of at-least pairs information (residue 1, residue 2) and optionally Direct Information to write to a file via numpy.savetxt().
521
+ delimiter : str
522
+ Delimiter to separate columns of the pairs ndarray when writing to a file.
523
+ fmt : tuple of str, default=('%d', '%d', '%.3f)
524
+ format passed as an argument to numpy.savetxt() to define type of column and output format. Set to ('%d', '%d') if only two columns are present in the ndarray.
525
+
526
+ Returns
527
+ -------
528
+ None
529
+ """
530
+ if len(pairs[0]) == 2:
531
+ fmt = ('%d', '%d')
532
+ np.savetxt(filepath, pairs, delimiter=delimiter, fmt=fmt)
533
+
534
+ class StructureInformation:
535
+ """
536
+ Information regarding a protein structure, obtained from a protein structure file.
537
+
538
+ Parameters
539
+ ----------
540
+ structure : biotite.structure
541
+ Structure obtained from an RCSB entry with a provided pdbx/mmcif file with a specified model number.
542
+ pdbx_file : biotite.io.pdbx.CIFFile
543
+ mmCIF file that contains generic information and atomic information of the protein structure categorized into mmCIF blocks.
544
+
545
+ Attributes
546
+ ----------
547
+ struct_ref_seq : zip
548
+ zipped version of parallel arrays that contain chain ids, beginning align indices of the sequence, and beginning align indices of the auth sequence.
549
+ """
550
+ def __init__(self, structure, pdbx_file: pdbx.CIFFile):
551
+ self.structure = structure
552
+ self.pdbx_file = pdbx_file
553
+ strand_ids = []
554
+ align_beg = []
555
+ auth_align_beg = []
556
+ # Obtain struct ref seq information (chain names, beginning residue, and the corresponding protein beginning residue)
557
+ for col_name, col in self.pdbx_file[list(self.pdbx_file.keys())[0]]['struct_ref_seq'].items():
558
+ if col_name == "pdbx_strand_id":
559
+ strand_ids = col.data.array
560
+ if col_name == "seq_align_beg":
561
+ align_beg = col.data.array
562
+ if col_name == "pdbx_auth_seq_align_beg":
563
+ auth_align_beg = col.data.array
564
+ # Put all three lists together
565
+ self.struct_ref_seq = zip(strand_ids, align_beg, auth_align_beg)
566
+
567
+ @staticmethod
568
+ def fetch_pdb(pdb_id: str, model_num: int=1, struc_format: str="mmcif") -> 'StructureInformation':
569
+ """
570
+ Fetches PDB as mmCIF file from RCSB and compiles the information into a StructureInformation instance.
571
+
572
+ Parameters
573
+ ----------
574
+ pdb_id : str
575
+ PDB ID to be fetched from the RCSB database.
576
+ model_num : int
577
+ The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
578
+ struc_format : str
579
+ The format of the file to pull from the RCSB database.
580
+
581
+ Returns
582
+ -------
583
+ StructureInformation
584
+ StructureInformation generated from pdbx.get_structure() function using the pdbx file fetched from RCSB. The pdbx file is also supplied as an argument.
585
+
586
+ Raises
587
+ ------
588
+ TypeError
589
+ Fetched data was not found and returned None instead.
590
+ """
591
+ fetched_data = rcsb.fetch(pdb_id, struc_format)
592
+ if fetched_data is None:
593
+ raise TypeError("RCSB fetch failed. Try fetch again.")
594
+ pdbx_file = pdbx.CIFFile.read(fetched_data)
595
+ return StructureInformation(pdbx.get_structure(pdbx_file=pdbx_file, model=model_num, use_author_fields=False), pdbx_file)
596
+
597
+ @staticmethod
598
+ def read_pdb_mmCIF(pdb_filepath: str, model_num: int=1) -> 'StructureInformation':
599
+ """
600
+ Reads PDB mmCIF file from filepath and compiles the information into a StructureInformation instance.
601
+
602
+ Parameters
603
+ ----------
604
+ pdb_filepath : str
605
+ Filepath of the PDB mmCIF file to be read.
606
+ model_num : int
607
+ The model number to access from the PDB to ensure an AtomArray is returned containing the atom information of the protein structure.
608
+
609
+ Returns
610
+ -------
611
+ StructureInformation
612
+ StructureInformation generated from pdbx.get_structure() function using the pdbx file fetched from RCSB. The pdbx file is also supplied as an argument.
613
+ """
614
+ pdbx_file = pdbx.CIFFile.read(pdb_filepath)
615
+ return StructureInformation(pdbx.get_structure(pdbx_file, model=model_num, use_author_fields=False), pdbx_file)
616
+
617
+ def get_chain_specific_structure(self, ca_only: bool, chain1: str, chain2: str, remove_hetero=True) -> tuple:
618
+ """
619
+ Subsets structure attribute to select for chain specific portions of the structure.
620
+
621
+ Parameters
622
+ ----------
623
+ ca_only : bool
624
+ If true, the structure will also be subsetted for atom entries where the atom_name annotation is "CA" (referring to alpha-carbons)
625
+ chain1 : str
626
+ Chain id corresponding to the first column of residues in the structure.
627
+ chain2 : str
628
+ Chain id corresponding to the second column of residues in the structure.
629
+ remove_hetero : bool, default=True
630
+ If true, the structure will also be subsetted for atom entries where the hetero annotation is False, thus removing heteroatoms.
631
+
632
+ Returns
633
+ -------
634
+ tuple of biotite.structure.AtomArray, biotite.structure.AtomArray
635
+ Two AtomArrays that refer to atoms in the first chain and second chain, respectively without accounting for the presence of heteroatoms if `remove_hetero` is True.
636
+ """
637
+ selected_structure = self.structure
638
+ if remove_hetero:
639
+ # Remove hetero atoms via hetero column of structure ndarray
640
+ selected_structure = self.structure[self.structure.hetero == False]
641
+ if ca_only:
642
+ # Consider selection of alpha-carbon atoms only
643
+ selected_structure = selected_structure[selected_structure.atom_name == "CA"]
644
+ chain1_structure = selected_structure[selected_structure.chain_id == chain1]
645
+ chain2_structure = selected_structure[selected_structure.chain_id == chain2]
646
+ return (chain1_structure, chain2_structure)
647
+
648
+ def generate_dist_matrix(self, ca_only: bool, chain1: str, chain2: str):
649
+ """
650
+ Generates distance matrix between two chains in the structure attribute.
651
+
652
+ Parameters
653
+ ----------
654
+ ca_only : bool
655
+ If True, only atoms that have the name "CA" are selected in the chains the distance matrix is calculated between.
656
+ chain1 : str
657
+ Chain id corresponding to the first column of residues in the structure.
658
+ chain2 : str
659
+ Chain id corresponding to the first column of residues in the structure.
660
+
661
+ Returns
662
+ -------
663
+ tuple of biotite.structure.AtomArray, biotite.structure.AtomArray, numpy.ndarray
664
+ Tuple containing the chain 1 structure, the chain 2 structure, and the distance matrix of chain 1 and chain 2's pairwise distances.
665
+ """
666
+ chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only, chain1, chain2, remove_hetero=True)
667
+ dist_matrix = cdist(chain1_structure.coord, chain2_structure.coord)
668
+ return (chain1_structure, chain2_structure, dist_matrix)
669
+
670
+ def get_min_dist_atom_info(self, pairs: npt.NDArray, chain1: str, chain2: str) -> npt.NDArray:
671
+ """
672
+ Generate a ndarray of residue ids and their corresponding atom names such that the distance is the minimum between the initial residues provided.
673
+
674
+ Parameters
675
+ ----------
676
+ pairs : numpy.ndarray
677
+ Pairs structured ndarray with "residue1" and "residue2" columns.
678
+ chain1 : str
679
+ Chain id corresponding to the first column of residues in the structure.
680
+ chain2 : str
681
+ Chain id corresponding to the second column of residues in the structure.
682
+
683
+ Returns
684
+ -------
685
+ min_dist_pairs_atoms_arr : numpy.ndarray
686
+ Structured ndarray that has residue indices, auth residue indices (corresponding to the protein numbering), and atomic names in the format {'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,str,str]}
687
+ """
688
+ shift1 = 0
689
+ shift2 = 0
690
+ for row in list(self.struct_ref_seq):
691
+ ref_seq_chain, ref_seq_beg, auth_ref_seq_beg = row
692
+ if ref_seq_chain == chain1:
693
+ shift1 = int(auth_ref_seq_beg) - int(ref_seq_beg)
694
+ if ref_seq_chain == chain2:
695
+ shift2 = int(auth_ref_seq_beg) - int(ref_seq_beg)
696
+
697
+ chain1_structure, chain2_structure = self.get_chain_specific_structure(ca_only=False, chain1=chain1, chain2=chain2, remove_hetero=True)
698
+ min_dist_pairs_atoms = []
699
+ for row in pairs:
700
+ # Obtain structure information for chains 1 and 2
701
+ chain1_res1_structure = chain1_structure[chain1_structure.res_id == row['residue1']]
702
+ chain2_res2_structure = chain2_structure[chain2_structure.res_id == row['residue2']]
703
+
704
+ # Calculate a distance matrix and find the indices of the minimal value in the matrix
705
+ dist_matrix = cdist(chain1_res1_structure.coord, chain2_res2_structure.coord)
706
+
707
+ ind = np.unravel_index(np.argmin(dist_matrix), dist_matrix.shape)
708
+ # Use the indices to access the atom in the atom array and get the correct atom name.
709
+ # Generate the auth ids of the residues in the pairs ndarray
710
+ auth_res_id1 = row['residue1'] + shift1
711
+ auth_res_id2 = row['residue2'] + shift2
712
+ min_dist_pairs_atoms.append((row['residue1'], row['residue2'],auth_res_id1, auth_res_id2, chain1_res1_structure[ind[0]].atom_name, chain2_res2_structure[ind[1]].atom_name))
713
+ min_dist_pairs_atoms_arr = np.array(min_dist_pairs_atoms, dtype={'names': ['residue1','residue2','auth_residue1','auth_residue2','atom_name1','atom_name2'], 'formats': [int,int,int,int,'<U10','<U10']})
714
+ return min_dist_pairs_atoms_arr
715
+
716
+ def get_contacts(self, ca_only: bool, threshold: float, chain1: str, chain2: str) -> set[tuple[int, int]]:
717
+ """
718
+ Get contacts from the structure attribute where the distance between two residues is less than the threshold.
719
+
720
+ Parameters
721
+ ----------
722
+ ca_only : bool
723
+ If true, only consider alpha-carbon to alpha-carbon distances.
724
+ threshold : float
725
+ Maximum distance to consider between two atoms.
726
+ chain1 : str
727
+ Chain id corresponding to the first column of residues in the structure.
728
+ chain2 : str
729
+ Chain id corresponding to the second column of residues in the structure.
730
+
731
+ Returns
732
+ -------
733
+ contacts_set : set of tuple of ints
734
+ Set of contacts, tuples with "residue1" and "residue2" from the structure that are within the distance threshold.
735
+ """
736
+ chain1_structure, chain2_structure, dist_matrix = self.generate_dist_matrix(ca_only, chain1, chain2)
737
+ contacts = np.argwhere(dist_matrix <= threshold)
738
+ contacts_set = set()
739
+ for contact in contacts:
740
+ contacts_set.add((chain1_structure[contact[0]].res_id, chain2_structure[contact[1]].res_id))
741
+ return contacts_set
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2024 Raheel Syed Ahmed
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,57 @@
1
+ Metadata-Version: 2.1
2
+ Name: dcatoolkit
3
+ Version: 0.1.0
4
+ Summary: Collection of useful modules and representations for managing DCA output data.
5
+ Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
6
+ Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
7
+ License: MIT License
8
+
9
+ Copyright (c) 2024 Raheel Syed Ahmed
10
+
11
+ Permission is hereby granted, free of charge, to any person obtaining a copy
12
+ of this software and associated documentation files (the "Software"), to deal
13
+ in the Software without restriction, including without limitation the rights
14
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
15
+ copies of the Software, and to permit persons to whom the Software is
16
+ furnished to do so, subject to the following conditions:
17
+
18
+ The above copyright notice and this permission notice shall be included in all
19
+ copies or substantial portions of the Software.
20
+
21
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
22
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
23
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
24
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
25
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
26
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
27
+ SOFTWARE.
28
+
29
+ Keywords: dca,toolkit,DI,coevolution
30
+ Classifier: Development Status :: 4 - Beta
31
+ Classifier: Intended Audience :: Science/Research
32
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
33
+ Classifier: License :: OSI Approved :: MIT License
34
+ Classifier: Programming Language :: Python :: 3
35
+ Classifier: Programming Language :: Python :: 3.10
36
+ Classifier: Programming Language :: Python :: 3.11
37
+ Classifier: Programming Language :: Python :: 3.12
38
+ Requires-Python: >=3.10
39
+ Description-Content-Type: text/markdown
40
+ License-File: LICENSE
41
+ Requires-Dist: biotite
42
+ Requires-Dist: matplotlib >=3.8.0
43
+ Requires-Dist: numpy >=1.26.0
44
+ Requires-Dist: pandas >=2.1.0
45
+ Requires-Dist: pyhmmer >=0.10.14
46
+ Requires-Dist: scikit-learn >=1.3
47
+ Requires-Dist: scipy >=1.11.0
48
+ Provides-Extra: docs
49
+ Requires-Dist: sphinx ; extra == 'docs'
50
+ Requires-Dist: numpydoc ; extra == 'docs'
51
+ Provides-Extra: lint
52
+ Requires-Dist: ruffle ; extra == 'lint'
53
+ Provides-Extra: tests
54
+ Requires-Dist: pytest ; extra == 'tests'
55
+
56
+ # dcatoolkit
57
+ Collection of useful modules and representations for managing DCA output data.
@@ -0,0 +1,7 @@
1
+ dcatoolkit/__init__.py,sha256=OLJId705w4u5t393Ua_Izedv2TNu6BQFVPJAP10Q45M,101
2
+ dcatoolkit/representation.py,sha256=0tp4nmbbAgHSV2ACtS_F6O4btCrDRnwCC6Xn6em-8Co,36010
3
+ dcatoolkit-0.1.0.dist-info/LICENSE,sha256=hcaC-91R_6xOkMHwpKEoOgwGo279fuTRWBK67aAo1wg,1074
4
+ dcatoolkit-0.1.0.dist-info/METADATA,sha256=Z9tUnnK020a0aXNV-SJkRTUxpJWDSuV2AzkSLOO066k,2584
5
+ dcatoolkit-0.1.0.dist-info/WHEEL,sha256=R0nc6qTxuoLk7ShA2_Y-UWkN8ZdfDBG2B6Eqpz2WXbs,91
6
+ dcatoolkit-0.1.0.dist-info/top_level.txt,sha256=wJ2JyTFXjRrHs2g4zgqZgT2mQHqd5kH2omd1Jj1lTwo,11
7
+ dcatoolkit-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (72.1.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1 @@
1
+ dcatoolkit