dcatoolkit 0.1.4__tar.gz → 0.1.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: dcatoolkit
3
- Version: 0.1.4
3
+ Version: 0.1.5
4
4
  Summary: Collection of useful modules and representations for managing DCA output data.
5
5
  Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
6
6
  Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "dcatoolkit"
7
- version = "0.1.4"
7
+ version = "0.1.5"
8
8
  description = "Collection of useful modules and representations for managing DCA output data."
9
9
  keywords = ["dca", "toolkit", "DI", "coevolution"]
10
10
 
@@ -1,4 +1,4 @@
1
1
 
2
- __version__ = "0.1.4"
2
+ __version__ = "0.1.5"
3
3
  from .representation import Pairs, DirectInformationData, StructureInformation, ResidueAlignment
4
4
  from .analytics import MSATools
@@ -1,12 +1,13 @@
1
1
  import numpy as np
2
2
  import pandas as pd
3
- from collections.abc import Iterable
4
3
  from scipy.spatial.distance import cdist
4
+
5
5
  import biotite.structure as struc
6
6
  import biotite.structure.io.pdbx as pdbx
7
7
  import biotite.database.rcsb as rcsb
8
8
 
9
- from typing import Optional
9
+ from collections.abc import Iterable
10
+ from typing import Optional, Union
10
11
  import numpy.typing as npt
11
12
 
12
13
 
@@ -43,6 +44,40 @@ class Pairs:
43
44
  elif ndarr is not None:
44
45
  self.pairs = ndarr
45
46
 
47
+ @staticmethod
48
+ def load_from_file(filepath: str):
49
+ """
50
+ Loads file containing whitespace-delimited data in columns of residues being column 1 and column 2.
51
+
52
+ Parameters
53
+ ----------
54
+ filepath : str
55
+ Filepath with residue columns corresponding to the indices of first and second components (proteins, chains, etc.) constituting a pair.
56
+
57
+ Returns
58
+ -------
59
+ Pairs
60
+ Pairs object with a loaded, structured ndarray with dtype=[('residue1', int), ('residue2', int)]
61
+ """
62
+ return Pairs(ndarr=np.loadtxt(filepath, dtype=[('residue1', int), ('residue2', int)]))
63
+
64
+ @staticmethod
65
+ def load_from_ndarray(ndarray: Union[npt.NDArray, Iterable[Iterable]]):
66
+ """
67
+ Loads 2d ndarray of residue pairs in columnar format into Pairs object.
68
+
69
+ Parameters
70
+ ----------
71
+ ndarray : numpy.ndarray or Iterable of Iterable (excluding dict)
72
+ Unstructured ndarray or iterable of iterables with pairs of residue indices with residue1 and residue 2 in separate columns or as two separate elements.
73
+
74
+ Returns
75
+ -------
76
+ Pairs
77
+ Pairs object with a loaded, structured ndarray with dtype=[('residue1', int), ('residue2', int)]
78
+ """
79
+ return Pairs(ndarr=np.array([tuple(x) for x in ndarray], dtype={'names': ('residue1', 'residue2'), 'formats': (int, int)}))
80
+
46
81
  @staticmethod
47
82
  def mirror_diagonal(pairs: npt.NDArray) -> npt.NDArray:
48
83
  """
@@ -58,7 +93,10 @@ class Pairs:
58
93
  numpy.ndarray
59
94
  Values flipped along the column axis.
60
95
  """
61
- return np.flip(pairs, axis=1)
96
+ if pairs.dtype.names is not None:
97
+ return np.array([tuple(x)[::-1] for x in pairs], dtype={'names': ('residue1', 'residue2'), 'formats': (int, int)})
98
+ else:
99
+ return np.flip(pairs, axis=1)
62
100
 
63
101
  @staticmethod
64
102
  def subset_pairs(pairs: npt.NDArray, number : Optional[int]=None) -> npt.NDArray:
@@ -104,7 +142,12 @@ class Pairs:
104
142
  The original pairs specified from the parameters section.
105
143
  """
106
144
  if mirror:
107
- return np.vstack((pairs, Pairs.mirror_diagonal(pairs)))
145
+ if pairs.dtype.names is not None:
146
+ unstruc_pairs = [tuple(x) for x in pairs]
147
+ unstruc_pairs_mirrored = [tuple(x)[::-1] for x in pairs]
148
+ return np.array(unstruc_pairs + unstruc_pairs_mirrored, dtype=[('residue1', int), ('residue2', int)])
149
+ else:
150
+ return np.vstack((pairs, Pairs.mirror_diagonal(pairs)))
108
151
  else:
109
152
  return pairs
110
153
 
@@ -130,8 +173,6 @@ class Pairs:
130
173
  # Check to see if user requested mirrored pairs, if so, add in pairs that are mirrored across diagonal
131
174
  pairs = Pairs.subset_pairs(pairs, number)
132
175
  if mirror:
133
- if pairs.dtype.names is not None:
134
- pairs = np.array([[*pair] for pair in pairs])
135
176
  pairs = Pairs.mirror_pairs(pairs, mirror)
136
177
  return pairs
137
178
 
@@ -345,21 +386,22 @@ class DirectInformationData:
345
386
  return DirectInformationData(np.loadtxt(DI_filepath, dtype={'names': ('residue1', 'residue2', 'DI'), 'formats': (int, int, float)}))
346
387
 
347
388
  @staticmethod
348
- def load_as_ndarray(ndarray: npt.NDArray) -> 'DirectInformationData':
389
+ def load_as_ndarray(ndarray: Union[npt.NDArray, Iterable[Iterable]]) -> 'DirectInformationData':
349
390
  """
350
391
  Function to generate a Direct Information object from a ndarray.
351
392
 
352
393
  Parameters
353
394
  ----------
354
- ndarray : numpy.ndarray
355
- An ndarray of shape (n,3) where its columns are (residue 1, residue 2, and Direct Information)
395
+ ndarray : numpy.ndarray or Iterable of Iterable (excluding dict)
396
+ An ndarray of shape (n,3) where its columns are (residue 1, residue 2, and Direct Information). Can also be parsed from an iterable of iterable provided that the aforementioned format is followed.
356
397
 
357
398
  Returns
358
399
  -------
359
400
  DirectInformationData
360
401
  DirectInformationData object with named structured array containing residue indices and the DI value of the pair.
361
402
  """
362
- if ndarray.shape[1] != 3:
403
+
404
+ if isinstance(ndarray, np.ndarray) and ndarray.shape[1] != 3:
363
405
  raise Exception("Dimensions of numpy array supplied are different from what is expected. Please supply residue1, residue2, and DI column in int, int, float format and with shape of (n, 3).")
364
406
  # Structured ndarrays require list of tuples for conversion.
365
407
  DI_data = np.array([tuple(x) for x in ndarray], dtype={'names': ('residue1', 'residue2', 'DI'), 'formats': (int, int, float)})
@@ -783,8 +825,8 @@ class StructureInformation:
783
825
  shift1 = 0
784
826
  shift2 = 0
785
827
  if self.atom_site_category:
786
- shift1 = abs(self.res_auth_dict[chain1][0] - self.res_auth_dict[chain1][1])
787
- shift2 = abs(self.res_auth_dict[chain2][0] - self.res_auth_dict[chain2][1])
828
+ shift1 = self.res_auth_dict[chain1][1] - self.res_auth_dict[chain1][0]
829
+ shift2 = self.res_auth_dict[chain2][1] - self.res_auth_dict[chain2][0]
788
830
  return shift1, shift2
789
831
  else:
790
832
  return shift1, shift2
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: dcatoolkit
3
- Version: 0.1.4
3
+ Version: 0.1.5
4
4
  Summary: Collection of useful modules and representations for managing DCA output data.
5
5
  Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
6
6
  Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
File without changes
File without changes
File without changes