dcatoolkit 0.1.6__tar.gz → 0.1.8__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dcatoolkit-0.1.6/src/dcatoolkit.egg-info → dcatoolkit-0.1.8}/PKG-INFO +1 -1
- {dcatoolkit-0.1.6 → dcatoolkit-0.1.8}/pyproject.toml +1 -1
- {dcatoolkit-0.1.6 → dcatoolkit-0.1.8}/src/dcatoolkit/__init__.py +1 -1
- {dcatoolkit-0.1.6 → dcatoolkit-0.1.8}/src/dcatoolkit/analytics.py +31 -16
- {dcatoolkit-0.1.6 → dcatoolkit-0.1.8}/src/dcatoolkit/representation.py +9 -11
- {dcatoolkit-0.1.6 → dcatoolkit-0.1.8/src/dcatoolkit.egg-info}/PKG-INFO +1 -1
- {dcatoolkit-0.1.6 → dcatoolkit-0.1.8}/LICENSE +0 -0
- {dcatoolkit-0.1.6 → dcatoolkit-0.1.8}/README.md +0 -0
- {dcatoolkit-0.1.6 → dcatoolkit-0.1.8}/setup.cfg +0 -0
- {dcatoolkit-0.1.6 → dcatoolkit-0.1.8}/src/dcatoolkit.egg-info/SOURCES.txt +0 -0
- {dcatoolkit-0.1.6 → dcatoolkit-0.1.8}/src/dcatoolkit.egg-info/dependency_links.txt +0 -0
- {dcatoolkit-0.1.6 → dcatoolkit-0.1.8}/src/dcatoolkit.egg-info/requires.txt +0 -0
- {dcatoolkit-0.1.6 → dcatoolkit-0.1.8}/src/dcatoolkit.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: dcatoolkit
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.8
|
|
4
4
|
Summary: Collection of useful modules and representations for managing DCA output data.
|
|
5
5
|
Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
6
6
|
Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "dcatoolkit"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.8"
|
|
8
8
|
description = "Collection of useful modules and representations for managing DCA output data."
|
|
9
9
|
keywords = ["dca", "toolkit", "DI", "coevolution"]
|
|
10
10
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import re
|
|
2
2
|
from collections import Counter
|
|
3
|
-
from typing import Optional
|
|
4
|
-
import string
|
|
3
|
+
from typing import Optional, Union
|
|
4
|
+
import string, io
|
|
5
5
|
|
|
6
6
|
class MSATools:
|
|
7
7
|
"""
|
|
@@ -16,23 +16,31 @@ class MSATools:
|
|
|
16
16
|
self.MSA = MSA
|
|
17
17
|
|
|
18
18
|
@staticmethod
|
|
19
|
-
def load_from_file(
|
|
19
|
+
def load_from_file(msa_source: Union[str, io.IOBase]) -> 'MSATools':
|
|
20
20
|
"""
|
|
21
21
|
Generates MSATools object from an MSA file in ".afa" format.
|
|
22
22
|
|
|
23
23
|
Parameters
|
|
24
24
|
----------
|
|
25
|
-
|
|
26
|
-
Filepath of the MSA in ".afa" format that is provided.
|
|
25
|
+
msa_source : str or io.IOBase
|
|
26
|
+
Filepath or IOBase of the MSA in ".afa" format that is provided.
|
|
27
27
|
|
|
28
28
|
Returns
|
|
29
29
|
-------
|
|
30
30
|
MSATools
|
|
31
31
|
An MSATools instance with the appropriate list of (header, sequence) tuples where sequences are simplified and converted to single line format.
|
|
32
32
|
"""
|
|
33
|
+
data = ""
|
|
33
34
|
msa_entries: list[tuple[str, str]] = []
|
|
34
|
-
|
|
35
|
-
|
|
35
|
+
if isinstance(msa_source, str):
|
|
36
|
+
with open(msa_source, 'r') as fs:
|
|
37
|
+
data = fs.read()
|
|
38
|
+
elif isinstance(msa_source, io.BytesIO):
|
|
39
|
+
data = msa_source.getvalue().decode()
|
|
40
|
+
elif isinstance(msa_source, io.StringIO):
|
|
41
|
+
data = msa_source.getvalue()
|
|
42
|
+
else:
|
|
43
|
+
raise Exception("msa_file is not bytesIO, stringIO, or a filepath.")
|
|
36
44
|
split_data = data.split(">")[1:]
|
|
37
45
|
for entry in split_data:
|
|
38
46
|
line_split_entry = entry.split("\n")
|
|
@@ -113,25 +121,32 @@ class MSATools:
|
|
|
113
121
|
kept_entries.append((header, sequence))
|
|
114
122
|
return kept_entries
|
|
115
123
|
|
|
116
|
-
def write(self,
|
|
124
|
+
def write(self, destination: Union[str, io.IOBase]) -> None:
|
|
117
125
|
"""
|
|
118
|
-
Writes this MSA's headers and sequences to the
|
|
126
|
+
Writes this MSA's headers and sequences to the destination specified.
|
|
119
127
|
|
|
120
128
|
Parameters
|
|
121
129
|
----------
|
|
122
|
-
|
|
123
|
-
Filepath to write the MSA supplied to.
|
|
130
|
+
destination : str or io.IOBase
|
|
131
|
+
Filepath or IO to write the MSA supplied to.
|
|
124
132
|
|
|
125
133
|
Returns
|
|
126
134
|
-------
|
|
127
135
|
None
|
|
128
136
|
"""
|
|
129
|
-
|
|
137
|
+
if isinstance(destination, str):
|
|
138
|
+
with open(destination, 'w') as fs:
|
|
139
|
+
for header, sequence in self.MSA:
|
|
140
|
+
fs.write(header)
|
|
141
|
+
fs.write("\n")
|
|
142
|
+
fs.write(sequence)
|
|
143
|
+
fs.write("\n")
|
|
144
|
+
elif isinstance(destination, io.IOBase):
|
|
130
145
|
for header, sequence in self.MSA:
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
146
|
+
destination.write(header)
|
|
147
|
+
destination.write("\n")
|
|
148
|
+
destination.write(sequence)
|
|
149
|
+
destination.write("\n")
|
|
135
150
|
|
|
136
151
|
def __len__(self):
|
|
137
152
|
"""
|
|
@@ -641,7 +641,9 @@ class StructureInformation:
|
|
|
641
641
|
self.unique_chains : numpy.ndarray, optional
|
|
642
642
|
Array of unique asym_id entries which corresponds to unique chain IDs.
|
|
643
643
|
self.chain_auth_dict : dict of str, str, optional
|
|
644
|
-
Uses chain id as a key and provides auth chain id as a value.
|
|
644
|
+
Uses chain id as a key and provides auth chain id as a value.
|
|
645
|
+
self.auth_chain_dict : dict of str, str, optional
|
|
646
|
+
Uses auth chain id as a key and provides original/label chain id as a value.
|
|
645
647
|
self.res_auth_dict : dict of str, tuple of int, int or optional
|
|
646
648
|
Uses chain id as a key and an array of residue index and auth residue index as a value.
|
|
647
649
|
"""
|
|
@@ -670,6 +672,7 @@ class StructureInformation:
|
|
|
670
672
|
self.first_block = list(self.pdbx_file)[0]
|
|
671
673
|
self.atom_site_category = self.pdbx_file[self.first_block].get('atom_site')
|
|
672
674
|
self.chain_auth_dict = {}
|
|
675
|
+
self.auth_chain_dict = {}
|
|
673
676
|
self.res_auth_dict = {}
|
|
674
677
|
if self.atom_site_category:
|
|
675
678
|
group_pdbs = []
|
|
@@ -700,6 +703,7 @@ class StructureInformation:
|
|
|
700
703
|
for unique_chain in self.unique_chains:
|
|
701
704
|
unique_entry = self.atom_data[self.atom_data[:,2] == unique_chain][0]
|
|
702
705
|
self.chain_auth_dict[unique_entry[2]] = unique_entry[4]
|
|
706
|
+
self.auth_chain_dict[unique_entry[4]] = unique_entry[2]
|
|
703
707
|
self.res_auth_dict[unique_entry[2]] = unique_entry[[1,3]].astype('int')
|
|
704
708
|
else:
|
|
705
709
|
self.atom_site_category = None
|
|
@@ -775,7 +779,6 @@ class StructureInformation:
|
|
|
775
779
|
else:
|
|
776
780
|
return str(self.full_sequences[self.chain_auth_dict[chain_id]])
|
|
777
781
|
|
|
778
|
-
|
|
779
782
|
def get_non_missing_sequence(self, chain_id: str, auth_chain_id_supplied: bool=False) -> str:
|
|
780
783
|
"""
|
|
781
784
|
Get sequence, including only non-missing residues, from the specified chain off of RCSB.
|
|
@@ -793,15 +796,10 @@ class StructureInformation:
|
|
|
793
796
|
The full sequence, including only non-missing residues, of the chain specified.
|
|
794
797
|
"""
|
|
795
798
|
if auth_chain_id_supplied:
|
|
796
|
-
original_chain_id =
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
if original_chain_id is None:
|
|
801
|
-
raise Exception("Auth chain id supplied not found in chain_auth_dict.")
|
|
802
|
-
else:
|
|
803
|
-
return self.non_missing_sequences[original_chain_id]
|
|
804
|
-
return self.non_missing_sequences[chain_id]
|
|
799
|
+
original_chain_id = self.auth_chain_dict[chain_id]
|
|
800
|
+
return self.non_missing_sequences[original_chain_id]
|
|
801
|
+
else:
|
|
802
|
+
return self.non_missing_sequences[chain_id]
|
|
805
803
|
|
|
806
804
|
def get_chain_specific_structure(self, ca_only: bool, chain1: str, chain2: str, remove_hetero=True) -> tuple:
|
|
807
805
|
"""
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: dcatoolkit
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.8
|
|
4
4
|
Summary: Collection of useful modules and representations for managing DCA output data.
|
|
5
5
|
Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
6
6
|
Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|