dcatoolkit 0.1.2__tar.gz → 0.1.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: dcatoolkit
3
- Version: 0.1.2
3
+ Version: 0.1.3
4
4
  Summary: Collection of useful modules and representations for managing DCA output data.
5
5
  Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
6
6
  Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "dcatoolkit"
7
- version = "0.1.2"
7
+ version = "0.1.3"
8
8
  description = "Collection of useful modules and representations for managing DCA output data."
9
9
  keywords = ["dca", "toolkit", "DI", "coevolution"]
10
10
 
@@ -0,0 +1,4 @@
1
+
2
+ __version__ = "0.1.3"
3
+ from .representation import Pairs, DirectInformationData, StructureInformation, ResidueAlignment
4
+ from .analytics import MSATools
@@ -0,0 +1,145 @@
1
+ import re
2
+ from collections import Counter
3
+ from typing import Optional
4
+ import string
5
+
6
+ class MSATools:
7
+ """
8
+ Tools and interface for encapsulating MSA data and providing functionality for filtering and analysis.
9
+
10
+ Parameters
11
+ ----------
12
+ MSA : list of tuple of str, str
13
+ Loaded MSA that is a list of tuples where the first element is the header and the second element is its corresponding sequence.
14
+ """
15
+ def __init__(self, MSA: list[tuple[str, str]]):
16
+ self.MSA = MSA
17
+
18
+ @staticmethod
19
+ def load_from_file(msa_filepath: str) -> 'MSATools':
20
+ """
21
+ Generates MSATools object from an MSA file in ".afa" format.
22
+
23
+ Parameters
24
+ ----------
25
+ msa_filepath : str
26
+ Filepath of the MSA in ".afa" format that is provided.
27
+
28
+ Returns
29
+ -------
30
+ MSATools
31
+ An MSATools instance with the appropriate list of (header, sequence) tuples where sequences are simplified and converted to single line format.
32
+ """
33
+ msa_entries: list[tuple[str, str]] = []
34
+ with open(msa_filepath, 'r') as fs:
35
+ data = fs.read()
36
+ split_data = data.split(">")[1:]
37
+ for entry in split_data:
38
+ line_split_entry = entry.split("\n")
39
+ header = line_split_entry[0]
40
+ sequence = "".join(line_split_entry[1:])
41
+ msa_entries.append((">"+header, sequence))
42
+ return MSATools(msa_entries)
43
+
44
+ @staticmethod
45
+ def get_sequence_max_cont_gaps(sequence: str) -> int:
46
+ """
47
+ Find maximum number of continuous gaps in a specific sequence.
48
+
49
+ Parameters
50
+ ----------
51
+ sequence : str
52
+ Sequence of characters, potentially containing multiple of '-', a gap character.
53
+
54
+ Returns
55
+ -------
56
+ int
57
+ The maximum number of continuous gaps in a sequence.
58
+ """
59
+ dash_match = re.findall(r"-+", sequence)
60
+ gap_counts = [len(match) for match in dash_match]
61
+ if len(gap_counts) > 0:
62
+ return max(gap_counts)
63
+ else:
64
+ return 0
65
+
66
+ def gap_frequency(self) -> tuple[dict[int, int], dict[int, float]]:
67
+ """
68
+ Calculates the frequency of maximum continuous gaps throughout the MSA where the key corresponds to the number of continous gaps and the value corresponds to the number of sequences or the cumulative percentage of their sequences.
69
+
70
+ Returns
71
+ -------
72
+ tuple of dict of int, int and dict of int, int
73
+ Two element tuple where first element is a frequency count dictionary and the second element is a cumulative percentage of sequences with a specific maximum number of continous gaps.
74
+ """
75
+ max_gap_counts = []
76
+ for header, sequence in self.MSA:
77
+ max_gap_counts.append(MSATools.get_sequence_max_cont_gaps(sequence))
78
+ frequency_count_dict = dict(Counter(max_gap_counts))
79
+ cumul_perc_dict = {}
80
+ cumul_count = 0
81
+ for key in sorted(frequency_count_dict.keys()):
82
+ value = frequency_count_dict[key]
83
+ cumul_count += value
84
+ cumul_perc_dict[key] = cumul_count / len(self.MSA)
85
+ return (frequency_count_dict, cumul_perc_dict)
86
+
87
+ def filter_by_continuous_gaps(self, max_gaps: Optional[int]=None) -> list[tuple[str, str]]:
88
+ """
89
+ Filter out entries in your MSA by the number of maximum continuous gaps specified unless None is provided. Also, removes .s and lowercase letters from the sequence.
90
+
91
+ Parameters
92
+ ----------
93
+ max_gaps : int
94
+ The maximum allowed number of continuous gaps in a sequence
95
+
96
+ Returns
97
+ -------
98
+ list of tuple of str, str
99
+ List of entries that are valid in that their sequences' number of maximum continuous gaps is within the threshold supplied as `max_gaps`.
100
+ """
101
+ table = str.maketrans('', '', string.ascii_lowercase+".")
102
+ if max_gaps == None:
103
+ kept_entries = []
104
+ for header, sequence in self.MSA:
105
+ sequence = sequence.translate(table)
106
+ kept_entries.append((header, sequence))
107
+ return kept_entries
108
+ else:
109
+ kept_entries = []
110
+ for header, sequence in self.MSA:
111
+ sequence = sequence.translate(table)
112
+ if MSATools.get_sequence_max_cont_gaps(sequence) <= max_gaps:
113
+ kept_entries.append((header, sequence))
114
+ return kept_entries
115
+
116
+ def write(self, filepath: str) -> None:
117
+ """
118
+ Writes this MSA's headers and sequences to the filepath specified.
119
+
120
+ Parameters
121
+ ----------
122
+ filepath : str
123
+ Filepath to write the MSA supplied to.
124
+
125
+ Returns
126
+ -------
127
+ None
128
+ """
129
+ with open(filepath, 'w') as fs:
130
+ for header, sequence in self.MSA:
131
+ fs.write(header)
132
+ fs.write("\n")
133
+ fs.write(sequence)
134
+ fs.write("\n")
135
+
136
+ def __len__(self):
137
+ """
138
+ Returns the number of sequences, and equivalently, the number of headers in the MSA.
139
+
140
+ Returns
141
+ -------
142
+ int
143
+ length of the MSA list of header, sequence tuples.
144
+ """
145
+ return len(self.MSA)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: dcatoolkit
3
- Version: 0.1.2
3
+ Version: 0.1.3
4
4
  Summary: Collection of useful modules and representations for managing DCA output data.
5
5
  Author-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
6
6
  Maintainer-email: Raheel Syed Ahmed <raheelsyedahmed@gmail.com>
@@ -2,6 +2,7 @@ LICENSE
2
2
  README.md
3
3
  pyproject.toml
4
4
  src/dcatoolkit/__init__.py
5
+ src/dcatoolkit/analytics.py
5
6
  src/dcatoolkit/representation.py
6
7
  src/dcatoolkit.egg-info/PKG-INFO
7
8
  src/dcatoolkit.egg-info/SOURCES.txt
@@ -1,3 +0,0 @@
1
-
2
- __version__ = "0.1.2"
3
- from .representation import Pairs, DirectInformationData, StructureInformation, ResidueAlignment
File without changes
File without changes
File without changes