drep 4.0.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,228 @@
1
+ import logging
2
+ import shutil
3
+
4
+ import pandas as pd
5
+
6
+ import drep
7
+ import drep.d_cluster.external
8
+ import drep.d_cluster.utils
9
+ import drep.d_cluster.compare_utils
10
+
11
+ class GenomeClusterController(object):
12
+ """
13
+ Handle the logic of comparing and clustering genomes
14
+ """
15
+ def __init__(self, wd, **kwargs):
16
+ self.wd = drep.WorkDirectory.WorkDirectory(wd)
17
+ self.kwargs = kwargs
18
+
19
+ # Handle special kwargs
20
+ self.debug = kwargs.get('debug', False)
21
+
22
+ def main(self, store_output=True):
23
+ """
24
+ Main entrypoint for the dRep cluster operation
25
+
26
+ Formerly called d_cluster_wrapper
27
+ """
28
+ # Get the arguments
29
+ self.parse_cluster_arguments()
30
+
31
+ # Run primary clustering
32
+ self.run_primary_clustering()
33
+
34
+ # Run secondary clustering
35
+ self.run_secondary_clustering()
36
+
37
+ # Save the output
38
+ if store_output:
39
+ self.store_output()
40
+ else:
41
+ self.return_output()
42
+
43
+ def parse_cluster_arguments(self):
44
+ """
45
+ Load the genomes and store Bdb in the wd
46
+ """
47
+ # Make sure the program this run actually needs is installed. Only the
48
+ # MASH primary path needs mash; the default (skani) does not.
49
+ primary_exe = 'mash' if self.kwargs.get('primary_algorithm', 'skani') == 'MASH' else 'skani'
50
+ if shutil.which(primary_exe) is None:
51
+ logging.error('Cannot locate the program {0}- make sure its in the system path' \
52
+ .format(primary_exe))
53
+
54
+ # If genomes are provided, load them
55
+ if self.kwargs.get('genomes', None) is not None:
56
+ assert self.wd.hasDb("Bdb") == False, \
57
+ "Don't provide new genomes- you already have them in the work directory"
58
+ Bdb = drep.d_cluster.utils.load_genomes(self.kwargs['genomes'])
59
+
60
+ # If genomes are not provided, don't load them
61
+ if self.kwargs.get('genomes', None) is None:
62
+ assert self.wd.hasDb("Bdb") != False, \
63
+ "Must either provide a genome list, or run the 'filter' operation with the same work directory"
64
+ Bdb = self.wd.get_db('Bdb')
65
+
66
+ # Make sure people weren't dumb with their cutoffs
67
+ for v in ['P_ani', 'S_ani']:
68
+ if self.kwargs.get(v) > 1:
69
+ logging.warning("{0} is set to {1}- this should be \
70
+ between 0-1, not 1-100".format(v, self.kwargs.get(v)))
71
+
72
+ # Load length and N50 if you need it
73
+ if self.kwargs.get('multiround_primary_clustering', False) | (self.kwargs.get('greedy_secondary_clustering', False)):
74
+ Bdb = drep.d_filter._add_lengthN50(Bdb, Bdb)
75
+
76
+ # Store the genomes
77
+ self.Bdb = Bdb
78
+
79
+ def run_primary_clustering(self):
80
+ """
81
+ Run primary clustering and end with an Mdb and an MCdb in this object
82
+ """
83
+ logging.info("Running primary clustering")
84
+ cached = (self.debug and self.wd.hasDb('Mdb') and self.wd.hasDb('CdbF'))
85
+
86
+ if self.kwargs.get('SkipMash', False):
87
+ logging.info("Nevermind! Skipping Mash")
88
+ # Make a "Cdb" where all genomes are in the same cluster
89
+ Cdb = drep.d_cluster.external._gen_nomash_cdb(self.Bdb)
90
+ # Make a blank "Mdb" for storage anyways
91
+ Mdb = pd.DataFrame({'Blank': []})
92
+
93
+ elif cached:
94
+ logging.info('Nevermind! Loading cached primary clustering')
95
+ Mdb = self.wd.get_db('Mdb')
96
+ Cdb = self.wd.get_db('CdbF')
97
+ logging.info('2. Primary clustering cache loaded')
98
+
99
+ elif len(self.Bdb) < 2:
100
+ logging.warning("Fewer than 2 genomes remain after filtering — skipping MASH clustering")
101
+ Cdb = drep.d_cluster.external._gen_nomash_cdb(self.Bdb)
102
+ Mdb = pd.DataFrame({'Blank': []})
103
+
104
+ else:
105
+ logging.info("Running pair-wise MASH clustering")
106
+ Mdb, Cdb, cluster_ret = drep.d_cluster.compare_utils.all_vs_all_primary(self.Bdb, self.wd.get_dir('MASH'), **self.kwargs)
107
+
108
+ if self.debug:
109
+ logging.debug("Debug mode on - saving Mdb ASAP")
110
+ self.wd.store_db(Mdb, 'Mdb')
111
+
112
+ logging.debug("Debug mode on - saving CdbF ASAP")
113
+ self.wd.store_db(Cdb, 'CdbF')
114
+
115
+ # Store the primary clustering results
116
+ self.wd.store_special('primary_linkage', cluster_ret)
117
+
118
+ logging.info("{0} primary clusters made".format(len(Cdb['primary_cluster'].unique())))
119
+ self.Mdb = Mdb
120
+ self.MCdb = Cdb
121
+
122
+ def run_secondary_clustering(self):
123
+ logging.info("Running secondary clustering")
124
+
125
+ # Get the arguments
126
+ algorithm = self.kwargs.get('S_algorithm', 'ANImf')
127
+ cached = (self.debug and self.wd.hasDb('Ndb') and self.wd.hasDb('Cdb'))
128
+ p = self.kwargs.get('processors', 6)
129
+ self.deal_with_nucmer_presets()
130
+
131
+ # Wipe any old secondary clusters
132
+ self.wd._wipe_secondary_clusters()
133
+
134
+ if len(self.Bdb) < 2:
135
+ logging.warning("Fewer than 2 genomes remain after filtering — skipping secondary clustering")
136
+ Cdb = drep.d_cluster.utils._gen_nomani_cdb(self.MCdb, data_folder=self.wd.get_dir('data'), **self.kwargs)
137
+ Ndb = pd.DataFrame({'Blank': []})
138
+
139
+ elif not self.kwargs.get('SkipSecondary', False):
140
+ if cached:
141
+ logging.info('3. Loading cached secondary clustering')
142
+ Ndb = self.wd.get_db('Ndb')
143
+ Cdb = self.wd.get_db('Cdb')
144
+
145
+ # Get rid of broken ones
146
+ base = len(Ndb)
147
+ Ndb = Ndb.dropna(subset=['reference'])
148
+ logging.info(f'!!! {len(Ndb) - base} lines from Ndb failed!')
149
+
150
+ logging.info('3. Secondary clustering cache loaded')
151
+
152
+ # Reuse primary's skani edges instead of re-running skani per cluster
153
+ elif self.can_reuse_primary_edges(algorithm):
154
+ logging.info("Reusing primary skani comparisons for secondary clustering")
155
+ Ndb, Cdb, c2ret = drep.d_cluster.compare_utils.secondary_clustering_from_primary_edges(
156
+ self.Bdb, self.MCdb, self.Mdb, **self.kwargs)
157
+ if self.debug:
158
+ self.wd.store_db(Ndb, 'Ndb')
159
+ self.wd.store_db(Cdb, 'Cdb')
160
+ self.wd.store_special('secondary_linkages', c2ret)
161
+
162
+ # Run comparisons, make Ndb
163
+ else:
164
+ drep.d_cluster.utils._print_time_estimate(self.Bdb, self.MCdb, algorithm, p)
165
+ Ndb, Cdb, c2ret = drep.d_cluster.compare_utils.secondary_clustering(self.Bdb, self.MCdb, algorithm, self.wd.get_dir('data'), wd=self.wd, **self.kwargs)
166
+ if self.debug:
167
+ logging.debug("Debug mode on - saving Ndb ASAP")
168
+ self.wd.store_db(Ndb, 'Ndb')
169
+ self.wd.store_db(Cdb, 'Cdb')
170
+
171
+ # Store the secondary clustering results
172
+ self.wd.store_special('secondary_linkages', c2ret)
173
+
174
+ else:
175
+ logging.info("3. Nevermind! Skipping secondary clustering")
176
+ Cdb = drep.d_cluster.utils._gen_nomani_cdb(self.MCdb, data_folder=self.wd.get_dir('data'), **self.kwargs)
177
+ Ndb = pd.DataFrame({'Blank': []})
178
+
179
+ logging.info(
180
+ "Step 4. Return output")
181
+
182
+ self.Cdb = Cdb
183
+ self.Ndb = Ndb
184
+
185
+ def can_reuse_primary_edges(self, algorithm):
186
+ """
187
+ Whether secondary clustering can be derived from primary's edges rather
188
+ than re-running comparisons.
189
+
190
+ This only holds when primary was skani (so Mdb contains real ANI plus
191
+ alignment fractions for every pair above the screen) and secondary wants
192
+ skani too. Any other secondary algorithm measures something different and
193
+ has to run for itself; greedy has its own code path.
194
+ """
195
+ if self.kwargs.get('reuse_primary_comparisons', True) is False:
196
+ return False
197
+ if self.kwargs.get('primary_algorithm', 'skani') != 'skani':
198
+ return False
199
+ if algorithm != 'skani':
200
+ return False
201
+ if self.kwargs.get('greedy_secondary_clustering', False):
202
+ return False
203
+ # Mdb must be the skani edge table, not a Mash table or a blank
204
+ return (self.Mdb is not None) and ('alignment_coverage' in self.Mdb.columns)
205
+
206
+ def store_output(self):
207
+ logging.debug("Main program run complete- saving output")
208
+ self.wd.store_db(self.Cdb, 'Cdb')
209
+ self.wd.store_db(self.Mdb, 'Mdb')
210
+ self.wd.store_db(self.Ndb, 'Ndb')
211
+ if not self.wd.hasDb('Bdb'):
212
+ self.wd.store_db(self.Bdb, 'Bdb')
213
+
214
+ # Log arguments
215
+ self.wd.store_special('cluster_log', self.kwargs)
216
+
217
+ def return_output(self):
218
+ return self.Cdb, self.Mdb, self.Ndb
219
+
220
+ def deal_with_nucmer_presets(self):
221
+ if self.kwargs.get('n_preset', None) != None:
222
+ self.kwargs['n_c'], self.kwargs['n_maxgap'], self.kwargs['n_noextend'], self.kwargs['n_method'] \
223
+ = drep.d_cluster.external._nucmer_preset(self.kwargs['n_PRESET'])
224
+
225
+ def d_cluster_wrapper(workDirectory, **kwargs):
226
+ GenomeClusterController(workDirectory, **kwargs).main()
227
+
228
+