drep 4.0.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
drep/d_evaluate.py ADDED
@@ -0,0 +1,355 @@
1
+ #!/usr/bin/env python3
2
+
3
+ import logging
4
+ import os
5
+ import pandas as pd
6
+ import sys
7
+
8
+ import drep
9
+ import drep.WorkDirectory
10
+ import drep.d_cluster.compare_utils
11
+ import drep.d_cluster.controller
12
+ import drep.d_filter
13
+ import drep.d_cluster
14
+ import drep.d_analyze
15
+
16
+ def d_evaluate_wrapper(wd,**kwargs):
17
+
18
+ logging.debug("Loading work directory")
19
+ wd = drep.WorkDirectory.WorkDirectory(wd)
20
+ logging.debug(str(wd))
21
+
22
+ # Run tertiary clustering if that's what you're here to do
23
+ if kwargs.get('run_tertiary_clustering', False):
24
+ logging.info("Running tertiary clustering on genome representatives")
25
+ run_tertiary_clustering(wd, **kwargs)
26
+
27
+ # Determine what to evaluate
28
+ evs = kwargs.get('evaluate')
29
+ options = ['1','2','3']
30
+ to_eval = drep.d_analyze._parse_plot_options(options, evs)
31
+ logging.debug("evaluating {0}".format(to_eval))
32
+
33
+ # 1) Evaluate de-replicated genome similarity
34
+ if '1' in to_eval:
35
+ logging.info('will compare winners')
36
+ Wmdb, Wndb = compare_winners(wd,**kwargs)
37
+
38
+ # Save databases
39
+ wd.store_db(Wmdb,'Wmdb',overwrite=True)
40
+ wd.store_db(Wndb,'Wndb',overwrite=True)
41
+
42
+ # 2) Throw warnings for clusters that were almost different
43
+ if '2' in to_eval:
44
+ logging.info('will provide warnings about clusters')
45
+ warnings = evaluate_warnings(wd, **kwargs)
46
+
47
+ # Save it
48
+ warn_log = os.path.join(wd.location, 'log/warnings.txt')
49
+ with open(warn_log, 'w') as file:
50
+ file.write('\n'.join(warnings))
51
+ file.write('\n')
52
+ logging.info("{0} warnings generated: saved to {1}".format(len(warnings),warn_log))
53
+
54
+ # 3) Generate a database of information on winning genomes
55
+ if '3' in to_eval:
56
+ logging.info('will produce Widb (winner information db)')
57
+ Widb = evaluate_winners(wd, **kwargs)
58
+
59
+ # Save it
60
+ wd.store_db(Widb,'Widb',overwrite=True)
61
+ loc = wd.location + 'data_tables/Widb.csv'
62
+ logging.info("Winner database saved to {0}".format(loc))
63
+
64
+ return
65
+
66
+ def compare_winners(wd, **kwargs):
67
+ Bdb = wd.get_db('Bdb')
68
+ Wdb = wd.get_db('Wdb')
69
+ Bdb = Bdb[Bdb['genome'].isin(Wdb['genome'].tolist())]
70
+ data_folder = os.path.join(wd.location, 'data/')
71
+ comp_method = kwargs.get('comp_method','ANIn')
72
+
73
+ # Generate MASH db
74
+ Wmdb = drep.d_cluster.compare_utils.all_vs_all_MASH(Bdb, data_folder)
75
+
76
+ # Generate ANIn db
77
+ Wndb = drep.d_cluster.compare_utils.compare_genomes(Bdb, comp_method, wd, **kwargs)
78
+
79
+ return Wmdb, Wndb
80
+
81
+ def evaluate_warnings(wd, **kwargs):
82
+ Cdb = wd.get_db('Cdb')
83
+ Ndb = wd.get_db('Ndb')
84
+ warn_dist = float(kwargs.get('warn_dist',.25))
85
+
86
+ if 'Blank' in Ndb:
87
+ logging.error("Ndb is blank (was run with --SkipSecondary) - "\
88
+ + "will skip cluster evaluation")
89
+
90
+ if wd.hasDb('Wmdb'):
91
+ Wmdb = wd.get_db('Wmdb')
92
+ Wndb = wd.get_db('Wndb')
93
+ warn_sim = float(kwargs.get('warn_sim',.98))
94
+ warn_aln = float(kwargs.get('warn_aln',.25))
95
+
96
+ warnings = []
97
+
98
+ # Check for cases where there is a close clustering
99
+ # That is, a case where a cluster was almost split or not-split
100
+ for cluster in sorted(Cdb['primary_cluster'].unique()):
101
+ d = Cdb[Cdb['primary_cluster'] == cluster]
102
+
103
+ if 'Blank' in Ndb:
104
+ continue
105
+
106
+ # Skip if it's a singleton
107
+ if len(d['genome'].unique()) == 1:
108
+ continue
109
+
110
+ # Load clustering information
111
+ linkI = wd.get_cluster("secondary_linkage_cluster_{0}".format(cluster))
112
+ db = linkI['db']
113
+ linkage = linkI['linkage']
114
+ args = linkI['arguments']
115
+ threshold = args['linkage_cutoff']
116
+ threshold = (1-threshold) * 100
117
+ names = list(db.columns)
118
+
119
+ if linkage is None:
120
+ continue
121
+
122
+ # Find cases where the cluster is close to the treshold used
123
+ for node in linkage:
124
+ t = node[2]
125
+ x = (1-t)*100
126
+ if abs(x - threshold) < warn_dist:
127
+
128
+ # This means that it wasn't split, but was almost
129
+ if x < threshold:
130
+ #c1 = Cdb['secondary_cluster'][Cdb['genome'] == names[node[0]]].tolist()[0]
131
+ #c2 = Cdb['secondary_cluster'][Cdb['genome'] == names[node[1]]].tolist()[0]
132
+
133
+ warning = "CLUSTERING WARNING: Primary cluster {0} was almost not split".\
134
+ format(cluster)
135
+ warnings.append(warning)
136
+
137
+ # This means that a cluster was split, but almost not
138
+ if x > threshold:
139
+ #print(node)
140
+ #c1 = Cdb['secondary_cluster'][Cdb['genome'] == names[int(node[0])]].tolist()[0]
141
+ #c2 = Cdb['secondary_cluster'][Cdb['genome'] == names[int(node[1])]].tolist()[0]
142
+ #assert c1 == c2
143
+
144
+ warning = "CLUSTERING WARNING: Primary cluster {0} was almost split".\
145
+ format(cluster)
146
+ warnings.append(warning)
147
+
148
+ # Find cases where you have a high self comparison
149
+ self_thresh = (1-drep.d_analyze.get_highest_self(Ndb, names))*100
150
+ if self_thresh <= threshold + warn_dist:
151
+ warning = "CLUSTERING WARNING: Primary cluster {0} has a high self-comparison value".\
152
+ format(cluster)
153
+ warnings.append(warning)
154
+
155
+ # Check for cases where winners are very similar
156
+ # Either based on MASH or ANIn
157
+ if wd.hasDb('Wmdb'):
158
+ # See if any MASH comparisons are too similar
159
+ Wmdb = Wmdb[(Wmdb['genome1'] != Wmdb['genome2']) & (Wmdb['similarity'] > warn_sim)\
160
+ & (Wmdb['genome1'] > Wmdb['genome2'])]
161
+ for i,row in Wmdb.iterrows():
162
+ c1 = Cdb['secondary_cluster'][Cdb['genome'] == row['genome1']].tolist()[0]
163
+ c2 = Cdb['secondary_cluster'][Cdb['genome'] == row['genome2']].tolist()[0]
164
+
165
+ warning = "WINNER WARNING: Genomes {0} ({3}) and {1} ({4}) have a high MASH score ({2:.2f}%)".format(\
166
+ row['genome1'], row['genome2'], row['similarity']*100, c1, c2)
167
+ warnings.append(warning)
168
+
169
+ # See if any secondary comparisons are too similar
170
+ Wndb = Wndb[(Wndb['reference'] != Wndb['querry']) & (Wndb['ani'] > warn_sim)\
171
+ & (Wndb['reference'] > Wndb['querry']) & (Wndb['alignment_coverage'] > warn_aln)]
172
+ for i,row in Wndb.iterrows():
173
+ c1 = Cdb['secondary_cluster'][Cdb['genome'] == row['reference']].tolist()[0]
174
+ c2 = Cdb['secondary_cluster'][Cdb['genome'] == row['querry']].tolist()[0]
175
+
176
+ warning = "WINNER WARNING: Genomes {0} ({3}) and {1} ({4}) have a high ANIn score ({2:.2f}% ANI ".format(\
177
+ row['reference'], row['querry'], row['ani']*100, c1, c2) + "; {0:.2f}% aligned)".format(row['alignment_coverage'])
178
+ warnings.append(warning)
179
+
180
+ return warnings
181
+
182
+ def comp_str(val):
183
+ if val == 100:
184
+ return('perfect')
185
+ if val > 90:
186
+ return("near")
187
+ if val > 70:
188
+ return("substantial")
189
+ if val > 50:
190
+ return("moderate")
191
+ if val <= 50:
192
+ return("partial")
193
+
194
+ def con_str(val):
195
+ if val == 0:
196
+ return ("none")
197
+ if val <= 5:
198
+ return("low")
199
+ if val < 10:
200
+ return ("medium")
201
+ if val < 15:
202
+ return("high")
203
+ if val >= 15:
204
+ return("very high")
205
+
206
+
207
+ def evaluate_winners(wd, **kwrags):
208
+ Wdb = wd.get_db('Wdb')
209
+ Cdb = wd.get_db('Cdb')
210
+ Ndb = wd.get_db('Ndb')
211
+
212
+ # For every winning genome, give some key stats based on the information available
213
+ Table = {'genome':[],'score':[],'completeness':[],'contamination':[],'strain_heterogeneity':[],\
214
+ 'size':[],'N50':[],'cluster':[], 'taxonomy':[], 'tax_confidence':[],\
215
+ 'cluster_members':[],'closest_cluster_member':[],'furthest_cluster_member':[],\
216
+ 'completeness_metric':[], 'contamination_metric':[]}
217
+
218
+ for i, row in Wdb.iterrows():
219
+ # Add info in Wdb
220
+ Table['genome'].append(row['genome'])
221
+ Table['score'].append(row['score'])
222
+ Table['cluster'].append(row['cluster'])
223
+
224
+ # Add clustering info
225
+ if 'Blank' not in Ndb:
226
+ d = Cdb[Cdb['secondary_cluster'] == row['cluster']]
227
+ ndb = Ndb[(Ndb['reference'] == row['genome']) & (Ndb['querry'].isin(d['genome'].tolist()))\
228
+ & (Ndb['reference'] != Ndb['querry'])]
229
+ members = len(d['genome'].unique())
230
+
231
+ if members > 1:
232
+ Table['closest_cluster_member'].append("{0:.2f}".format(ndb['ani'].max()*100))
233
+ Table['furthest_cluster_member'].append("{0:.2f}".format(ndb['ani'].min()*100))
234
+ Table['cluster_members'].append(len(d['genome'].unique()))
235
+ else:
236
+ Table['closest_cluster_member'].append("NA")
237
+ Table['furthest_cluster_member'].append("NA")
238
+ Table['cluster_members'].append(len(d['genome'].unique()))
239
+ else:
240
+ d = Cdb[Cdb['secondary_cluster'] == row['cluster']]
241
+ Table['closest_cluster_member'].append("NA")
242
+ Table['furthest_cluster_member'].append("NA")
243
+ Table['cluster_members'].append(len(d['genome'].unique()))
244
+
245
+ # Add checkM info
246
+ if wd.hasDb('Chdb'):
247
+ Chdb = wd.get_db('Chdb')
248
+ d = Chdb[Chdb['Bin Id'] == row['genome']]
249
+ Table['completeness'].append(d['Completeness'].tolist()[0])
250
+ Table['contamination'].append(d['Contamination'].tolist()[0])
251
+ Table['strain_heterogeneity'].append(d['Strain heterogeneity'].tolist()[0])
252
+ Table['size'].append(d['Genome size (bp)'].tolist()[0])
253
+ Table['N50'].append(d['N50 (scaffolds)'].tolist()[0])
254
+ Table['completeness_metric'].append(comp_str(d['Completeness'].tolist()[0]))
255
+ Table['contamination_metric'].append(con_str(d['Contamination'].tolist()[0]))
256
+ elif wd.hasDb('genomeInfo'):
257
+ Gdb = wd.get_db('genomeInfo')
258
+ d = Gdb[Gdb['genome'] == row['genome']]
259
+ if len(d) > 0:
260
+ comp = d['completeness'].tolist()[0]
261
+ con = d['contamination'].tolist()[0]
262
+ Table['completeness'].append(comp)
263
+ Table['contamination'].append(con)
264
+ Table['strain_heterogeneity'].append(d['strain_heterogeneity'].tolist()[0] if 'strain_heterogeneity' in d.columns else "NA")
265
+ Table['size'].append(d['length'].tolist()[0] if 'length' in d.columns else "NA")
266
+ Table['N50'].append(d['N50'].tolist()[0] if 'N50' in d.columns else "NA")
267
+ Table['completeness_metric'].append(comp_str(comp))
268
+ Table['contamination_metric'].append(con_str(con))
269
+ else:
270
+ Table['completeness'].append("NA")
271
+ Table['contamination'].append("NA")
272
+ Table['strain_heterogeneity'].append("NA")
273
+ Table['size'].append("NA")
274
+ Table['N50'].append("NA")
275
+ Table['completeness_metric'].append("NA")
276
+ Table['contamination_metric'].append("NA")
277
+ else:
278
+ Table['completeness'].append("NA")
279
+ Table['contamination'].append("NA")
280
+ Table['strain_heterogeneity'].append("NA")
281
+ Table['size'].append("NA")
282
+ Table['N50'].append("NA")
283
+ Table['completeness_metric'].append("NA")
284
+ Table['contamination_metric'].append("NA")
285
+
286
+ # Add taxonomy info
287
+ if wd.hasDb('Tdb'):
288
+ Tdb = wd.get_db('Tdb')
289
+ d = Tdb[Tdb['genome'] == row['genome']]
290
+ Table['taxonomy'].append(d['taxonomy'][d['tax_confidence'] == d['tax_confidence'].max()].tolist()[0])
291
+ Table['tax_confidence'].append(d['tax_confidence'].max())
292
+ else:
293
+ Table['taxonomy'].append('NA')
294
+ Table['tax_confidence'].append('NA')
295
+
296
+ Widb = pd.DataFrame(Table)
297
+ return Widb
298
+
299
+ def run_tertiary_clustering(wd, **kwargs):
300
+ # Create a new workdir inside data
301
+ new_wd_loc = wd.get_dir('data') + 'tertiary_clustering'
302
+ nWd = drep.WorkDirectory.WorkDirectory(new_wd_loc)
303
+
304
+ # Make a copy of the kwargs
305
+ kwargs_copy = kwargs.copy()
306
+ if 'genomes' in kwargs_copy:
307
+ del kwargs_copy['genomes']
308
+ kwargs_copy['P_ani'] = kwargs_copy['P_ani'] - 0.05
309
+ kwargs_copy['multiround_primary_clustering'] = False
310
+ kwargs_copy['greedy_secondary_clustering'] = False
311
+
312
+ # Figure out what genomes you're going to compare and make a new Bdb
313
+ Bdb = wd.get_db('Bdb')
314
+ Cdb = wd.get_db('Cdb')
315
+ Wdb = wd.get_db('Wdb')
316
+ BBdb = Bdb[Bdb['genome'].isin(Wdb['genome'].tolist())]
317
+ nWd.store_db(BBdb, 'Bdb')
318
+
319
+ # Copy over genome info for choose
320
+ for name in ['genomeInfo', 'Chdb']:
321
+ if wd.hasDb(name):
322
+ nWd.store_db(wd.get_db(name), name)
323
+
324
+ # Run the comparison
325
+ drep.d_cluster.controller.d_cluster_wrapper(new_wd_loc, **kwargs_copy)
326
+
327
+ # Reconsile results for Cdb
328
+ nWd = drep.WorkDirectory.WorkDirectory(new_wd_loc)
329
+ nCdb = nWd.get_db('Cdb')
330
+
331
+ Cdb['original_secondary_cluster'] = Cdb['secondary_cluster']
332
+ del Cdb['secondary_cluster']
333
+ Cdb = pd.merge(Cdb, nCdb[['genome', 'secondary_cluster']], on='genome', how='left')
334
+
335
+ rep_cdb = Cdb.dropna()
336
+ old2new = rep_cdb.set_index('original_secondary_cluster')['secondary_cluster'].to_dict()
337
+ Cdb['secondary_cluster'] = Cdb['original_secondary_cluster'].map(old2new)
338
+
339
+ # Rename clusters
340
+ old2new_names = {}
341
+ for clust, db in Cdb.groupby('secondary_cluster'):
342
+ if len(db['original_secondary_cluster'].unique()) == 1:
343
+ old2new_names[clust] = db['original_secondary_cluster'].iloc[0]
344
+ else:
345
+ new_name = f"{'.'.join(sorted(list(set([x.split('_')[0] for x in list(db['original_secondary_cluster'].unique())]))))}" + \
346
+ f"_{'.'.join(sorted(list(set([x.split('_')[1] for x in list(db['original_secondary_cluster'].unique())]))))}"
347
+ old2new_names[clust] = new_name
348
+ logging.debug(f"Clusters {list(db['original_secondary_cluster'].unique())} were merged into {new_name}")
349
+ Cdb['secondary_cluster'] = Cdb['secondary_cluster'].map(old2new_names)
350
+
351
+ # Store new Cdb
352
+ wd.store_db(Cdb, 'Cdb')
353
+
354
+ # Re-run choose
355
+ drep.d_choose.d_choose_wrapper(wd.location, **kwargs_copy)