drep 4.0.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- drep/VERSION +1 -0
- drep/WorkDirectory.py +355 -0
- drep/__init__.py +101 -0
- drep/argumentParser.py +279 -0
- drep/controller.py +105 -0
- drep/d_adjust.py +272 -0
- drep/d_analyze.py +1613 -0
- drep/d_bonus.py +429 -0
- drep/d_choose.py +362 -0
- drep/d_cluster/__init__.py +0 -0
- drep/d_cluster/cluster_utils.py +126 -0
- drep/d_cluster/compare_utils.py +636 -0
- drep/d_cluster/controller.py +228 -0
- drep/d_cluster/external.py +765 -0
- drep/d_cluster/greedy_clustering.py +181 -0
- drep/d_cluster/parsers.py +0 -0
- drep/d_cluster/union_find.py +543 -0
- drep/d_cluster/utils.py +687 -0
- drep/d_evaluate.py +355 -0
- drep/d_filter.py +831 -0
- drep/d_workflows.py +135 -0
- drep-4.0.2.data/scripts/ScaffoldLevel_dRep.py +1101 -0
- drep-4.0.2.data/scripts/dRep +32 -0
- drep-4.0.2.data/scripts/parse_stb.py +140 -0
- drep-4.0.2.dist-info/METADATA +23 -0
- drep-4.0.2.dist-info/RECORD +28 -0
- drep-4.0.2.dist-info/WHEEL +5 -0
- drep-4.0.2.dist-info/top_level.txt +1 -0
drep/d_evaluate.py
ADDED
|
@@ -0,0 +1,355 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
import os
|
|
5
|
+
import pandas as pd
|
|
6
|
+
import sys
|
|
7
|
+
|
|
8
|
+
import drep
|
|
9
|
+
import drep.WorkDirectory
|
|
10
|
+
import drep.d_cluster.compare_utils
|
|
11
|
+
import drep.d_cluster.controller
|
|
12
|
+
import drep.d_filter
|
|
13
|
+
import drep.d_cluster
|
|
14
|
+
import drep.d_analyze
|
|
15
|
+
|
|
16
|
+
def d_evaluate_wrapper(wd,**kwargs):
|
|
17
|
+
|
|
18
|
+
logging.debug("Loading work directory")
|
|
19
|
+
wd = drep.WorkDirectory.WorkDirectory(wd)
|
|
20
|
+
logging.debug(str(wd))
|
|
21
|
+
|
|
22
|
+
# Run tertiary clustering if that's what you're here to do
|
|
23
|
+
if kwargs.get('run_tertiary_clustering', False):
|
|
24
|
+
logging.info("Running tertiary clustering on genome representatives")
|
|
25
|
+
run_tertiary_clustering(wd, **kwargs)
|
|
26
|
+
|
|
27
|
+
# Determine what to evaluate
|
|
28
|
+
evs = kwargs.get('evaluate')
|
|
29
|
+
options = ['1','2','3']
|
|
30
|
+
to_eval = drep.d_analyze._parse_plot_options(options, evs)
|
|
31
|
+
logging.debug("evaluating {0}".format(to_eval))
|
|
32
|
+
|
|
33
|
+
# 1) Evaluate de-replicated genome similarity
|
|
34
|
+
if '1' in to_eval:
|
|
35
|
+
logging.info('will compare winners')
|
|
36
|
+
Wmdb, Wndb = compare_winners(wd,**kwargs)
|
|
37
|
+
|
|
38
|
+
# Save databases
|
|
39
|
+
wd.store_db(Wmdb,'Wmdb',overwrite=True)
|
|
40
|
+
wd.store_db(Wndb,'Wndb',overwrite=True)
|
|
41
|
+
|
|
42
|
+
# 2) Throw warnings for clusters that were almost different
|
|
43
|
+
if '2' in to_eval:
|
|
44
|
+
logging.info('will provide warnings about clusters')
|
|
45
|
+
warnings = evaluate_warnings(wd, **kwargs)
|
|
46
|
+
|
|
47
|
+
# Save it
|
|
48
|
+
warn_log = os.path.join(wd.location, 'log/warnings.txt')
|
|
49
|
+
with open(warn_log, 'w') as file:
|
|
50
|
+
file.write('\n'.join(warnings))
|
|
51
|
+
file.write('\n')
|
|
52
|
+
logging.info("{0} warnings generated: saved to {1}".format(len(warnings),warn_log))
|
|
53
|
+
|
|
54
|
+
# 3) Generate a database of information on winning genomes
|
|
55
|
+
if '3' in to_eval:
|
|
56
|
+
logging.info('will produce Widb (winner information db)')
|
|
57
|
+
Widb = evaluate_winners(wd, **kwargs)
|
|
58
|
+
|
|
59
|
+
# Save it
|
|
60
|
+
wd.store_db(Widb,'Widb',overwrite=True)
|
|
61
|
+
loc = wd.location + 'data_tables/Widb.csv'
|
|
62
|
+
logging.info("Winner database saved to {0}".format(loc))
|
|
63
|
+
|
|
64
|
+
return
|
|
65
|
+
|
|
66
|
+
def compare_winners(wd, **kwargs):
|
|
67
|
+
Bdb = wd.get_db('Bdb')
|
|
68
|
+
Wdb = wd.get_db('Wdb')
|
|
69
|
+
Bdb = Bdb[Bdb['genome'].isin(Wdb['genome'].tolist())]
|
|
70
|
+
data_folder = os.path.join(wd.location, 'data/')
|
|
71
|
+
comp_method = kwargs.get('comp_method','ANIn')
|
|
72
|
+
|
|
73
|
+
# Generate MASH db
|
|
74
|
+
Wmdb = drep.d_cluster.compare_utils.all_vs_all_MASH(Bdb, data_folder)
|
|
75
|
+
|
|
76
|
+
# Generate ANIn db
|
|
77
|
+
Wndb = drep.d_cluster.compare_utils.compare_genomes(Bdb, comp_method, wd, **kwargs)
|
|
78
|
+
|
|
79
|
+
return Wmdb, Wndb
|
|
80
|
+
|
|
81
|
+
def evaluate_warnings(wd, **kwargs):
|
|
82
|
+
Cdb = wd.get_db('Cdb')
|
|
83
|
+
Ndb = wd.get_db('Ndb')
|
|
84
|
+
warn_dist = float(kwargs.get('warn_dist',.25))
|
|
85
|
+
|
|
86
|
+
if 'Blank' in Ndb:
|
|
87
|
+
logging.error("Ndb is blank (was run with --SkipSecondary) - "\
|
|
88
|
+
+ "will skip cluster evaluation")
|
|
89
|
+
|
|
90
|
+
if wd.hasDb('Wmdb'):
|
|
91
|
+
Wmdb = wd.get_db('Wmdb')
|
|
92
|
+
Wndb = wd.get_db('Wndb')
|
|
93
|
+
warn_sim = float(kwargs.get('warn_sim',.98))
|
|
94
|
+
warn_aln = float(kwargs.get('warn_aln',.25))
|
|
95
|
+
|
|
96
|
+
warnings = []
|
|
97
|
+
|
|
98
|
+
# Check for cases where there is a close clustering
|
|
99
|
+
# That is, a case where a cluster was almost split or not-split
|
|
100
|
+
for cluster in sorted(Cdb['primary_cluster'].unique()):
|
|
101
|
+
d = Cdb[Cdb['primary_cluster'] == cluster]
|
|
102
|
+
|
|
103
|
+
if 'Blank' in Ndb:
|
|
104
|
+
continue
|
|
105
|
+
|
|
106
|
+
# Skip if it's a singleton
|
|
107
|
+
if len(d['genome'].unique()) == 1:
|
|
108
|
+
continue
|
|
109
|
+
|
|
110
|
+
# Load clustering information
|
|
111
|
+
linkI = wd.get_cluster("secondary_linkage_cluster_{0}".format(cluster))
|
|
112
|
+
db = linkI['db']
|
|
113
|
+
linkage = linkI['linkage']
|
|
114
|
+
args = linkI['arguments']
|
|
115
|
+
threshold = args['linkage_cutoff']
|
|
116
|
+
threshold = (1-threshold) * 100
|
|
117
|
+
names = list(db.columns)
|
|
118
|
+
|
|
119
|
+
if linkage is None:
|
|
120
|
+
continue
|
|
121
|
+
|
|
122
|
+
# Find cases where the cluster is close to the treshold used
|
|
123
|
+
for node in linkage:
|
|
124
|
+
t = node[2]
|
|
125
|
+
x = (1-t)*100
|
|
126
|
+
if abs(x - threshold) < warn_dist:
|
|
127
|
+
|
|
128
|
+
# This means that it wasn't split, but was almost
|
|
129
|
+
if x < threshold:
|
|
130
|
+
#c1 = Cdb['secondary_cluster'][Cdb['genome'] == names[node[0]]].tolist()[0]
|
|
131
|
+
#c2 = Cdb['secondary_cluster'][Cdb['genome'] == names[node[1]]].tolist()[0]
|
|
132
|
+
|
|
133
|
+
warning = "CLUSTERING WARNING: Primary cluster {0} was almost not split".\
|
|
134
|
+
format(cluster)
|
|
135
|
+
warnings.append(warning)
|
|
136
|
+
|
|
137
|
+
# This means that a cluster was split, but almost not
|
|
138
|
+
if x > threshold:
|
|
139
|
+
#print(node)
|
|
140
|
+
#c1 = Cdb['secondary_cluster'][Cdb['genome'] == names[int(node[0])]].tolist()[0]
|
|
141
|
+
#c2 = Cdb['secondary_cluster'][Cdb['genome'] == names[int(node[1])]].tolist()[0]
|
|
142
|
+
#assert c1 == c2
|
|
143
|
+
|
|
144
|
+
warning = "CLUSTERING WARNING: Primary cluster {0} was almost split".\
|
|
145
|
+
format(cluster)
|
|
146
|
+
warnings.append(warning)
|
|
147
|
+
|
|
148
|
+
# Find cases where you have a high self comparison
|
|
149
|
+
self_thresh = (1-drep.d_analyze.get_highest_self(Ndb, names))*100
|
|
150
|
+
if self_thresh <= threshold + warn_dist:
|
|
151
|
+
warning = "CLUSTERING WARNING: Primary cluster {0} has a high self-comparison value".\
|
|
152
|
+
format(cluster)
|
|
153
|
+
warnings.append(warning)
|
|
154
|
+
|
|
155
|
+
# Check for cases where winners are very similar
|
|
156
|
+
# Either based on MASH or ANIn
|
|
157
|
+
if wd.hasDb('Wmdb'):
|
|
158
|
+
# See if any MASH comparisons are too similar
|
|
159
|
+
Wmdb = Wmdb[(Wmdb['genome1'] != Wmdb['genome2']) & (Wmdb['similarity'] > warn_sim)\
|
|
160
|
+
& (Wmdb['genome1'] > Wmdb['genome2'])]
|
|
161
|
+
for i,row in Wmdb.iterrows():
|
|
162
|
+
c1 = Cdb['secondary_cluster'][Cdb['genome'] == row['genome1']].tolist()[0]
|
|
163
|
+
c2 = Cdb['secondary_cluster'][Cdb['genome'] == row['genome2']].tolist()[0]
|
|
164
|
+
|
|
165
|
+
warning = "WINNER WARNING: Genomes {0} ({3}) and {1} ({4}) have a high MASH score ({2:.2f}%)".format(\
|
|
166
|
+
row['genome1'], row['genome2'], row['similarity']*100, c1, c2)
|
|
167
|
+
warnings.append(warning)
|
|
168
|
+
|
|
169
|
+
# See if any secondary comparisons are too similar
|
|
170
|
+
Wndb = Wndb[(Wndb['reference'] != Wndb['querry']) & (Wndb['ani'] > warn_sim)\
|
|
171
|
+
& (Wndb['reference'] > Wndb['querry']) & (Wndb['alignment_coverage'] > warn_aln)]
|
|
172
|
+
for i,row in Wndb.iterrows():
|
|
173
|
+
c1 = Cdb['secondary_cluster'][Cdb['genome'] == row['reference']].tolist()[0]
|
|
174
|
+
c2 = Cdb['secondary_cluster'][Cdb['genome'] == row['querry']].tolist()[0]
|
|
175
|
+
|
|
176
|
+
warning = "WINNER WARNING: Genomes {0} ({3}) and {1} ({4}) have a high ANIn score ({2:.2f}% ANI ".format(\
|
|
177
|
+
row['reference'], row['querry'], row['ani']*100, c1, c2) + "; {0:.2f}% aligned)".format(row['alignment_coverage'])
|
|
178
|
+
warnings.append(warning)
|
|
179
|
+
|
|
180
|
+
return warnings
|
|
181
|
+
|
|
182
|
+
def comp_str(val):
|
|
183
|
+
if val == 100:
|
|
184
|
+
return('perfect')
|
|
185
|
+
if val > 90:
|
|
186
|
+
return("near")
|
|
187
|
+
if val > 70:
|
|
188
|
+
return("substantial")
|
|
189
|
+
if val > 50:
|
|
190
|
+
return("moderate")
|
|
191
|
+
if val <= 50:
|
|
192
|
+
return("partial")
|
|
193
|
+
|
|
194
|
+
def con_str(val):
|
|
195
|
+
if val == 0:
|
|
196
|
+
return ("none")
|
|
197
|
+
if val <= 5:
|
|
198
|
+
return("low")
|
|
199
|
+
if val < 10:
|
|
200
|
+
return ("medium")
|
|
201
|
+
if val < 15:
|
|
202
|
+
return("high")
|
|
203
|
+
if val >= 15:
|
|
204
|
+
return("very high")
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def evaluate_winners(wd, **kwrags):
|
|
208
|
+
Wdb = wd.get_db('Wdb')
|
|
209
|
+
Cdb = wd.get_db('Cdb')
|
|
210
|
+
Ndb = wd.get_db('Ndb')
|
|
211
|
+
|
|
212
|
+
# For every winning genome, give some key stats based on the information available
|
|
213
|
+
Table = {'genome':[],'score':[],'completeness':[],'contamination':[],'strain_heterogeneity':[],\
|
|
214
|
+
'size':[],'N50':[],'cluster':[], 'taxonomy':[], 'tax_confidence':[],\
|
|
215
|
+
'cluster_members':[],'closest_cluster_member':[],'furthest_cluster_member':[],\
|
|
216
|
+
'completeness_metric':[], 'contamination_metric':[]}
|
|
217
|
+
|
|
218
|
+
for i, row in Wdb.iterrows():
|
|
219
|
+
# Add info in Wdb
|
|
220
|
+
Table['genome'].append(row['genome'])
|
|
221
|
+
Table['score'].append(row['score'])
|
|
222
|
+
Table['cluster'].append(row['cluster'])
|
|
223
|
+
|
|
224
|
+
# Add clustering info
|
|
225
|
+
if 'Blank' not in Ndb:
|
|
226
|
+
d = Cdb[Cdb['secondary_cluster'] == row['cluster']]
|
|
227
|
+
ndb = Ndb[(Ndb['reference'] == row['genome']) & (Ndb['querry'].isin(d['genome'].tolist()))\
|
|
228
|
+
& (Ndb['reference'] != Ndb['querry'])]
|
|
229
|
+
members = len(d['genome'].unique())
|
|
230
|
+
|
|
231
|
+
if members > 1:
|
|
232
|
+
Table['closest_cluster_member'].append("{0:.2f}".format(ndb['ani'].max()*100))
|
|
233
|
+
Table['furthest_cluster_member'].append("{0:.2f}".format(ndb['ani'].min()*100))
|
|
234
|
+
Table['cluster_members'].append(len(d['genome'].unique()))
|
|
235
|
+
else:
|
|
236
|
+
Table['closest_cluster_member'].append("NA")
|
|
237
|
+
Table['furthest_cluster_member'].append("NA")
|
|
238
|
+
Table['cluster_members'].append(len(d['genome'].unique()))
|
|
239
|
+
else:
|
|
240
|
+
d = Cdb[Cdb['secondary_cluster'] == row['cluster']]
|
|
241
|
+
Table['closest_cluster_member'].append("NA")
|
|
242
|
+
Table['furthest_cluster_member'].append("NA")
|
|
243
|
+
Table['cluster_members'].append(len(d['genome'].unique()))
|
|
244
|
+
|
|
245
|
+
# Add checkM info
|
|
246
|
+
if wd.hasDb('Chdb'):
|
|
247
|
+
Chdb = wd.get_db('Chdb')
|
|
248
|
+
d = Chdb[Chdb['Bin Id'] == row['genome']]
|
|
249
|
+
Table['completeness'].append(d['Completeness'].tolist()[0])
|
|
250
|
+
Table['contamination'].append(d['Contamination'].tolist()[0])
|
|
251
|
+
Table['strain_heterogeneity'].append(d['Strain heterogeneity'].tolist()[0])
|
|
252
|
+
Table['size'].append(d['Genome size (bp)'].tolist()[0])
|
|
253
|
+
Table['N50'].append(d['N50 (scaffolds)'].tolist()[0])
|
|
254
|
+
Table['completeness_metric'].append(comp_str(d['Completeness'].tolist()[0]))
|
|
255
|
+
Table['contamination_metric'].append(con_str(d['Contamination'].tolist()[0]))
|
|
256
|
+
elif wd.hasDb('genomeInfo'):
|
|
257
|
+
Gdb = wd.get_db('genomeInfo')
|
|
258
|
+
d = Gdb[Gdb['genome'] == row['genome']]
|
|
259
|
+
if len(d) > 0:
|
|
260
|
+
comp = d['completeness'].tolist()[0]
|
|
261
|
+
con = d['contamination'].tolist()[0]
|
|
262
|
+
Table['completeness'].append(comp)
|
|
263
|
+
Table['contamination'].append(con)
|
|
264
|
+
Table['strain_heterogeneity'].append(d['strain_heterogeneity'].tolist()[0] if 'strain_heterogeneity' in d.columns else "NA")
|
|
265
|
+
Table['size'].append(d['length'].tolist()[0] if 'length' in d.columns else "NA")
|
|
266
|
+
Table['N50'].append(d['N50'].tolist()[0] if 'N50' in d.columns else "NA")
|
|
267
|
+
Table['completeness_metric'].append(comp_str(comp))
|
|
268
|
+
Table['contamination_metric'].append(con_str(con))
|
|
269
|
+
else:
|
|
270
|
+
Table['completeness'].append("NA")
|
|
271
|
+
Table['contamination'].append("NA")
|
|
272
|
+
Table['strain_heterogeneity'].append("NA")
|
|
273
|
+
Table['size'].append("NA")
|
|
274
|
+
Table['N50'].append("NA")
|
|
275
|
+
Table['completeness_metric'].append("NA")
|
|
276
|
+
Table['contamination_metric'].append("NA")
|
|
277
|
+
else:
|
|
278
|
+
Table['completeness'].append("NA")
|
|
279
|
+
Table['contamination'].append("NA")
|
|
280
|
+
Table['strain_heterogeneity'].append("NA")
|
|
281
|
+
Table['size'].append("NA")
|
|
282
|
+
Table['N50'].append("NA")
|
|
283
|
+
Table['completeness_metric'].append("NA")
|
|
284
|
+
Table['contamination_metric'].append("NA")
|
|
285
|
+
|
|
286
|
+
# Add taxonomy info
|
|
287
|
+
if wd.hasDb('Tdb'):
|
|
288
|
+
Tdb = wd.get_db('Tdb')
|
|
289
|
+
d = Tdb[Tdb['genome'] == row['genome']]
|
|
290
|
+
Table['taxonomy'].append(d['taxonomy'][d['tax_confidence'] == d['tax_confidence'].max()].tolist()[0])
|
|
291
|
+
Table['tax_confidence'].append(d['tax_confidence'].max())
|
|
292
|
+
else:
|
|
293
|
+
Table['taxonomy'].append('NA')
|
|
294
|
+
Table['tax_confidence'].append('NA')
|
|
295
|
+
|
|
296
|
+
Widb = pd.DataFrame(Table)
|
|
297
|
+
return Widb
|
|
298
|
+
|
|
299
|
+
def run_tertiary_clustering(wd, **kwargs):
|
|
300
|
+
# Create a new workdir inside data
|
|
301
|
+
new_wd_loc = wd.get_dir('data') + 'tertiary_clustering'
|
|
302
|
+
nWd = drep.WorkDirectory.WorkDirectory(new_wd_loc)
|
|
303
|
+
|
|
304
|
+
# Make a copy of the kwargs
|
|
305
|
+
kwargs_copy = kwargs.copy()
|
|
306
|
+
if 'genomes' in kwargs_copy:
|
|
307
|
+
del kwargs_copy['genomes']
|
|
308
|
+
kwargs_copy['P_ani'] = kwargs_copy['P_ani'] - 0.05
|
|
309
|
+
kwargs_copy['multiround_primary_clustering'] = False
|
|
310
|
+
kwargs_copy['greedy_secondary_clustering'] = False
|
|
311
|
+
|
|
312
|
+
# Figure out what genomes you're going to compare and make a new Bdb
|
|
313
|
+
Bdb = wd.get_db('Bdb')
|
|
314
|
+
Cdb = wd.get_db('Cdb')
|
|
315
|
+
Wdb = wd.get_db('Wdb')
|
|
316
|
+
BBdb = Bdb[Bdb['genome'].isin(Wdb['genome'].tolist())]
|
|
317
|
+
nWd.store_db(BBdb, 'Bdb')
|
|
318
|
+
|
|
319
|
+
# Copy over genome info for choose
|
|
320
|
+
for name in ['genomeInfo', 'Chdb']:
|
|
321
|
+
if wd.hasDb(name):
|
|
322
|
+
nWd.store_db(wd.get_db(name), name)
|
|
323
|
+
|
|
324
|
+
# Run the comparison
|
|
325
|
+
drep.d_cluster.controller.d_cluster_wrapper(new_wd_loc, **kwargs_copy)
|
|
326
|
+
|
|
327
|
+
# Reconsile results for Cdb
|
|
328
|
+
nWd = drep.WorkDirectory.WorkDirectory(new_wd_loc)
|
|
329
|
+
nCdb = nWd.get_db('Cdb')
|
|
330
|
+
|
|
331
|
+
Cdb['original_secondary_cluster'] = Cdb['secondary_cluster']
|
|
332
|
+
del Cdb['secondary_cluster']
|
|
333
|
+
Cdb = pd.merge(Cdb, nCdb[['genome', 'secondary_cluster']], on='genome', how='left')
|
|
334
|
+
|
|
335
|
+
rep_cdb = Cdb.dropna()
|
|
336
|
+
old2new = rep_cdb.set_index('original_secondary_cluster')['secondary_cluster'].to_dict()
|
|
337
|
+
Cdb['secondary_cluster'] = Cdb['original_secondary_cluster'].map(old2new)
|
|
338
|
+
|
|
339
|
+
# Rename clusters
|
|
340
|
+
old2new_names = {}
|
|
341
|
+
for clust, db in Cdb.groupby('secondary_cluster'):
|
|
342
|
+
if len(db['original_secondary_cluster'].unique()) == 1:
|
|
343
|
+
old2new_names[clust] = db['original_secondary_cluster'].iloc[0]
|
|
344
|
+
else:
|
|
345
|
+
new_name = f"{'.'.join(sorted(list(set([x.split('_')[0] for x in list(db['original_secondary_cluster'].unique())]))))}" + \
|
|
346
|
+
f"_{'.'.join(sorted(list(set([x.split('_')[1] for x in list(db['original_secondary_cluster'].unique())]))))}"
|
|
347
|
+
old2new_names[clust] = new_name
|
|
348
|
+
logging.debug(f"Clusters {list(db['original_secondary_cluster'].unique())} were merged into {new_name}")
|
|
349
|
+
Cdb['secondary_cluster'] = Cdb['secondary_cluster'].map(old2new_names)
|
|
350
|
+
|
|
351
|
+
# Store new Cdb
|
|
352
|
+
wd.store_db(Cdb, 'Cdb')
|
|
353
|
+
|
|
354
|
+
# Re-run choose
|
|
355
|
+
drep.d_choose.d_choose_wrapper(wd.location, **kwargs_copy)
|