drep 4.0.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
drep/d_choose.py ADDED
@@ -0,0 +1,362 @@
1
+ #!/usr/bin/env python3
2
+ '''
3
+ d_choose - a subset of drep
4
+
5
+ Choose best genome from each cluster
6
+ '''
7
+
8
+ import logging
9
+ import glob
10
+ import pandas as pd
11
+ import os
12
+ import sys
13
+ import shutil
14
+ import numpy as np
15
+
16
+ import drep
17
+ import drep.d_cluster
18
+ import drep.d_filter
19
+ import drep.WorkDirectory
20
+
21
+ def d_choose_wrapper(wd, **kwargs):
22
+ '''
23
+ Controller for the dRep choose operation
24
+
25
+ Based off of the formula:
26
+ A*Completeness - B*Contamination + C*(Contamination *
27
+ (strain_heterogeneity/100)) + D*log(N50) + E*log(size)
28
+
29
+ A = completeness_weight;
30
+ B = contamination_weight;
31
+ C = strain_heterogeneity_weight;
32
+ D = N50_weight;
33
+ E = size_weight
34
+
35
+ Args:
36
+ wd (WorkDirectory): The current workDirectory
37
+ **kwargs: Command line arguments
38
+
39
+ Keyword args:
40
+ genomeInfo: .csv or .tsv genomeInfo file
41
+ ignoreGenomeQuality: Don't run checkM or do any quality-based filtering (not recommended)
42
+ checkM_method: Either lineage_wf (more accurate) or taxonomy_wf (faster)
43
+
44
+ completeness_weight: see formula
45
+ contamination_weight: see formula
46
+ strain_heterogeneity_weight: see formula
47
+ N50_weight: see formula
48
+ size_weight: see formula
49
+
50
+ Returns:
51
+ Makes Sdb (scoreDb) in the workDirectory
52
+ '''
53
+ # Load the WorkDirectory.
54
+ logging.info("Loading work directory")
55
+ workDirectory = drep.WorkDirectory.WorkDirectory(wd)
56
+ logging.debug(str(workDirectory))
57
+ wd = workDirectory
58
+
59
+ # Make sure you have enough
60
+ kwargs = _validate_choose_arguments(workDirectory, kwargs)
61
+ Cdb = wd.get_db('Cdb')
62
+ bdb = wd.get_db('Bdb')
63
+
64
+ # Get comp/con information
65
+ if kwargs.get('ignoreGenomeQuality', False):
66
+ logging.debug("Skipping all quality-based filtering")
67
+ Gdb = drep.d_filter.calc_genome_info(bdb['location'].tolist())
68
+ else:
69
+ Gdb = drep.d_filter._get_run_genomeInfo(wd, bdb, **kwargs)
70
+
71
+ if kwargs.get('centrality_weight', 0) > 0:
72
+ Gdb = add_centrality(wd, Gdb, **kwargs)
73
+
74
+ wd.store_db(Gdb, 'genomeInformation', overwrite=True)
75
+
76
+ # Call a method with Cdb and Chdb, returning Sdb (scored db) and Wdb (winner db)
77
+
78
+ Sdb, Wdb = choose_winners(Cdb, Gdb, **kwargs)
79
+
80
+ # Save Sdb and Wdb
81
+ workDirectory.store_db(Sdb,'Sdb',overwrite=True)
82
+ workDirectory.store_db(Wdb,'Wdb',overwrite=True)
83
+
84
+ # Make a "winning genomes" folder and populate it
85
+ logging.debug('saving dereplicated genomes')
86
+ g2l = bdb.set_index('genome')['location'].to_dict()
87
+ derep_genomes = [g2l[g] for g in list(Wdb['genome'].unique())]
88
+ wd.store_special('dereplicated_genomes', derep_genomes)
89
+
90
+ def choose_winners(Cdb, Gdb, **kwargs):
91
+ '''
92
+ Make a scoring database and pick the winner of each cluster
93
+
94
+ Args:
95
+ Cdb: clustering database
96
+ Gdb: genome information database
97
+
98
+ Keyword Args:
99
+ See wrapper
100
+
101
+ Returns:
102
+ List: [Sdb (scoring database), Wdb (winner database)]
103
+ '''
104
+ # If you have extra genome weights, load them
105
+ ew_loc = kwargs.get('extra_weight_table', None)
106
+ if ew_loc is not None:
107
+ Edb = load_extra_weight_table(ew_loc, list(Cdb['genome'].unique()), **kwargs)
108
+ else:
109
+ Edb = None
110
+
111
+ # Generate Sdb
112
+ genomes = list(Cdb['genome'].unique())
113
+ Sdb = score_genomes(genomes, Gdb, Edb=Edb, **kwargs)
114
+ logging.debug("Sdb finished")
115
+
116
+ # Generate Wdb
117
+ Wdb = pick_winners(Sdb,Cdb)
118
+ logging.debug("Wdb finished")
119
+
120
+ return Sdb, Wdb
121
+
122
+ def load_extra_weight_table(loc, genomes, **kwargs):
123
+ """
124
+
125
+ Args:
126
+ loc: location of extra weight table
127
+ genomes: list of genomes you have in Cdb
128
+ **kwargs: nothing realld
129
+
130
+ Returns:
131
+ dataframe with columns "genome" and "extra_weight"
132
+
133
+ """
134
+ if not os.path.join(loc):
135
+ logging.error(f"COULD NOT FIND FILE {loc}; WILL NOT PROCESS EXTRA WEIGHTS")
136
+ return None
137
+ else:
138
+ try:
139
+ db = pd.read_csv(loc, sep='\t', names=['genome', 'extra_weight'])
140
+ db['extra_weight'] = db['extra_weight'].astype(float)
141
+ except:
142
+ f = ''
143
+ with open(loc, 'r') as o:
144
+ x = 0
145
+ for line in o.readlines():
146
+ f += line + '\n'
147
+ x += 1
148
+
149
+ if x > 10:
150
+ break
151
+
152
+ logging.error(f"COULD NOT LOAD FILE {loc}; WILL NOT PROCESS EXTRA WEIGHTS. FILE LOOKS LIKE:\n{f}")
153
+ return None
154
+
155
+ wg = set(db['genome'].tolist())
156
+ gg = set(genomes)
157
+
158
+ missing = "\n".join(list(wg-gg)[:10])
159
+
160
+ logging.info(f'Loaded {len(db)} extra weights. {len(gg.intersection(wg))} of {len(gg)} genomes have an extra weight. {len(wg - gg)} genomes have a weight but ARE NOT KNOWN BY DREP; here are some examples:\n{missing}')
161
+
162
+ return db[db['genome'].isin(gg)]
163
+
164
+ def pick_winners(Sdb, Cdb):
165
+ '''
166
+ Based on clustering and scores, pick the best genome from every cluster
167
+
168
+ Args:
169
+ Sdb: score of every genome
170
+ Cdb: clustering
171
+
172
+ Returns:
173
+ DataFrame: Wdb (winner database)
174
+ '''
175
+ Table = {'genome':[],'cluster':[],'score':[]}
176
+ for cluster in Cdb['secondary_cluster'].unique():
177
+ d = Sdb[Sdb['genome'].isin(Cdb['genome'][Cdb['secondary_cluster'] == cluster].tolist())]
178
+ score = d['score'].max()
179
+ genome = d['genome'][d['score'] == score].tolist()[0]
180
+ Table['genome'].append(genome)
181
+ Table['score'].append(score)
182
+ Table['cluster'].append(cluster)
183
+
184
+ Wdb = pd.DataFrame(Table)
185
+ return Wdb
186
+
187
+ def score_genomes(genomes, Gdb, Edb=None, **kwargs):
188
+ '''
189
+ Calculate the scores for a list of genomes
190
+
191
+ Args:
192
+ genomes: list of genomes
193
+ Gdb: genome information database
194
+
195
+ Keyword Args:
196
+ See wrapper
197
+
198
+ Returns:
199
+ DataFrame: Sdb (scoring database)
200
+ '''
201
+
202
+ if Edb is not None:
203
+ g2e = Edb.set_index('genome')['extra_weight'].to_dict()
204
+ else:
205
+ g2e = {}
206
+
207
+ # Test for NaNs in Gdb['strain_heterogeneity']
208
+ if 'strain_heterogeneity' in Gdb.columns:
209
+ if Gdb['strain_heterogeneity'].isnull().values.any():
210
+ kwargs['strain_heterogeneity_weight'] = 0
211
+ logging.warning ('NaNs found in Gdb strain_heterogeneity! '
212
+ 'This may have happened if CheckM results '
213
+ 'were imported for some but not all '
214
+ 'genomes. \n\n'
215
+ 'Setting strain_heterogeneity_weight to zero')
216
+ Gdb.drop('strain_heterogeneity',
217
+ axis=1, inplace=True)
218
+
219
+ Table = {'genome':[],'score':[]}
220
+ for genome in genomes:
221
+ if genome in g2e:
222
+ extra = g2e[genome]
223
+ else:
224
+ extra = 0
225
+ row = Gdb[Gdb['genome'] == genome]
226
+
227
+ score = score_row(row, extra=extra, **kwargs)
228
+ Table['genome'].append(genome)
229
+ Table['score'].append(score)
230
+
231
+ Sdb = pd.DataFrame(Table)
232
+ return Sdb
233
+
234
+ def score_row(row, extra=0, **kwargs):
235
+ '''
236
+ Perform the scoring of a row based on kwargs
237
+
238
+ Args:
239
+ row: row of genome information
240
+
241
+ Keyword Args:
242
+ ignoreGenomeQuality: Don't run checkM or do any quality-based filtering (not recommended)
243
+
244
+ completeness_weight: see formula
245
+ contamination_weight: see formula
246
+ strain_heterogeneity_weight: see formula
247
+ N50_weight: see formula
248
+ size_weight: see formula
249
+ extra = extra weight to apply
250
+
251
+ Returns:
252
+ float: score
253
+ '''
254
+ comW = kwargs.get('completeness_weight',1)
255
+ conW = kwargs.get('contamination_weight',1)
256
+ n50W = kwargs.get('N50_weight',1)
257
+ sizeW = kwargs.get('size_weight',1)
258
+ strW = kwargs.get('strain_heterogeneity_weight',1)
259
+ centW = kwargs.get('centrality_weight', 0)
260
+
261
+ # For centrality calculations
262
+ S_ani = kwargs.get('S_ani', 0.99)
263
+ if centW > 0:
264
+ cent = row['centrality'].tolist()[0]
265
+ else:
266
+ cent = 0
267
+
268
+ n50 = float(row['N50'].tolist()[0])
269
+ size = float(row['length'].tolist()[0])
270
+
271
+ if kwargs.get('ignoreGenomeQuality', False):
272
+ score = (np.log10(n50) * n50W) + (np.log10(size) * sizeW) + ((cent - S_ani) * centW) + float(extra)
273
+ return score
274
+
275
+ com = float(row['completeness'].tolist()[0])
276
+ con = float(row['contamination'].tolist()[0])
277
+
278
+ if 'strain_heterogeneity' in row:
279
+ strh = float(row['strain_heterogeneity'].tolist()[0])
280
+ else:
281
+ strh = 0
282
+
283
+ score = (com * comW) - (con * conW) + (strW * (con * (strh/100))) \
284
+ + (np.log10(n50) * n50W) + (np.log10(size) * sizeW) + ((cent - S_ani) * centW) + float(extra)
285
+ return score
286
+
287
+ def _validate_choose_arguments(wd, kwargs):
288
+ '''
289
+ Validate choose arguments
290
+
291
+ Make sure you have a Cdb
292
+
293
+ Args:
294
+ wd: WorkDirectory
295
+ kwargs: keyword arguments
296
+ '''
297
+ if not wd.hasDb('Cdb'):
298
+ logging.error("Can't find Cdb- quitting")
299
+ logging.error("Cdb is not found in the work directory- you must run cluster before you ",
300
+ + "can choose")
301
+ sys.exit()
302
+
303
+ # Validate centrality arguments
304
+ if (kwargs.get('SkipSecondary', True)) & (kwargs.get('centrality_weight', 0) > 0):
305
+ logging.error(
306
+ "You skipped secondary clustering but have centrality weight above 0. You cant do that. I will now set the centrality weight to 0 to avoid a crash")
307
+ kwargs['centrality_weight'] = 0
308
+
309
+ return kwargs
310
+
311
+ def add_centrality(wd, Gdb, **kwargs):
312
+ """
313
+ Add a columns named "centrality" to genome info
314
+ """
315
+ Ndb = wd.get_db('Ndb')
316
+ Cdb = wd.get_db('Cdb')
317
+
318
+ if Cdb['cluster_method'].iloc[0] == 'greedy':
319
+ Ndb = calc_centrality_from_scratch(wd.get_db('Bdb'), Cdb, os.path.join(wd.get_dir('MASH'), 'centrality_calculations/'))
320
+
321
+ g2c = Cdb.set_index('genome')['secondary_cluster'].to_dict()
322
+ c2s = Cdb['secondary_cluster'].value_counts().to_dict()
323
+
324
+ Ndb['cluster_1'] = Ndb['reference'].map(g2c)
325
+ Ndb['cluster_2'] = Ndb['querry'].map(g2c)
326
+ Ndb = Ndb[Ndb['cluster_1'] == Ndb['cluster_2']]
327
+ Ndb = Ndb[Ndb['reference'] != Ndb['querry']]
328
+
329
+ genome2centrality = {}
330
+ for cluster, ndb in Ndb.groupby('cluster_1'):
331
+ #print(f"Cluster {cluster} has {c2s[cluster]} members and {len(ndb)} comps")
332
+
333
+ mlen = c2s[cluster]
334
+ assert len(ndb) == (mlen * mlen) - mlen
335
+ for genome, db in ndb.groupby('reference'):
336
+ genome2centrality[genome] = db['ani'].mean()
337
+
338
+ Gdb['centrality'] = Gdb['genome'].map(genome2centrality).fillna(0)
339
+ return Gdb
340
+
341
+ def calc_centrality_from_scratch(Bdb, Cdb, data_folder):
342
+ """
343
+ Calculate centrality from scratch using Mash
344
+ """
345
+ logging.info("Calculating centrality using Mash")
346
+
347
+ # 1) Run calculations
348
+ dbs = []
349
+ Xdb = pd.merge(Cdb, Bdb, on='genome', how='left')
350
+ for cluster, bdb in Xdb.groupby('secondary_cluster'):
351
+ if len(bdb) <= 1:
352
+ continue
353
+
354
+ df = os.path.join(data_folder, cluster + '/')
355
+ mdb, cdb, cluster_ret = drep.d_cluster.compare_utils.all_vs_all_MASH(bdb, df, MASH_sketch=10000)
356
+ mdb['ani'] = 1 - mdb['dist']
357
+ mdb['cluster'] = cluster
358
+ mdb = mdb.rename(columns={'genome1':'reference', 'genome2':'querry'})
359
+ dbs.append(mdb[['reference', 'querry', 'ani', 'cluster']])
360
+
361
+ Mdb = pd.concat(dbs).reset_index(drop=True)
362
+ return Mdb
File without changes
@@ -0,0 +1,126 @@
1
+ import logging
2
+ import os
3
+ import sys
4
+
5
+ import numpy as np
6
+ import pandas as pd
7
+ import scipy.cluster
8
+ from scipy.spatial import distance as ssd
9
+
10
+ import drep.d_cluster.utils
11
+
12
+ def genome_hierarchical_clustering(Ndb, **kwargs):
13
+ '''
14
+ Cluster ANI database
15
+
16
+ Args:
17
+ Ndb: result of secondary clustering
18
+
19
+ Keyword arguments:
20
+ clusterAlg: how to cluster the database (default = single)
21
+ S_ani: thershold to cluster at (default = .99)
22
+ cov_thresh: minumum coverage to be included in clustering (default = .5)
23
+ cluster: name of the cluster
24
+ comp_method: comparison algorithm used
25
+
26
+ Returns:
27
+ list: [Cdb, {cluster:[linkage, linkage_db, arguments]}]
28
+ '''
29
+ logging.debug('Clustering ANIn database')
30
+
31
+ S_Lmethod = kwargs.get('clusterAlg', 'single')
32
+ S_Lcutoff = 1 - kwargs.get('S_ani', .99)
33
+ cov_thresh = float(kwargs.get('cov_thresh',0.5))
34
+ cluster = kwargs.get('cluster','')
35
+ comp_method = kwargs.get('comp_method', 'unk')
36
+
37
+ Table = {'genome':[],'secondary_cluster':[]}
38
+
39
+ # Handle the case where there's only one genome
40
+ if len(Ndb['reference'].unique()) == 1:
41
+ Table['genome'].append(os.path.basename(Ndb['reference'].unique().tolist()[0]))
42
+ Table['secondary_cluster'].append("{0}_0".format(cluster))
43
+ cluster_ret = []
44
+
45
+ else:
46
+ # Make linkage Ndb
47
+ Ldb = drep.d_cluster.utils.make_linkage_Ndb(Ndb, **kwargs)
48
+
49
+ # 3) Cluster the linkagedb
50
+ Gdb, linkage = cluster_hierarchical(Ldb, linkage_method= S_Lmethod, \
51
+ linkage_cutoff= S_Lcutoff)
52
+
53
+ # 4) Extract secondary clusters
54
+ for clust, d in Gdb.groupby('cluster'):
55
+ for genome in d['genome'].tolist():
56
+ Table['genome'].append(genome)
57
+ Table['secondary_cluster'].append("{0}_{1}".format(cluster,clust))
58
+
59
+ # 5) Save the linkage
60
+ arguments = {'linkage_method':S_Lmethod,'linkage_cutoff':S_Lcutoff,\
61
+ 'comparison_algorithm':comp_method,'minimum_coverage':cov_thresh}
62
+ cluster_ret = [linkage, Ldb, arguments]
63
+
64
+ # Return the database
65
+ Gdb = pd.DataFrame(Table)
66
+ Gdb['threshold'] = S_Lcutoff
67
+ Gdb['cluster_method'] = S_Lmethod
68
+ Gdb['comparison_algorithm'] = comp_method
69
+
70
+ return Gdb, cluster_ret
71
+
72
+
73
+ def iteratre_clusters(Bdb, Cdb, id='primary_cluster'):
74
+ '''
75
+ An iterator: Given Bdb and Cdb, yeild smaller Bdb's in the same cluster
76
+
77
+ Args:
78
+ Bdb: [genome, location]
79
+ Cdb: [genome, id]
80
+ id: what to iterate on (default = 'primary_cluster')
81
+
82
+ Returns:
83
+ list: [d(subset of b), cluster(name of cluster)]
84
+ '''
85
+ Bdb = pd.merge(Bdb,Cdb)
86
+ for cluster, d in Bdb.groupby(id):
87
+ yield d, cluster
88
+
89
+
90
+ def cluster_hierarchical(db, linkage_method= 'single', linkage_cutoff= 0.10):
91
+ '''
92
+ Perform hierarchical clustering on a symmetrical distiance matrix
93
+
94
+ Note this builds a dense matrix and is O(N^2) in memory. Single-linkage
95
+ primary clustering goes through drep.d_cluster.union_find instead, which is
96
+ equivalent but does not need the matrix.
97
+
98
+ Args:
99
+ db: result of db.pivot usually
100
+ linkage_method: passed to scipy.cluster.hierarchy.fcluster
101
+ linkage_cutoff: distance to draw the clustering line (default = .1)
102
+
103
+ Returns:
104
+ list: [Cdb, linkage]
105
+ '''
106
+ # Save names
107
+ names = list(db.columns)
108
+
109
+ # Generate linkage dataframe
110
+ arr = np.asarray(db)
111
+ try:
112
+ arr = ssd.squareform(arr)
113
+ except:
114
+ logging.error("The database passed in is not symmetrical!")
115
+ logging.error(arr)
116
+ logging.error(names)
117
+ sys.exit()
118
+ linkage = scipy.cluster.hierarchy.linkage(arr, method= linkage_method)
119
+
120
+ # Form clusters
121
+ fclust = scipy.cluster.hierarchy.fcluster(linkage,linkage_cutoff, \
122
+ criterion='distance')
123
+ # Make Cdb
124
+ Cdb = drep.d_cluster.utils._gen_cdb_from_fclust(fclust,names)
125
+
126
+ return Cdb, linkage