drep 4.0.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- drep/VERSION +1 -0
- drep/WorkDirectory.py +355 -0
- drep/__init__.py +101 -0
- drep/argumentParser.py +279 -0
- drep/controller.py +105 -0
- drep/d_adjust.py +272 -0
- drep/d_analyze.py +1613 -0
- drep/d_bonus.py +429 -0
- drep/d_choose.py +362 -0
- drep/d_cluster/__init__.py +0 -0
- drep/d_cluster/cluster_utils.py +126 -0
- drep/d_cluster/compare_utils.py +636 -0
- drep/d_cluster/controller.py +228 -0
- drep/d_cluster/external.py +765 -0
- drep/d_cluster/greedy_clustering.py +181 -0
- drep/d_cluster/parsers.py +0 -0
- drep/d_cluster/union_find.py +543 -0
- drep/d_cluster/utils.py +687 -0
- drep/d_evaluate.py +355 -0
- drep/d_filter.py +831 -0
- drep/d_workflows.py +135 -0
- drep-4.0.2.data/scripts/ScaffoldLevel_dRep.py +1101 -0
- drep-4.0.2.data/scripts/dRep +32 -0
- drep-4.0.2.data/scripts/parse_stb.py +140 -0
- drep-4.0.2.dist-info/METADATA +23 -0
- drep-4.0.2.dist-info/RECORD +28 -0
- drep-4.0.2.dist-info/WHEEL +5 -0
- drep-4.0.2.dist-info/top_level.txt +1 -0
drep/d_choose.py
ADDED
|
@@ -0,0 +1,362 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
'''
|
|
3
|
+
d_choose - a subset of drep
|
|
4
|
+
|
|
5
|
+
Choose best genome from each cluster
|
|
6
|
+
'''
|
|
7
|
+
|
|
8
|
+
import logging
|
|
9
|
+
import glob
|
|
10
|
+
import pandas as pd
|
|
11
|
+
import os
|
|
12
|
+
import sys
|
|
13
|
+
import shutil
|
|
14
|
+
import numpy as np
|
|
15
|
+
|
|
16
|
+
import drep
|
|
17
|
+
import drep.d_cluster
|
|
18
|
+
import drep.d_filter
|
|
19
|
+
import drep.WorkDirectory
|
|
20
|
+
|
|
21
|
+
def d_choose_wrapper(wd, **kwargs):
|
|
22
|
+
'''
|
|
23
|
+
Controller for the dRep choose operation
|
|
24
|
+
|
|
25
|
+
Based off of the formula:
|
|
26
|
+
A*Completeness - B*Contamination + C*(Contamination *
|
|
27
|
+
(strain_heterogeneity/100)) + D*log(N50) + E*log(size)
|
|
28
|
+
|
|
29
|
+
A = completeness_weight;
|
|
30
|
+
B = contamination_weight;
|
|
31
|
+
C = strain_heterogeneity_weight;
|
|
32
|
+
D = N50_weight;
|
|
33
|
+
E = size_weight
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
wd (WorkDirectory): The current workDirectory
|
|
37
|
+
**kwargs: Command line arguments
|
|
38
|
+
|
|
39
|
+
Keyword args:
|
|
40
|
+
genomeInfo: .csv or .tsv genomeInfo file
|
|
41
|
+
ignoreGenomeQuality: Don't run checkM or do any quality-based filtering (not recommended)
|
|
42
|
+
checkM_method: Either lineage_wf (more accurate) or taxonomy_wf (faster)
|
|
43
|
+
|
|
44
|
+
completeness_weight: see formula
|
|
45
|
+
contamination_weight: see formula
|
|
46
|
+
strain_heterogeneity_weight: see formula
|
|
47
|
+
N50_weight: see formula
|
|
48
|
+
size_weight: see formula
|
|
49
|
+
|
|
50
|
+
Returns:
|
|
51
|
+
Makes Sdb (scoreDb) in the workDirectory
|
|
52
|
+
'''
|
|
53
|
+
# Load the WorkDirectory.
|
|
54
|
+
logging.info("Loading work directory")
|
|
55
|
+
workDirectory = drep.WorkDirectory.WorkDirectory(wd)
|
|
56
|
+
logging.debug(str(workDirectory))
|
|
57
|
+
wd = workDirectory
|
|
58
|
+
|
|
59
|
+
# Make sure you have enough
|
|
60
|
+
kwargs = _validate_choose_arguments(workDirectory, kwargs)
|
|
61
|
+
Cdb = wd.get_db('Cdb')
|
|
62
|
+
bdb = wd.get_db('Bdb')
|
|
63
|
+
|
|
64
|
+
# Get comp/con information
|
|
65
|
+
if kwargs.get('ignoreGenomeQuality', False):
|
|
66
|
+
logging.debug("Skipping all quality-based filtering")
|
|
67
|
+
Gdb = drep.d_filter.calc_genome_info(bdb['location'].tolist())
|
|
68
|
+
else:
|
|
69
|
+
Gdb = drep.d_filter._get_run_genomeInfo(wd, bdb, **kwargs)
|
|
70
|
+
|
|
71
|
+
if kwargs.get('centrality_weight', 0) > 0:
|
|
72
|
+
Gdb = add_centrality(wd, Gdb, **kwargs)
|
|
73
|
+
|
|
74
|
+
wd.store_db(Gdb, 'genomeInformation', overwrite=True)
|
|
75
|
+
|
|
76
|
+
# Call a method with Cdb and Chdb, returning Sdb (scored db) and Wdb (winner db)
|
|
77
|
+
|
|
78
|
+
Sdb, Wdb = choose_winners(Cdb, Gdb, **kwargs)
|
|
79
|
+
|
|
80
|
+
# Save Sdb and Wdb
|
|
81
|
+
workDirectory.store_db(Sdb,'Sdb',overwrite=True)
|
|
82
|
+
workDirectory.store_db(Wdb,'Wdb',overwrite=True)
|
|
83
|
+
|
|
84
|
+
# Make a "winning genomes" folder and populate it
|
|
85
|
+
logging.debug('saving dereplicated genomes')
|
|
86
|
+
g2l = bdb.set_index('genome')['location'].to_dict()
|
|
87
|
+
derep_genomes = [g2l[g] for g in list(Wdb['genome'].unique())]
|
|
88
|
+
wd.store_special('dereplicated_genomes', derep_genomes)
|
|
89
|
+
|
|
90
|
+
def choose_winners(Cdb, Gdb, **kwargs):
|
|
91
|
+
'''
|
|
92
|
+
Make a scoring database and pick the winner of each cluster
|
|
93
|
+
|
|
94
|
+
Args:
|
|
95
|
+
Cdb: clustering database
|
|
96
|
+
Gdb: genome information database
|
|
97
|
+
|
|
98
|
+
Keyword Args:
|
|
99
|
+
See wrapper
|
|
100
|
+
|
|
101
|
+
Returns:
|
|
102
|
+
List: [Sdb (scoring database), Wdb (winner database)]
|
|
103
|
+
'''
|
|
104
|
+
# If you have extra genome weights, load them
|
|
105
|
+
ew_loc = kwargs.get('extra_weight_table', None)
|
|
106
|
+
if ew_loc is not None:
|
|
107
|
+
Edb = load_extra_weight_table(ew_loc, list(Cdb['genome'].unique()), **kwargs)
|
|
108
|
+
else:
|
|
109
|
+
Edb = None
|
|
110
|
+
|
|
111
|
+
# Generate Sdb
|
|
112
|
+
genomes = list(Cdb['genome'].unique())
|
|
113
|
+
Sdb = score_genomes(genomes, Gdb, Edb=Edb, **kwargs)
|
|
114
|
+
logging.debug("Sdb finished")
|
|
115
|
+
|
|
116
|
+
# Generate Wdb
|
|
117
|
+
Wdb = pick_winners(Sdb,Cdb)
|
|
118
|
+
logging.debug("Wdb finished")
|
|
119
|
+
|
|
120
|
+
return Sdb, Wdb
|
|
121
|
+
|
|
122
|
+
def load_extra_weight_table(loc, genomes, **kwargs):
|
|
123
|
+
"""
|
|
124
|
+
|
|
125
|
+
Args:
|
|
126
|
+
loc: location of extra weight table
|
|
127
|
+
genomes: list of genomes you have in Cdb
|
|
128
|
+
**kwargs: nothing realld
|
|
129
|
+
|
|
130
|
+
Returns:
|
|
131
|
+
dataframe with columns "genome" and "extra_weight"
|
|
132
|
+
|
|
133
|
+
"""
|
|
134
|
+
if not os.path.join(loc):
|
|
135
|
+
logging.error(f"COULD NOT FIND FILE {loc}; WILL NOT PROCESS EXTRA WEIGHTS")
|
|
136
|
+
return None
|
|
137
|
+
else:
|
|
138
|
+
try:
|
|
139
|
+
db = pd.read_csv(loc, sep='\t', names=['genome', 'extra_weight'])
|
|
140
|
+
db['extra_weight'] = db['extra_weight'].astype(float)
|
|
141
|
+
except:
|
|
142
|
+
f = ''
|
|
143
|
+
with open(loc, 'r') as o:
|
|
144
|
+
x = 0
|
|
145
|
+
for line in o.readlines():
|
|
146
|
+
f += line + '\n'
|
|
147
|
+
x += 1
|
|
148
|
+
|
|
149
|
+
if x > 10:
|
|
150
|
+
break
|
|
151
|
+
|
|
152
|
+
logging.error(f"COULD NOT LOAD FILE {loc}; WILL NOT PROCESS EXTRA WEIGHTS. FILE LOOKS LIKE:\n{f}")
|
|
153
|
+
return None
|
|
154
|
+
|
|
155
|
+
wg = set(db['genome'].tolist())
|
|
156
|
+
gg = set(genomes)
|
|
157
|
+
|
|
158
|
+
missing = "\n".join(list(wg-gg)[:10])
|
|
159
|
+
|
|
160
|
+
logging.info(f'Loaded {len(db)} extra weights. {len(gg.intersection(wg))} of {len(gg)} genomes have an extra weight. {len(wg - gg)} genomes have a weight but ARE NOT KNOWN BY DREP; here are some examples:\n{missing}')
|
|
161
|
+
|
|
162
|
+
return db[db['genome'].isin(gg)]
|
|
163
|
+
|
|
164
|
+
def pick_winners(Sdb, Cdb):
|
|
165
|
+
'''
|
|
166
|
+
Based on clustering and scores, pick the best genome from every cluster
|
|
167
|
+
|
|
168
|
+
Args:
|
|
169
|
+
Sdb: score of every genome
|
|
170
|
+
Cdb: clustering
|
|
171
|
+
|
|
172
|
+
Returns:
|
|
173
|
+
DataFrame: Wdb (winner database)
|
|
174
|
+
'''
|
|
175
|
+
Table = {'genome':[],'cluster':[],'score':[]}
|
|
176
|
+
for cluster in Cdb['secondary_cluster'].unique():
|
|
177
|
+
d = Sdb[Sdb['genome'].isin(Cdb['genome'][Cdb['secondary_cluster'] == cluster].tolist())]
|
|
178
|
+
score = d['score'].max()
|
|
179
|
+
genome = d['genome'][d['score'] == score].tolist()[0]
|
|
180
|
+
Table['genome'].append(genome)
|
|
181
|
+
Table['score'].append(score)
|
|
182
|
+
Table['cluster'].append(cluster)
|
|
183
|
+
|
|
184
|
+
Wdb = pd.DataFrame(Table)
|
|
185
|
+
return Wdb
|
|
186
|
+
|
|
187
|
+
def score_genomes(genomes, Gdb, Edb=None, **kwargs):
|
|
188
|
+
'''
|
|
189
|
+
Calculate the scores for a list of genomes
|
|
190
|
+
|
|
191
|
+
Args:
|
|
192
|
+
genomes: list of genomes
|
|
193
|
+
Gdb: genome information database
|
|
194
|
+
|
|
195
|
+
Keyword Args:
|
|
196
|
+
See wrapper
|
|
197
|
+
|
|
198
|
+
Returns:
|
|
199
|
+
DataFrame: Sdb (scoring database)
|
|
200
|
+
'''
|
|
201
|
+
|
|
202
|
+
if Edb is not None:
|
|
203
|
+
g2e = Edb.set_index('genome')['extra_weight'].to_dict()
|
|
204
|
+
else:
|
|
205
|
+
g2e = {}
|
|
206
|
+
|
|
207
|
+
# Test for NaNs in Gdb['strain_heterogeneity']
|
|
208
|
+
if 'strain_heterogeneity' in Gdb.columns:
|
|
209
|
+
if Gdb['strain_heterogeneity'].isnull().values.any():
|
|
210
|
+
kwargs['strain_heterogeneity_weight'] = 0
|
|
211
|
+
logging.warning ('NaNs found in Gdb strain_heterogeneity! '
|
|
212
|
+
'This may have happened if CheckM results '
|
|
213
|
+
'were imported for some but not all '
|
|
214
|
+
'genomes. \n\n'
|
|
215
|
+
'Setting strain_heterogeneity_weight to zero')
|
|
216
|
+
Gdb.drop('strain_heterogeneity',
|
|
217
|
+
axis=1, inplace=True)
|
|
218
|
+
|
|
219
|
+
Table = {'genome':[],'score':[]}
|
|
220
|
+
for genome in genomes:
|
|
221
|
+
if genome in g2e:
|
|
222
|
+
extra = g2e[genome]
|
|
223
|
+
else:
|
|
224
|
+
extra = 0
|
|
225
|
+
row = Gdb[Gdb['genome'] == genome]
|
|
226
|
+
|
|
227
|
+
score = score_row(row, extra=extra, **kwargs)
|
|
228
|
+
Table['genome'].append(genome)
|
|
229
|
+
Table['score'].append(score)
|
|
230
|
+
|
|
231
|
+
Sdb = pd.DataFrame(Table)
|
|
232
|
+
return Sdb
|
|
233
|
+
|
|
234
|
+
def score_row(row, extra=0, **kwargs):
|
|
235
|
+
'''
|
|
236
|
+
Perform the scoring of a row based on kwargs
|
|
237
|
+
|
|
238
|
+
Args:
|
|
239
|
+
row: row of genome information
|
|
240
|
+
|
|
241
|
+
Keyword Args:
|
|
242
|
+
ignoreGenomeQuality: Don't run checkM or do any quality-based filtering (not recommended)
|
|
243
|
+
|
|
244
|
+
completeness_weight: see formula
|
|
245
|
+
contamination_weight: see formula
|
|
246
|
+
strain_heterogeneity_weight: see formula
|
|
247
|
+
N50_weight: see formula
|
|
248
|
+
size_weight: see formula
|
|
249
|
+
extra = extra weight to apply
|
|
250
|
+
|
|
251
|
+
Returns:
|
|
252
|
+
float: score
|
|
253
|
+
'''
|
|
254
|
+
comW = kwargs.get('completeness_weight',1)
|
|
255
|
+
conW = kwargs.get('contamination_weight',1)
|
|
256
|
+
n50W = kwargs.get('N50_weight',1)
|
|
257
|
+
sizeW = kwargs.get('size_weight',1)
|
|
258
|
+
strW = kwargs.get('strain_heterogeneity_weight',1)
|
|
259
|
+
centW = kwargs.get('centrality_weight', 0)
|
|
260
|
+
|
|
261
|
+
# For centrality calculations
|
|
262
|
+
S_ani = kwargs.get('S_ani', 0.99)
|
|
263
|
+
if centW > 0:
|
|
264
|
+
cent = row['centrality'].tolist()[0]
|
|
265
|
+
else:
|
|
266
|
+
cent = 0
|
|
267
|
+
|
|
268
|
+
n50 = float(row['N50'].tolist()[0])
|
|
269
|
+
size = float(row['length'].tolist()[0])
|
|
270
|
+
|
|
271
|
+
if kwargs.get('ignoreGenomeQuality', False):
|
|
272
|
+
score = (np.log10(n50) * n50W) + (np.log10(size) * sizeW) + ((cent - S_ani) * centW) + float(extra)
|
|
273
|
+
return score
|
|
274
|
+
|
|
275
|
+
com = float(row['completeness'].tolist()[0])
|
|
276
|
+
con = float(row['contamination'].tolist()[0])
|
|
277
|
+
|
|
278
|
+
if 'strain_heterogeneity' in row:
|
|
279
|
+
strh = float(row['strain_heterogeneity'].tolist()[0])
|
|
280
|
+
else:
|
|
281
|
+
strh = 0
|
|
282
|
+
|
|
283
|
+
score = (com * comW) - (con * conW) + (strW * (con * (strh/100))) \
|
|
284
|
+
+ (np.log10(n50) * n50W) + (np.log10(size) * sizeW) + ((cent - S_ani) * centW) + float(extra)
|
|
285
|
+
return score
|
|
286
|
+
|
|
287
|
+
def _validate_choose_arguments(wd, kwargs):
|
|
288
|
+
'''
|
|
289
|
+
Validate choose arguments
|
|
290
|
+
|
|
291
|
+
Make sure you have a Cdb
|
|
292
|
+
|
|
293
|
+
Args:
|
|
294
|
+
wd: WorkDirectory
|
|
295
|
+
kwargs: keyword arguments
|
|
296
|
+
'''
|
|
297
|
+
if not wd.hasDb('Cdb'):
|
|
298
|
+
logging.error("Can't find Cdb- quitting")
|
|
299
|
+
logging.error("Cdb is not found in the work directory- you must run cluster before you ",
|
|
300
|
+
+ "can choose")
|
|
301
|
+
sys.exit()
|
|
302
|
+
|
|
303
|
+
# Validate centrality arguments
|
|
304
|
+
if (kwargs.get('SkipSecondary', True)) & (kwargs.get('centrality_weight', 0) > 0):
|
|
305
|
+
logging.error(
|
|
306
|
+
"You skipped secondary clustering but have centrality weight above 0. You cant do that. I will now set the centrality weight to 0 to avoid a crash")
|
|
307
|
+
kwargs['centrality_weight'] = 0
|
|
308
|
+
|
|
309
|
+
return kwargs
|
|
310
|
+
|
|
311
|
+
def add_centrality(wd, Gdb, **kwargs):
|
|
312
|
+
"""
|
|
313
|
+
Add a columns named "centrality" to genome info
|
|
314
|
+
"""
|
|
315
|
+
Ndb = wd.get_db('Ndb')
|
|
316
|
+
Cdb = wd.get_db('Cdb')
|
|
317
|
+
|
|
318
|
+
if Cdb['cluster_method'].iloc[0] == 'greedy':
|
|
319
|
+
Ndb = calc_centrality_from_scratch(wd.get_db('Bdb'), Cdb, os.path.join(wd.get_dir('MASH'), 'centrality_calculations/'))
|
|
320
|
+
|
|
321
|
+
g2c = Cdb.set_index('genome')['secondary_cluster'].to_dict()
|
|
322
|
+
c2s = Cdb['secondary_cluster'].value_counts().to_dict()
|
|
323
|
+
|
|
324
|
+
Ndb['cluster_1'] = Ndb['reference'].map(g2c)
|
|
325
|
+
Ndb['cluster_2'] = Ndb['querry'].map(g2c)
|
|
326
|
+
Ndb = Ndb[Ndb['cluster_1'] == Ndb['cluster_2']]
|
|
327
|
+
Ndb = Ndb[Ndb['reference'] != Ndb['querry']]
|
|
328
|
+
|
|
329
|
+
genome2centrality = {}
|
|
330
|
+
for cluster, ndb in Ndb.groupby('cluster_1'):
|
|
331
|
+
#print(f"Cluster {cluster} has {c2s[cluster]} members and {len(ndb)} comps")
|
|
332
|
+
|
|
333
|
+
mlen = c2s[cluster]
|
|
334
|
+
assert len(ndb) == (mlen * mlen) - mlen
|
|
335
|
+
for genome, db in ndb.groupby('reference'):
|
|
336
|
+
genome2centrality[genome] = db['ani'].mean()
|
|
337
|
+
|
|
338
|
+
Gdb['centrality'] = Gdb['genome'].map(genome2centrality).fillna(0)
|
|
339
|
+
return Gdb
|
|
340
|
+
|
|
341
|
+
def calc_centrality_from_scratch(Bdb, Cdb, data_folder):
|
|
342
|
+
"""
|
|
343
|
+
Calculate centrality from scratch using Mash
|
|
344
|
+
"""
|
|
345
|
+
logging.info("Calculating centrality using Mash")
|
|
346
|
+
|
|
347
|
+
# 1) Run calculations
|
|
348
|
+
dbs = []
|
|
349
|
+
Xdb = pd.merge(Cdb, Bdb, on='genome', how='left')
|
|
350
|
+
for cluster, bdb in Xdb.groupby('secondary_cluster'):
|
|
351
|
+
if len(bdb) <= 1:
|
|
352
|
+
continue
|
|
353
|
+
|
|
354
|
+
df = os.path.join(data_folder, cluster + '/')
|
|
355
|
+
mdb, cdb, cluster_ret = drep.d_cluster.compare_utils.all_vs_all_MASH(bdb, df, MASH_sketch=10000)
|
|
356
|
+
mdb['ani'] = 1 - mdb['dist']
|
|
357
|
+
mdb['cluster'] = cluster
|
|
358
|
+
mdb = mdb.rename(columns={'genome1':'reference', 'genome2':'querry'})
|
|
359
|
+
dbs.append(mdb[['reference', 'querry', 'ani', 'cluster']])
|
|
360
|
+
|
|
361
|
+
Mdb = pd.concat(dbs).reset_index(drop=True)
|
|
362
|
+
return Mdb
|
|
File without changes
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import os
|
|
3
|
+
import sys
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
import pandas as pd
|
|
7
|
+
import scipy.cluster
|
|
8
|
+
from scipy.spatial import distance as ssd
|
|
9
|
+
|
|
10
|
+
import drep.d_cluster.utils
|
|
11
|
+
|
|
12
|
+
def genome_hierarchical_clustering(Ndb, **kwargs):
|
|
13
|
+
'''
|
|
14
|
+
Cluster ANI database
|
|
15
|
+
|
|
16
|
+
Args:
|
|
17
|
+
Ndb: result of secondary clustering
|
|
18
|
+
|
|
19
|
+
Keyword arguments:
|
|
20
|
+
clusterAlg: how to cluster the database (default = single)
|
|
21
|
+
S_ani: thershold to cluster at (default = .99)
|
|
22
|
+
cov_thresh: minumum coverage to be included in clustering (default = .5)
|
|
23
|
+
cluster: name of the cluster
|
|
24
|
+
comp_method: comparison algorithm used
|
|
25
|
+
|
|
26
|
+
Returns:
|
|
27
|
+
list: [Cdb, {cluster:[linkage, linkage_db, arguments]}]
|
|
28
|
+
'''
|
|
29
|
+
logging.debug('Clustering ANIn database')
|
|
30
|
+
|
|
31
|
+
S_Lmethod = kwargs.get('clusterAlg', 'single')
|
|
32
|
+
S_Lcutoff = 1 - kwargs.get('S_ani', .99)
|
|
33
|
+
cov_thresh = float(kwargs.get('cov_thresh',0.5))
|
|
34
|
+
cluster = kwargs.get('cluster','')
|
|
35
|
+
comp_method = kwargs.get('comp_method', 'unk')
|
|
36
|
+
|
|
37
|
+
Table = {'genome':[],'secondary_cluster':[]}
|
|
38
|
+
|
|
39
|
+
# Handle the case where there's only one genome
|
|
40
|
+
if len(Ndb['reference'].unique()) == 1:
|
|
41
|
+
Table['genome'].append(os.path.basename(Ndb['reference'].unique().tolist()[0]))
|
|
42
|
+
Table['secondary_cluster'].append("{0}_0".format(cluster))
|
|
43
|
+
cluster_ret = []
|
|
44
|
+
|
|
45
|
+
else:
|
|
46
|
+
# Make linkage Ndb
|
|
47
|
+
Ldb = drep.d_cluster.utils.make_linkage_Ndb(Ndb, **kwargs)
|
|
48
|
+
|
|
49
|
+
# 3) Cluster the linkagedb
|
|
50
|
+
Gdb, linkage = cluster_hierarchical(Ldb, linkage_method= S_Lmethod, \
|
|
51
|
+
linkage_cutoff= S_Lcutoff)
|
|
52
|
+
|
|
53
|
+
# 4) Extract secondary clusters
|
|
54
|
+
for clust, d in Gdb.groupby('cluster'):
|
|
55
|
+
for genome in d['genome'].tolist():
|
|
56
|
+
Table['genome'].append(genome)
|
|
57
|
+
Table['secondary_cluster'].append("{0}_{1}".format(cluster,clust))
|
|
58
|
+
|
|
59
|
+
# 5) Save the linkage
|
|
60
|
+
arguments = {'linkage_method':S_Lmethod,'linkage_cutoff':S_Lcutoff,\
|
|
61
|
+
'comparison_algorithm':comp_method,'minimum_coverage':cov_thresh}
|
|
62
|
+
cluster_ret = [linkage, Ldb, arguments]
|
|
63
|
+
|
|
64
|
+
# Return the database
|
|
65
|
+
Gdb = pd.DataFrame(Table)
|
|
66
|
+
Gdb['threshold'] = S_Lcutoff
|
|
67
|
+
Gdb['cluster_method'] = S_Lmethod
|
|
68
|
+
Gdb['comparison_algorithm'] = comp_method
|
|
69
|
+
|
|
70
|
+
return Gdb, cluster_ret
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def iteratre_clusters(Bdb, Cdb, id='primary_cluster'):
|
|
74
|
+
'''
|
|
75
|
+
An iterator: Given Bdb and Cdb, yeild smaller Bdb's in the same cluster
|
|
76
|
+
|
|
77
|
+
Args:
|
|
78
|
+
Bdb: [genome, location]
|
|
79
|
+
Cdb: [genome, id]
|
|
80
|
+
id: what to iterate on (default = 'primary_cluster')
|
|
81
|
+
|
|
82
|
+
Returns:
|
|
83
|
+
list: [d(subset of b), cluster(name of cluster)]
|
|
84
|
+
'''
|
|
85
|
+
Bdb = pd.merge(Bdb,Cdb)
|
|
86
|
+
for cluster, d in Bdb.groupby(id):
|
|
87
|
+
yield d, cluster
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def cluster_hierarchical(db, linkage_method= 'single', linkage_cutoff= 0.10):
|
|
91
|
+
'''
|
|
92
|
+
Perform hierarchical clustering on a symmetrical distiance matrix
|
|
93
|
+
|
|
94
|
+
Note this builds a dense matrix and is O(N^2) in memory. Single-linkage
|
|
95
|
+
primary clustering goes through drep.d_cluster.union_find instead, which is
|
|
96
|
+
equivalent but does not need the matrix.
|
|
97
|
+
|
|
98
|
+
Args:
|
|
99
|
+
db: result of db.pivot usually
|
|
100
|
+
linkage_method: passed to scipy.cluster.hierarchy.fcluster
|
|
101
|
+
linkage_cutoff: distance to draw the clustering line (default = .1)
|
|
102
|
+
|
|
103
|
+
Returns:
|
|
104
|
+
list: [Cdb, linkage]
|
|
105
|
+
'''
|
|
106
|
+
# Save names
|
|
107
|
+
names = list(db.columns)
|
|
108
|
+
|
|
109
|
+
# Generate linkage dataframe
|
|
110
|
+
arr = np.asarray(db)
|
|
111
|
+
try:
|
|
112
|
+
arr = ssd.squareform(arr)
|
|
113
|
+
except:
|
|
114
|
+
logging.error("The database passed in is not symmetrical!")
|
|
115
|
+
logging.error(arr)
|
|
116
|
+
logging.error(names)
|
|
117
|
+
sys.exit()
|
|
118
|
+
linkage = scipy.cluster.hierarchy.linkage(arr, method= linkage_method)
|
|
119
|
+
|
|
120
|
+
# Form clusters
|
|
121
|
+
fclust = scipy.cluster.hierarchy.fcluster(linkage,linkage_cutoff, \
|
|
122
|
+
criterion='distance')
|
|
123
|
+
# Make Cdb
|
|
124
|
+
Cdb = drep.d_cluster.utils._gen_cdb_from_fclust(fclust,names)
|
|
125
|
+
|
|
126
|
+
return Cdb, linkage
|