drep 4.0.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- drep/VERSION +1 -0
- drep/WorkDirectory.py +355 -0
- drep/__init__.py +101 -0
- drep/argumentParser.py +279 -0
- drep/controller.py +105 -0
- drep/d_adjust.py +272 -0
- drep/d_analyze.py +1613 -0
- drep/d_bonus.py +429 -0
- drep/d_choose.py +362 -0
- drep/d_cluster/__init__.py +0 -0
- drep/d_cluster/cluster_utils.py +126 -0
- drep/d_cluster/compare_utils.py +636 -0
- drep/d_cluster/controller.py +228 -0
- drep/d_cluster/external.py +765 -0
- drep/d_cluster/greedy_clustering.py +181 -0
- drep/d_cluster/parsers.py +0 -0
- drep/d_cluster/union_find.py +543 -0
- drep/d_cluster/utils.py +687 -0
- drep/d_evaluate.py +355 -0
- drep/d_filter.py +831 -0
- drep/d_workflows.py +135 -0
- drep-4.0.2.data/scripts/ScaffoldLevel_dRep.py +1101 -0
- drep-4.0.2.data/scripts/dRep +32 -0
- drep-4.0.2.data/scripts/parse_stb.py +140 -0
- drep-4.0.2.dist-info/METADATA +23 -0
- drep-4.0.2.dist-info/RECORD +28 -0
- drep-4.0.2.dist-info/WHEEL +5 -0
- drep-4.0.2.dist-info/top_level.txt +1 -0
drep/VERSION
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
4.0.2
|
drep/WorkDirectory.py
ADDED
|
@@ -0,0 +1,355 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
This module provides access to the workDirectory
|
|
4
|
+
|
|
5
|
+
The directory layout::
|
|
6
|
+
|
|
7
|
+
workDirectory
|
|
8
|
+
./data
|
|
9
|
+
...../MASH_files/
|
|
10
|
+
...../ANIn_files/
|
|
11
|
+
...../gANI_files/
|
|
12
|
+
...../Clustering_files/
|
|
13
|
+
...../checkM/
|
|
14
|
+
........./genomes/
|
|
15
|
+
........./checkM_outdir/
|
|
16
|
+
...../prodigal/
|
|
17
|
+
./figures
|
|
18
|
+
./data_tables
|
|
19
|
+
...../Bdb.csv # Sequence locations and filenames
|
|
20
|
+
...../Mdb.csv # Raw results of MASH comparisons
|
|
21
|
+
...../Ndb.csv # Raw results of ANIn comparisons
|
|
22
|
+
...../Cdb.csv # Genomes and cluster designations
|
|
23
|
+
...../Chdb.csv # CheckM results for Bdb
|
|
24
|
+
...../Sdb.csv # Scoring information
|
|
25
|
+
...../Wdb.csv # Winning genomes
|
|
26
|
+
./dereplicated_genomes
|
|
27
|
+
./log
|
|
28
|
+
...../logger.log
|
|
29
|
+
...../cluster_arguments.json
|
|
30
|
+
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
import os
|
|
34
|
+
import logging
|
|
35
|
+
import pandas as pd
|
|
36
|
+
import pickle
|
|
37
|
+
import json
|
|
38
|
+
import sys
|
|
39
|
+
import shutil
|
|
40
|
+
import glob
|
|
41
|
+
import numpy as np
|
|
42
|
+
|
|
43
|
+
import drep
|
|
44
|
+
|
|
45
|
+
class WorkDirectory(object):
|
|
46
|
+
'''
|
|
47
|
+
Object to interact with the workDirectory
|
|
48
|
+
|
|
49
|
+
Args:
|
|
50
|
+
location (str): location to make the workDirectory
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
'''
|
|
54
|
+
firstLevels = ['data','figures','data_tables','dereplicated_genomes','log']
|
|
55
|
+
|
|
56
|
+
def __init__(self, location):
|
|
57
|
+
self.location = os.path.abspath(location)
|
|
58
|
+
self.data_tables = {}
|
|
59
|
+
self.clusters = {}
|
|
60
|
+
self.arguments = {}
|
|
61
|
+
self.overwrite = True
|
|
62
|
+
self.name = None
|
|
63
|
+
|
|
64
|
+
self.make_fileStructure()
|
|
65
|
+
self.load_cached()
|
|
66
|
+
|
|
67
|
+
def __str__(self):
|
|
68
|
+
string = "Located: {0}\nDatatables: {1}\nCluster files: {2}\nArguments: {3}".format(\
|
|
69
|
+
self.location,list(self.data_tables.keys()),list(self.clusters.keys()),\
|
|
70
|
+
list(self.arguments.keys()))
|
|
71
|
+
|
|
72
|
+
return string
|
|
73
|
+
|
|
74
|
+
def make_fileStructure(self):
|
|
75
|
+
'''
|
|
76
|
+
Make the top level file structure
|
|
77
|
+
'''
|
|
78
|
+
location = self.location
|
|
79
|
+
|
|
80
|
+
if not os.path.exists(location):
|
|
81
|
+
os.makedirs(location)
|
|
82
|
+
|
|
83
|
+
for l in WorkDirectory.firstLevels:
|
|
84
|
+
loc = location + '/' + l
|
|
85
|
+
if not os.path.exists(loc):
|
|
86
|
+
os.makedirs(loc)
|
|
87
|
+
|
|
88
|
+
def load_cached(self):
|
|
89
|
+
'''
|
|
90
|
+
The wrapper to load everything it has into attributes
|
|
91
|
+
'''
|
|
92
|
+
# Import data_tables
|
|
93
|
+
loc = self.get_dir('data_tables')
|
|
94
|
+
self.import_data_tables(loc)
|
|
95
|
+
|
|
96
|
+
# Import pickles
|
|
97
|
+
loc = self.location + '/data/Clustering_files/'
|
|
98
|
+
if not os.path.exists(loc):
|
|
99
|
+
os.makedirs(loc)
|
|
100
|
+
self.import_clusters(loc)
|
|
101
|
+
|
|
102
|
+
# Import arguments
|
|
103
|
+
loc = self.location + '/log/'
|
|
104
|
+
if not os.path.exists(loc):
|
|
105
|
+
os.makedirs(loc)
|
|
106
|
+
self.import_arguments(loc)
|
|
107
|
+
|
|
108
|
+
def import_data_tables(self, loc):
|
|
109
|
+
'''
|
|
110
|
+
Given the location of the datatables, load them
|
|
111
|
+
'''
|
|
112
|
+
tables = [os.path.join(loc, t) for t in os.listdir(loc) if \
|
|
113
|
+
os.path.isfile(os.path.join(loc, t))]
|
|
114
|
+
for t in tables:
|
|
115
|
+
self.data_tables[os.path.basename(t).replace('.csv','').\
|
|
116
|
+
replace('.pickle', '')] = t
|
|
117
|
+
|
|
118
|
+
def import_clusters(self,loc):
|
|
119
|
+
'''
|
|
120
|
+
Given the location of the cluster files, load them
|
|
121
|
+
'''
|
|
122
|
+
pickles = [os.path.join(loc, t) for t in os.listdir(loc) if os.path.isfile(os.path.join(loc, t))]
|
|
123
|
+
for p in pickles:
|
|
124
|
+
assert p.endswith('.pickle'), "{0} is incorrectly in the data/Clustering_files folder".format(p)
|
|
125
|
+
|
|
126
|
+
f = open(p,'rb')
|
|
127
|
+
try:
|
|
128
|
+
linkage = pickle.load(f)
|
|
129
|
+
db = pickle.load(f)
|
|
130
|
+
args = pickle.load(f)
|
|
131
|
+
name = os.path.basename(p).replace('.pickle','')
|
|
132
|
+
self.clusters[name] = {'linkage':linkage,'db':db,'arguments':args}
|
|
133
|
+
except:
|
|
134
|
+
print("{0} is an imporperly made pickle- skipping import".format(p))
|
|
135
|
+
|
|
136
|
+
def import_arguments(self,loc):
|
|
137
|
+
'''
|
|
138
|
+
Given the location of the log directory, load it
|
|
139
|
+
'''
|
|
140
|
+
args = [os.path.join(loc, t) for t in os.listdir(loc) if os.path.isfile(os.path.join(loc, t))]
|
|
141
|
+
for j in args:
|
|
142
|
+
if j.endswith('_arguments.json'):
|
|
143
|
+
with open(j) as data_file:
|
|
144
|
+
data = json.load(data_file)
|
|
145
|
+
self.arguments[os.path.basename(j).replace('_arguments.json','')] = data
|
|
146
|
+
|
|
147
|
+
def hasDb(self,db):
|
|
148
|
+
'''
|
|
149
|
+
If db is in the data_tables, return True
|
|
150
|
+
'''
|
|
151
|
+
return (db in self.data_tables)
|
|
152
|
+
|
|
153
|
+
def get_cluster(self, name):
|
|
154
|
+
'''
|
|
155
|
+
Get the cluster passed in
|
|
156
|
+
|
|
157
|
+
Args:
|
|
158
|
+
name: name of the cluster
|
|
159
|
+
|
|
160
|
+
Returns:
|
|
161
|
+
cluster
|
|
162
|
+
'''
|
|
163
|
+
# If the whole cluster name is passed in
|
|
164
|
+
if name in self.clusters:
|
|
165
|
+
return self.clusters[name]
|
|
166
|
+
|
|
167
|
+
else:
|
|
168
|
+
n = 'secondary_linkage_cluster_{0}'.format(name)
|
|
169
|
+
if n in self.clusters:
|
|
170
|
+
return self.clusters[n]
|
|
171
|
+
|
|
172
|
+
print("Could not find cluster {0} - could it be a singleton?".format(name))
|
|
173
|
+
print("The clusters I have are: {0}".format(list(self.clusters.keys())))
|
|
174
|
+
print("Quitting")
|
|
175
|
+
sys.exit()
|
|
176
|
+
|
|
177
|
+
def get_primary_linkage(self):
|
|
178
|
+
'''
|
|
179
|
+
Get the primary linkage cluster
|
|
180
|
+
'''
|
|
181
|
+
return self.clusters['primary_linkage']
|
|
182
|
+
|
|
183
|
+
def store_db(self,db,name,overwrite=None):
|
|
184
|
+
'''
|
|
185
|
+
Store a dataframe in the workDirectory
|
|
186
|
+
|
|
187
|
+
Will make a physical copy in the datatables folder
|
|
188
|
+
|
|
189
|
+
Args:
|
|
190
|
+
db: pandas dataframe to store
|
|
191
|
+
name: name to store it under (will add .csv automatically)
|
|
192
|
+
overwrite: if True, overwrite if DataFrame with same name already exists
|
|
193
|
+
'''
|
|
194
|
+
loc = self.get_dir('data_tables')
|
|
195
|
+
|
|
196
|
+
if overwrite == None:
|
|
197
|
+
overwrite = self.overwrite
|
|
198
|
+
|
|
199
|
+
if os.path.isfile(loc + name + '.csv'):
|
|
200
|
+
assert overwrite == True, "data_table {0} already exists".format(name)
|
|
201
|
+
|
|
202
|
+
if name == 'Mdb':
|
|
203
|
+
floc = loc + name + '.csv'
|
|
204
|
+
db.to_csv(floc, index=False)
|
|
205
|
+
self.data_tables[name] = floc
|
|
206
|
+
|
|
207
|
+
else:
|
|
208
|
+
floc = loc + name + '.csv'
|
|
209
|
+
db.to_csv(floc, index=False)
|
|
210
|
+
self.data_tables[name] = floc
|
|
211
|
+
|
|
212
|
+
def get_db(self, name, return_none=True, forPlotting=False):
|
|
213
|
+
'''
|
|
214
|
+
Get database from self.data_tables
|
|
215
|
+
|
|
216
|
+
Args:
|
|
217
|
+
name: name of dataframe
|
|
218
|
+
return_none: if True will return None if database not found; otherwise assert False
|
|
219
|
+
forPlotting: if True don't do fancy dType loading; it messes with order of names for dendrograms
|
|
220
|
+
'''
|
|
221
|
+
if name in self.data_tables:
|
|
222
|
+
if name == 'Mdb':
|
|
223
|
+
if forPlotting:
|
|
224
|
+
return pd.read_csv(self.data_tables[name])
|
|
225
|
+
else:
|
|
226
|
+
dTypes={'genome1':'category', 'genome2':'category', 'dist':np.float32,\
|
|
227
|
+
'similarity':np.float32}
|
|
228
|
+
return pd.read_csv(self.data_tables[name], dtype=dTypes)
|
|
229
|
+
|
|
230
|
+
else:
|
|
231
|
+
return pd.read_csv(self.data_tables[name])
|
|
232
|
+
else:
|
|
233
|
+
if return_none:
|
|
234
|
+
return None
|
|
235
|
+
else:
|
|
236
|
+
assert False, "Datatable {0} is not in the work directory {1}".format(\
|
|
237
|
+
name, self.location)
|
|
238
|
+
|
|
239
|
+
def get_dir(self, dir):
|
|
240
|
+
'''
|
|
241
|
+
Get the location of one of the named directory types
|
|
242
|
+
|
|
243
|
+
Args:
|
|
244
|
+
dir: Name of directory to find
|
|
245
|
+
|
|
246
|
+
Returns:
|
|
247
|
+
string: Location of requested directory
|
|
248
|
+
'''
|
|
249
|
+
d = None
|
|
250
|
+
if dir == 'data_tables':
|
|
251
|
+
d = self.location + '/data_tables/'
|
|
252
|
+
if dir == 'prodigal':
|
|
253
|
+
d = self.location + '/data/prodigal/'
|
|
254
|
+
elif dir == 'centrifuge':
|
|
255
|
+
d = self.location + '/data/centrifuge/'
|
|
256
|
+
elif dir == 'ESOM':
|
|
257
|
+
d = self.location + '/data/ESOM/'
|
|
258
|
+
elif dir == 'log':
|
|
259
|
+
d = self.location + '/log/'
|
|
260
|
+
elif dir == 'cmd_logs':
|
|
261
|
+
d = self.location + '/log/cmd_logs/'
|
|
262
|
+
elif dir == 'MASH':
|
|
263
|
+
d = self.location + '/data/MASH_files/'
|
|
264
|
+
elif dir == 'checkM':
|
|
265
|
+
d = self.location + '/data/checkM/'
|
|
266
|
+
elif dir == 'data':
|
|
267
|
+
d = self.location + '/data/'
|
|
268
|
+
elif dir == 'clustering':
|
|
269
|
+
d = self.location + '/data/Clustering_files/'
|
|
270
|
+
elif dir == 'dereplicated_genomes':
|
|
271
|
+
d = self.location + '/dereplicated_genomes/'
|
|
272
|
+
elif dir == 'figures':
|
|
273
|
+
d = self.location + '/figures/'
|
|
274
|
+
|
|
275
|
+
if d == None:
|
|
276
|
+
assert False, "{0} is not a directory I know about".format(dir)
|
|
277
|
+
|
|
278
|
+
if not os.path.exists(d):
|
|
279
|
+
os.makedirs(d)
|
|
280
|
+
|
|
281
|
+
return d
|
|
282
|
+
|
|
283
|
+
def get_loc(self, what):
|
|
284
|
+
'''
|
|
285
|
+
Get the location of Things
|
|
286
|
+
|
|
287
|
+
Args:
|
|
288
|
+
what: string of what to get the location of
|
|
289
|
+
|
|
290
|
+
Returns:
|
|
291
|
+
string: location of what
|
|
292
|
+
'''
|
|
293
|
+
if what == 'log':
|
|
294
|
+
return self.location + '/log/logger.log'
|
|
295
|
+
|
|
296
|
+
def store_special(self, name, thing):
|
|
297
|
+
'''
|
|
298
|
+
Store special items in the work directory
|
|
299
|
+
|
|
300
|
+
Args:
|
|
301
|
+
name: what to store
|
|
302
|
+
thing: actual thing to store
|
|
303
|
+
'''
|
|
304
|
+
if name == 'primary_linkage':
|
|
305
|
+
assert len(thing) == 3
|
|
306
|
+
store_loc = self.get_dir('clustering')
|
|
307
|
+
logging.debug('Saving primary_linkage pickle to {0}'.format(store_loc))
|
|
308
|
+
with open(store_loc + 'primary_linkage.pickle', 'wb') as handle:
|
|
309
|
+
pickle.dump(thing[0], handle, protocol=4)
|
|
310
|
+
pickle.dump(thing[1], handle, protocol=4)
|
|
311
|
+
pickle.dump(thing[2], handle, protocol=4)
|
|
312
|
+
|
|
313
|
+
elif name == 'secondary_linkages':
|
|
314
|
+
assert type(thing) == dict
|
|
315
|
+
store_loc = self.get_dir('clustering')
|
|
316
|
+
for name, cluster_ret in thing.items():
|
|
317
|
+
#print(name)
|
|
318
|
+
if len(cluster_ret) == 0:
|
|
319
|
+
continue
|
|
320
|
+
assert len(cluster_ret) == 3
|
|
321
|
+
|
|
322
|
+
pickle_name = "secondary_linkage_cluster_{0}.pickle".format(name)
|
|
323
|
+
logging.debug('Saving secondary_linkage pickle {1} to {0}'.format(pickle_name,\
|
|
324
|
+
store_loc))
|
|
325
|
+
with open(store_loc + pickle_name, 'wb') as handle:
|
|
326
|
+
pickle.dump(cluster_ret[0],handle, protocol=4)
|
|
327
|
+
pickle.dump(cluster_ret[1],handle, protocol=4)
|
|
328
|
+
pickle.dump(cluster_ret[2],handle, protocol=4)
|
|
329
|
+
|
|
330
|
+
elif name == 'dereplicated_genomes':
|
|
331
|
+
output_folder = self.get_dir('dereplicated_genomes')
|
|
332
|
+
if os.path.exists(output_folder):
|
|
333
|
+
#logging.debug("{0} already exists: removing and remaking".format(output_folder))
|
|
334
|
+
shutil.rmtree(output_folder)
|
|
335
|
+
os.makedirs(output_folder)
|
|
336
|
+
|
|
337
|
+
for loc in thing:
|
|
338
|
+
genome = os.path.basename(loc)
|
|
339
|
+
shutil.copy2(loc, "{0}{1}".format(output_folder,genome))
|
|
340
|
+
#logging.info("Done! Dereplicated genomes saved at {0}".format(output_folder))
|
|
341
|
+
|
|
342
|
+
elif name == 'cluster_log':
|
|
343
|
+
cluster_log = os.path.join(self.get_dir('log') + 'cluster_arguments.json')
|
|
344
|
+
with open(cluster_log, 'w') as fp:
|
|
345
|
+
json.dump(thing, fp)
|
|
346
|
+
fp.close()
|
|
347
|
+
|
|
348
|
+
def _wipe_secondary_clusters(self):
|
|
349
|
+
'''
|
|
350
|
+
Wipe any and all secondary clusters present in the workDirectory
|
|
351
|
+
'''
|
|
352
|
+
# Clear out clustering folder
|
|
353
|
+
c_folder = self.get_dir('clustering')
|
|
354
|
+
for fn in glob.glob(c_folder + 'secondary_linkage_cluster*'):
|
|
355
|
+
os.remove(fn)
|
drep/__init__.py
ADDED
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
|
|
3
|
+
from subprocess import call
|
|
4
|
+
import os
|
|
5
|
+
# from Bio import SeqIO
|
|
6
|
+
import shutil
|
|
7
|
+
import multiprocessing
|
|
8
|
+
import multiprocessing.dummy
|
|
9
|
+
import datetime
|
|
10
|
+
|
|
11
|
+
import drep.d_filter
|
|
12
|
+
import drep.d_cluster
|
|
13
|
+
import drep.d_bonus
|
|
14
|
+
|
|
15
|
+
def run_cmd(cmd, dry=False, shell=True, logdir=False):
|
|
16
|
+
# if dry, just print cmd
|
|
17
|
+
if dry:
|
|
18
|
+
if shell:
|
|
19
|
+
print(cmd)
|
|
20
|
+
else:
|
|
21
|
+
print(' '.join(cmd))
|
|
22
|
+
return
|
|
23
|
+
|
|
24
|
+
# figure out what files you're going to store the output
|
|
25
|
+
if logdir == False:
|
|
26
|
+
devnull = open(os.devnull, 'w')
|
|
27
|
+
sto = devnull
|
|
28
|
+
ste = devnull
|
|
29
|
+
else:
|
|
30
|
+
uniq_filename = str(datetime.datetime.now().date()) + '_' + \
|
|
31
|
+
str(datetime.datetime.now().time()).replace(':', '.')
|
|
32
|
+
sto = open(os.path.join(logdir + uniq_filename + '.STDOUT'), 'w')
|
|
33
|
+
ste = open(os.path.join(logdir + uniq_filename + '.STDERR'), 'w')
|
|
34
|
+
|
|
35
|
+
# log the command
|
|
36
|
+
now = open(os.path.join(logdir + uniq_filename + '.CMD'), 'w')
|
|
37
|
+
if shell:
|
|
38
|
+
now.write(str(cmd) + '\n')
|
|
39
|
+
else:
|
|
40
|
+
now.write(' '.join(cmd) + '\n')
|
|
41
|
+
now.close()
|
|
42
|
+
|
|
43
|
+
# run the command
|
|
44
|
+
if shell:
|
|
45
|
+
call(cmd,shell=True,stdout=sto, stderr=ste)
|
|
46
|
+
else:
|
|
47
|
+
call(cmd,stdout=sto, stderr=ste)
|
|
48
|
+
return
|
|
49
|
+
|
|
50
|
+
def thread_cmd_wrapper(tup):
|
|
51
|
+
run_cmd(*tup)
|
|
52
|
+
|
|
53
|
+
def thread_cmds(cmds, dry=False, shell=False, logdir=False, t=10):
|
|
54
|
+
pool = multiprocessing.dummy.Pool(processes=t)
|
|
55
|
+
tups = [(cmd, dry, shell, logdir) for cmd in cmds]
|
|
56
|
+
pool.map(thread_cmd_wrapper, tups)
|
|
57
|
+
pool.close()
|
|
58
|
+
pool.join()
|
|
59
|
+
return
|
|
60
|
+
|
|
61
|
+
def make_dir(outdirname,dry=False,overwrite=False):
|
|
62
|
+
if dry:
|
|
63
|
+
return
|
|
64
|
+
if os.path.exists(outdirname):
|
|
65
|
+
if overwrite:
|
|
66
|
+
pass
|
|
67
|
+
#shutil.rmtree(outdirname, ignore_errors=True)
|
|
68
|
+
#os.makedirs(outdirname)
|
|
69
|
+
else:
|
|
70
|
+
raise ValueError("{0} already exsists! Will not overwrite with current settings".\
|
|
71
|
+
format(outdirname))
|
|
72
|
+
else:
|
|
73
|
+
os.makedirs(outdirname)
|
|
74
|
+
|
|
75
|
+
def clobber_dir(outdirname,dry=False,overwrite=False):
|
|
76
|
+
if dry:
|
|
77
|
+
return
|
|
78
|
+
if os.path.exists(outdirname):
|
|
79
|
+
if overwrite:
|
|
80
|
+
shutil.rmtree(outdirname, ignore_errors=True)
|
|
81
|
+
os.makedirs(outdirname)
|
|
82
|
+
else:
|
|
83
|
+
raise ValueError("{0} already exsists! Will not overwrite with current settings".\
|
|
84
|
+
format(outdirname))
|
|
85
|
+
else:
|
|
86
|
+
os.makedirs(outdirname)
|
|
87
|
+
|
|
88
|
+
def get_exe(name):
|
|
89
|
+
'''
|
|
90
|
+
Given the name of a program, return the path to the exe if I can find it
|
|
91
|
+
|
|
92
|
+
Args:
|
|
93
|
+
name: name of program
|
|
94
|
+
|
|
95
|
+
Returns:
|
|
96
|
+
string: location to program exe
|
|
97
|
+
'''
|
|
98
|
+
loc, works = drep.d_bonus.find_program(name)
|
|
99
|
+
if works == False:
|
|
100
|
+
raise ValueError("{0} isn't working- make sure its installed".format(name))
|
|
101
|
+
return loc
|