drep 4.0.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
drep/VERSION ADDED
@@ -0,0 +1 @@
1
+ 4.0.2
drep/WorkDirectory.py ADDED
@@ -0,0 +1,355 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ This module provides access to the workDirectory
4
+
5
+ The directory layout::
6
+
7
+ workDirectory
8
+ ./data
9
+ ...../MASH_files/
10
+ ...../ANIn_files/
11
+ ...../gANI_files/
12
+ ...../Clustering_files/
13
+ ...../checkM/
14
+ ........./genomes/
15
+ ........./checkM_outdir/
16
+ ...../prodigal/
17
+ ./figures
18
+ ./data_tables
19
+ ...../Bdb.csv # Sequence locations and filenames
20
+ ...../Mdb.csv # Raw results of MASH comparisons
21
+ ...../Ndb.csv # Raw results of ANIn comparisons
22
+ ...../Cdb.csv # Genomes and cluster designations
23
+ ...../Chdb.csv # CheckM results for Bdb
24
+ ...../Sdb.csv # Scoring information
25
+ ...../Wdb.csv # Winning genomes
26
+ ./dereplicated_genomes
27
+ ./log
28
+ ...../logger.log
29
+ ...../cluster_arguments.json
30
+
31
+ """
32
+
33
+ import os
34
+ import logging
35
+ import pandas as pd
36
+ import pickle
37
+ import json
38
+ import sys
39
+ import shutil
40
+ import glob
41
+ import numpy as np
42
+
43
+ import drep
44
+
45
+ class WorkDirectory(object):
46
+ '''
47
+ Object to interact with the workDirectory
48
+
49
+ Args:
50
+ location (str): location to make the workDirectory
51
+
52
+
53
+ '''
54
+ firstLevels = ['data','figures','data_tables','dereplicated_genomes','log']
55
+
56
+ def __init__(self, location):
57
+ self.location = os.path.abspath(location)
58
+ self.data_tables = {}
59
+ self.clusters = {}
60
+ self.arguments = {}
61
+ self.overwrite = True
62
+ self.name = None
63
+
64
+ self.make_fileStructure()
65
+ self.load_cached()
66
+
67
+ def __str__(self):
68
+ string = "Located: {0}\nDatatables: {1}\nCluster files: {2}\nArguments: {3}".format(\
69
+ self.location,list(self.data_tables.keys()),list(self.clusters.keys()),\
70
+ list(self.arguments.keys()))
71
+
72
+ return string
73
+
74
+ def make_fileStructure(self):
75
+ '''
76
+ Make the top level file structure
77
+ '''
78
+ location = self.location
79
+
80
+ if not os.path.exists(location):
81
+ os.makedirs(location)
82
+
83
+ for l in WorkDirectory.firstLevels:
84
+ loc = location + '/' + l
85
+ if not os.path.exists(loc):
86
+ os.makedirs(loc)
87
+
88
+ def load_cached(self):
89
+ '''
90
+ The wrapper to load everything it has into attributes
91
+ '''
92
+ # Import data_tables
93
+ loc = self.get_dir('data_tables')
94
+ self.import_data_tables(loc)
95
+
96
+ # Import pickles
97
+ loc = self.location + '/data/Clustering_files/'
98
+ if not os.path.exists(loc):
99
+ os.makedirs(loc)
100
+ self.import_clusters(loc)
101
+
102
+ # Import arguments
103
+ loc = self.location + '/log/'
104
+ if not os.path.exists(loc):
105
+ os.makedirs(loc)
106
+ self.import_arguments(loc)
107
+
108
+ def import_data_tables(self, loc):
109
+ '''
110
+ Given the location of the datatables, load them
111
+ '''
112
+ tables = [os.path.join(loc, t) for t in os.listdir(loc) if \
113
+ os.path.isfile(os.path.join(loc, t))]
114
+ for t in tables:
115
+ self.data_tables[os.path.basename(t).replace('.csv','').\
116
+ replace('.pickle', '')] = t
117
+
118
+ def import_clusters(self,loc):
119
+ '''
120
+ Given the location of the cluster files, load them
121
+ '''
122
+ pickles = [os.path.join(loc, t) for t in os.listdir(loc) if os.path.isfile(os.path.join(loc, t))]
123
+ for p in pickles:
124
+ assert p.endswith('.pickle'), "{0} is incorrectly in the data/Clustering_files folder".format(p)
125
+
126
+ f = open(p,'rb')
127
+ try:
128
+ linkage = pickle.load(f)
129
+ db = pickle.load(f)
130
+ args = pickle.load(f)
131
+ name = os.path.basename(p).replace('.pickle','')
132
+ self.clusters[name] = {'linkage':linkage,'db':db,'arguments':args}
133
+ except:
134
+ print("{0} is an imporperly made pickle- skipping import".format(p))
135
+
136
+ def import_arguments(self,loc):
137
+ '''
138
+ Given the location of the log directory, load it
139
+ '''
140
+ args = [os.path.join(loc, t) for t in os.listdir(loc) if os.path.isfile(os.path.join(loc, t))]
141
+ for j in args:
142
+ if j.endswith('_arguments.json'):
143
+ with open(j) as data_file:
144
+ data = json.load(data_file)
145
+ self.arguments[os.path.basename(j).replace('_arguments.json','')] = data
146
+
147
+ def hasDb(self,db):
148
+ '''
149
+ If db is in the data_tables, return True
150
+ '''
151
+ return (db in self.data_tables)
152
+
153
+ def get_cluster(self, name):
154
+ '''
155
+ Get the cluster passed in
156
+
157
+ Args:
158
+ name: name of the cluster
159
+
160
+ Returns:
161
+ cluster
162
+ '''
163
+ # If the whole cluster name is passed in
164
+ if name in self.clusters:
165
+ return self.clusters[name]
166
+
167
+ else:
168
+ n = 'secondary_linkage_cluster_{0}'.format(name)
169
+ if n in self.clusters:
170
+ return self.clusters[n]
171
+
172
+ print("Could not find cluster {0} - could it be a singleton?".format(name))
173
+ print("The clusters I have are: {0}".format(list(self.clusters.keys())))
174
+ print("Quitting")
175
+ sys.exit()
176
+
177
+ def get_primary_linkage(self):
178
+ '''
179
+ Get the primary linkage cluster
180
+ '''
181
+ return self.clusters['primary_linkage']
182
+
183
+ def store_db(self,db,name,overwrite=None):
184
+ '''
185
+ Store a dataframe in the workDirectory
186
+
187
+ Will make a physical copy in the datatables folder
188
+
189
+ Args:
190
+ db: pandas dataframe to store
191
+ name: name to store it under (will add .csv automatically)
192
+ overwrite: if True, overwrite if DataFrame with same name already exists
193
+ '''
194
+ loc = self.get_dir('data_tables')
195
+
196
+ if overwrite == None:
197
+ overwrite = self.overwrite
198
+
199
+ if os.path.isfile(loc + name + '.csv'):
200
+ assert overwrite == True, "data_table {0} already exists".format(name)
201
+
202
+ if name == 'Mdb':
203
+ floc = loc + name + '.csv'
204
+ db.to_csv(floc, index=False)
205
+ self.data_tables[name] = floc
206
+
207
+ else:
208
+ floc = loc + name + '.csv'
209
+ db.to_csv(floc, index=False)
210
+ self.data_tables[name] = floc
211
+
212
+ def get_db(self, name, return_none=True, forPlotting=False):
213
+ '''
214
+ Get database from self.data_tables
215
+
216
+ Args:
217
+ name: name of dataframe
218
+ return_none: if True will return None if database not found; otherwise assert False
219
+ forPlotting: if True don't do fancy dType loading; it messes with order of names for dendrograms
220
+ '''
221
+ if name in self.data_tables:
222
+ if name == 'Mdb':
223
+ if forPlotting:
224
+ return pd.read_csv(self.data_tables[name])
225
+ else:
226
+ dTypes={'genome1':'category', 'genome2':'category', 'dist':np.float32,\
227
+ 'similarity':np.float32}
228
+ return pd.read_csv(self.data_tables[name], dtype=dTypes)
229
+
230
+ else:
231
+ return pd.read_csv(self.data_tables[name])
232
+ else:
233
+ if return_none:
234
+ return None
235
+ else:
236
+ assert False, "Datatable {0} is not in the work directory {1}".format(\
237
+ name, self.location)
238
+
239
+ def get_dir(self, dir):
240
+ '''
241
+ Get the location of one of the named directory types
242
+
243
+ Args:
244
+ dir: Name of directory to find
245
+
246
+ Returns:
247
+ string: Location of requested directory
248
+ '''
249
+ d = None
250
+ if dir == 'data_tables':
251
+ d = self.location + '/data_tables/'
252
+ if dir == 'prodigal':
253
+ d = self.location + '/data/prodigal/'
254
+ elif dir == 'centrifuge':
255
+ d = self.location + '/data/centrifuge/'
256
+ elif dir == 'ESOM':
257
+ d = self.location + '/data/ESOM/'
258
+ elif dir == 'log':
259
+ d = self.location + '/log/'
260
+ elif dir == 'cmd_logs':
261
+ d = self.location + '/log/cmd_logs/'
262
+ elif dir == 'MASH':
263
+ d = self.location + '/data/MASH_files/'
264
+ elif dir == 'checkM':
265
+ d = self.location + '/data/checkM/'
266
+ elif dir == 'data':
267
+ d = self.location + '/data/'
268
+ elif dir == 'clustering':
269
+ d = self.location + '/data/Clustering_files/'
270
+ elif dir == 'dereplicated_genomes':
271
+ d = self.location + '/dereplicated_genomes/'
272
+ elif dir == 'figures':
273
+ d = self.location + '/figures/'
274
+
275
+ if d == None:
276
+ assert False, "{0} is not a directory I know about".format(dir)
277
+
278
+ if not os.path.exists(d):
279
+ os.makedirs(d)
280
+
281
+ return d
282
+
283
+ def get_loc(self, what):
284
+ '''
285
+ Get the location of Things
286
+
287
+ Args:
288
+ what: string of what to get the location of
289
+
290
+ Returns:
291
+ string: location of what
292
+ '''
293
+ if what == 'log':
294
+ return self.location + '/log/logger.log'
295
+
296
+ def store_special(self, name, thing):
297
+ '''
298
+ Store special items in the work directory
299
+
300
+ Args:
301
+ name: what to store
302
+ thing: actual thing to store
303
+ '''
304
+ if name == 'primary_linkage':
305
+ assert len(thing) == 3
306
+ store_loc = self.get_dir('clustering')
307
+ logging.debug('Saving primary_linkage pickle to {0}'.format(store_loc))
308
+ with open(store_loc + 'primary_linkage.pickle', 'wb') as handle:
309
+ pickle.dump(thing[0], handle, protocol=4)
310
+ pickle.dump(thing[1], handle, protocol=4)
311
+ pickle.dump(thing[2], handle, protocol=4)
312
+
313
+ elif name == 'secondary_linkages':
314
+ assert type(thing) == dict
315
+ store_loc = self.get_dir('clustering')
316
+ for name, cluster_ret in thing.items():
317
+ #print(name)
318
+ if len(cluster_ret) == 0:
319
+ continue
320
+ assert len(cluster_ret) == 3
321
+
322
+ pickle_name = "secondary_linkage_cluster_{0}.pickle".format(name)
323
+ logging.debug('Saving secondary_linkage pickle {1} to {0}'.format(pickle_name,\
324
+ store_loc))
325
+ with open(store_loc + pickle_name, 'wb') as handle:
326
+ pickle.dump(cluster_ret[0],handle, protocol=4)
327
+ pickle.dump(cluster_ret[1],handle, protocol=4)
328
+ pickle.dump(cluster_ret[2],handle, protocol=4)
329
+
330
+ elif name == 'dereplicated_genomes':
331
+ output_folder = self.get_dir('dereplicated_genomes')
332
+ if os.path.exists(output_folder):
333
+ #logging.debug("{0} already exists: removing and remaking".format(output_folder))
334
+ shutil.rmtree(output_folder)
335
+ os.makedirs(output_folder)
336
+
337
+ for loc in thing:
338
+ genome = os.path.basename(loc)
339
+ shutil.copy2(loc, "{0}{1}".format(output_folder,genome))
340
+ #logging.info("Done! Dereplicated genomes saved at {0}".format(output_folder))
341
+
342
+ elif name == 'cluster_log':
343
+ cluster_log = os.path.join(self.get_dir('log') + 'cluster_arguments.json')
344
+ with open(cluster_log, 'w') as fp:
345
+ json.dump(thing, fp)
346
+ fp.close()
347
+
348
+ def _wipe_secondary_clusters(self):
349
+ '''
350
+ Wipe any and all secondary clusters present in the workDirectory
351
+ '''
352
+ # Clear out clustering folder
353
+ c_folder = self.get_dir('clustering')
354
+ for fn in glob.glob(c_folder + 'secondary_linkage_cluster*'):
355
+ os.remove(fn)
drep/__init__.py ADDED
@@ -0,0 +1,101 @@
1
+ #!/usr/bin/env python3
2
+
3
+ from subprocess import call
4
+ import os
5
+ # from Bio import SeqIO
6
+ import shutil
7
+ import multiprocessing
8
+ import multiprocessing.dummy
9
+ import datetime
10
+
11
+ import drep.d_filter
12
+ import drep.d_cluster
13
+ import drep.d_bonus
14
+
15
+ def run_cmd(cmd, dry=False, shell=True, logdir=False):
16
+ # if dry, just print cmd
17
+ if dry:
18
+ if shell:
19
+ print(cmd)
20
+ else:
21
+ print(' '.join(cmd))
22
+ return
23
+
24
+ # figure out what files you're going to store the output
25
+ if logdir == False:
26
+ devnull = open(os.devnull, 'w')
27
+ sto = devnull
28
+ ste = devnull
29
+ else:
30
+ uniq_filename = str(datetime.datetime.now().date()) + '_' + \
31
+ str(datetime.datetime.now().time()).replace(':', '.')
32
+ sto = open(os.path.join(logdir + uniq_filename + '.STDOUT'), 'w')
33
+ ste = open(os.path.join(logdir + uniq_filename + '.STDERR'), 'w')
34
+
35
+ # log the command
36
+ now = open(os.path.join(logdir + uniq_filename + '.CMD'), 'w')
37
+ if shell:
38
+ now.write(str(cmd) + '\n')
39
+ else:
40
+ now.write(' '.join(cmd) + '\n')
41
+ now.close()
42
+
43
+ # run the command
44
+ if shell:
45
+ call(cmd,shell=True,stdout=sto, stderr=ste)
46
+ else:
47
+ call(cmd,stdout=sto, stderr=ste)
48
+ return
49
+
50
+ def thread_cmd_wrapper(tup):
51
+ run_cmd(*tup)
52
+
53
+ def thread_cmds(cmds, dry=False, shell=False, logdir=False, t=10):
54
+ pool = multiprocessing.dummy.Pool(processes=t)
55
+ tups = [(cmd, dry, shell, logdir) for cmd in cmds]
56
+ pool.map(thread_cmd_wrapper, tups)
57
+ pool.close()
58
+ pool.join()
59
+ return
60
+
61
+ def make_dir(outdirname,dry=False,overwrite=False):
62
+ if dry:
63
+ return
64
+ if os.path.exists(outdirname):
65
+ if overwrite:
66
+ pass
67
+ #shutil.rmtree(outdirname, ignore_errors=True)
68
+ #os.makedirs(outdirname)
69
+ else:
70
+ raise ValueError("{0} already exsists! Will not overwrite with current settings".\
71
+ format(outdirname))
72
+ else:
73
+ os.makedirs(outdirname)
74
+
75
+ def clobber_dir(outdirname,dry=False,overwrite=False):
76
+ if dry:
77
+ return
78
+ if os.path.exists(outdirname):
79
+ if overwrite:
80
+ shutil.rmtree(outdirname, ignore_errors=True)
81
+ os.makedirs(outdirname)
82
+ else:
83
+ raise ValueError("{0} already exsists! Will not overwrite with current settings".\
84
+ format(outdirname))
85
+ else:
86
+ os.makedirs(outdirname)
87
+
88
+ def get_exe(name):
89
+ '''
90
+ Given the name of a program, return the path to the exe if I can find it
91
+
92
+ Args:
93
+ name: name of program
94
+
95
+ Returns:
96
+ string: location to program exe
97
+ '''
98
+ loc, works = drep.d_bonus.find_program(name)
99
+ if works == False:
100
+ raise ValueError("{0} isn't working- make sure its installed".format(name))
101
+ return loc