drep 4.0.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
drep/argumentParser.py ADDED
@@ -0,0 +1,279 @@
1
+ #!/usr/bin/env python
2
+
3
+ '''
4
+ dRep- parse command-line arguemnts
5
+ '''
6
+
7
+ __author__ = "Matt Olm"
8
+ __license__ = "MIT"
9
+ __email__ = "mattolm@gmail.com"
10
+ __status__ = "Development"
11
+
12
+ import argparse
13
+ import os
14
+ import sys
15
+
16
+ import drep
17
+ from drep.controller import Controller
18
+
19
+
20
+ def version():
21
+ versionFile = open(os.path.join(drep.__path__[0], 'VERSION'))
22
+ return versionFile.read().strip()
23
+
24
+
25
+ VERSION = version()
26
+
27
+ """
28
+ ########################################
29
+ # Argument Parsing #
30
+ ########################################
31
+ """
32
+
33
+
34
+ class SmartFormatter(argparse.ArgumentDefaultsHelpFormatter):
35
+ def _split_lines(self, text, width):
36
+ if text.startswith('R|'):
37
+ return text[2:].splitlines()
38
+ # this is the RawTextHelpFormatter._split_lines
39
+ return argparse.HelpFormatter._split_lines(self, text, width)
40
+
41
+
42
+ def printHelp():
43
+ print('')
44
+ print(' ...::: dRep v' + VERSION + ' :::...''')
45
+ print('''\
46
+
47
+ Matt Olm. MIT License. Banfield Lab, UC Berkeley. 2017 (last updated 2026)
48
+
49
+ See https://drep.readthedocs.io/en/latest/index.html for documentation
50
+ Choose one of the operations below for more detailed help.
51
+
52
+ Example: dRep dereplicate -h
53
+
54
+ Commands:
55
+ compare -> Compare and cluster a set of genomes
56
+ dereplicate -> De-replicate a set of genomes
57
+ check_dependencies -> Check which dependencies are properly installed
58
+ ''')
59
+
60
+
61
+ def parse_args(args):
62
+ parser = argparse.ArgumentParser(formatter_class=SmartFormatter)
63
+ subparsers = parser.add_subparsers(help='Desired operation', dest='operation')
64
+
65
+ # Make a parent parser for all of the subparsers
66
+ parent_parser = argparse.ArgumentParser(add_help=False)
67
+ parent_parser.add_argument("work_directory", help="R|Directory where data and output are stored\
68
+ \n*** USE THE SAME WORK DIRECTORY FOR ALL DREP OPERATIONS ***")
69
+
70
+ Bflags = parent_parser.add_argument_group('SYSTEM PARAMETERS')
71
+ Bflags.add_argument('-p', '--processors', help='threads', default=6, type=int)
72
+ Bflags.add_argument('-d', '--debug', help='make extra debugging output', default=False,
73
+ action="store_true")
74
+ Bflags.add_argument("-h", "--help", action="help", help="show this help message and exit")
75
+
76
+ # Make a parent parser for genome input
77
+ genome_parser = argparse.ArgumentParser(add_help=False)
78
+ Gflags = genome_parser.add_argument_group('GENOME INPUT')
79
+ Gflags.add_argument('-g', '--genomes', nargs='*', help='genomes to filter in .fasta format.\
80
+ Not necessary if Bdb or Wdb already exist. Can also input a text file with paths to genomes, which results in fewer OS issues than wildcard expansion')
81
+
82
+ # Make a parent parser for filter operation
83
+ filtering_parent = argparse.ArgumentParser(add_help=False)
84
+ fiflags = filtering_parent.add_argument_group('GENOME FILTERING OPTIONS')
85
+ fiflags.add_argument("-l", "--length", help="Minimum genome length", default=50000,
86
+ type=float)
87
+ fiflags.add_argument("-comp", "--completeness", help="Minimum genome completeness",
88
+ default=75, type=float)
89
+ fiflags.add_argument("-con", "--contamination", help="Maximum genome contamination",
90
+ default=25, type=float)
91
+
92
+ quality_parent = argparse.ArgumentParser(add_help=False)
93
+ Iflags = quality_parent.add_argument_group('GENOME QUALITY ASSESSMENT OPTIONS')
94
+ Iflags.add_argument("--ignoreGenomeQuality", help="Don't run checkM or do any \
95
+ quality filtering. NOT RECOMMENDED! This is useful for use with bacteriophages\
96
+ or eukaryotes or things where checkM scoring does not work. Will only \
97
+ choose genomes based on length and N50", action='store_true')
98
+ Iflags.add_argument('--genomeInfo', help='location of .csv or .tsv file containing quality \
99
+ information on the genomes (the delimiter is detected automatically). Must contain: ["genome"(filename of .fasta file \
100
+ of that genome, including extension e.g. genome.fasta), "completeness"(0-100 value for completeness of the genome), \
101
+ "contamination"(0-100 value of the contamination of the genome)]. Raw CheckM2 and CheckM1 \
102
+ output can also be provided directly')
103
+ Iflags.add_argument("--checkM_method", help="Either lineage_wf (more accurate) \
104
+ or taxonomy_wf (faster)", choices={'taxonomy_wf', 'lineage_wf'}, \
105
+ default='lineage_wf')
106
+ Iflags.add_argument("--set_recursion", help="Increases the python recursion limit. " \
107
+ + "NOT RECOMMENDED unless checkM is crashing due to recursion issues. " \
108
+ + "Recommended to set to 2000 if needed, but setting this could crash python", \
109
+ default='0')
110
+ Iflags.add_argument("--checkm_group_size", help="The number of genomes passed to checkM at a time. Increasing this increases RAM but makes checkM faster", \
111
+ default=2000, type=int)
112
+
113
+ # Make a parent parser for the cluster operation
114
+ cluster_parent = argparse.ArgumentParser(add_help=False)
115
+
116
+ Clustflags = cluster_parent.add_argument_group('GENOME COMPARISON OPTIONS')
117
+ Clustflags.add_argument("--S_algorithm", help="R|Algorithm for secondary clustering comaprisons:\n" \
118
+ + "skani = (DEFAULT) Kmer-based approach; fastest and most accurate.\n" \
119
+ + " When paired with --primary_algorithm skani, secondary reuses\n" \
120
+ + " the comparisons already done during primary clustering.\n" \
121
+ + "fastANI = Kmer-based approach; very fast\n" \
122
+ + "ANImf = Align whole genomes with nucmer; filter alignment; compare aligned regions\n" \
123
+ + "ANIn = Align whole genomes with nucmer; compare aligned regions\n" \
124
+ + "gANI = Identify and align ORFs; compare aligned ORFS\n" \
125
+ + "goANI = Open source version of gANI; requires nsmimscan\n",
126
+ default='skani', choices={'ANIn', 'gANI', 'ANImf', 'goANI', 'fastANI', 'skani'})
127
+ Clustflags.add_argument("--primary_algorithm", help="R|Program to use for primary clustering.\n" \
128
+ + "skani = (DEFAULT) skani triangle --sparse. Only above-threshold pairs\n" \
129
+ + " are produced, so there is no N^2 matrix in RAM or on disk, and\n" \
130
+ + " a skani --S_algorithm can reuse these comparisons instead of\n" \
131
+ + " recomputing them.\n" \
132
+ + "MASH = all-vs-all Mash. Pre-v4 behavior; builds the full N^2 table.",
133
+ default='skani', choices={'MASH', 'skani'})
134
+ Clustflags.add_argument("--no_reuse_primary_comparisons", dest='reuse_primary_comparisons',
135
+ help="Re-run skani during secondary clustering instead of reusing the "
136
+ "comparisons already computed during primary clustering. Only "
137
+ "relevant with --primary_algorithm skani and a skani --S_algorithm, "
138
+ "where the two stages otherwise compute the same ANI values twice. "
139
+ "Reuse is exact, so this is mostly a debugging escape hatch.",
140
+ action='store_false', default=True)
141
+ Clustflags.add_argument("--primary_skani_min_af",
142
+ help="Minimum percent of a genome that must align for a pair to form a "
143
+ "primary-clustering edge (--primary_algorithm skani only). skani's ANI "
144
+ "is measured within aligned regions only, so without this filter genomes "
145
+ "sharing just a small conserved region become edges and single linkage "
146
+ "chains them into one huge cluster. The default reproduces the MASH "
147
+ "partition closely; lower it only if you have very fragmented genomes "
148
+ "and understand the chaining risk.",
149
+ default=15, type=float)
150
+ Clustflags.add_argument("-ms", "--MASH_sketch", help="MASH sketch size", default=1000)
151
+ Clustflags.add_argument("--SkipMash", help="Skip primary clustering entirely and run secondary\
152
+ clustering on all genomes at once. (Named for when primary clustering was\
153
+ always MASH; it applies to whichever --primary_algorithm is in use.)",
154
+ action='store_true')
155
+ Clustflags.add_argument("--SkipSecondary", help="Skip secondary clustering, just perform MASH\
156
+ clustering", action='store_true')
157
+ Clustflags.add_argument("--skani_extra",
158
+ help="Extra arguments to pass to skani triangle",
159
+ default="", type=str)
160
+ Clustflags.add_argument("--n_PRESET", help="R|Presets to pass to nucmer\n" \
161
+ + "tight = only align highly conserved regions\n" \
162
+ + "normal = default ANIn parameters", choices=['normal', 'tight'],
163
+ default='normal')
164
+
165
+ Compflags = cluster_parent.add_argument_group('GENOME CLUSTERING OPTIONS')
166
+ Compflags.add_argument("-pa", "--P_ani", help="ANI threshold to form primary (MASH) clusters",
167
+ default=0.9, type=float)
168
+ Compflags.add_argument("-sa", "--S_ani", help="ANI threshold to form secondary clusters",
169
+ default=0.95, type=float)
170
+ Compflags.add_argument("-nc", "--cov_thresh", help="Minmum level of overlap between\
171
+ genomes when doing secondary comparisons", default=0.1, type=float)
172
+ Compflags.add_argument("-cm", "--coverage_method", help="R|Method to calculate coverage of an alignment\n" \
173
+ + "(for ANIn/ANImf only; gANI and fastANI can only do larger method)\n"
174
+ + "total = 2*(aligned length) / (sum of total genome lengths)\n" \
175
+ + "larger = max((aligned length / genome 1), (aligned_length / genome2))\n",
176
+ choices=['total', 'larger'], default='larger')
177
+ Compflags.add_argument("--clusterAlg", help="Algorithm used to cluster genomes during SECONDARY\
178
+ clustering (passed to scipy.cluster.hierarchy.linkage)", default='average',
179
+ choices={'single', 'complete', 'average', 'weighted', 'centroid', 'median', 'ward'})
180
+ Compflags.add_argument("--primary_clusterAlg", help="R|Algorithm used to cluster genomes during PRIMARY\n" \
181
+ "(MASH/skani) clustering. The default 'single' is equivalent to connected\n" \
182
+ "components and is computed with a fast, low-memory streaming algorithm that\n" \
183
+ "scales to very large genome sets. Any other choice falls back to the classic\n" \
184
+ "dense scipy path (see --classic_primary_clustering).", default='single',
185
+ choices={'single', 'complete', 'average', 'weighted', 'centroid', 'median', 'ward'})
186
+ Compflags.add_argument("--classic_primary_clustering", help="Force the classic dense (scipy) primary\
187
+ clustering path instead of the streaming single-linkage algorithm. Uses much more\
188
+ RAM at scale but reproduces pre-v4 behavior and allows non-single linkage methods\
189
+ and the primary dendrogram plot.",
190
+ action='store_true', default=False)
191
+
192
+ GRflags = cluster_parent.add_argument_group('GREEDY CLUSTERING OPTIONS\n'
193
+ 'These decrease RAM use and runtime at the expense of a minor loss in '
194
+ 'accuracy.\nRecommended when clustering 5000+ genomes')
195
+ GRflags.add_argument("--multiround_primary_clustering",
196
+ help='Cluster each primary clunk separately and '
197
+ 'merge at the end with single linkage. Decreases '
198
+ 'RAM usage and increases speed, and the cost of a minor '
199
+ 'loss in precision and the inability to plot '
200
+ 'primary_clustering_dendrograms. Especially helpful '
201
+ 'when clustering 5000+ genomes. Will be done with '
202
+ 'single linkage clustering',
203
+ action='store_true')
204
+ GRflags.add_argument("--primary_chunksize",
205
+ help="Impacts multiround_primary_clustering. If you have more than this many genomes, "
206
+ "process them in chunks of this size.",
207
+ default=5000, type=int)
208
+ GRflags.add_argument("--greedy_secondary_clustering",
209
+ help="Use a heuristic to avoid pair-wise comparisons when doing secondary clustering. Will "
210
+ "be done with single linkage clustering. Only works for fastANI S_algorithm option at "
211
+ "the moment",
212
+ action='store_true')
213
+ GRflags.add_argument("--run_tertiary_clustering",
214
+ help="Run an additional round of clustering on the final genome set. This is especially "
215
+ "useful when greedy clustering is performed and/or to handle cases where similar genomes "
216
+ "end up in different primary clusters. Only works with dereplicate, not compare.",
217
+ action='store_true')
218
+
219
+ # Make a parent parser for scoring
220
+ scoring_parent = argparse.ArgumentParser(add_help=False)
221
+ Sflags = scoring_parent.add_argument_group(
222
+ "SCORING CRITERIA\nBased off of the formula: \nA*Completeness - B*Contamination + C*(Contamination * ("
223
+ "strain_heterogeneity/100)) + D*log(N50) + E*log(size) + F*(centrality - S_ani)\n\nA = completeness_weight; B = "
224
+ "contamination_weight; C = strain_heterogeneity_weight; D = N50_weight; E = size_weight; F = cent_weight")
225
+
226
+ Sflags.add_argument("-comW", "--completeness_weight", default=1, type=float,
227
+ help='completeness weight')
228
+ Sflags.add_argument("-conW", "--contamination_weight", default=5, type=float,
229
+ help='contamination weight')
230
+ Sflags.add_argument("-strW", "--strain_heterogeneity_weight", default=1, type=float,
231
+ help='strain heterogeneity weight')
232
+ Sflags.add_argument("-N50W", "--N50_weight", default=0.5, type=float,
233
+ help='weight of log(genome N50)')
234
+ Sflags.add_argument("-sizeW", "--size_weight", default=0, type=float,
235
+ help='weight of log(genome size)')
236
+ Sflags.add_argument("-centW", "--centrality_weight", default=1, type=float,
237
+ help='Weight of (centrality - S_ani)')
238
+ Sflags.add_argument("-extraW", "--extra_weight_table", default=None, type=str,
239
+ help='Path to a tab-separated file with two-columns, no headers, listing genome and extra score to apply to that genome')
240
+
241
+ # Make a parent parser for evaluate
242
+ evaluate_parent = argparse.ArgumentParser(add_help=False)
243
+ Fflags = evaluate_parent.add_argument_group('WARNINGS')
244
+ Fflags.add_argument("--gen_warnings", default=False, help="Generate warnings", action='store_true')
245
+ Fflags.add_argument("--warn_dist", default=0.25, help="How far from the threshold " + \
246
+ " to throw cluster warnings")
247
+ Fflags.add_argument("--warn_sim", default=0.98, help="Similarity threshold for " + \
248
+ " warnings between dereplicated genomes")
249
+ Fflags.add_argument("--warn_aln", default=0.25, help="Minimum aligned fraction for " + \
250
+ " warnings between dereplicated genomes (ANIn)")
251
+
252
+ # Make a parent parser for analyze
253
+ analyze_parent = argparse.ArgumentParser(add_help=False)
254
+ Aflags = analyze_parent.add_argument_group('ANALYZE')
255
+ Aflags.add_argument("--skip_plots", default=False, help="Dont make plots", action='store_true')
256
+
257
+ dereplicate_parser = subparsers.add_parser("dereplicate", formatter_class=SmartFormatter, \
258
+ parents=[parent_parser, genome_parser, filtering_parent, quality_parent,
259
+ cluster_parent, scoring_parent, \
260
+ evaluate_parent, analyze_parent], add_help=False, epilog= \
261
+ "Example: dRep dereplicate output_dir/ -g /path/to/genomes/*.fasta")
262
+
263
+ dereplicate_parser = subparsers.add_parser("compare", formatter_class=SmartFormatter, \
264
+ parents=[parent_parser, genome_parser, cluster_parent, evaluate_parent], \
265
+ add_help=False, epilog= \
266
+ "Example: dRep compare output_dir/ -g /path/to/genomes/*.fasta")
267
+
268
+ dep_parser = subparsers.add_parser("check_dependencies", formatter_class=SmartFormatter)
269
+
270
+ '''
271
+ ####### PARSE THE ARGUMENTS ######
272
+ '''
273
+
274
+ # Handle the situation where the user wants the raw help
275
+ if len(args) == 0 or args[0] == '-h' or args[0] == '--help':
276
+ printHelp()
277
+ sys.exit(0)
278
+ else:
279
+ return parser.parse_args(args)
drep/controller.py ADDED
@@ -0,0 +1,105 @@
1
+ #!/usr/bin/env python3
2
+
3
+ ###############################################################################
4
+ #
5
+ # dRep - main program entry point
6
+ #
7
+ ###############################################################################
8
+
9
+ '''
10
+ Controller- takes input from argparse and calls correct modules
11
+ '''
12
+
13
+
14
+ __author__ = "Matt Olm"
15
+ __license__ = "MIT"
16
+ __email__ = "mattolm@gmail.com"
17
+ __status__ = "Development"
18
+
19
+ import argparse
20
+ import logging
21
+ import os
22
+ import sys
23
+
24
+ import drep
25
+ import drep.d_cluster.controller
26
+ from drep.WorkDirectory import WorkDirectory
27
+ import drep.d_cluster
28
+ import drep.d_analyze
29
+ import drep.d_filter
30
+ import drep.d_choose
31
+ import drep.d_adjust
32
+ import drep.d_bonus
33
+ import drep.d_evaluate
34
+ import drep.d_workflows
35
+
36
+ def version():
37
+ versionFile = open(os.path.join(drep.__path__[0], 'VERSION'))
38
+ return versionFile.read().strip()
39
+
40
+ VERSION = version()
41
+
42
+ class Controller():
43
+ def __init__(self):
44
+ self.logger = logging.getLogger()
45
+
46
+ def dereplicate_operation(self, **kwargs):
47
+ logging.debug("Starting the dereplicate operation")
48
+ drep.d_workflows.dereplicate_wrapper(kwargs['work_directory'],**kwargs)
49
+ logging.debug("Finished the dereplicate operation!")
50
+
51
+ def compare_operation(self, **kwargs):
52
+ logging.debug("Starting the compare operation")
53
+ drep.d_workflows.compare_wrapper(kwargs['work_directory'],**kwargs)
54
+ logging.debug("!!! Finished the compare operation !!!")
55
+
56
+ def setup_logger(self,loc):
57
+ ''' set up logger such that DEBUG goes only to file, rest go to file and console '''
58
+
59
+ # set up logging everything to file
60
+ logging.basicConfig(level=logging.DEBUG,
61
+ format='%(asctime)s %(levelname)-8s %(message)s',
62
+ datefmt='%m-%d %H:%M',
63
+ filename=loc)
64
+
65
+ # set up logging of INFO or higher to sys.stderr
66
+ console = logging.StreamHandler()
67
+ console.setLevel(logging.INFO)
68
+ formatter = logging.Formatter('%(message)s')
69
+ console.setFormatter(formatter)
70
+ logging.getLogger('').addHandler(console)
71
+
72
+ logging.debug("!"*80)
73
+ logging.debug("***Logger started up at {0}***".format(loc))
74
+ logging.debug("Command to run dRep was: {0}\n".format(' '.join(sys.argv)))
75
+ logging.debug("dRep version {0} was run \n".format(VERSION))
76
+ logging.debug("!"*80 + '\n')
77
+
78
+ def parseArguments(self, args):
79
+ ''' Parse user options and call the correct pipeline'''
80
+ if args.operation == 'check_dependencies':
81
+ drep.d_bonus.check_dependencies(print_out=True)
82
+ return
83
+
84
+ # Load the workDirectory
85
+ wd_loc = str(os.path.abspath(args.work_directory))
86
+ wd = WorkDirectory(wd_loc)
87
+
88
+ # Set up the logger
89
+ self.setup_logger(wd.get_loc('log'))
90
+ logging.debug(str(args))
91
+
92
+ # Do some testing
93
+ if args.run_tertiary_clustering:
94
+ if args.operation != "dereplicate":
95
+ raise ValueError("Can only run tertiary clustering with dereplicate")
96
+
97
+
98
+ # Call the appropriate workflow
99
+ if args.operation == "dereplicate":
100
+ self.dereplicate_operation(**vars(args))
101
+ if args.operation == "compare":
102
+ self.compare_operation(**vars(args))
103
+
104
+ def loadDefaultArgs(self):
105
+ pass
drep/d_adjust.py ADDED
@@ -0,0 +1,272 @@
1
+ #!/usr/bin/env python3
2
+
3
+ import logging
4
+ import glob
5
+ import pandas as pd
6
+ import os
7
+ import shutil
8
+ import numpy as np
9
+ import pickle
10
+ import sys
11
+
12
+ import drep.WorkDirectory
13
+ import drep as dm
14
+ import drep.d_cluster as dClust
15
+ import drep.d_cluster.cluster_utils
16
+ import drep.d_cluster.compare_utils
17
+ import drep.d_cluster.controller
18
+ import drep.d_filter as d_filter
19
+ import drep.d_choose as dChoose
20
+
21
+ def d_adjust_wrapper(wd,**kwargs):
22
+ # Load the WorkDirectory.
23
+ logging.info("Loading work directory")
24
+ wd = drep.WorkDirectory.WorkDirectory(wd)
25
+ logging.debug(str(wd))
26
+
27
+ if kwargs.get('cluster') != None:
28
+ logging.info('adjusting cluster {0}'.format(kwargs.get('cluster')))
29
+ adjust_cluster_wrapper(wd, **kwargs)
30
+
31
+ elif kwargs.get('rm_cluster') != None:
32
+ logging.info('removing cluster(s) {0}'.format(' '.join(kwargs.get('rm_cluster'))))
33
+ remove_cluster_wrapper(wd, **kwargs)
34
+
35
+ def remove_cluster_wrapper(wd, **kwargs):
36
+ # Get information
37
+ Cdb = wd.get_db('Cdb')
38
+ Wdb = wd.get_db('Wdb')
39
+
40
+ for cluster in kwargs.get('rm_cluster'):
41
+ type = cluster_type(cluster)
42
+
43
+ if type == 'primary_cluster':
44
+ # Make sure cluster is in Cdb
45
+ if int(cluster) not in Cdb[type].tolist():
46
+ logging.error("{0} is not in Cdb- quitting".format(cluster))
47
+ sys.exit()
48
+
49
+ # Make sure cluster is Wdb
50
+ W_Pclusters = [x.split('_')[0] for x in Wdb['cluster'].tolist()]
51
+ if cluster not in W_Pclusters:
52
+ logging.error("{0} is not in Wdb- quitting".format(cluster))
53
+ sys.exit()
54
+
55
+ # Remove the cluster
56
+ remove_primary_cluster(cluster, wd, **kwargs)
57
+
58
+ elif type == 'secondary_cluster':
59
+
60
+ # Make sure cluster is in Cdb
61
+ if cluster not in Cdb[type].tolist():
62
+ logging.error("{0} is not in Cdb- quitting".format(cluster))
63
+ sys.exit()
64
+
65
+ # Make sure cluster is in Wdb
66
+ if cluster not in Wdb['cluster'].tolist():
67
+ logging.error("{0} is not in Wdb- quitting".format(cluster))
68
+ sys.exit()
69
+
70
+ # Remove the cluster
71
+ remove_secondary_cluster(cluster, wd, **kwargs)
72
+
73
+
74
+ def remove_primary_cluster(Rcluster, wd, **kwargs):
75
+ # Get things to alter
76
+ Cdb = wd.get_db('Cdb')
77
+ Wdb = wd.get_db('Wdb')
78
+
79
+ # If this is a singleton, just remove the associated secondary cluster
80
+ if len(Cdb['genome'][Cdb['primary_cluster'] == int(Rcluster)].unique()) == 1:
81
+ Rsecondary = "{0}_0".format(Rcluster)
82
+ logging.error("{0} is a primary singleton- will remove {1} instead".format(Rcluster, Rsecondary))
83
+ remove_secondary_cluster(Rsecondary, wd, **kwargs)
84
+ return
85
+
86
+ # Find the genome(s) to remove
87
+ Wdb['pclust'] = [x.split('_')[0] for x in Wdb['cluster'].tolist()]
88
+ Rgenomes = Wdb['genome'][Wdb['pclust'] == Rcluster].tolist()
89
+ for Rgenome in Rgenomes:
90
+ if not os.path.isfile('{0}/dereplicated_genomes/{1}'.format(wd.location, Rgenome)):
91
+ raise ValueError()
92
+
93
+ # Find the pickle to remove
94
+ Rpickle = "{0}/data/Clustering_files/secondary_linkage_cluster_{1}.pickle".format(\
95
+ wd.location, Rcluster)
96
+ if not os.path.isfile(Rpickle):
97
+ raise ValueError()
98
+
99
+ logging.info("will remove cluster {0}, genomes {1}, and pickle {2}".format(Rcluster, \
100
+ ' '.join(Rgenomes), os.path.basename(Rpickle)))
101
+
102
+ # Remove cluster from Wdb
103
+ newWdb = Wdb[Wdb['pclust'] != Rcluster]
104
+ del newWdb['pclust']
105
+
106
+ # Remove cluster from Cdb
107
+ newCdb = Cdb[Cdb['primary_cluster'] != int(Rcluster)]
108
+
109
+ # Remove genomes
110
+ for Rgenome in Rgenomes:
111
+ os.remove('{0}/dereplicated_genomes/{1}'.format(wd.location, Rgenome))
112
+
113
+ # Remove pickle
114
+ os.remove(Rpickle)
115
+
116
+ # Save removed stuff
117
+ wd.store_db(newCdb,'Cdb',overwrite=True)
118
+ wd.store_db(newWdb,'Wdb',overwrite=True)
119
+
120
+ logging.info("done")
121
+
122
+
123
+ def remove_secondary_cluster(Rcluster, wd, **kwargs):
124
+ # Get things to alter
125
+ Cdb = wd.get_db('Cdb')
126
+ Wdb = wd.get_db('Wdb')
127
+
128
+ # Find the genome to remove
129
+ Rgenome = Wdb['genome'][Wdb['cluster'] == Rcluster].tolist()[0]
130
+ if not os.path.isfile('{0}/dereplicated_genomes/{1}'.format(wd.location, Rgenome)):
131
+ raise ValueError('Bug')
132
+
133
+ logging.info("will remove {0} and genome {1}".format(Rcluster, Rgenome))
134
+
135
+ # Remove cluster from Wdb
136
+ newWdb = Wdb[Wdb['cluster'] != Rcluster]
137
+
138
+ # Remove cluster from Cdb
139
+ newCdb = Cdb[Cdb['secondary_cluster'] != Rcluster]
140
+
141
+ # Remove genome
142
+ os.remove('{0}/dereplicated_genomes/{1}'.format(wd.location, Rgenome))
143
+
144
+ # Save removed stuff
145
+ wd.store_db(newCdb,'Cdb',overwrite=True)
146
+ wd.store_db(newWdb,'Wdb',overwrite=True)
147
+
148
+ logging.info("done")
149
+
150
+ def cluster_type(cluster):
151
+ if '_' in cluster:
152
+ return 'secondary_cluster'
153
+ else:
154
+ return 'primary_cluster'
155
+
156
+ def adjust_cluster_wrapper(wd, **kwargs):
157
+ # Validate arguments
158
+ cluster = kwargs.get('cluster')
159
+ if cluster == None:
160
+ logging.error("Must specify a cluster")
161
+ sys.exit()
162
+ comp_method = kwargs.get('clustering_method')
163
+ clust_method = kwargs.get('clusterAlg')
164
+ threshold = kwargs.pop('threshold',None)
165
+ cov_thresh = float(kwargs.get('minimum_coverage'))
166
+ if threshold != None: threshold = 1- float(threshold)
167
+
168
+ # Make a bdb listing the genomes to cluster
169
+ Cdb = wd.get_db('Cdb')
170
+ Bdb = wd.get_db('Bdb')
171
+ genomes = Cdb['genome'][Cdb['primary_cluster'] == int(cluster)].tolist()
172
+ bdb = Bdb[Bdb['genome'].isin(genomes)]
173
+
174
+ # Make the comparison database
175
+ Xdb = drep.d_cluster.compare_utils.compare_genomes(bdb, comp_method, wd, **kwargs)
176
+
177
+ # Remove values without enough coverage
178
+ Xdb.loc[Xdb['alignment_coverage'] <= cov_thresh, 'ani'] = 0
179
+
180
+ # Make it symmetrical
181
+ Xdb['av_ani'] = Xdb.apply(lambda row: dClust.average_ani (row,Xdb),axis=1)
182
+ Xdb['dist'] = 1 - Xdb['av_ani']
183
+ db = Xdb.pivot(index="reference", columns="querry", values="dist")
184
+
185
+ # Cluster it
186
+ if threshold == None:
187
+ threshold = float(0)
188
+ cdb, linkage = drep.d_cluster.cluster_utils.cluster_hierarchical(db, linkage_method = clust_method, \
189
+ linkage_cutoff = threshold)
190
+
191
+ # Save the pickle
192
+ data_folder = wd.location + '/data/Clustering_files/'
193
+ arguments = {'linkage_method':clust_method,'linkage_cutoff':threshold,\
194
+ 'comparison_algorithm':comp_method,'minimum_coverage':cov_thresh}
195
+ pickle_name = "secondary_linkage_cluster_{0}.pickle".format(cluster)
196
+ logging.debug('Saving secondary_linkage pickle {1} to {0}'.format(data_folder,\
197
+ pickle_name))
198
+ with open(data_folder + pickle_name, 'wb') as handle:
199
+ pickle.dump(linkage, handle)
200
+ pickle.dump(db,handle)
201
+ pickle.dump(arguments,handle)
202
+
203
+ # Change the cdb
204
+ cdb['cluster_method'] = clust_method
205
+ cdb['comparison_algorithm'] = comp_method
206
+ cdb['threshold'] = str(threshold)
207
+ cdb['primary_cluster'] = int(cluster)
208
+ cdb['secondary_cluster'] = ["{0}_{1}".format(cluster, x) for x in cdb['cluster']]
209
+ del cdb['cluster']
210
+
211
+ # Incorporate it into the existing Cdb
212
+ Cdb = Cdb[Cdb['primary_cluster'] != int(cluster)]
213
+ Cdb = pd.concat([Cdb,cdb], ignore_index=True)
214
+
215
+ # Save it
216
+ wd.store_db(Cdb,'Cdb',overwrite=True)
217
+
218
+ if (not kwargs.get('skip_winner',False)) and wd.hasDb('Wdb'):
219
+
220
+ # Remake Wdb based on Sdb
221
+ oriWdb = wd.get_db('Wdb')
222
+ newWdb = dChoose.pick_winners(wd.get_db('Sdb'),Cdb)
223
+ change = accounce_changes(newWdb, oriWdb)
224
+ wd.store_db(newWdb,'Wdb',overwrite=True)
225
+
226
+ if change:
227
+ # Change the ./dereplicated genomes thing
228
+ logging.info("Remaking ./dereplicated genomes...")
229
+ output_folder = wd.location + '/dereplicated_genomes/'
230
+ dm.clobber_dir(output_folder,dry=kwargs.get('dry',False),\
231
+ overwrite=True)
232
+
233
+ for genome in newWdb['genome'].unique():
234
+ loc = Bdb['location'][Bdb['genome'] == genome].tolist()[0]
235
+ shutil.copy2(loc, "{0}{1}".format(output_folder,genome))
236
+
237
+ def accounce_changes(newWdb, oriWdb):
238
+ changes = False
239
+
240
+ for cluster in newWdb['cluster'].unique():
241
+ w1 = newWdb['genome'][newWdb['cluster'] == cluster].unique()[0]
242
+ if cluster not in oriWdb['cluster'].tolist():
243
+ change = "{0} is the winner of the new cluster {1}".format(w1,cluster)
244
+ logging.info(change)
245
+ changes = True
246
+ continue
247
+ w2 = oriWdb['genome'][oriWdb['cluster'] == cluster].unique()[0]
248
+ if w1 != w2:
249
+ change = "the winner of cluster {0} is now {1} (was {2})".format(\
250
+ cluster, w1, w2)
251
+ logging.info(change)
252
+ changes = True
253
+ continue
254
+
255
+ for cluster in oriWdb['cluster'].unique():
256
+ if cluster not in newWdb['cluster'].tolist():
257
+ change = "cluster {0} no longer exists".format(cluster)
258
+ logging.info(change)
259
+ changes = True
260
+
261
+ if not changes:
262
+ change = "No secondary clusters were made or destroyed, and none have new winners"
263
+ logging.info(change)
264
+
265
+ return changes
266
+
267
+
268
+ def test_adjust():
269
+ print("Write this you lazy bum")
270
+
271
+ if __name__ == '__main__':
272
+ test_adjust()