drep 4.0.0__tar.gz → 4.0.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {drep-4.0.0 → drep-4.0.2}/PKG-INFO +8 -2
  2. {drep-4.0.0 → drep-4.0.2}/README.md +1 -1
  3. drep-4.0.2/drep/VERSION +1 -0
  4. {drep-4.0.0 → drep-4.0.2}/drep/argumentParser.py +4 -3
  5. {drep-4.0.0 → drep-4.0.2}/drep/d_choose.py +1 -1
  6. {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/external.py +25 -11
  7. {drep-4.0.0 → drep-4.0.2}/drep/d_filter.py +138 -6
  8. {drep-4.0.0 → drep-4.0.2}/drep.egg-info/PKG-INFO +8 -2
  9. drep-4.0.0/drep/VERSION +0 -1
  10. {drep-4.0.0 → drep-4.0.2}/bin/dRep +0 -0
  11. {drep-4.0.0 → drep-4.0.2}/drep/WorkDirectory.py +0 -0
  12. {drep-4.0.0 → drep-4.0.2}/drep/__init__.py +0 -0
  13. {drep-4.0.0 → drep-4.0.2}/drep/controller.py +0 -0
  14. {drep-4.0.0 → drep-4.0.2}/drep/d_adjust.py +0 -0
  15. {drep-4.0.0 → drep-4.0.2}/drep/d_analyze.py +0 -0
  16. {drep-4.0.0 → drep-4.0.2}/drep/d_bonus.py +0 -0
  17. {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/__init__.py +0 -0
  18. {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/cluster_utils.py +0 -0
  19. {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/compare_utils.py +0 -0
  20. {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/controller.py +0 -0
  21. {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/greedy_clustering.py +0 -0
  22. {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/parsers.py +0 -0
  23. {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/union_find.py +0 -0
  24. {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/utils.py +0 -0
  25. {drep-4.0.0 → drep-4.0.2}/drep/d_evaluate.py +0 -0
  26. {drep-4.0.0 → drep-4.0.2}/drep/d_workflows.py +0 -0
  27. {drep-4.0.0 → drep-4.0.2}/drep.egg-info/SOURCES.txt +0 -0
  28. {drep-4.0.0 → drep-4.0.2}/drep.egg-info/dependency_links.txt +0 -0
  29. {drep-4.0.0 → drep-4.0.2}/drep.egg-info/not-zip-safe +0 -0
  30. {drep-4.0.0 → drep-4.0.2}/drep.egg-info/requires.txt +0 -0
  31. {drep-4.0.0 → drep-4.0.2}/drep.egg-info/top_level.txt +0 -0
  32. {drep-4.0.0 → drep-4.0.2}/helper_scripts/ScaffoldLevel_dRep.py +0 -0
  33. {drep-4.0.0 → drep-4.0.2}/helper_scripts/parse_stb.py +0 -0
  34. {drep-4.0.0 → drep-4.0.2}/pyproject.toml +0 -0
  35. {drep-4.0.0 → drep-4.0.2}/setup.cfg +0 -0
  36. {drep-4.0.0 → drep-4.0.2}/setup.py +0 -0
  37. {drep-4.0.0 → drep-4.0.2}/tests/test_suite.py +0 -0
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.1
1
+ Metadata-Version: 2.4
2
2
  Name: drep
3
- Version: 4.0.0
3
+ Version: 4.0.2
4
4
  Summary: De-replication of microbial genomes assembled from multiple samples
5
5
  Home-page: https://github.com/MrOlm/drep
6
6
  Author: Matt Olm
@@ -15,3 +15,9 @@ Requires-Dist: scikit-learn
15
15
  Requires-Dist: tqdm
16
16
  Requires-Dist: setuptools
17
17
  Requires-Dist: pytest
18
+ Dynamic: author
19
+ Dynamic: author-email
20
+ Dynamic: home-page
21
+ Dynamic: license
22
+ Dynamic: requires-dist
23
+ Dynamic: summary
@@ -52,7 +52,7 @@ $ dRep check_dependencies
52
52
  ## Dependencies
53
53
  ### Near Essential
54
54
  * [skani](https://github.com/bluenote-1577/skani) - Makes primary clusters and performs the default secondary comparison (v0.2+ confirmed works)
55
- * [CheckM](http://ecogenomics.github.io/CheckM/) - Determines contamination and completeness of genomes (v1.0.7 confirmed works). Only needed for `dereplicate`; skip it with `--genomeInfo` or `--ignoreGenomeQuality`
55
+ * [CheckM](http://ecogenomics.github.io/CheckM/) - Determines contamination and completeness of genomes (v1.0.7 confirmed works). Only needed for `dereplicate`; skip it with `--genomeInfo` (raw CheckM2 or CheckM1 output can be passed directly) or `--ignoreGenomeQuality`
56
56
 
57
57
  ### Optional
58
58
 
@@ -0,0 +1 @@
1
+ 4.0.2
@@ -95,10 +95,11 @@ def parse_args(args):
95
95
  quality filtering. NOT RECOMMENDED! This is useful for use with bacteriophages\
96
96
  or eukaryotes or things where checkM scoring does not work. Will only \
97
97
  choose genomes based on length and N50", action='store_true')
98
- Iflags.add_argument('--genomeInfo', help='location of .csv file containing quality \
99
- information on the genomes. Must contain: ["genome"(filename of .fasta file \
98
+ Iflags.add_argument('--genomeInfo', help='location of .csv or .tsv file containing quality \
99
+ information on the genomes (the delimiter is detected automatically). Must contain: ["genome"(filename of .fasta file \
100
100
  of that genome, including extension e.g. genome.fasta), "completeness"(0-100 value for completeness of the genome), \
101
- "contamination"(0-100 value of the contamination of the genome)]')
101
+ "contamination"(0-100 value of the contamination of the genome)]. Raw CheckM2 and CheckM1 \
102
+ output can also be provided directly')
102
103
  Iflags.add_argument("--checkM_method", help="Either lineage_wf (more accurate) \
103
104
  or taxonomy_wf (faster)", choices={'taxonomy_wf', 'lineage_wf'}, \
104
105
  default='lineage_wf')
@@ -37,7 +37,7 @@ def d_choose_wrapper(wd, **kwargs):
37
37
  **kwargs: Command line arguments
38
38
 
39
39
  Keyword args:
40
- genomeInfo: .csv genomeInfo file
40
+ genomeInfo: .csv or .tsv genomeInfo file
41
41
  ignoreGenomeQuality: Don't run checkM or do any quality-based filtering (not recommended)
42
42
  checkM_method: Either lineage_wf (more accurate) or taxonomy_wf (faster)
43
43
 
@@ -717,17 +717,31 @@ def add_avani(db):
717
717
  db: dataframe
718
718
  '''
719
719
 
720
- logging.debug('making dictionary for average_ani')
721
- combo2value = {}
722
- for i, row in db.iterrows():
723
- combo2value["{0}-vs-{1}".format(row['querry'], row['reference'])] \
724
- = row['ani']
725
-
726
- logging.debug('list comprehension for average_ani')
727
- db['av_ani'] = [np.mean([combo2value["{0}-vs-{1}".format(q, r)],
728
- combo2value["{0}-vs-{1}".format(r, q)]]) if r != q else 1\
729
- for q, r in zip(db['querry'].tolist(),
730
- db['reference'].tolist())]
720
+ # Vectorized: this used to iterrows() into a string-keyed dict, which took
721
+ # hours and hundreds of GB once a single primary cluster held tens of
722
+ # thousands of genomes (issue #308). Each (reference, querry) pair is encoded
723
+ # as one integer so its reverse can be looked up in bulk.
724
+ logging.debug('averaging reciprocal ANI values')
725
+ ref = db['reference'].values
726
+ qry = db['querry'].values
727
+ names = pd.Index(pd.unique(np.concatenate([ref, qry])))
728
+ n = len(names)
729
+ r = names.get_indexer(ref).astype(np.int64)
730
+ q = names.get_indexer(qry).astype(np.int64)
731
+
732
+ # When a pair appears more than once the last value wins, as before
733
+ ani = pd.Series(db['ani'].values, index=r * n + q)
734
+ ani = ani[~ani.index.duplicated(keep='last')]
735
+ forward = ani.reindex(r * n + q).values
736
+ reverse = ani.reindex(q * n + r).values
737
+
738
+ same = r == q
739
+ missing = ~np.isin(q * n + r, ani.index.values) & ~same
740
+ if missing.any():
741
+ i = np.argmax(missing)
742
+ raise KeyError("{0}-vs-{1}".format(ref[i], qry[i]))
743
+
744
+ db['av_ani'] = np.where(same, 1, (forward + reverse) / 2)
731
745
 
732
746
  logging.debug('averageing done')
733
747
 
@@ -28,7 +28,7 @@ def d_filter_wrapper(wd, **kwargs):
28
28
 
29
29
  Keyword Args:
30
30
  genomes: genomes to filter in .fasta format
31
- genomeInfo: location of .csv file with the columns: ["genome"(basename of .fasta file of that genome), "completeness"(0-100 value for completeness of the genome), "contamination"(0-100 value of the contamination of the genome)]
31
+ genomeInfo: location of .csv or .tsv file with the columns: ["genome"(basename of .fasta file of that genome), "completeness"(0-100 value for completeness of the genome), "contamination"(0-100 value of the contamination of the genome)]
32
32
 
33
33
  processors: Threads to use with checkM / prodigal
34
34
  overwrite: Overwrite existing data in the work folder
@@ -127,6 +127,118 @@ def sanity_check(bdb, **kwargs):
127
127
  logging.info(f"Hey! You're running multiround_primary_clustering but not run_tertiary_clustering! You should add --run_tertiary_clustering when running with multiround_primary_clustering to avoid weird placement; see https://drep.readthedocs.io/en/latest/choosing_parameters.html#using-greedy-algorithms for more info")
128
128
 
129
129
 
130
+ GENOMEINFO_REQUIRED_COLUMNS = ['genome', 'completeness', 'contamination']
131
+ GENOMEINFO_DELIMITERS = [(',', 'comma'), ('\t', 'tab'), (';', 'semicolon'), ('|', 'pipe')]
132
+
133
+ # Columns that dRep knows what to do with in a genomeInfo table
134
+ GENOMEINFO_KNOWN_COLUMNS = ['genome', 'completeness', 'contamination',
135
+ 'strain_heterogeneity', 'length', 'N50',
136
+ 'location', 'centrality']
137
+
138
+ # Raw CheckM2 (quality_report.tsv) and CheckM1 (--tab_table) column names
139
+ GENOMEINFO_COLUMN_TRANSLATION = {
140
+ 'Name': 'genome', # CheckM2
141
+ 'Bin Id': 'genome', # CheckM1
142
+ 'Completeness': 'completeness', # both
143
+ 'Contamination': 'contamination', # both
144
+ 'Strain heterogeneity': 'strain_heterogeneity', # CheckM1
145
+ }
146
+
147
+ # CheckM2 run with --general / --specific / --allmodels reports these instead
148
+ # of a plain "Completeness" column; the first one present is used
149
+ GENOMEINFO_COMPLETENESS_FALLBACKS = ['Completeness_General', 'Completeness_Specific']
150
+
151
+ def _genomeInfo_translation(columns):
152
+ '''
153
+ Figure out how to rename raw CheckM1 / CheckM2 columns into genomeInfo columns
154
+
155
+ Raw CheckM2 output is a perfectly good genomeInfo file except that its
156
+ columns are named "Name", "Completeness", and "Contamination", so translate
157
+ it rather than making the user do it by hand (issue #305). Columns that are
158
+ already there under the dRep name are never overwritten
159
+
160
+ Args:
161
+ columns: the columns of the table that was loaded
162
+
163
+ Returns:
164
+ dict: {current column name: dRep column name}
165
+ '''
166
+ columns = list(columns)
167
+ translation = {c: n for c, n in GENOMEINFO_COLUMN_TRANSLATION.items()
168
+ if (c in columns) and (n not in columns)}
169
+
170
+ if ('completeness' not in columns) and ('Completeness' not in columns):
171
+ for c in GENOMEINFO_COMPLETENESS_FALLBACKS:
172
+ if c in columns:
173
+ translation[c] = 'completeness'
174
+ break
175
+
176
+ # Never rename two columns into the same one; first in the file wins
177
+ deduped = {}
178
+ for c in columns:
179
+ n = translation.get(c, None)
180
+ if (n is not None) and (n not in deduped.values()):
181
+ deduped[c] = n
182
+
183
+ return deduped
184
+
185
+ def load_genomeInfo(location):
186
+ '''
187
+ Load a user-provided genomeInfo file, figuring out the format as you go
188
+
189
+ pandas happily parses a .tsv with sep=',' (you just get one mangled column),
190
+ so the delimiter can't be picked based on which read_csv call fails to raise
191
+ an exception. Instead try each delimiter and keep the one that actually
192
+ yields the columns dRep needs (issue #305). Raw CheckM1 / CheckM2 output is
193
+ then translated into genomeInfo columns
194
+
195
+ Args:
196
+ location: location of the genomeInfo file
197
+
198
+ Returns:
199
+ DataFrame: genomeInfo
200
+ '''
201
+ best = None
202
+ for sep, sep_name in GENOMEINFO_DELIMITERS:
203
+ try:
204
+ db = pd.read_csv(location, sep=sep)
205
+ except Exception:
206
+ continue
207
+ translation = _genomeInfo_translation(db.columns)
208
+ columns = [translation.get(c, c) for c in db.columns]
209
+ found = len([c for c in GENOMEINFO_REQUIRED_COLUMNS if c in columns])
210
+ score = (found, len(db.columns))
211
+ if (best is None) or (score > best[0]):
212
+ best = (score, sep_name, db, translation)
213
+
214
+ if best is None:
215
+ raise Exception("Cannot parse the genomeInfo file {0}; it must be a "
216
+ "comma- or tab-delimited table".format(location))
217
+
218
+ (found, _), sep_name, Idb, translation = best
219
+ logging.debug("Parsed genomeInfo file {0} as {1}-delimited; the columns are "
220
+ "{2}".format(location, sep_name, list(Idb.columns)))
221
+
222
+ # Translate raw CheckM1 / CheckM2 output
223
+ if len(translation) > 0:
224
+ logging.info("Translating the columns of the provided genomeInfo: {0}".format(
225
+ ', '.join(['{0} -> {1}'.format(c, n) for c, n in translation.items()])))
226
+ for c in GENOMEINFO_COMPLETENESS_FALLBACKS:
227
+ if translation.get(c, None) == 'completeness':
228
+ logging.warning('This CheckM2 output has no "Completeness" column, '
229
+ 'so "{0}" is being used as completeness'.format(c))
230
+ Idb = Idb.rename(columns=translation)
231
+ Idb = Idb[[c for c in GENOMEINFO_KNOWN_COLUMNS if c in Idb.columns]]
232
+
233
+ if found < len(GENOMEINFO_REQUIRED_COLUMNS):
234
+ logging.warning("Could not find the columns {0} in the genomeInfo file "
235
+ "{1}; the best guess of its format is {2}-delimited "
236
+ "with the columns {3}".format(
237
+ GENOMEINFO_REQUIRED_COLUMNS, location, sep_name,
238
+ list(Idb.columns)))
239
+
240
+ return Idb
241
+
130
242
  def _get_run_genomeInfo(workDirectory, bdb, **kwargs):
131
243
  '''
132
244
  Through kwargs and the wd, get genomeInfo
@@ -145,10 +257,7 @@ def _get_run_genomeInfo(workDirectory, bdb, **kwargs):
145
257
 
146
258
  if kwargs.get('genomeInfo', None) != None:
147
259
  logging.debug("Loading provided genome quality information")
148
- try:
149
- Idb = pd.read_csv(kwargs.get('genomeInfo'))
150
- except:
151
- Idb = pd.read_csv(kwargs.get('genomeInfo'), sep='\t')
260
+ Idb = load_genomeInfo(kwargs.get('genomeInfo'))
152
261
  try:
153
262
  Tdb = _validate_genomeInfo(Idb, bdb)
154
263
  except IndexError:
@@ -212,7 +321,8 @@ def _validate_genomeInfo(Idb, bdb):
212
321
  # Make sure it has required columns
213
322
  for r in ['completeness', 'contamination', 'genome']:
214
323
  if r not in Idb.columns:
215
- raise KeyError("{0} missing from GenomeInfo".format(r))
324
+ raise KeyError("{0} missing from GenomeInfo; the columns that are "
325
+ "there are {1}".format(r, list(Idb.columns)))
216
326
 
217
327
  # Make sure correct datatypes
218
328
  for r in ['completeness', 'contamination', 'strain_heterogeneity']:
@@ -228,6 +338,28 @@ def _validate_genomeInfo(Idb, bdb):
228
338
  Idb['genome'] = [os.path.basename(x) for x in Idb['location']]
229
339
  break
230
340
 
341
+ # See if the file extension was stripped (this is what CheckM2 does)
342
+ have = set(Idb['genome'].astype(str))
343
+ missing = [g for g in bdb['genome'].unique() if g not in have]
344
+ if len(missing) > 0:
345
+ matches = {}
346
+ for genome in missing:
347
+ base = os.path.splitext(genome)[0]
348
+ if base in have:
349
+ matches.setdefault(base, []).append(genome)
350
+
351
+ # Only correct names that map back to a single genome
352
+ ambiguous = {b: g for b, g in matches.items() if len(g) > 1}
353
+ if len(ambiguous) > 0:
354
+ logging.warning("Provided genome info is missing the file extension on "
355
+ "genomes that differ only by extension, so it can't be "
356
+ "corrected: {0}".format(sorted(ambiguous.keys())))
357
+ translation = {b: g[0] for b, g in matches.items() if len(g) == 1}
358
+ if len(translation) > 0:
359
+ logging.warning("Provided genome info is missing the file extension on "
360
+ "{0} genomes- correcting".format(len(translation)))
361
+ Idb['genome'] = [translation.get(g, g) for g in Idb['genome'].astype(str)]
362
+
231
363
  # Make sure it matchs up with bdb
232
364
  for genome in list(bdb['genome'].unique()):
233
365
  if genome not in Idb['genome'].tolist():
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.1
1
+ Metadata-Version: 2.4
2
2
  Name: drep
3
- Version: 4.0.0
3
+ Version: 4.0.2
4
4
  Summary: De-replication of microbial genomes assembled from multiple samples
5
5
  Home-page: https://github.com/MrOlm/drep
6
6
  Author: Matt Olm
@@ -15,3 +15,9 @@ Requires-Dist: scikit-learn
15
15
  Requires-Dist: tqdm
16
16
  Requires-Dist: setuptools
17
17
  Requires-Dist: pytest
18
+ Dynamic: author
19
+ Dynamic: author-email
20
+ Dynamic: home-page
21
+ Dynamic: license
22
+ Dynamic: requires-dist
23
+ Dynamic: summary
drep-4.0.0/drep/VERSION DELETED
@@ -1 +0,0 @@
1
- 4.0.0
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes