drep 4.0.0__tar.gz → 4.0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {drep-4.0.0 → drep-4.0.2}/PKG-INFO +8 -2
- {drep-4.0.0 → drep-4.0.2}/README.md +1 -1
- drep-4.0.2/drep/VERSION +1 -0
- {drep-4.0.0 → drep-4.0.2}/drep/argumentParser.py +4 -3
- {drep-4.0.0 → drep-4.0.2}/drep/d_choose.py +1 -1
- {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/external.py +25 -11
- {drep-4.0.0 → drep-4.0.2}/drep/d_filter.py +138 -6
- {drep-4.0.0 → drep-4.0.2}/drep.egg-info/PKG-INFO +8 -2
- drep-4.0.0/drep/VERSION +0 -1
- {drep-4.0.0 → drep-4.0.2}/bin/dRep +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/WorkDirectory.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/__init__.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/controller.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/d_adjust.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/d_analyze.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/d_bonus.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/__init__.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/cluster_utils.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/compare_utils.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/controller.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/greedy_clustering.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/parsers.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/union_find.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/d_cluster/utils.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/d_evaluate.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep/d_workflows.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep.egg-info/SOURCES.txt +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep.egg-info/dependency_links.txt +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep.egg-info/not-zip-safe +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep.egg-info/requires.txt +0 -0
- {drep-4.0.0 → drep-4.0.2}/drep.egg-info/top_level.txt +0 -0
- {drep-4.0.0 → drep-4.0.2}/helper_scripts/ScaffoldLevel_dRep.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/helper_scripts/parse_stb.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/pyproject.toml +0 -0
- {drep-4.0.0 → drep-4.0.2}/setup.cfg +0 -0
- {drep-4.0.0 → drep-4.0.2}/setup.py +0 -0
- {drep-4.0.0 → drep-4.0.2}/tests/test_suite.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: drep
|
|
3
|
-
Version: 4.0.
|
|
3
|
+
Version: 4.0.2
|
|
4
4
|
Summary: De-replication of microbial genomes assembled from multiple samples
|
|
5
5
|
Home-page: https://github.com/MrOlm/drep
|
|
6
6
|
Author: Matt Olm
|
|
@@ -15,3 +15,9 @@ Requires-Dist: scikit-learn
|
|
|
15
15
|
Requires-Dist: tqdm
|
|
16
16
|
Requires-Dist: setuptools
|
|
17
17
|
Requires-Dist: pytest
|
|
18
|
+
Dynamic: author
|
|
19
|
+
Dynamic: author-email
|
|
20
|
+
Dynamic: home-page
|
|
21
|
+
Dynamic: license
|
|
22
|
+
Dynamic: requires-dist
|
|
23
|
+
Dynamic: summary
|
|
@@ -52,7 +52,7 @@ $ dRep check_dependencies
|
|
|
52
52
|
## Dependencies
|
|
53
53
|
### Near Essential
|
|
54
54
|
* [skani](https://github.com/bluenote-1577/skani) - Makes primary clusters and performs the default secondary comparison (v0.2+ confirmed works)
|
|
55
|
-
* [CheckM](http://ecogenomics.github.io/CheckM/) - Determines contamination and completeness of genomes (v1.0.7 confirmed works). Only needed for `dereplicate`; skip it with `--genomeInfo` or `--ignoreGenomeQuality`
|
|
55
|
+
* [CheckM](http://ecogenomics.github.io/CheckM/) - Determines contamination and completeness of genomes (v1.0.7 confirmed works). Only needed for `dereplicate`; skip it with `--genomeInfo` (raw CheckM2 or CheckM1 output can be passed directly) or `--ignoreGenomeQuality`
|
|
56
56
|
|
|
57
57
|
### Optional
|
|
58
58
|
|
drep-4.0.2/drep/VERSION
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
4.0.2
|
|
@@ -95,10 +95,11 @@ def parse_args(args):
|
|
|
95
95
|
quality filtering. NOT RECOMMENDED! This is useful for use with bacteriophages\
|
|
96
96
|
or eukaryotes or things where checkM scoring does not work. Will only \
|
|
97
97
|
choose genomes based on length and N50", action='store_true')
|
|
98
|
-
Iflags.add_argument('--genomeInfo', help='location of .csv file containing quality \
|
|
99
|
-
information on the genomes. Must contain: ["genome"(filename of .fasta file \
|
|
98
|
+
Iflags.add_argument('--genomeInfo', help='location of .csv or .tsv file containing quality \
|
|
99
|
+
information on the genomes (the delimiter is detected automatically). Must contain: ["genome"(filename of .fasta file \
|
|
100
100
|
of that genome, including extension e.g. genome.fasta), "completeness"(0-100 value for completeness of the genome), \
|
|
101
|
-
"contamination"(0-100 value of the contamination of the genome)]
|
|
101
|
+
"contamination"(0-100 value of the contamination of the genome)]. Raw CheckM2 and CheckM1 \
|
|
102
|
+
output can also be provided directly')
|
|
102
103
|
Iflags.add_argument("--checkM_method", help="Either lineage_wf (more accurate) \
|
|
103
104
|
or taxonomy_wf (faster)", choices={'taxonomy_wf', 'lineage_wf'}, \
|
|
104
105
|
default='lineage_wf')
|
|
@@ -37,7 +37,7 @@ def d_choose_wrapper(wd, **kwargs):
|
|
|
37
37
|
**kwargs: Command line arguments
|
|
38
38
|
|
|
39
39
|
Keyword args:
|
|
40
|
-
genomeInfo: .csv genomeInfo file
|
|
40
|
+
genomeInfo: .csv or .tsv genomeInfo file
|
|
41
41
|
ignoreGenomeQuality: Don't run checkM or do any quality-based filtering (not recommended)
|
|
42
42
|
checkM_method: Either lineage_wf (more accurate) or taxonomy_wf (faster)
|
|
43
43
|
|
|
@@ -717,17 +717,31 @@ def add_avani(db):
|
|
|
717
717
|
db: dataframe
|
|
718
718
|
'''
|
|
719
719
|
|
|
720
|
-
|
|
721
|
-
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
|
|
728
|
-
|
|
729
|
-
|
|
730
|
-
|
|
720
|
+
# Vectorized: this used to iterrows() into a string-keyed dict, which took
|
|
721
|
+
# hours and hundreds of GB once a single primary cluster held tens of
|
|
722
|
+
# thousands of genomes (issue #308). Each (reference, querry) pair is encoded
|
|
723
|
+
# as one integer so its reverse can be looked up in bulk.
|
|
724
|
+
logging.debug('averaging reciprocal ANI values')
|
|
725
|
+
ref = db['reference'].values
|
|
726
|
+
qry = db['querry'].values
|
|
727
|
+
names = pd.Index(pd.unique(np.concatenate([ref, qry])))
|
|
728
|
+
n = len(names)
|
|
729
|
+
r = names.get_indexer(ref).astype(np.int64)
|
|
730
|
+
q = names.get_indexer(qry).astype(np.int64)
|
|
731
|
+
|
|
732
|
+
# When a pair appears more than once the last value wins, as before
|
|
733
|
+
ani = pd.Series(db['ani'].values, index=r * n + q)
|
|
734
|
+
ani = ani[~ani.index.duplicated(keep='last')]
|
|
735
|
+
forward = ani.reindex(r * n + q).values
|
|
736
|
+
reverse = ani.reindex(q * n + r).values
|
|
737
|
+
|
|
738
|
+
same = r == q
|
|
739
|
+
missing = ~np.isin(q * n + r, ani.index.values) & ~same
|
|
740
|
+
if missing.any():
|
|
741
|
+
i = np.argmax(missing)
|
|
742
|
+
raise KeyError("{0}-vs-{1}".format(ref[i], qry[i]))
|
|
743
|
+
|
|
744
|
+
db['av_ani'] = np.where(same, 1, (forward + reverse) / 2)
|
|
731
745
|
|
|
732
746
|
logging.debug('averageing done')
|
|
733
747
|
|
|
@@ -28,7 +28,7 @@ def d_filter_wrapper(wd, **kwargs):
|
|
|
28
28
|
|
|
29
29
|
Keyword Args:
|
|
30
30
|
genomes: genomes to filter in .fasta format
|
|
31
|
-
genomeInfo: location of .csv file with the columns: ["genome"(basename of .fasta file of that genome), "completeness"(0-100 value for completeness of the genome), "contamination"(0-100 value of the contamination of the genome)]
|
|
31
|
+
genomeInfo: location of .csv or .tsv file with the columns: ["genome"(basename of .fasta file of that genome), "completeness"(0-100 value for completeness of the genome), "contamination"(0-100 value of the contamination of the genome)]
|
|
32
32
|
|
|
33
33
|
processors: Threads to use with checkM / prodigal
|
|
34
34
|
overwrite: Overwrite existing data in the work folder
|
|
@@ -127,6 +127,118 @@ def sanity_check(bdb, **kwargs):
|
|
|
127
127
|
logging.info(f"Hey! You're running multiround_primary_clustering but not run_tertiary_clustering! You should add --run_tertiary_clustering when running with multiround_primary_clustering to avoid weird placement; see https://drep.readthedocs.io/en/latest/choosing_parameters.html#using-greedy-algorithms for more info")
|
|
128
128
|
|
|
129
129
|
|
|
130
|
+
GENOMEINFO_REQUIRED_COLUMNS = ['genome', 'completeness', 'contamination']
|
|
131
|
+
GENOMEINFO_DELIMITERS = [(',', 'comma'), ('\t', 'tab'), (';', 'semicolon'), ('|', 'pipe')]
|
|
132
|
+
|
|
133
|
+
# Columns that dRep knows what to do with in a genomeInfo table
|
|
134
|
+
GENOMEINFO_KNOWN_COLUMNS = ['genome', 'completeness', 'contamination',
|
|
135
|
+
'strain_heterogeneity', 'length', 'N50',
|
|
136
|
+
'location', 'centrality']
|
|
137
|
+
|
|
138
|
+
# Raw CheckM2 (quality_report.tsv) and CheckM1 (--tab_table) column names
|
|
139
|
+
GENOMEINFO_COLUMN_TRANSLATION = {
|
|
140
|
+
'Name': 'genome', # CheckM2
|
|
141
|
+
'Bin Id': 'genome', # CheckM1
|
|
142
|
+
'Completeness': 'completeness', # both
|
|
143
|
+
'Contamination': 'contamination', # both
|
|
144
|
+
'Strain heterogeneity': 'strain_heterogeneity', # CheckM1
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
# CheckM2 run with --general / --specific / --allmodels reports these instead
|
|
148
|
+
# of a plain "Completeness" column; the first one present is used
|
|
149
|
+
GENOMEINFO_COMPLETENESS_FALLBACKS = ['Completeness_General', 'Completeness_Specific']
|
|
150
|
+
|
|
151
|
+
def _genomeInfo_translation(columns):
|
|
152
|
+
'''
|
|
153
|
+
Figure out how to rename raw CheckM1 / CheckM2 columns into genomeInfo columns
|
|
154
|
+
|
|
155
|
+
Raw CheckM2 output is a perfectly good genomeInfo file except that its
|
|
156
|
+
columns are named "Name", "Completeness", and "Contamination", so translate
|
|
157
|
+
it rather than making the user do it by hand (issue #305). Columns that are
|
|
158
|
+
already there under the dRep name are never overwritten
|
|
159
|
+
|
|
160
|
+
Args:
|
|
161
|
+
columns: the columns of the table that was loaded
|
|
162
|
+
|
|
163
|
+
Returns:
|
|
164
|
+
dict: {current column name: dRep column name}
|
|
165
|
+
'''
|
|
166
|
+
columns = list(columns)
|
|
167
|
+
translation = {c: n for c, n in GENOMEINFO_COLUMN_TRANSLATION.items()
|
|
168
|
+
if (c in columns) and (n not in columns)}
|
|
169
|
+
|
|
170
|
+
if ('completeness' not in columns) and ('Completeness' not in columns):
|
|
171
|
+
for c in GENOMEINFO_COMPLETENESS_FALLBACKS:
|
|
172
|
+
if c in columns:
|
|
173
|
+
translation[c] = 'completeness'
|
|
174
|
+
break
|
|
175
|
+
|
|
176
|
+
# Never rename two columns into the same one; first in the file wins
|
|
177
|
+
deduped = {}
|
|
178
|
+
for c in columns:
|
|
179
|
+
n = translation.get(c, None)
|
|
180
|
+
if (n is not None) and (n not in deduped.values()):
|
|
181
|
+
deduped[c] = n
|
|
182
|
+
|
|
183
|
+
return deduped
|
|
184
|
+
|
|
185
|
+
def load_genomeInfo(location):
|
|
186
|
+
'''
|
|
187
|
+
Load a user-provided genomeInfo file, figuring out the format as you go
|
|
188
|
+
|
|
189
|
+
pandas happily parses a .tsv with sep=',' (you just get one mangled column),
|
|
190
|
+
so the delimiter can't be picked based on which read_csv call fails to raise
|
|
191
|
+
an exception. Instead try each delimiter and keep the one that actually
|
|
192
|
+
yields the columns dRep needs (issue #305). Raw CheckM1 / CheckM2 output is
|
|
193
|
+
then translated into genomeInfo columns
|
|
194
|
+
|
|
195
|
+
Args:
|
|
196
|
+
location: location of the genomeInfo file
|
|
197
|
+
|
|
198
|
+
Returns:
|
|
199
|
+
DataFrame: genomeInfo
|
|
200
|
+
'''
|
|
201
|
+
best = None
|
|
202
|
+
for sep, sep_name in GENOMEINFO_DELIMITERS:
|
|
203
|
+
try:
|
|
204
|
+
db = pd.read_csv(location, sep=sep)
|
|
205
|
+
except Exception:
|
|
206
|
+
continue
|
|
207
|
+
translation = _genomeInfo_translation(db.columns)
|
|
208
|
+
columns = [translation.get(c, c) for c in db.columns]
|
|
209
|
+
found = len([c for c in GENOMEINFO_REQUIRED_COLUMNS if c in columns])
|
|
210
|
+
score = (found, len(db.columns))
|
|
211
|
+
if (best is None) or (score > best[0]):
|
|
212
|
+
best = (score, sep_name, db, translation)
|
|
213
|
+
|
|
214
|
+
if best is None:
|
|
215
|
+
raise Exception("Cannot parse the genomeInfo file {0}; it must be a "
|
|
216
|
+
"comma- or tab-delimited table".format(location))
|
|
217
|
+
|
|
218
|
+
(found, _), sep_name, Idb, translation = best
|
|
219
|
+
logging.debug("Parsed genomeInfo file {0} as {1}-delimited; the columns are "
|
|
220
|
+
"{2}".format(location, sep_name, list(Idb.columns)))
|
|
221
|
+
|
|
222
|
+
# Translate raw CheckM1 / CheckM2 output
|
|
223
|
+
if len(translation) > 0:
|
|
224
|
+
logging.info("Translating the columns of the provided genomeInfo: {0}".format(
|
|
225
|
+
', '.join(['{0} -> {1}'.format(c, n) for c, n in translation.items()])))
|
|
226
|
+
for c in GENOMEINFO_COMPLETENESS_FALLBACKS:
|
|
227
|
+
if translation.get(c, None) == 'completeness':
|
|
228
|
+
logging.warning('This CheckM2 output has no "Completeness" column, '
|
|
229
|
+
'so "{0}" is being used as completeness'.format(c))
|
|
230
|
+
Idb = Idb.rename(columns=translation)
|
|
231
|
+
Idb = Idb[[c for c in GENOMEINFO_KNOWN_COLUMNS if c in Idb.columns]]
|
|
232
|
+
|
|
233
|
+
if found < len(GENOMEINFO_REQUIRED_COLUMNS):
|
|
234
|
+
logging.warning("Could not find the columns {0} in the genomeInfo file "
|
|
235
|
+
"{1}; the best guess of its format is {2}-delimited "
|
|
236
|
+
"with the columns {3}".format(
|
|
237
|
+
GENOMEINFO_REQUIRED_COLUMNS, location, sep_name,
|
|
238
|
+
list(Idb.columns)))
|
|
239
|
+
|
|
240
|
+
return Idb
|
|
241
|
+
|
|
130
242
|
def _get_run_genomeInfo(workDirectory, bdb, **kwargs):
|
|
131
243
|
'''
|
|
132
244
|
Through kwargs and the wd, get genomeInfo
|
|
@@ -145,10 +257,7 @@ def _get_run_genomeInfo(workDirectory, bdb, **kwargs):
|
|
|
145
257
|
|
|
146
258
|
if kwargs.get('genomeInfo', None) != None:
|
|
147
259
|
logging.debug("Loading provided genome quality information")
|
|
148
|
-
|
|
149
|
-
Idb = pd.read_csv(kwargs.get('genomeInfo'))
|
|
150
|
-
except:
|
|
151
|
-
Idb = pd.read_csv(kwargs.get('genomeInfo'), sep='\t')
|
|
260
|
+
Idb = load_genomeInfo(kwargs.get('genomeInfo'))
|
|
152
261
|
try:
|
|
153
262
|
Tdb = _validate_genomeInfo(Idb, bdb)
|
|
154
263
|
except IndexError:
|
|
@@ -212,7 +321,8 @@ def _validate_genomeInfo(Idb, bdb):
|
|
|
212
321
|
# Make sure it has required columns
|
|
213
322
|
for r in ['completeness', 'contamination', 'genome']:
|
|
214
323
|
if r not in Idb.columns:
|
|
215
|
-
raise KeyError("{0} missing from GenomeInfo"
|
|
324
|
+
raise KeyError("{0} missing from GenomeInfo; the columns that are "
|
|
325
|
+
"there are {1}".format(r, list(Idb.columns)))
|
|
216
326
|
|
|
217
327
|
# Make sure correct datatypes
|
|
218
328
|
for r in ['completeness', 'contamination', 'strain_heterogeneity']:
|
|
@@ -228,6 +338,28 @@ def _validate_genomeInfo(Idb, bdb):
|
|
|
228
338
|
Idb['genome'] = [os.path.basename(x) for x in Idb['location']]
|
|
229
339
|
break
|
|
230
340
|
|
|
341
|
+
# See if the file extension was stripped (this is what CheckM2 does)
|
|
342
|
+
have = set(Idb['genome'].astype(str))
|
|
343
|
+
missing = [g for g in bdb['genome'].unique() if g not in have]
|
|
344
|
+
if len(missing) > 0:
|
|
345
|
+
matches = {}
|
|
346
|
+
for genome in missing:
|
|
347
|
+
base = os.path.splitext(genome)[0]
|
|
348
|
+
if base in have:
|
|
349
|
+
matches.setdefault(base, []).append(genome)
|
|
350
|
+
|
|
351
|
+
# Only correct names that map back to a single genome
|
|
352
|
+
ambiguous = {b: g for b, g in matches.items() if len(g) > 1}
|
|
353
|
+
if len(ambiguous) > 0:
|
|
354
|
+
logging.warning("Provided genome info is missing the file extension on "
|
|
355
|
+
"genomes that differ only by extension, so it can't be "
|
|
356
|
+
"corrected: {0}".format(sorted(ambiguous.keys())))
|
|
357
|
+
translation = {b: g[0] for b, g in matches.items() if len(g) == 1}
|
|
358
|
+
if len(translation) > 0:
|
|
359
|
+
logging.warning("Provided genome info is missing the file extension on "
|
|
360
|
+
"{0} genomes- correcting".format(len(translation)))
|
|
361
|
+
Idb['genome'] = [translation.get(g, g) for g in Idb['genome'].astype(str)]
|
|
362
|
+
|
|
231
363
|
# Make sure it matchs up with bdb
|
|
232
364
|
for genome in list(bdb['genome'].unique()):
|
|
233
365
|
if genome not in Idb['genome'].tolist():
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: drep
|
|
3
|
-
Version: 4.0.
|
|
3
|
+
Version: 4.0.2
|
|
4
4
|
Summary: De-replication of microbial genomes assembled from multiple samples
|
|
5
5
|
Home-page: https://github.com/MrOlm/drep
|
|
6
6
|
Author: Matt Olm
|
|
@@ -15,3 +15,9 @@ Requires-Dist: scikit-learn
|
|
|
15
15
|
Requires-Dist: tqdm
|
|
16
16
|
Requires-Dist: setuptools
|
|
17
17
|
Requires-Dist: pytest
|
|
18
|
+
Dynamic: author
|
|
19
|
+
Dynamic: author-email
|
|
20
|
+
Dynamic: home-page
|
|
21
|
+
Dynamic: license
|
|
22
|
+
Dynamic: requires-dist
|
|
23
|
+
Dynamic: summary
|
drep-4.0.0/drep/VERSION
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
4.0.0
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|