diverse-seq 2024.8.26a5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diverse_seq-2024.8.26a5/LICENSE +29 -0
- diverse_seq-2024.8.26a5/PKG-INFO +357 -0
- diverse_seq-2024.8.26a5/README.md +297 -0
- diverse_seq-2024.8.26a5/pyproject.toml +180 -0
- diverse_seq-2024.8.26a5/src/diverse_seq/__init__.py +7 -0
- diverse_seq-2024.8.26a5/src/diverse_seq/cli.py +392 -0
- diverse_seq-2024.8.26a5/src/diverse_seq/data_store.py +237 -0
- diverse_seq-2024.8.26a5/src/diverse_seq/distance.py +28 -0
- diverse_seq-2024.8.26a5/src/diverse_seq/io.py +176 -0
- diverse_seq-2024.8.26a5/src/diverse_seq/record.py +518 -0
- diverse_seq-2024.8.26a5/src/diverse_seq/records.py +742 -0
- diverse_seq-2024.8.26a5/src/diverse_seq/util.py +164 -0
- diverse_seq-2024.8.26a5/tests/conftest.py +13 -0
- diverse_seq-2024.8.26a5/tests/data/brca1.dvseqs +0 -0
- diverse_seq-2024.8.26a5/tests/data/brca1.fasta +2623 -0
- diverse_seq-2024.8.26a5/tests/test_cli.py +233 -0
- diverse_seq-2024.8.26a5/tests/test_data_store.py +90 -0
- diverse_seq-2024.8.26a5/tests/test_distance.py +29 -0
- diverse_seq-2024.8.26a5/tests/test_record.py +428 -0
- diverse_seq-2024.8.26a5/tests/test_records.py +192 -0
- diverse_seq-2024.8.26a5/tests/test_util.py +84 -0
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
BSD 3-Clause License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2022, GavinHuttley
|
|
4
|
+
All rights reserved.
|
|
5
|
+
|
|
6
|
+
Redistribution and use in source and binary forms, with or without
|
|
7
|
+
modification, are permitted provided that the following conditions are met:
|
|
8
|
+
|
|
9
|
+
1. Redistributions of source code must retain the above copyright notice, this
|
|
10
|
+
list of conditions and the following disclaimer.
|
|
11
|
+
|
|
12
|
+
2. Redistributions in binary form must reproduce the above copyright notice,
|
|
13
|
+
this list of conditions and the following disclaimer in the documentation
|
|
14
|
+
and/or other materials provided with the distribution.
|
|
15
|
+
|
|
16
|
+
3. Neither the name of the copyright holder nor the names of its
|
|
17
|
+
contributors may be used to endorse or promote products derived from
|
|
18
|
+
this software without specific prior written permission.
|
|
19
|
+
|
|
20
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
21
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
22
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
23
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
24
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
25
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
26
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
27
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
28
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
29
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
|
@@ -0,0 +1,357 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: diverse_seq
|
|
3
|
+
Version: 2024.8.26a5
|
|
4
|
+
Summary: diverse_seq: a tool for sampling diverse biological sequences
|
|
5
|
+
Keywords: biology,genomics,statistics,phylogeny,evolution,bioinformatics
|
|
6
|
+
Author-email: Gavin Huttley <Gavin.Huttley@anu.edu.au>
|
|
7
|
+
Requires-Python: >=3.10,<3.13
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
Classifier: Development Status :: 3 - Alpha
|
|
10
|
+
Classifier: Intended Audience :: Science/Research
|
|
11
|
+
Classifier: License :: OSI Approved :: BSD License
|
|
12
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
13
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Requires-Dist: attrs
|
|
18
|
+
Requires-Dist: click
|
|
19
|
+
Requires-Dist: cogent3
|
|
20
|
+
Requires-Dist: hdf5plugin
|
|
21
|
+
Requires-Dist: h5py
|
|
22
|
+
Requires-Dist: numpy>=2.0
|
|
23
|
+
Requires-Dist: rich
|
|
24
|
+
Requires-Dist: scitrack
|
|
25
|
+
Requires-Dist: cogapp ; extra == "dev"
|
|
26
|
+
Requires-Dist: docformatter ; extra == "dev"
|
|
27
|
+
Requires-Dist: flit ; extra == "dev"
|
|
28
|
+
Requires-Dist: nox ; extra == "dev"
|
|
29
|
+
Requires-Dist: pytest ; extra == "dev"
|
|
30
|
+
Requires-Dist: pytest-cov ; extra == "dev"
|
|
31
|
+
Requires-Dist: pytest-xdist ; extra == "dev"
|
|
32
|
+
Requires-Dist: ruff==0.6.2 ; extra == "dev"
|
|
33
|
+
Requires-Dist: click ; extra == "doc"
|
|
34
|
+
Requires-Dist: ipykernel ; extra == "doc"
|
|
35
|
+
Requires-Dist: ipython ; extra == "doc"
|
|
36
|
+
Requires-Dist: ipywidgets ; extra == "doc"
|
|
37
|
+
Requires-Dist: jupyter-sphinx ; extra == "doc"
|
|
38
|
+
Requires-Dist: jupyter_client ; extra == "doc"
|
|
39
|
+
Requires-Dist: jupyterlab ; extra == "doc"
|
|
40
|
+
Requires-Dist: jupytext ; extra == "doc"
|
|
41
|
+
Requires-Dist: kaleido ; extra == "doc"
|
|
42
|
+
Requires-Dist: nbconvert>5.4 ; extra == "doc"
|
|
43
|
+
Requires-Dist: nbformat ; extra == "doc"
|
|
44
|
+
Requires-Dist: nbsphinx ; extra == "doc"
|
|
45
|
+
Requires-Dist: numpydoc ; extra == "doc"
|
|
46
|
+
Requires-Dist: pandas ; extra == "doc"
|
|
47
|
+
Requires-Dist: plotly ; extra == "doc"
|
|
48
|
+
Requires-Dist: nox ; extra == "test"
|
|
49
|
+
Requires-Dist: pytest ; extra == "test"
|
|
50
|
+
Requires-Dist: pytest-cov ; extra == "test"
|
|
51
|
+
Requires-Dist: pytest-xdist ; extra == "test"
|
|
52
|
+
Requires-Dist: ruff==0.6.2 ; extra == "test"
|
|
53
|
+
Project-URL: Bug Tracker, https://github.com/HuttleyLab/DiverseSeq/issues
|
|
54
|
+
Project-URL: Documentation, https://github.com/HuttleyLab/DiverseSeq
|
|
55
|
+
Project-URL: Source Code, https://github.com/HuttleyLab/DiverseSeq/
|
|
56
|
+
Provides-Extra: dev
|
|
57
|
+
Provides-Extra: doc
|
|
58
|
+
Provides-Extra: test
|
|
59
|
+
|
|
60
|
+
[](https://github.com/HuttleyLab/DiverseSeq/actions/workflows/ci.yml)
|
|
61
|
+
[](https://coveralls.io/github/HuttleyLab/DiverseSeq?branch=main)
|
|
62
|
+
[](https://app.codacy.com/gh/HuttleyLab/DiverseSeq/dashboard?utm_source=gh&utm_medium=referral&utm_content=&utm_campaign=Badge_grade)
|
|
63
|
+
[](https://github.com/HuttleyLab/DiverseSeq/actions/workflows/codeql.yml)
|
|
64
|
+
[](https://github.com/astral-sh/ruff)
|
|
65
|
+
|
|
66
|
+
# DiverseSeq identifies the most diverse biological sequences from a collection
|
|
67
|
+
|
|
68
|
+
`diverse_seq` provides tools for selecting a representative subset of sequences from a larger collection. It is an alignment-free method which scales linearly with the number of sequences. It identifies the subset of sequences that maximize diversity as measured using Jensen-Shannon divergence. `DiverseSeq` provides a command-line tool (`dvs`) and plugins to the Cogent3 app system (prefixed by `dvs_`) allowing users to embed code in their own scripts. The command-line tools can be run in parallel.
|
|
69
|
+
|
|
70
|
+
## The available commands
|
|
71
|
+
|
|
72
|
+
<!-- [[[cog
|
|
73
|
+
import cog
|
|
74
|
+
from diverse_seq.cli import main
|
|
75
|
+
from click.testing import CliRunner
|
|
76
|
+
runner = CliRunner()
|
|
77
|
+
result = runner.invoke(main, ["--help"])
|
|
78
|
+
help = result.output.replace("Usage: main", "Usage: dvs")
|
|
79
|
+
cog.out(
|
|
80
|
+
"```\n{}\n```".format(help)
|
|
81
|
+
)
|
|
82
|
+
]]] -->
|
|
83
|
+
```
|
|
84
|
+
Usage: dvs [OPTIONS] COMMAND [ARGS]...
|
|
85
|
+
|
|
86
|
+
dvs -- alignment free detection of the most diverse sequences using JSD
|
|
87
|
+
|
|
88
|
+
Options:
|
|
89
|
+
--version Show the version and exit.
|
|
90
|
+
--help Show this message and exit.
|
|
91
|
+
|
|
92
|
+
Commands:
|
|
93
|
+
prep Writes processed sequences to a <HDF5 file>.dvseqs.
|
|
94
|
+
max Identify the seqs that maximise average delta JSD
|
|
95
|
+
nmost Identify n seqs that maximise average delta JSD
|
|
96
|
+
|
|
97
|
+
```
|
|
98
|
+
<!-- [[[end]]] -->
|
|
99
|
+
|
|
100
|
+
### `dvs prep`: Preparing the sequence data
|
|
101
|
+
|
|
102
|
+
Convert sequence data into a more efficient format for the diversity assessment. This must be done before running either the `nmost` or `max` commands.
|
|
103
|
+
|
|
104
|
+
#### Usage:
|
|
105
|
+
|
|
106
|
+
<!-- [[[cog
|
|
107
|
+
import cog
|
|
108
|
+
from diverse_seq.cli import main
|
|
109
|
+
from click.testing import CliRunner
|
|
110
|
+
runner = CliRunner()
|
|
111
|
+
result = runner.invoke(main, ["prep", "--help"])
|
|
112
|
+
help = result.output.replace("Usage: main", "Usage: dvs")
|
|
113
|
+
cog.out(
|
|
114
|
+
"```\n{}\n```".format(help)
|
|
115
|
+
)
|
|
116
|
+
]]] -->
|
|
117
|
+
```
|
|
118
|
+
Usage: dvs prep [OPTIONS]
|
|
119
|
+
|
|
120
|
+
Writes processed sequences to a <HDF5 file>.dvseqs.
|
|
121
|
+
|
|
122
|
+
Options:
|
|
123
|
+
-s, --seqdir PATH directory containing sequence files [required]
|
|
124
|
+
-sf, --suffix TEXT sequence file suffix [default: fa]
|
|
125
|
+
-o, --outpath PATH write processed seqs to this filename [required]
|
|
126
|
+
-np, --numprocs INTEGER number of processes [default: 1]
|
|
127
|
+
-F, --force_overwrite Overwrite existing file if it exists
|
|
128
|
+
-m, --moltype [dna|rna] Molecular type of sequences, defaults to DNA
|
|
129
|
+
[default: dna]
|
|
130
|
+
-L, --limit INTEGER number of sequences to process
|
|
131
|
+
--help Show this message and exit.
|
|
132
|
+
|
|
133
|
+
```
|
|
134
|
+
<!-- [[[end]]] -->
|
|
135
|
+
|
|
136
|
+
### `dvs nmost`: Select the n-most diverse sequences
|
|
137
|
+
|
|
138
|
+
We recommend using `nmost` for large datasets.
|
|
139
|
+
|
|
140
|
+
> **Note**
|
|
141
|
+
> A fuller explanation is coming soon!
|
|
142
|
+
|
|
143
|
+
#### Command line usage:
|
|
144
|
+
|
|
145
|
+
<!-- [[[cog
|
|
146
|
+
import cog
|
|
147
|
+
from diverse_seq.cli import main
|
|
148
|
+
from click.testing import CliRunner
|
|
149
|
+
runner = CliRunner()
|
|
150
|
+
result = runner.invoke(main, ["nmost", "--help"])
|
|
151
|
+
help = result.output.replace("Usage: main", "Usage: dvs")
|
|
152
|
+
cog.out(
|
|
153
|
+
"```\n{}\n```".format(help)
|
|
154
|
+
)
|
|
155
|
+
]]] -->
|
|
156
|
+
```
|
|
157
|
+
Usage: dvs nmost [OPTIONS]
|
|
158
|
+
|
|
159
|
+
Identify n seqs that maximise average delta JSD
|
|
160
|
+
|
|
161
|
+
Options:
|
|
162
|
+
-s, --seqfile PATH path to .dvtgseqs file [required]
|
|
163
|
+
-o, --outpath PATH the input string will be cast to Path instance
|
|
164
|
+
-n, --number INTEGER number of seqs in divergent set [required]
|
|
165
|
+
-k INTEGER k-mer size [default: 6]
|
|
166
|
+
-i, --include TEXT seqnames to include in divergent set
|
|
167
|
+
-np, --numprocs INTEGER number of processes [default: 1]
|
|
168
|
+
-L, --limit INTEGER number of sequences to process
|
|
169
|
+
-v, --verbose is an integer indicating number of cl occurrences
|
|
170
|
+
[default: 0]
|
|
171
|
+
--help Show this message and exit.
|
|
172
|
+
|
|
173
|
+
```
|
|
174
|
+
<!-- [[[end]]] -->
|
|
175
|
+
|
|
176
|
+
#### As a cogent3 plugin:
|
|
177
|
+
|
|
178
|
+
The `dvs_select_nmost` is also available as a [cogent3 app](https://cogent3.org/doc/app/index.html). The result of using `cogent3.app_help("dvs_select_nmost")` is shown below.
|
|
179
|
+
|
|
180
|
+
<!-- [[[cog
|
|
181
|
+
import cog
|
|
182
|
+
import contextlib
|
|
183
|
+
import io
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
from cogent3 import app_help
|
|
187
|
+
|
|
188
|
+
buffer = io.StringIO()
|
|
189
|
+
|
|
190
|
+
with contextlib.redirect_stdout(buffer):
|
|
191
|
+
app_help("dvs_select_nmost")
|
|
192
|
+
cog.out(
|
|
193
|
+
"```\n{}\n```".format(buffer.getvalue())
|
|
194
|
+
)
|
|
195
|
+
]]] -->
|
|
196
|
+
```
|
|
197
|
+
Overview
|
|
198
|
+
--------
|
|
199
|
+
selects the n-most diverse seqs from a sequence collection
|
|
200
|
+
|
|
201
|
+
Options for making the app
|
|
202
|
+
--------------------------
|
|
203
|
+
dvs_select_nmost_app = get_app(
|
|
204
|
+
'dvs_select_nmost',
|
|
205
|
+
n=3,
|
|
206
|
+
moltype='dna',
|
|
207
|
+
include=None,
|
|
208
|
+
k=6,
|
|
209
|
+
seed=None,
|
|
210
|
+
)
|
|
211
|
+
|
|
212
|
+
Parameters
|
|
213
|
+
----------
|
|
214
|
+
n
|
|
215
|
+
the number of divergent sequences
|
|
216
|
+
moltype
|
|
217
|
+
molecular type of the sequences
|
|
218
|
+
k
|
|
219
|
+
k-mer size
|
|
220
|
+
include
|
|
221
|
+
sequence names to include in the final result
|
|
222
|
+
seed
|
|
223
|
+
random number seed
|
|
224
|
+
|
|
225
|
+
Notes
|
|
226
|
+
-----
|
|
227
|
+
If called with an alignment, the ungapped sequences are used.
|
|
228
|
+
The order of the sequences is randomised. If include is not None, the
|
|
229
|
+
named sequences are added to the final result.
|
|
230
|
+
|
|
231
|
+
Input type
|
|
232
|
+
----------
|
|
233
|
+
SequenceCollection, Alignment, ArrayAlignment
|
|
234
|
+
|
|
235
|
+
Output type
|
|
236
|
+
-----------
|
|
237
|
+
SequenceCollection, Alignment, ArrayAlignment
|
|
238
|
+
|
|
239
|
+
```
|
|
240
|
+
<!-- [[[end]]] -->
|
|
241
|
+
|
|
242
|
+
### `dvs max`: Maximise average delta JSD
|
|
243
|
+
|
|
244
|
+
The result of the `max` command is typically a set that are modestly more diverse than that fron `nmost`.
|
|
245
|
+
|
|
246
|
+
> **Note**
|
|
247
|
+
> A fuller explanation is coming soon!
|
|
248
|
+
|
|
249
|
+
#### Command line usage:
|
|
250
|
+
|
|
251
|
+
<!-- [[[cog
|
|
252
|
+
import cog
|
|
253
|
+
from diverse_seq.cli import main
|
|
254
|
+
from click.testing import CliRunner
|
|
255
|
+
runner = CliRunner()
|
|
256
|
+
result = runner.invoke(main, ["max", "--help"])
|
|
257
|
+
help = result.output.replace("Usage: main", "Usage: dvs")
|
|
258
|
+
cog.out(
|
|
259
|
+
"```\n{}\n```".format(help)
|
|
260
|
+
)
|
|
261
|
+
]]] -->
|
|
262
|
+
```
|
|
263
|
+
Usage: dvs max [OPTIONS]
|
|
264
|
+
|
|
265
|
+
Identify the seqs that maximise average delta JSD
|
|
266
|
+
|
|
267
|
+
Options:
|
|
268
|
+
-s, --seqfile PATH path to .dvtgseqs file [required]
|
|
269
|
+
-o, --outpath PATH the input string will be cast to Path instance
|
|
270
|
+
-z, --min_size INTEGER minimum size of divergent set [default: 7]
|
|
271
|
+
-zp, --max_size INTEGER maximum size of divergent set
|
|
272
|
+
-k INTEGER k-mer size [default: 6]
|
|
273
|
+
-st, --stat [stdev|cov] statistic to maximise [default: stdev]
|
|
274
|
+
-i, --include TEXT seqnames to include in divergent set
|
|
275
|
+
-np, --numprocs INTEGER number of processes [default: 1]
|
|
276
|
+
-L, --limit INTEGER number of sequences to process
|
|
277
|
+
-T, --test_run reduce number of paths and size of query seqs
|
|
278
|
+
-v, --verbose is an integer indicating number of cl occurrences
|
|
279
|
+
[default: 0]
|
|
280
|
+
--help Show this message and exit.
|
|
281
|
+
|
|
282
|
+
```
|
|
283
|
+
<!-- [[[end]]] -->
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
#### As a cogent3 plugin:
|
|
287
|
+
|
|
288
|
+
The `dvs_select_nmost` is also available as a [cogent3 app](https://cogent3.org/doc/app/index.html). The result of using `cogent3.app_help("dvs_select_nmost")` is shown below.
|
|
289
|
+
|
|
290
|
+
<!-- [[[cog
|
|
291
|
+
import cog
|
|
292
|
+
import contextlib
|
|
293
|
+
import io
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
from cogent3 import app_help
|
|
297
|
+
|
|
298
|
+
buffer = io.StringIO()
|
|
299
|
+
|
|
300
|
+
with contextlib.redirect_stdout(buffer):
|
|
301
|
+
app_help("dvs_select_max")
|
|
302
|
+
cog.out(
|
|
303
|
+
"```\n{}\n```".format(buffer.getvalue())
|
|
304
|
+
)
|
|
305
|
+
]]] -->
|
|
306
|
+
```
|
|
307
|
+
Overview
|
|
308
|
+
--------
|
|
309
|
+
selects the maximally diverse seqs from a sequence collection
|
|
310
|
+
|
|
311
|
+
Options for making the app
|
|
312
|
+
--------------------------
|
|
313
|
+
dvs_select_max_app = get_app(
|
|
314
|
+
'dvs_select_max',
|
|
315
|
+
min_size=3,
|
|
316
|
+
max_size=10,
|
|
317
|
+
stat='stdev',
|
|
318
|
+
moltype='dna',
|
|
319
|
+
include=None,
|
|
320
|
+
k=6,
|
|
321
|
+
seed=None,
|
|
322
|
+
)
|
|
323
|
+
|
|
324
|
+
Parameters
|
|
325
|
+
----------
|
|
326
|
+
min_size
|
|
327
|
+
minimum size of the divergent set
|
|
328
|
+
max_size
|
|
329
|
+
the maximum size if the divergent set
|
|
330
|
+
stat
|
|
331
|
+
statistic for maximising the set, either mean_delta_jsd, mean_jsd, total_jsd
|
|
332
|
+
moltype
|
|
333
|
+
molecular type of the sequences
|
|
334
|
+
include
|
|
335
|
+
sequence names to include in the final result
|
|
336
|
+
k
|
|
337
|
+
k-mer size
|
|
338
|
+
seed
|
|
339
|
+
random number seed
|
|
340
|
+
|
|
341
|
+
Notes
|
|
342
|
+
-----
|
|
343
|
+
If called with an alignment, the ungapped sequences are used.
|
|
344
|
+
The order of the sequences is randomised. If include is not None, the
|
|
345
|
+
named sequences are added to the final result.
|
|
346
|
+
|
|
347
|
+
Input type
|
|
348
|
+
----------
|
|
349
|
+
SequenceCollection, Alignment, ArrayAlignment
|
|
350
|
+
|
|
351
|
+
Output type
|
|
352
|
+
-----------
|
|
353
|
+
SequenceCollection, Alignment, ArrayAlignment
|
|
354
|
+
|
|
355
|
+
```
|
|
356
|
+
<!-- [[[end]]] -->
|
|
357
|
+
|
|
@@ -0,0 +1,297 @@
|
|
|
1
|
+
[](https://github.com/HuttleyLab/DiverseSeq/actions/workflows/ci.yml)
|
|
2
|
+
[](https://coveralls.io/github/HuttleyLab/DiverseSeq?branch=main)
|
|
3
|
+
[](https://app.codacy.com/gh/HuttleyLab/DiverseSeq/dashboard?utm_source=gh&utm_medium=referral&utm_content=&utm_campaign=Badge_grade)
|
|
4
|
+
[](https://github.com/HuttleyLab/DiverseSeq/actions/workflows/codeql.yml)
|
|
5
|
+
[](https://github.com/astral-sh/ruff)
|
|
6
|
+
|
|
7
|
+
# DiverseSeq identifies the most diverse biological sequences from a collection
|
|
8
|
+
|
|
9
|
+
`diverse_seq` provides tools for selecting a representative subset of sequences from a larger collection. It is an alignment-free method which scales linearly with the number of sequences. It identifies the subset of sequences that maximize diversity as measured using Jensen-Shannon divergence. `DiverseSeq` provides a command-line tool (`dvs`) and plugins to the Cogent3 app system (prefixed by `dvs_`) allowing users to embed code in their own scripts. The command-line tools can be run in parallel.
|
|
10
|
+
|
|
11
|
+
## The available commands
|
|
12
|
+
|
|
13
|
+
<!-- [[[cog
|
|
14
|
+
import cog
|
|
15
|
+
from diverse_seq.cli import main
|
|
16
|
+
from click.testing import CliRunner
|
|
17
|
+
runner = CliRunner()
|
|
18
|
+
result = runner.invoke(main, ["--help"])
|
|
19
|
+
help = result.output.replace("Usage: main", "Usage: dvs")
|
|
20
|
+
cog.out(
|
|
21
|
+
"```\n{}\n```".format(help)
|
|
22
|
+
)
|
|
23
|
+
]]] -->
|
|
24
|
+
```
|
|
25
|
+
Usage: dvs [OPTIONS] COMMAND [ARGS]...
|
|
26
|
+
|
|
27
|
+
dvs -- alignment free detection of the most diverse sequences using JSD
|
|
28
|
+
|
|
29
|
+
Options:
|
|
30
|
+
--version Show the version and exit.
|
|
31
|
+
--help Show this message and exit.
|
|
32
|
+
|
|
33
|
+
Commands:
|
|
34
|
+
prep Writes processed sequences to a <HDF5 file>.dvseqs.
|
|
35
|
+
max Identify the seqs that maximise average delta JSD
|
|
36
|
+
nmost Identify n seqs that maximise average delta JSD
|
|
37
|
+
|
|
38
|
+
```
|
|
39
|
+
<!-- [[[end]]] -->
|
|
40
|
+
|
|
41
|
+
### `dvs prep`: Preparing the sequence data
|
|
42
|
+
|
|
43
|
+
Convert sequence data into a more efficient format for the diversity assessment. This must be done before running either the `nmost` or `max` commands.
|
|
44
|
+
|
|
45
|
+
#### Usage:
|
|
46
|
+
|
|
47
|
+
<!-- [[[cog
|
|
48
|
+
import cog
|
|
49
|
+
from diverse_seq.cli import main
|
|
50
|
+
from click.testing import CliRunner
|
|
51
|
+
runner = CliRunner()
|
|
52
|
+
result = runner.invoke(main, ["prep", "--help"])
|
|
53
|
+
help = result.output.replace("Usage: main", "Usage: dvs")
|
|
54
|
+
cog.out(
|
|
55
|
+
"```\n{}\n```".format(help)
|
|
56
|
+
)
|
|
57
|
+
]]] -->
|
|
58
|
+
```
|
|
59
|
+
Usage: dvs prep [OPTIONS]
|
|
60
|
+
|
|
61
|
+
Writes processed sequences to a <HDF5 file>.dvseqs.
|
|
62
|
+
|
|
63
|
+
Options:
|
|
64
|
+
-s, --seqdir PATH directory containing sequence files [required]
|
|
65
|
+
-sf, --suffix TEXT sequence file suffix [default: fa]
|
|
66
|
+
-o, --outpath PATH write processed seqs to this filename [required]
|
|
67
|
+
-np, --numprocs INTEGER number of processes [default: 1]
|
|
68
|
+
-F, --force_overwrite Overwrite existing file if it exists
|
|
69
|
+
-m, --moltype [dna|rna] Molecular type of sequences, defaults to DNA
|
|
70
|
+
[default: dna]
|
|
71
|
+
-L, --limit INTEGER number of sequences to process
|
|
72
|
+
--help Show this message and exit.
|
|
73
|
+
|
|
74
|
+
```
|
|
75
|
+
<!-- [[[end]]] -->
|
|
76
|
+
|
|
77
|
+
### `dvs nmost`: Select the n-most diverse sequences
|
|
78
|
+
|
|
79
|
+
We recommend using `nmost` for large datasets.
|
|
80
|
+
|
|
81
|
+
> **Note**
|
|
82
|
+
> A fuller explanation is coming soon!
|
|
83
|
+
|
|
84
|
+
#### Command line usage:
|
|
85
|
+
|
|
86
|
+
<!-- [[[cog
|
|
87
|
+
import cog
|
|
88
|
+
from diverse_seq.cli import main
|
|
89
|
+
from click.testing import CliRunner
|
|
90
|
+
runner = CliRunner()
|
|
91
|
+
result = runner.invoke(main, ["nmost", "--help"])
|
|
92
|
+
help = result.output.replace("Usage: main", "Usage: dvs")
|
|
93
|
+
cog.out(
|
|
94
|
+
"```\n{}\n```".format(help)
|
|
95
|
+
)
|
|
96
|
+
]]] -->
|
|
97
|
+
```
|
|
98
|
+
Usage: dvs nmost [OPTIONS]
|
|
99
|
+
|
|
100
|
+
Identify n seqs that maximise average delta JSD
|
|
101
|
+
|
|
102
|
+
Options:
|
|
103
|
+
-s, --seqfile PATH path to .dvtgseqs file [required]
|
|
104
|
+
-o, --outpath PATH the input string will be cast to Path instance
|
|
105
|
+
-n, --number INTEGER number of seqs in divergent set [required]
|
|
106
|
+
-k INTEGER k-mer size [default: 6]
|
|
107
|
+
-i, --include TEXT seqnames to include in divergent set
|
|
108
|
+
-np, --numprocs INTEGER number of processes [default: 1]
|
|
109
|
+
-L, --limit INTEGER number of sequences to process
|
|
110
|
+
-v, --verbose is an integer indicating number of cl occurrences
|
|
111
|
+
[default: 0]
|
|
112
|
+
--help Show this message and exit.
|
|
113
|
+
|
|
114
|
+
```
|
|
115
|
+
<!-- [[[end]]] -->
|
|
116
|
+
|
|
117
|
+
#### As a cogent3 plugin:
|
|
118
|
+
|
|
119
|
+
The `dvs_select_nmost` is also available as a [cogent3 app](https://cogent3.org/doc/app/index.html). The result of using `cogent3.app_help("dvs_select_nmost")` is shown below.
|
|
120
|
+
|
|
121
|
+
<!-- [[[cog
|
|
122
|
+
import cog
|
|
123
|
+
import contextlib
|
|
124
|
+
import io
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
from cogent3 import app_help
|
|
128
|
+
|
|
129
|
+
buffer = io.StringIO()
|
|
130
|
+
|
|
131
|
+
with contextlib.redirect_stdout(buffer):
|
|
132
|
+
app_help("dvs_select_nmost")
|
|
133
|
+
cog.out(
|
|
134
|
+
"```\n{}\n```".format(buffer.getvalue())
|
|
135
|
+
)
|
|
136
|
+
]]] -->
|
|
137
|
+
```
|
|
138
|
+
Overview
|
|
139
|
+
--------
|
|
140
|
+
selects the n-most diverse seqs from a sequence collection
|
|
141
|
+
|
|
142
|
+
Options for making the app
|
|
143
|
+
--------------------------
|
|
144
|
+
dvs_select_nmost_app = get_app(
|
|
145
|
+
'dvs_select_nmost',
|
|
146
|
+
n=3,
|
|
147
|
+
moltype='dna',
|
|
148
|
+
include=None,
|
|
149
|
+
k=6,
|
|
150
|
+
seed=None,
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
Parameters
|
|
154
|
+
----------
|
|
155
|
+
n
|
|
156
|
+
the number of divergent sequences
|
|
157
|
+
moltype
|
|
158
|
+
molecular type of the sequences
|
|
159
|
+
k
|
|
160
|
+
k-mer size
|
|
161
|
+
include
|
|
162
|
+
sequence names to include in the final result
|
|
163
|
+
seed
|
|
164
|
+
random number seed
|
|
165
|
+
|
|
166
|
+
Notes
|
|
167
|
+
-----
|
|
168
|
+
If called with an alignment, the ungapped sequences are used.
|
|
169
|
+
The order of the sequences is randomised. If include is not None, the
|
|
170
|
+
named sequences are added to the final result.
|
|
171
|
+
|
|
172
|
+
Input type
|
|
173
|
+
----------
|
|
174
|
+
SequenceCollection, Alignment, ArrayAlignment
|
|
175
|
+
|
|
176
|
+
Output type
|
|
177
|
+
-----------
|
|
178
|
+
SequenceCollection, Alignment, ArrayAlignment
|
|
179
|
+
|
|
180
|
+
```
|
|
181
|
+
<!-- [[[end]]] -->
|
|
182
|
+
|
|
183
|
+
### `dvs max`: Maximise average delta JSD
|
|
184
|
+
|
|
185
|
+
The result of the `max` command is typically a set that are modestly more diverse than that fron `nmost`.
|
|
186
|
+
|
|
187
|
+
> **Note**
|
|
188
|
+
> A fuller explanation is coming soon!
|
|
189
|
+
|
|
190
|
+
#### Command line usage:
|
|
191
|
+
|
|
192
|
+
<!-- [[[cog
|
|
193
|
+
import cog
|
|
194
|
+
from diverse_seq.cli import main
|
|
195
|
+
from click.testing import CliRunner
|
|
196
|
+
runner = CliRunner()
|
|
197
|
+
result = runner.invoke(main, ["max", "--help"])
|
|
198
|
+
help = result.output.replace("Usage: main", "Usage: dvs")
|
|
199
|
+
cog.out(
|
|
200
|
+
"```\n{}\n```".format(help)
|
|
201
|
+
)
|
|
202
|
+
]]] -->
|
|
203
|
+
```
|
|
204
|
+
Usage: dvs max [OPTIONS]
|
|
205
|
+
|
|
206
|
+
Identify the seqs that maximise average delta JSD
|
|
207
|
+
|
|
208
|
+
Options:
|
|
209
|
+
-s, --seqfile PATH path to .dvtgseqs file [required]
|
|
210
|
+
-o, --outpath PATH the input string will be cast to Path instance
|
|
211
|
+
-z, --min_size INTEGER minimum size of divergent set [default: 7]
|
|
212
|
+
-zp, --max_size INTEGER maximum size of divergent set
|
|
213
|
+
-k INTEGER k-mer size [default: 6]
|
|
214
|
+
-st, --stat [stdev|cov] statistic to maximise [default: stdev]
|
|
215
|
+
-i, --include TEXT seqnames to include in divergent set
|
|
216
|
+
-np, --numprocs INTEGER number of processes [default: 1]
|
|
217
|
+
-L, --limit INTEGER number of sequences to process
|
|
218
|
+
-T, --test_run reduce number of paths and size of query seqs
|
|
219
|
+
-v, --verbose is an integer indicating number of cl occurrences
|
|
220
|
+
[default: 0]
|
|
221
|
+
--help Show this message and exit.
|
|
222
|
+
|
|
223
|
+
```
|
|
224
|
+
<!-- [[[end]]] -->
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
#### As a cogent3 plugin:
|
|
228
|
+
|
|
229
|
+
The `dvs_select_nmost` is also available as a [cogent3 app](https://cogent3.org/doc/app/index.html). The result of using `cogent3.app_help("dvs_select_nmost")` is shown below.
|
|
230
|
+
|
|
231
|
+
<!-- [[[cog
|
|
232
|
+
import cog
|
|
233
|
+
import contextlib
|
|
234
|
+
import io
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
from cogent3 import app_help
|
|
238
|
+
|
|
239
|
+
buffer = io.StringIO()
|
|
240
|
+
|
|
241
|
+
with contextlib.redirect_stdout(buffer):
|
|
242
|
+
app_help("dvs_select_max")
|
|
243
|
+
cog.out(
|
|
244
|
+
"```\n{}\n```".format(buffer.getvalue())
|
|
245
|
+
)
|
|
246
|
+
]]] -->
|
|
247
|
+
```
|
|
248
|
+
Overview
|
|
249
|
+
--------
|
|
250
|
+
selects the maximally diverse seqs from a sequence collection
|
|
251
|
+
|
|
252
|
+
Options for making the app
|
|
253
|
+
--------------------------
|
|
254
|
+
dvs_select_max_app = get_app(
|
|
255
|
+
'dvs_select_max',
|
|
256
|
+
min_size=3,
|
|
257
|
+
max_size=10,
|
|
258
|
+
stat='stdev',
|
|
259
|
+
moltype='dna',
|
|
260
|
+
include=None,
|
|
261
|
+
k=6,
|
|
262
|
+
seed=None,
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
Parameters
|
|
266
|
+
----------
|
|
267
|
+
min_size
|
|
268
|
+
minimum size of the divergent set
|
|
269
|
+
max_size
|
|
270
|
+
the maximum size if the divergent set
|
|
271
|
+
stat
|
|
272
|
+
statistic for maximising the set, either mean_delta_jsd, mean_jsd, total_jsd
|
|
273
|
+
moltype
|
|
274
|
+
molecular type of the sequences
|
|
275
|
+
include
|
|
276
|
+
sequence names to include in the final result
|
|
277
|
+
k
|
|
278
|
+
k-mer size
|
|
279
|
+
seed
|
|
280
|
+
random number seed
|
|
281
|
+
|
|
282
|
+
Notes
|
|
283
|
+
-----
|
|
284
|
+
If called with an alignment, the ungapped sequences are used.
|
|
285
|
+
The order of the sequences is randomised. If include is not None, the
|
|
286
|
+
named sequences are added to the final result.
|
|
287
|
+
|
|
288
|
+
Input type
|
|
289
|
+
----------
|
|
290
|
+
SequenceCollection, Alignment, ArrayAlignment
|
|
291
|
+
|
|
292
|
+
Output type
|
|
293
|
+
-----------
|
|
294
|
+
SequenceCollection, Alignment, ArrayAlignment
|
|
295
|
+
|
|
296
|
+
```
|
|
297
|
+
<!-- [[[end]]] -->
|