timon-gui 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- timon/__init__.py +7 -0
- timon/app/__init__.py +14 -0
- timon/app/config.py +568 -0
- timon/app/events.py +315 -0
- timon/app/model/__init__.py +35 -0
- timon/app/model/downloads.py +417 -0
- timon/app/model/experiment.py +476 -0
- timon/app/model/files.py +184 -0
- timon/app/model/history.py +279 -0
- timon/app/model/live.py +416 -0
- timon/app/model/nextflow.py +726 -0
- timon/app/model/nfstate.py +117 -0
- timon/app/model/params.py +247 -0
- timon/app/model/results.py +286 -0
- timon/app/model/samples.py +90 -0
- timon/app/model/validation.py +180 -0
- timon/app/presenters.py +716 -0
- timon/app/routes.py +446 -0
- timon/app/static/fonts/LICENSE-IBMPlexMono.txt +93 -0
- timon/app/static/fonts/LICENSE-Inter.txt +93 -0
- timon/app/static/fonts/ibm-plex-mono-400-latin-ext.woff2 +0 -0
- timon/app/static/fonts/ibm-plex-mono-400-latin.woff2 +0 -0
- timon/app/static/fonts/ibm-plex-mono-500-latin-ext.woff2 +0 -0
- timon/app/static/fonts/ibm-plex-mono-500-latin.woff2 +0 -0
- timon/app/static/fonts/ibm-plex-mono-600-latin-ext.woff2 +0 -0
- timon/app/static/fonts/ibm-plex-mono-600-latin.woff2 +0 -0
- timon/app/static/fonts/inter-300-700-latin-ext.woff2 +0 -0
- timon/app/static/fonts/inter-300-700-latin.woff2 +0 -0
- timon/app/static/img/bloom_orig.png +0 -0
- timon/app/static/img/isolate-icon.png +0 -0
- timon/app/static/img/mag-icon.png +0 -0
- timon/app/static/img/wheel1.svg +1 -0
- timon/app/static/img/wheel2.svg +1 -0
- timon/app/static/js/socket.io.js +6046 -0
- timon/app/static/style/app.css +2 -0
- timon/app/templates/_brand.html +20 -0
- timon/app/templates/_params.html +173 -0
- timon/app/templates/index.html +2604 -0
- timon/cli.py +132 -0
- timon/paths.py +197 -0
- timon_gui-0.1.0.dist-info/METADATA +76 -0
- timon_gui-0.1.0.dist-info/RECORD +45 -0
- timon_gui-0.1.0.dist-info/WHEEL +4 -0
- timon_gui-0.1.0.dist-info/entry_points.txt +2 -0
- timon_gui-0.1.0.dist-info/licenses/LICENSE +21 -0
timon/__init__.py
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
"""timon — Toolkit of Integrated MicrobiOme analysis with support for loNg reads."""
|
|
2
|
+
|
|
3
|
+
__version__ = "0.1.0"
|
|
4
|
+
# The year __version__ was released, shown in the page footer. Bumped with the
|
|
5
|
+
# version, not read from the clock: an old install should keep saying the year
|
|
6
|
+
# it is from rather than quietly claiming to be current.
|
|
7
|
+
__release_year__ = "2026"
|
timon/app/__init__.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
from flask import Flask
|
|
2
|
+
from flask_socketio import SocketIO
|
|
3
|
+
from .config import Config
|
|
4
|
+
|
|
5
|
+
socketio = SocketIO()
|
|
6
|
+
|
|
7
|
+
def create_app():
|
|
8
|
+
app = Flask(__name__)
|
|
9
|
+
app.config.from_object(Config)
|
|
10
|
+
socketio.init_app(app)
|
|
11
|
+
|
|
12
|
+
with app.app_context():
|
|
13
|
+
from . import routes, events
|
|
14
|
+
return app
|
timon/app/config.py
ADDED
|
@@ -0,0 +1,568 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import secrets
|
|
3
|
+
|
|
4
|
+
class Config:
|
|
5
|
+
SECRET_KEY = os.getenv("TIMON_SECRET_KEY") or secrets.token_hex(32)
|
|
6
|
+
IMPORT_FOLDER = os.getenv("INPUT_DIR", "imports")
|
|
7
|
+
# Where runs write, which is deliberately not where the reads are: a
|
|
8
|
+
# folder called "imports" holding the results of an analysis is a folder
|
|
9
|
+
# nobody can read the layout of. Everything of timon's own that a run
|
|
10
|
+
# needs and nobody asked for — nextflow's work directory, the resource
|
|
11
|
+
# ceiling, the sample sheet — goes in here under a leading dot, so what
|
|
12
|
+
# the results view walks is the runs themselves.
|
|
13
|
+
OUTPUT_FOLDER = os.getenv("OUTPUT_DIR", "timon_results")
|
|
14
|
+
# Container engine Nextflow provisions tasks with. timon supports docker,
|
|
15
|
+
# singularity, apptainer and conda (model.nextflow.CONTAINER_BINARIES), and
|
|
16
|
+
# every pipeline declares which of those it supports; see "profiles" below.
|
|
17
|
+
PROFILE = os.getenv("TIMON_PROFILE", "docker")
|
|
18
|
+
|
|
19
|
+
# ── reference databases ──────────────────────────────────────────────────────
|
|
20
|
+
#
|
|
21
|
+
# Every database a pipeline reads is timon's to find: timon.paths.REFERENCE_DATA
|
|
22
|
+
# says where each one is (db_root(), unless its environment variable names a
|
|
23
|
+
# copy elsewhere), and a pipeline entry only lists which it reads under
|
|
24
|
+
# "reference_data". Nothing is asked for in the form — one database, one place,
|
|
25
|
+
# whichever pipeline reads it.
|
|
26
|
+
#
|
|
27
|
+
# A database timon cannot find is the commonest thing between a fresh install
|
|
28
|
+
# and a first run, and the fix is always the same act: fetch a large file and
|
|
29
|
+
# leave it where timon looks. So where each one can be got is declared here,
|
|
30
|
+
# and model/downloads.py is the one place that carries a download out.
|
|
31
|
+
#
|
|
32
|
+
# Keyed by the database, a key of timon.paths.REFERENCE_DATA. A database with
|
|
33
|
+
# no entry is still reported missing; it is simply not something the button can
|
|
34
|
+
# fetch.
|
|
35
|
+
#
|
|
36
|
+
# label what the page calls the download
|
|
37
|
+
# size how large it is. Someone deciding whether to start one at all
|
|
38
|
+
# is deciding on this number, so it is written by hand from the
|
|
39
|
+
# publisher's own page — a HEAD request would report it only
|
|
40
|
+
# after the user had committed to asking
|
|
41
|
+
# url what is fetched, over https
|
|
42
|
+
# archive True for a `.tar.gz` unpacked once it has landed; False for a
|
|
43
|
+
# file that is used exactly as it arrives
|
|
44
|
+
# install_as the name it takes under timon.paths.db_root(). It *must* be
|
|
45
|
+
# one of the names paths.py looks for, or the download would
|
|
46
|
+
# land somewhere timon never looks (tests/test_config.py)
|
|
47
|
+
# description one line: what is in it and what it is enough for
|
|
48
|
+
#
|
|
49
|
+
# Optional, and only where the publisher offers the same database in more than
|
|
50
|
+
# one build:
|
|
51
|
+
#
|
|
52
|
+
# variants the builds, each its own source. A variant carries "id" and
|
|
53
|
+
# whatever it changes — url, size, install_as, label, note — and
|
|
54
|
+
# inherits the rest from the entry around it, so what the builds
|
|
55
|
+
# share is written once. The page offers them as a choice and
|
|
56
|
+
# sends back an id; ``default`` names the one chosen for a user
|
|
57
|
+
# who expresses no preference, and the bundle button fetches
|
|
58
|
+
# that one
|
|
59
|
+
# default the id of that default variant
|
|
60
|
+
#
|
|
61
|
+
# A database published one way declares neither, and is its own single variant
|
|
62
|
+
# — model/downloads.py normalises both shapes to the same thing, so nothing
|
|
63
|
+
# past it has two to reason about.
|
|
64
|
+
DB_SOURCES = {
|
|
65
|
+
# As published at https://benlangmead.github.io/aws-indexes/k2. Dated
|
|
66
|
+
# builds: the date is part of the URL and of the name each is installed
|
|
67
|
+
# under, so a newer index is a new variant beside these rather than a
|
|
68
|
+
# silent change. Sizes in GiB, as the publisher's page gives them; both
|
|
69
|
+
# variants are kept on the same date, since which of the two someone
|
|
70
|
+
# picked should not also decide how current their index is.
|
|
71
|
+
"kraken_db": {
|
|
72
|
+
"label": "Kraken2 PlusPF",
|
|
73
|
+
"archive": True,
|
|
74
|
+
"description": "Archaea, bacteria, viruses, plasmids, human, vectors, "
|
|
75
|
+
"protozoa and fungi.",
|
|
76
|
+
# Two builds of the same collection, from the same date. The cap is
|
|
77
|
+
# how much memory Kraken2 needs to hold the index while it classifies,
|
|
78
|
+
# so this is a choice about the machine and not about the analysis: a
|
|
79
|
+
# laptop with 8 GB cannot run the larger one at all, and the smaller
|
|
80
|
+
# one is more heavily hashed, so it classifies a little less. Which
|
|
81
|
+
# was installed is not recorded anywhere and no run reads it — both
|
|
82
|
+
# land under a name paths.py looks for, and a Kraken2 index is a
|
|
83
|
+
# Kraken2 index to the pipeline.
|
|
84
|
+
"default": "16GB",
|
|
85
|
+
"variants": [
|
|
86
|
+
{
|
|
87
|
+
"id": "16GB",
|
|
88
|
+
"label": "Kraken2 PlusPF-16",
|
|
89
|
+
"size": "11.1 GB",
|
|
90
|
+
"url": "https://genome-idx.s3.amazonaws.com/kraken/k2_pluspf_16_GB_20260626.tar.gz",
|
|
91
|
+
"install_as": "k2_pluspf_16_GB_20260626",
|
|
92
|
+
"note": "needs 16 GB of memory to classify with",
|
|
93
|
+
},
|
|
94
|
+
{
|
|
95
|
+
"id": "8GB",
|
|
96
|
+
"label": "Kraken2 PlusPF-8",
|
|
97
|
+
"size": "5.5 GB",
|
|
98
|
+
"url": "https://genome-idx.s3.amazonaws.com/kraken/k2_pluspf_08_GB_20260626.tar.gz",
|
|
99
|
+
"install_as": "k2_pluspf_08_GB_20260626",
|
|
100
|
+
"note": "same organisms, hashed down to 8 GB — for a machine "
|
|
101
|
+
"that cannot spare 16",
|
|
102
|
+
},
|
|
103
|
+
],
|
|
104
|
+
},
|
|
105
|
+
# One package, and the project publishes it as "latest" rather than per
|
|
106
|
+
# release: GTDB-Tk checks the data version it is given at startup, so the
|
|
107
|
+
# pipeline's own image decides what it will accept.
|
|
108
|
+
"gtdbtk_db": {
|
|
109
|
+
"label": "GTDB-Tk reference data",
|
|
110
|
+
"size": "~100 GB",
|
|
111
|
+
"url": "https://data.ace.uq.edu.au/public/gtdb/data/releases/latest/"
|
|
112
|
+
"auxillary_files/gtdbtk_package/full_package/gtdbtk_data.tar.gz",
|
|
113
|
+
"archive": True,
|
|
114
|
+
"install_as": "gtdbtk_data",
|
|
115
|
+
"description": "The full package, which is the only one GTDB-Tk classify "
|
|
116
|
+
"takes. Allow for a long download and twice the space "
|
|
117
|
+
"while it unpacks.",
|
|
118
|
+
},
|
|
119
|
+
# timon's own: 220 dereplicated cyanobacterial genomes, published on
|
|
120
|
+
# Zenodo under a record that names this exact set, so the URL is as fixed
|
|
121
|
+
# as a dated Kraken2 build. The tarball wraps them in one directory, which
|
|
122
|
+
# is unwrapped on install — CoverM is handed the directory and names each
|
|
123
|
+
# genome after its file, so what it reads is the .fna files themselves.
|
|
124
|
+
"genomes_db": {
|
|
125
|
+
"label": "cyanobacterial genome set",
|
|
126
|
+
"size": "278 MB",
|
|
127
|
+
"url": "https://zenodo.org/records/19522349/files/"
|
|
128
|
+
"cyanobacteriota_ncbi_dRep_n220.tar.gz",
|
|
129
|
+
"archive": True,
|
|
130
|
+
"install_as": "cyanobacteriota_ncbi_dRep_n220",
|
|
131
|
+
"description": "220 dereplicated cyanobacterial genomes, which CoverM "
|
|
132
|
+
"maps the reads against. Any set of FASTA files is a "
|
|
133
|
+
"valid answer, so TIMON_GENOMES_DB can name a smaller one.",
|
|
134
|
+
},
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
# What the **install databases** button fetches: installed once on a machine
|
|
138
|
+
# and read by every run after. A run that reads one of them cannot start until
|
|
139
|
+
# it is there, so the button is the first thing a fresh install offers and is
|
|
140
|
+
# frozen once there is nothing left for it to do.
|
|
141
|
+
#
|
|
142
|
+
# GTDB-Tk is deliberately not in it. At ~100 GB it would make every install
|
|
143
|
+
# pay for a step only mag-ont's bin QA takes, and a run that skips that step
|
|
144
|
+
# never reads it; it is offered on its own, and only to a run that needs it.
|
|
145
|
+
DATABASE_BUNDLE = ["kraken_db", "genomes_db"]
|
|
146
|
+
|
|
147
|
+
# ── linked pipelines ─────────────────────────────────────────────────────────
|
|
148
|
+
#
|
|
149
|
+
# Each entry links a Nextflow pipeline to timon. Required keys:
|
|
150
|
+
#
|
|
151
|
+
# name, description, icon what the workflow card shows
|
|
152
|
+
# pipeline GitHub "owner/repo", passed to `nextflow run`
|
|
153
|
+
# revision the release tag the run is pinned to — a tag, never
|
|
154
|
+
# a branch or a bare commit, so what a run used has a
|
|
155
|
+
# name upstream (tests/test_config.py holds that).
|
|
156
|
+
# None refuses to run (see model.nextflow.build_command)
|
|
157
|
+
# profiles container engines the pipeline supports, of those
|
|
158
|
+
# timon does (model.nextflow.CONTAINER_BINARIES): a
|
|
159
|
+
# pipeline declaring more is listed with the
|
|
160
|
+
# intersection, and a profile absent here is refused
|
|
161
|
+
# columns, file_column sample-sheet columns, and which one holds reads
|
|
162
|
+
# reference_data the databases the pipeline reads, by the keys of
|
|
163
|
+
# timon.paths.REFERENCE_DATA, each passed under its
|
|
164
|
+
# flag in model.nextflow.REFERENCE_FLAGS. Never
|
|
165
|
+
# asked for: timon knows where they are, and a run
|
|
166
|
+
# that reads a missing one cannot start. So they
|
|
167
|
+
# are never declared in "params" either — that
|
|
168
|
+
# would be a second source
|
|
169
|
+
# params the run-configuration form
|
|
170
|
+
#
|
|
171
|
+
# "params" is written by hand, and is a *selection*: a pipeline accepts far
|
|
172
|
+
# more than a user should have to look at, and everything left out simply
|
|
173
|
+
# keeps the pipeline's own default. Add a parameter here when a run actually
|
|
174
|
+
# turns on it. Each entry is:
|
|
175
|
+
#
|
|
176
|
+
# id the Nextflow parameter, passed as --<id>. Required.
|
|
177
|
+
# type "bool" | "number" | "text" | "select" (default: text)
|
|
178
|
+
# label what the form shows; defaults to the id
|
|
179
|
+
# default must match the pipeline's own default at the pinned
|
|
180
|
+
# revision, or the form will lie about what a run does.
|
|
181
|
+
# None, for a number, is a pipeline that sets no value of its
|
|
182
|
+
# own: the field may be left empty and nothing is passed
|
|
183
|
+
# description one line under the field
|
|
184
|
+
# help_text the part that actually decides a value; disclosed on demand
|
|
185
|
+
# min/max/step numbers only. step "any" means the value is a float, not an
|
|
186
|
+
# integer, and is cast as one (see model.params.coerce)
|
|
187
|
+
# enum selects only: the whole set of accepted values
|
|
188
|
+
# options_from selects only, instead of enum: the options are what an
|
|
189
|
+
# installed database holds, read by the reader of that name in
|
|
190
|
+
# timon.paths.DATABASE_OPTIONS. Until the database can be
|
|
191
|
+
# looked into, the default is the only option
|
|
192
|
+
# placeholder text fields only: hint shown while the field is empty
|
|
193
|
+
# path text fields only: the value is a filesystem path, so the
|
|
194
|
+
# field gets the same browse button the sample sheet's path
|
|
195
|
+
# columns have. "dir" picks a folder, "file" picks a file,
|
|
196
|
+
# "any" takes either — a database shipped as a directory or
|
|
197
|
+
# as a tarball
|
|
198
|
+
# required refuse to save the configuration without a value
|
|
199
|
+
# required_when {other param id: [values]} — required only while that other
|
|
200
|
+
# parameter holds one of those values (a checkbox is True)
|
|
201
|
+
# active_when {other param id: [values]} — part of the run only while
|
|
202
|
+
# *every* one of those parameters holds one of the listed
|
|
203
|
+
# values. Anything else drops out of the form, out of
|
|
204
|
+
# validation and off the command line: a QC threshold under
|
|
205
|
+
# "skip QC", the assembly options of a route this mode does
|
|
206
|
+
# not take. Conditions chain — a parameter whose controller
|
|
207
|
+
# has itself dropped out goes with it
|
|
208
|
+
# group section heading; blurbs come from "param_groups" below
|
|
209
|
+
#
|
|
210
|
+
# Optional keys:
|
|
211
|
+
#
|
|
212
|
+
# required_columns sample-sheet columns that must be filled on every row.
|
|
213
|
+
# Defaults to *all* of "columns". Declared only where the
|
|
214
|
+
# pipeline's own assets/schema_input.json asks for less:
|
|
215
|
+
# a column timon insists on that the pipeline treats as
|
|
216
|
+
# optional locks the user out of a sheet the pipeline
|
|
217
|
+
# would have accepted — a nanopore run with no short reads
|
|
218
|
+
# beside it, say
|
|
219
|
+
# one_of_columns groups of columns of which at least one must be filled on
|
|
220
|
+
# each row, for a schema's "anyOf" — a row needing either
|
|
221
|
+
# reads to assemble or an assembly already made
|
|
222
|
+
# param_groups {group: blurb} for the section headings used by "params"
|
|
223
|
+
# db_optional_when {db key: [flags]} — a database under "reference_data"
|
|
224
|
+
# that is only read while none of those flags is set: not
|
|
225
|
+
# passed, and not missing, for a run that skips its step
|
|
226
|
+
# test_profile the pipeline's own smoke-test profile, launched as
|
|
227
|
+
# `-profile <test_profile>,<engine>` by the quick test
|
|
228
|
+
# button. Declared only for a pipeline whose test profile
|
|
229
|
+
# fetches its own reads and its own databases: timon hands
|
|
230
|
+
# it no sample sheet, no parameters and no database, so a
|
|
231
|
+
# profile expecting any of them would fail at once. A
|
|
232
|
+
# pipeline without this key is offered no quick test
|
|
233
|
+
# selectable False leaves a declared pipeline out of the workflow
|
|
234
|
+
# picker, and refuses a switch to it. For one written down
|
|
235
|
+
# before it can be run — the entry stays, so nothing has to
|
|
236
|
+
# be reconstructed when it can
|
|
237
|
+
PIPELINES = {
|
|
238
|
+
"roshab-cli": {
|
|
239
|
+
"name": "roshab-cli",
|
|
240
|
+
"description": "Taxonomic classification and evaluation of cyanotoxin biosynthesis potential from nanopore reads",
|
|
241
|
+
"icon": "img/bloom_orig.png",
|
|
242
|
+
"pipeline": "dsamoht/roshab-cli",
|
|
243
|
+
# The first release. Bumping it is a timon release, and the defaults
|
|
244
|
+
# below have to be rechecked against that tag's nextflow_schema.json.
|
|
245
|
+
"revision": "v0.1.0",
|
|
246
|
+
"profiles": ["docker", "singularity", "apptainer"],
|
|
247
|
+
# conf/test.config at this revision names its own sample sheet and
|
|
248
|
+
# every database it reads — the viral Kraken2 index by URL, the CoverM
|
|
249
|
+
# genomes and the gene database from the pipeline itself — so
|
|
250
|
+
# the quick test needs nothing from the form. It checks the install
|
|
251
|
+
# rather than a configuration, and what it downloads is the same
|
|
252
|
+
# 569 MB index the roshab-cli example run asks the user for.
|
|
253
|
+
"test_profile": "test",
|
|
254
|
+
"file_column": "reads",
|
|
255
|
+
# Found by timon and installed by the install databases button — see
|
|
256
|
+
# timon.paths.REFERENCE_DATA. Not the gene database:
|
|
257
|
+
# from v0.1.0 the pipeline ships its own (assets/cyanotoxin_genes_
|
|
258
|
+
# mibig-4.0_two-class_v2.faa), whose sixth header field tells toxin
|
|
259
|
+
# genes from the rest, and the reports are written against that panel.
|
|
260
|
+
"reference_data": ["kraken_db", "genomes_db"],
|
|
261
|
+
# Matches assets/schema_input.json at this revision, which requires
|
|
262
|
+
# every one of them — so no "required_columns" here.
|
|
263
|
+
"columns": ["sample_id", "group", "info", "date", "reads"],
|
|
264
|
+
# The `--db_dir` install route is deliberately absent: setting it runs
|
|
265
|
+
# the database download *instead of* an analysis, which is not what
|
|
266
|
+
# this form builds.
|
|
267
|
+
"param_groups": {
|
|
268
|
+
"workflow": "Which screening route the run takes. Taxonomic profiling runs in every mode.",
|
|
269
|
+
"read QC": "Chopper trimming and filtering thresholds.",
|
|
270
|
+
"taxonomic profiling": "Kraken2, Bracken and CoverM options.",
|
|
271
|
+
"read screening": "The `reads` route: cyanotoxin genes called straight off the QC'd reads with DIAMOND.",
|
|
272
|
+
"assembly": "The `assembly` and `both` routes: how the reads are assembled before they are screened.",
|
|
273
|
+
"BGC screening": "Detection of biosynthetic gene clusters in the assembled contigs.",
|
|
274
|
+
"reference databases": "Needed only by the steps named. Kraken2 and the genome database are "
|
|
275
|
+
"timon's own — installed once, from the databases card — and the gene "
|
|
276
|
+
"database ships with the pipeline, so none of them is asked for here.",
|
|
277
|
+
},
|
|
278
|
+
"params": [
|
|
279
|
+
{"id": "mode", "label": "screening mode", "type": "select", "default": "reads",
|
|
280
|
+
"enum": ["reads", "assembly", "both"], "group": "workflow",
|
|
281
|
+
"description": "Cyanotoxin screening route.",
|
|
282
|
+
"help_text": "`reads` runs `diamond blastx` on the QC reads (minutes). `assembly` assembles the reads "
|
|
283
|
+
"and screens the contigs for biosynthetic gene clusters (hours, high memory). `both` runs "
|
|
284
|
+
"the two routes on the same reads."},
|
|
285
|
+
{"id": "skip_qc", "label": "skip QC", "type": "bool", "default": False, "group": "workflow",
|
|
286
|
+
"description": "Skip the read QC steps (NanoPlot and Chopper)."},
|
|
287
|
+
{"id": "skip_nanoplot", "label": "skip Nanoplot", "type": "bool", "default": False, "group": "workflow",
|
|
288
|
+
"active_when": {"skip_qc": [False]},
|
|
289
|
+
"description": "Skip the NanoPlot quality assessment but still run Chopper."},
|
|
290
|
+
|
|
291
|
+
# Chopper is the whole of this group, so skipping QC empties it.
|
|
292
|
+
#
|
|
293
|
+
# Every number declares both bounds (tests/test_config.py insists).
|
|
294
|
+
# Where the pipeline sets none, "max" is timon's own ceiling on what
|
|
295
|
+
# a typo can send: generous past any real run, short of absurd.
|
|
296
|
+
{"id": "chopper_headcrop", "label": "headcrop", "type": "number", "default": 80, "min": 0, "max": 10000,
|
|
297
|
+
"group": "read QC", "active_when": {"skip_qc": [False]},
|
|
298
|
+
"description": "Bases trimmed from the start of each read."},
|
|
299
|
+
{"id": "chopper_tailcrop", "label": "tailcrop", "type": "number", "default": 50, "min": 0, "max": 10000,
|
|
300
|
+
"group": "read QC", "active_when": {"skip_qc": [False]},
|
|
301
|
+
"description": "Bases trimmed from the end of each read."},
|
|
302
|
+
{"id": "chopper_minlength", "label": "min. length", "type": "number", "default": 500, "min": 1, "max": 1000000,
|
|
303
|
+
"group": "read QC", "active_when": {"skip_qc": [False]},
|
|
304
|
+
"description": "Minimum read length kept after trimming."},
|
|
305
|
+
{"id": "chopper_minq", "label": "min. Q-score (Phred)", "type": "number", "default": 9, "min": 0, "max": 60,
|
|
306
|
+
"group": "read QC", "active_when": {"skip_qc": [False]},
|
|
307
|
+
"description": "Minimum average read quality kept."},
|
|
308
|
+
|
|
309
|
+
# A fraction the pipeline declares as a float: step "any" keeps
|
|
310
|
+
# 0.05 from being saved back as 0.
|
|
311
|
+
{"id": "kraken_confidence", "label": "Kraken2 confidence", "type": "number", "default": 0.0,
|
|
312
|
+
"min": 0, "max": 1, "step": "any", "group": "taxonomic profiling",
|
|
313
|
+
"description": "Kraken2 confidence score threshold, between 0 and 1.",
|
|
314
|
+
"help_text": "Fraction of a read's k-mers that must map to a taxon's clade for the read to be assigned "
|
|
315
|
+
"to it. `0` (the default) keeps Kraken2's own behaviour; higher values make classification "
|
|
316
|
+
"more conservative at the cost of sensitivity."},
|
|
317
|
+
# Only a length the installed Kraken2 index has a distribution
|
|
318
|
+
# for is a length Bracken can run with, so the options are read
|
|
319
|
+
# off the index (timon.paths.DATABASE_OPTIONS) and not written here.
|
|
320
|
+
{"id": "bracken_length", "label": "Bracken read length", "type": "select", "default": "300",
|
|
321
|
+
"options_from": "bracken_lengths", "group": "taxonomic profiling",
|
|
322
|
+
"description": "Read length Bracken was built for, also used as the SeqKit window and step size.",
|
|
323
|
+
"help_text": "Long reads are cut into non-overlapping windows of this length before Kraken2 "
|
|
324
|
+
"classification, so that the read lengths match the Bracken k-mer distribution."},
|
|
325
|
+
|
|
326
|
+
{"id": "diamond_blastx_id", "label": "blastx min. identity (%)", "type": "number", "default": 70,
|
|
327
|
+
"min": 0, "max": 100, "group": "read screening",
|
|
328
|
+
"active_when": {"mode": ["reads", "both"]},
|
|
329
|
+
"description": "Minimum percentage identity of the read-level `diamond blastx` alignments."},
|
|
330
|
+
# Both read by PLOT_GENE_DIAMOND_READS alone (conf/modules.config),
|
|
331
|
+
# so they go with the read route.
|
|
332
|
+
{"id": "diamond_min_aln_length", "label": "min. alignment length (aa)", "type": "number", "default": 25,
|
|
333
|
+
"min": 1, "max": 10000, "group": "read screening", "active_when": {"mode": ["reads", "both"]},
|
|
334
|
+
"description": "Minimum alignment length kept when summarising the DIAMOND hits.",
|
|
335
|
+
"help_text": "A short alignment over a conserved NRPS/PKS domain is not diagnostic on its own: 25 aa "
|
|
336
|
+
"is roughly the span of a single adenylation-domain core motif, shared across "
|
|
337
|
+
"essentially every NRPS."},
|
|
338
|
+
# A fraction the pipeline declares as a float, as kraken_confidence.
|
|
339
|
+
{"id": "diamond_range_overlap_frac", "label": "range overlap fraction", "type": "number", "default": 0.5,
|
|
340
|
+
"min": 0, "max": 1, "step": "any", "group": "read screening",
|
|
341
|
+
"active_when": {"mode": ["reads", "both"]},
|
|
342
|
+
"description": "Overlap at which two alignments on one read count as the same gene.",
|
|
343
|
+
"help_text": "One long read yields several alignments along different segments. Alignments "
|
|
344
|
+
"overlapping by at least this fraction of the shorter one compete for the same range "
|
|
345
|
+
"and only the best-scoring is counted; the rest count as separate gene occurrences."},
|
|
346
|
+
|
|
347
|
+
# The assembly and BGC groups belong to the contig route: the
|
|
348
|
+
# `reads` mode never assembles, so nothing in them is asked for.
|
|
349
|
+
{"id": "assembler", "label": "assembler", "type": "select", "default": "flye",
|
|
350
|
+
"enum": ["flye", "metamdbg"], "group": "assembly",
|
|
351
|
+
"active_when": {"mode": ["assembly", "both"]},
|
|
352
|
+
"description": "Assembler used for the contig-level route."},
|
|
353
|
+
# Each passed to its own assembler only (conf/modules.config), and
|
|
354
|
+
# typically starts with a dash — build_command joins such a value
|
|
355
|
+
# to its flag so nextflow does not read it as a flag of its own.
|
|
356
|
+
{"id": "flye_args", "label": "extra Flye arguments", "type": "text", "default": "",
|
|
357
|
+
"placeholder": "e.g. --iterations 2", "group": "assembly",
|
|
358
|
+
"active_when": {"mode": ["assembly", "both"], "assembler": ["flye"]},
|
|
359
|
+
"description": "Added to `flye`, after the `--meta` the pipeline always passes."},
|
|
360
|
+
{"id": "metamdbg_args", "label": "extra metaMDBG arguments", "type": "text", "default": "",
|
|
361
|
+
"placeholder": "e.g. --min-read-quality 12", "group": "assembly",
|
|
362
|
+
"active_when": {"mode": ["assembly", "both"], "assembler": ["metamdbg"]},
|
|
363
|
+
"description": "Added to `metaMDBG asm`."},
|
|
364
|
+
# No default of its own: unset, the heavy steps keep the
|
|
365
|
+
# process_high label's CPUs.
|
|
366
|
+
{"id": "assembly_cpus", "label": "assembly CPUs", "type": "number", "default": None, "min": 1, "max": 1024,
|
|
367
|
+
"group": "assembly", "active_when": {"mode": ["assembly", "both"]},
|
|
368
|
+
"description": "CPUs given to the assembly and antiSMASH steps. Leave empty for the pipeline's own.",
|
|
369
|
+
"help_text": "Overrides the `process_high` CPU default for Flye, metaMDBG and antiSMASH. Memory and "
|
|
370
|
+
"time still come from that label."},
|
|
371
|
+
{"id": "min_contig_length", "label": "min. contig length", "type": "number", "default": 1000, "min": 1, "max": 10000000,
|
|
372
|
+
"group": "assembly", "active_when": {"mode": ["assembly", "both"]},
|
|
373
|
+
"description": "Minimum contig length kept for screening, also passed to antiSMASH as `--minlength`."},
|
|
374
|
+
{"id": "coassemble_by_group", "label": "co-assemble by group", "type": "bool", "default": False,
|
|
375
|
+
"group": "assembly", "active_when": {"mode": ["assembly", "both"]},
|
|
376
|
+
"description": "Co-assemble all the samples of a group instead of one assembly per sample."},
|
|
377
|
+
|
|
378
|
+
{"id": "antismash_genefinding", "label": "antiSMASH gene finding", "type": "select", "default": "prodigal-m",
|
|
379
|
+
"enum": ["glimmerhmm", "prodigal", "prodigal-m", "none", "error"], "group": "BGC screening",
|
|
380
|
+
"active_when": {"mode": ["assembly", "both"]},
|
|
381
|
+
"description": "antiSMASH `--genefinding-tool`."},
|
|
382
|
+
# v0.1.0 took the BGC route down to antiSMASH alone: DIAMOND
|
|
383
|
+
# blastp, GECCO, DeepBGC, BiG-SCAPE and the merge between them are
|
|
384
|
+
# gone, and so are their parameters and databases.
|
|
385
|
+
|
|
386
|
+
# A database only the contig route reads. Asked for here rather
|
|
387
|
+
# than through reference_data because whether a run needs it follows
|
|
388
|
+
# from the mode, not from the pipeline choice, and an empty value
|
|
389
|
+
# is simply not passed on the command line. active_when and
|
|
390
|
+
# required_when say the same thing: it is asked for exactly when
|
|
391
|
+
# it is read.
|
|
392
|
+
{"id": "antismash_db", "label": "antiSMASH databases", "type": "text", "default": "",
|
|
393
|
+
"placeholder": "/path/to/antismash_db", "path": "any", "group": "reference databases",
|
|
394
|
+
"active_when": {"mode": ["assembly", "both"]},
|
|
395
|
+
"required_when": {"mode": ["assembly", "both"]},
|
|
396
|
+
"description": "Required with screening mode `assembly` or `both`.",
|
|
397
|
+
"help_text": "A directory or a `.tar.gz` tarball, created with `download-antismash-databases` "
|
|
398
|
+
"from the antiSMASH distribution."},
|
|
399
|
+
]
|
|
400
|
+
},
|
|
401
|
+
"mag-ont": {
|
|
402
|
+
"name": "mag-ont",
|
|
403
|
+
"description": "Automation of metagenome assembly and binning with support for nanopore reads",
|
|
404
|
+
"icon": "img/mag-icon.png",
|
|
405
|
+
"pipeline": "dsamoht/mag-ont",
|
|
406
|
+
# The nf-core template release. Bumping it is a timon release, and the
|
|
407
|
+
# defaults below have to be rechecked against that tag's
|
|
408
|
+
# nextflow_schema.json. The pipeline refuses a run that names
|
|
409
|
+
# --medaka_model or --skip_medaka with any assembler but flye, which
|
|
410
|
+
# is what the active_when on those two keeps off the command line.
|
|
411
|
+
"revision": "v1.4.0",
|
|
412
|
+
# Those of the tag's nextflow.config that timon supports (see
|
|
413
|
+
# model.nextflow.CONTAINER_BINARIES). The pipeline declares more —
|
|
414
|
+
# podman, shifter, charliecloud, wave, mamba — and "debug", "gpu" and
|
|
415
|
+
# "drac" besides, which are not container engines at all.
|
|
416
|
+
"profiles": ["docker", "singularity", "apptainer", "conda"],
|
|
417
|
+
"file_column": "long_reads",
|
|
418
|
+
"reference_data": ["gtdbtk_db"],
|
|
419
|
+
# The pipeline's own rule for gtdbtk_db: required unless --skip_gtdbtk
|
|
420
|
+
# is set. Demanding it from a run that skips the step it feeds would
|
|
421
|
+
# lock the user out of a configuration the pipeline accepts.
|
|
422
|
+
# --only_qc stops before binning, so nothing reads it then either.
|
|
423
|
+
"db_optional_when": {"gtdbtk_db": ["skip_gtdbtk", "skip_bin_qa", "only_qc"]},
|
|
424
|
+
# Matches assets/schema_input.json at this revision, which requires
|
|
425
|
+
# only the two identifiers and then either something to assemble or
|
|
426
|
+
# something already assembled. Short reads are an optional extra used
|
|
427
|
+
# for binning coverage, so a long-read-only sheet — the ordinary
|
|
428
|
+
# nanopore case — has to be accepted.
|
|
429
|
+
"columns": ["sample_id", "group", "assembly_fasta", "long_reads", "short_reads_1", "short_reads_2"],
|
|
430
|
+
"required_columns": ["sample_id", "group"],
|
|
431
|
+
"one_of_columns": [["assembly_fasta", "long_reads"]],
|
|
432
|
+
"param_groups": {
|
|
433
|
+
"read QC": "Quality control and filtering of the long reads.",
|
|
434
|
+
"assembly": "How the long reads are assembled, and — with Flye — polished.",
|
|
435
|
+
"binning": "Which binners run, and how they are configured.",
|
|
436
|
+
"bin QC & reporting": "Quality assessment and classification of the recovered bins, and the run report. "
|
|
437
|
+
"GTDB-Tk reads timon's own copy of its database; skipping it, or bin QA, "
|
|
438
|
+
"means the database is not needed at all.",
|
|
439
|
+
},
|
|
440
|
+
"params": [
|
|
441
|
+
# "skip QC" is the switch for the whole group: with it set, the
|
|
442
|
+
# thresholds and the per-tool skips below are all moot.
|
|
443
|
+
{"id": "chopper_minlength", "label": "min. length", "type": "number", "default": 1000, "min": 0, "max": 1000000,
|
|
444
|
+
"group": "read QC", "active_when": {"skip_qc": [False]},
|
|
445
|
+
"description": "Minimum read length kept by Chopper."},
|
|
446
|
+
{"id": "chopper_minq", "label": "min. Q-score (Phred)", "type": "number", "default": 10, "min": 0, "max": 60,
|
|
447
|
+
"group": "read QC", "active_when": {"skip_qc": [False]},
|
|
448
|
+
"description": "Minimum average read quality kept by Chopper."},
|
|
449
|
+
{"id": "skip_qc", "label": "skip QC", "type": "bool", "default": False, "group": "read QC",
|
|
450
|
+
"description": "Skip long read quality control entirely."},
|
|
451
|
+
# The switch for everything *after* QC. Not made to hang off
|
|
452
|
+
# "skip QC", though the pipeline refuses the two together: a
|
|
453
|
+
# condition that drops takes whatever hangs off it along, and
|
|
454
|
+
# every assembly and binning field hangs off this one — so
|
|
455
|
+
# ticking "skip QC" would silently take the rest of the run with
|
|
456
|
+
# it. The pipeline says so at startup instead.
|
|
457
|
+
{"id": "only_qc", "label": "only QC", "type": "bool", "default": False, "group": "read QC",
|
|
458
|
+
"description": "Stop after read QC: no assembly, binning or bin QA. Cannot be combined with skip QC.",
|
|
459
|
+
"help_text": "Runs NanoPlot, Porechop_ABI and Chopper, and publishes the QC'd reads and the MultiQC "
|
|
460
|
+
"report. Samples that come with their own assembly skip read QC, so nothing runs for them."},
|
|
461
|
+
{"id": "skip_nanoplot", "label": "skip Nanoplot", "type": "bool", "default": False, "group": "read QC",
|
|
462
|
+
"active_when": {"skip_qc": [False]},
|
|
463
|
+
"description": "Skip NanoPlot read quality reports."},
|
|
464
|
+
{"id": "skip_porechop", "label": "skip Porechop", "type": "bool", "default": False, "group": "read QC",
|
|
465
|
+
"active_when": {"skip_qc": [False]},
|
|
466
|
+
"description": "Skip Porechop_ABI adapter removal."},
|
|
467
|
+
|
|
468
|
+
{"id": "assembler", "label": "assembler", "type": "select", "default": "flye",
|
|
469
|
+
"enum": ["flye", "metamdbg"], "active_when": {"only_qc": [False]}, "group": "assembly",
|
|
470
|
+
"description": "Long read assembler to use."},
|
|
471
|
+
# Medaka polishes Flye assemblies only — metaMDBG produces a
|
|
472
|
+
# consensus of its own — and the pipeline refuses a run that names
|
|
473
|
+
# either Medaka parameter with any other assembler rather than
|
|
474
|
+
# ignoring it. So both leave the form with `flye`, which is also
|
|
475
|
+
# what keeps them off the command line. The model is asked for on
|
|
476
|
+
# top of that only by a run that actually polishes.
|
|
477
|
+
{"id": "medaka_model", "label": "Medaka model", "type": "text", "default": "r1041_e82_400bps_hac_v5.2.0",
|
|
478
|
+
"group": "assembly", "required": True,
|
|
479
|
+
"active_when": {"only_qc": [False], "assembler": ["flye"], "skip_medaka": [False]},
|
|
480
|
+
"description": "Medaka model used to polish the assembly.",
|
|
481
|
+
"help_text": "Only used once 'skip Medaka' is unticked. Must match the flow cell, kit and basecaller used to produce the reads. "
|
|
482
|
+
"Run `medaka tools list_models` to see the models available in the container."},
|
|
483
|
+
# Skipped by default from v1.4.0: Medaka is a haploid consensus
|
|
484
|
+
# model, and on a co-assembly it can rewrite a low-coverage genome
|
|
485
|
+
# with the sequence of a high-coverage relative.
|
|
486
|
+
{"id": "skip_medaka", "label": "skip Medaka", "type": "bool", "default": True, "group": "assembly",
|
|
487
|
+
"active_when": {"only_qc": [False], "assembler": ["flye"]},
|
|
488
|
+
"description": "Skip Medaka polishing of the assembly.",
|
|
489
|
+
"help_text": "On by default: Medaka is a haploid consensus model, and on a co-assembly it can "
|
|
490
|
+
"rewrite a low-coverage genome with the sequence of a high-coverage relative. Untick "
|
|
491
|
+
"to polish anyway, with a Medaka model that matches the basecaller."},
|
|
492
|
+
|
|
493
|
+
# Read by the MAXBIN process alone (conf/modules.config), so it
|
|
494
|
+
# goes with the binner.
|
|
495
|
+
{"id": "maxbin_minlen", "label": "MaxBin2 min. contig length", "type": "number", "default": 2500, "min": 0, "max": 10000000,
|
|
496
|
+
"group": "binning", "active_when": {"only_qc": [False], "skip_maxbin": [False]},
|
|
497
|
+
"description": "Minimum contig length considered by MaxBin2."},
|
|
498
|
+
# The single-contig MAG hold-out is decided by CheckM2, so the
|
|
499
|
+
# whole of it sits under "skip bin QA" in the pipeline (main.nf).
|
|
500
|
+
{"id": "sc_mag_minimum", "label": "single-contig MAG length", "type": "number", "default": 500000, "min": 0, "max": 100000000,
|
|
501
|
+
"group": "binning", "active_when": {"only_qc": [False], "skip_bin_qa": [False]},
|
|
502
|
+
"description": "Contigs at least this long are assessed on their own as candidate single-contig MAGs."},
|
|
503
|
+
# A percentage the pipeline declares as a float: step "any" keeps
|
|
504
|
+
# 92.5 from being saved back as 92.
|
|
505
|
+
{"id": "sc_mag_min_completeness", "label": "single-contig MAG min. completeness (%)", "type": "number",
|
|
506
|
+
"default": 90, "min": 0, "max": 100, "step": "any", "group": "binning",
|
|
507
|
+
"active_when": {"only_qc": [False], "skip_bin_qa": [False]},
|
|
508
|
+
"description": "CheckM2 completeness a long contig must reach to be held out of binning "
|
|
509
|
+
"and kept as a single-contig MAG."},
|
|
510
|
+
# Co-binning across groups. Its rules — one read type for the
|
|
511
|
+
# whole run, sample ids unique across it — are the pipeline's to
|
|
512
|
+
# enforce, and it does so before anything is submitted.
|
|
513
|
+
{"id": "binning_map_mode", "label": "coverage mapping", "type": "select", "default": "group",
|
|
514
|
+
"enum": ["group", "all"], "group": "binning", "active_when": {"only_qc": [False]},
|
|
515
|
+
"description": "`group` maps a group's own samples against its assembly; `all` maps every sample "
|
|
516
|
+
"of the run against every assembly.",
|
|
517
|
+
"help_text": "With `all`, each binning run sees one coverage column per sample of the run instead of "
|
|
518
|
+
"one per sample of the group, which gives the binners a differential coverage signal even "
|
|
519
|
+
"for a group holding a single sample. It costs one mapping job per assembly and per "
|
|
520
|
+
"sample, and every group of the run has to use the same read type."},
|
|
521
|
+
{"id": "skip_maxbin", "label": "skip MaxBin2", "type": "bool", "default": False, "active_when": {"only_qc": [False]}, "group": "binning",
|
|
522
|
+
"description": "Skip binning with MaxBin2."},
|
|
523
|
+
{"id": "skip_concoct", "label": "skip CONCOCT", "type": "bool", "default": False, "active_when": {"only_qc": [False]}, "group": "binning",
|
|
524
|
+
"description": "Skip binning with CONCOCT."},
|
|
525
|
+
{"id": "skip_semibin", "label": "skip SemiBin2", "type": "bool", "default": False, "active_when": {"only_qc": [False]}, "group": "binning",
|
|
526
|
+
"description": "Skip binning with SemiBin2."},
|
|
527
|
+
|
|
528
|
+
{"id": "skip_bin_qa", "label": "skip bin QA", "type": "bool", "default": False, "active_when": {"only_qc": [False]}, "group": "bin QC & reporting",
|
|
529
|
+
"description": "Skip CheckM2, GTDB-Tk and the MAG summary."},
|
|
530
|
+
# Under "skip bin QA", which already takes GTDB-Tk with it — and
|
|
531
|
+
# with it goes gtdbtk_db, through "db_optional_when" above.
|
|
532
|
+
{"id": "skip_gtdbtk", "label": "skip GTDB-Tk", "type": "bool", "default": False, "group": "bin QC & reporting",
|
|
533
|
+
"active_when": {"only_qc": [False], "skip_bin_qa": [False]},
|
|
534
|
+
"description": "Skip taxonomic classification with GTDB-Tk."},
|
|
535
|
+
{"id": "skip_multiqc", "label": "skip MultiQC", "type": "bool", "default": False, "group": "bin QC & reporting",
|
|
536
|
+
"description": "Skip the MultiQC report."},
|
|
537
|
+
]
|
|
538
|
+
},
|
|
539
|
+
"isolate-wf": {
|
|
540
|
+
"name": "isolate-wf",
|
|
541
|
+
"description": "Workflow for consensus isolate genome assembly",
|
|
542
|
+
"icon": "img/isolate-icon.png",
|
|
543
|
+
"pipeline": "dsamoht/isolate-wf",
|
|
544
|
+
# The repository is not published yet; runs are refused until it exists
|
|
545
|
+
# and a revision is pinned here. Until then it is not offered either:
|
|
546
|
+
# a card whose only answer is a refusal is worse than no card, and the
|
|
547
|
+
# entry is kept so that publishing the pipeline is a revision and a
|
|
548
|
+
# flag rather than a rewrite.
|
|
549
|
+
"revision": None,
|
|
550
|
+
"selectable": False,
|
|
551
|
+
"profiles": ["docker", "singularity", "apptainer"],
|
|
552
|
+
"file_column": "long_reads",
|
|
553
|
+
"reference_data": [],
|
|
554
|
+
"columns": ["sample_id", "long_reads", "short_reads_1", "short_reads_2"],
|
|
555
|
+
# Short reads are what makes the assembly a consensus one, but the
|
|
556
|
+
# pipeline also runs without them.
|
|
557
|
+
"required_columns": ["sample_id", "long_reads"],
|
|
558
|
+
"params": [
|
|
559
|
+
{"id": "skip_qc", "label": "skip QC", "type": "bool", "default": False},
|
|
560
|
+
{"id": "skip_porechop", "label": "skip Porechop", "type": "bool", "default": False,
|
|
561
|
+
"active_when": {"skip_qc": [False]}},
|
|
562
|
+
{"id": "chopper_minq", "label": "min. Q-score (Phred)", "type": "number", "default": 10, "min": 1, "max": 60,
|
|
563
|
+
"active_when": {"skip_qc": [False]}},
|
|
564
|
+
{"id": "chopper_minlength","label": "min. length", "type": "number", "default": 1000, "min": 0, "max": 10000,
|
|
565
|
+
"active_when": {"skip_qc": [False]}},
|
|
566
|
+
]
|
|
567
|
+
}
|
|
568
|
+
}
|