timon-gui 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. timon/__init__.py +7 -0
  2. timon/app/__init__.py +14 -0
  3. timon/app/config.py +568 -0
  4. timon/app/events.py +315 -0
  5. timon/app/model/__init__.py +35 -0
  6. timon/app/model/downloads.py +417 -0
  7. timon/app/model/experiment.py +476 -0
  8. timon/app/model/files.py +184 -0
  9. timon/app/model/history.py +279 -0
  10. timon/app/model/live.py +416 -0
  11. timon/app/model/nextflow.py +726 -0
  12. timon/app/model/nfstate.py +117 -0
  13. timon/app/model/params.py +247 -0
  14. timon/app/model/results.py +286 -0
  15. timon/app/model/samples.py +90 -0
  16. timon/app/model/validation.py +180 -0
  17. timon/app/presenters.py +716 -0
  18. timon/app/routes.py +446 -0
  19. timon/app/static/fonts/LICENSE-IBMPlexMono.txt +93 -0
  20. timon/app/static/fonts/LICENSE-Inter.txt +93 -0
  21. timon/app/static/fonts/ibm-plex-mono-400-latin-ext.woff2 +0 -0
  22. timon/app/static/fonts/ibm-plex-mono-400-latin.woff2 +0 -0
  23. timon/app/static/fonts/ibm-plex-mono-500-latin-ext.woff2 +0 -0
  24. timon/app/static/fonts/ibm-plex-mono-500-latin.woff2 +0 -0
  25. timon/app/static/fonts/ibm-plex-mono-600-latin-ext.woff2 +0 -0
  26. timon/app/static/fonts/ibm-plex-mono-600-latin.woff2 +0 -0
  27. timon/app/static/fonts/inter-300-700-latin-ext.woff2 +0 -0
  28. timon/app/static/fonts/inter-300-700-latin.woff2 +0 -0
  29. timon/app/static/img/bloom_orig.png +0 -0
  30. timon/app/static/img/isolate-icon.png +0 -0
  31. timon/app/static/img/mag-icon.png +0 -0
  32. timon/app/static/img/wheel1.svg +1 -0
  33. timon/app/static/img/wheel2.svg +1 -0
  34. timon/app/static/js/socket.io.js +6046 -0
  35. timon/app/static/style/app.css +2 -0
  36. timon/app/templates/_brand.html +20 -0
  37. timon/app/templates/_params.html +173 -0
  38. timon/app/templates/index.html +2604 -0
  39. timon/cli.py +132 -0
  40. timon/paths.py +197 -0
  41. timon_gui-0.1.0.dist-info/METADATA +76 -0
  42. timon_gui-0.1.0.dist-info/RECORD +45 -0
  43. timon_gui-0.1.0.dist-info/WHEEL +4 -0
  44. timon_gui-0.1.0.dist-info/entry_points.txt +2 -0
  45. timon_gui-0.1.0.dist-info/licenses/LICENSE +21 -0
timon/__init__.py ADDED
@@ -0,0 +1,7 @@
1
+ """timon — Toolkit of Integrated MicrobiOme analysis with support for loNg reads."""
2
+
3
+ __version__ = "0.1.0"
4
+ # The year __version__ was released, shown in the page footer. Bumped with the
5
+ # version, not read from the clock: an old install should keep saying the year
6
+ # it is from rather than quietly claiming to be current.
7
+ __release_year__ = "2026"
timon/app/__init__.py ADDED
@@ -0,0 +1,14 @@
1
+ from flask import Flask
2
+ from flask_socketio import SocketIO
3
+ from .config import Config
4
+
5
+ socketio = SocketIO()
6
+
7
+ def create_app():
8
+ app = Flask(__name__)
9
+ app.config.from_object(Config)
10
+ socketio.init_app(app)
11
+
12
+ with app.app_context():
13
+ from . import routes, events
14
+ return app
timon/app/config.py ADDED
@@ -0,0 +1,568 @@
1
+ import os
2
+ import secrets
3
+
4
+ class Config:
5
+ SECRET_KEY = os.getenv("TIMON_SECRET_KEY") or secrets.token_hex(32)
6
+ IMPORT_FOLDER = os.getenv("INPUT_DIR", "imports")
7
+ # Where runs write, which is deliberately not where the reads are: a
8
+ # folder called "imports" holding the results of an analysis is a folder
9
+ # nobody can read the layout of. Everything of timon's own that a run
10
+ # needs and nobody asked for — nextflow's work directory, the resource
11
+ # ceiling, the sample sheet — goes in here under a leading dot, so what
12
+ # the results view walks is the runs themselves.
13
+ OUTPUT_FOLDER = os.getenv("OUTPUT_DIR", "timon_results")
14
+ # Container engine Nextflow provisions tasks with. timon supports docker,
15
+ # singularity, apptainer and conda (model.nextflow.CONTAINER_BINARIES), and
16
+ # every pipeline declares which of those it supports; see "profiles" below.
17
+ PROFILE = os.getenv("TIMON_PROFILE", "docker")
18
+
19
+ # ── reference databases ──────────────────────────────────────────────────────
20
+ #
21
+ # Every database a pipeline reads is timon's to find: timon.paths.REFERENCE_DATA
22
+ # says where each one is (db_root(), unless its environment variable names a
23
+ # copy elsewhere), and a pipeline entry only lists which it reads under
24
+ # "reference_data". Nothing is asked for in the form — one database, one place,
25
+ # whichever pipeline reads it.
26
+ #
27
+ # A database timon cannot find is the commonest thing between a fresh install
28
+ # and a first run, and the fix is always the same act: fetch a large file and
29
+ # leave it where timon looks. So where each one can be got is declared here,
30
+ # and model/downloads.py is the one place that carries a download out.
31
+ #
32
+ # Keyed by the database, a key of timon.paths.REFERENCE_DATA. A database with
33
+ # no entry is still reported missing; it is simply not something the button can
34
+ # fetch.
35
+ #
36
+ # label what the page calls the download
37
+ # size how large it is. Someone deciding whether to start one at all
38
+ # is deciding on this number, so it is written by hand from the
39
+ # publisher's own page — a HEAD request would report it only
40
+ # after the user had committed to asking
41
+ # url what is fetched, over https
42
+ # archive True for a `.tar.gz` unpacked once it has landed; False for a
43
+ # file that is used exactly as it arrives
44
+ # install_as the name it takes under timon.paths.db_root(). It *must* be
45
+ # one of the names paths.py looks for, or the download would
46
+ # land somewhere timon never looks (tests/test_config.py)
47
+ # description one line: what is in it and what it is enough for
48
+ #
49
+ # Optional, and only where the publisher offers the same database in more than
50
+ # one build:
51
+ #
52
+ # variants the builds, each its own source. A variant carries "id" and
53
+ # whatever it changes — url, size, install_as, label, note — and
54
+ # inherits the rest from the entry around it, so what the builds
55
+ # share is written once. The page offers them as a choice and
56
+ # sends back an id; ``default`` names the one chosen for a user
57
+ # who expresses no preference, and the bundle button fetches
58
+ # that one
59
+ # default the id of that default variant
60
+ #
61
+ # A database published one way declares neither, and is its own single variant
62
+ # — model/downloads.py normalises both shapes to the same thing, so nothing
63
+ # past it has two to reason about.
64
+ DB_SOURCES = {
65
+ # As published at https://benlangmead.github.io/aws-indexes/k2. Dated
66
+ # builds: the date is part of the URL and of the name each is installed
67
+ # under, so a newer index is a new variant beside these rather than a
68
+ # silent change. Sizes in GiB, as the publisher's page gives them; both
69
+ # variants are kept on the same date, since which of the two someone
70
+ # picked should not also decide how current their index is.
71
+ "kraken_db": {
72
+ "label": "Kraken2 PlusPF",
73
+ "archive": True,
74
+ "description": "Archaea, bacteria, viruses, plasmids, human, vectors, "
75
+ "protozoa and fungi.",
76
+ # Two builds of the same collection, from the same date. The cap is
77
+ # how much memory Kraken2 needs to hold the index while it classifies,
78
+ # so this is a choice about the machine and not about the analysis: a
79
+ # laptop with 8 GB cannot run the larger one at all, and the smaller
80
+ # one is more heavily hashed, so it classifies a little less. Which
81
+ # was installed is not recorded anywhere and no run reads it — both
82
+ # land under a name paths.py looks for, and a Kraken2 index is a
83
+ # Kraken2 index to the pipeline.
84
+ "default": "16GB",
85
+ "variants": [
86
+ {
87
+ "id": "16GB",
88
+ "label": "Kraken2 PlusPF-16",
89
+ "size": "11.1 GB",
90
+ "url": "https://genome-idx.s3.amazonaws.com/kraken/k2_pluspf_16_GB_20260626.tar.gz",
91
+ "install_as": "k2_pluspf_16_GB_20260626",
92
+ "note": "needs 16 GB of memory to classify with",
93
+ },
94
+ {
95
+ "id": "8GB",
96
+ "label": "Kraken2 PlusPF-8",
97
+ "size": "5.5 GB",
98
+ "url": "https://genome-idx.s3.amazonaws.com/kraken/k2_pluspf_08_GB_20260626.tar.gz",
99
+ "install_as": "k2_pluspf_08_GB_20260626",
100
+ "note": "same organisms, hashed down to 8 GB — for a machine "
101
+ "that cannot spare 16",
102
+ },
103
+ ],
104
+ },
105
+ # One package, and the project publishes it as "latest" rather than per
106
+ # release: GTDB-Tk checks the data version it is given at startup, so the
107
+ # pipeline's own image decides what it will accept.
108
+ "gtdbtk_db": {
109
+ "label": "GTDB-Tk reference data",
110
+ "size": "~100 GB",
111
+ "url": "https://data.ace.uq.edu.au/public/gtdb/data/releases/latest/"
112
+ "auxillary_files/gtdbtk_package/full_package/gtdbtk_data.tar.gz",
113
+ "archive": True,
114
+ "install_as": "gtdbtk_data",
115
+ "description": "The full package, which is the only one GTDB-Tk classify "
116
+ "takes. Allow for a long download and twice the space "
117
+ "while it unpacks.",
118
+ },
119
+ # timon's own: 220 dereplicated cyanobacterial genomes, published on
120
+ # Zenodo under a record that names this exact set, so the URL is as fixed
121
+ # as a dated Kraken2 build. The tarball wraps them in one directory, which
122
+ # is unwrapped on install — CoverM is handed the directory and names each
123
+ # genome after its file, so what it reads is the .fna files themselves.
124
+ "genomes_db": {
125
+ "label": "cyanobacterial genome set",
126
+ "size": "278 MB",
127
+ "url": "https://zenodo.org/records/19522349/files/"
128
+ "cyanobacteriota_ncbi_dRep_n220.tar.gz",
129
+ "archive": True,
130
+ "install_as": "cyanobacteriota_ncbi_dRep_n220",
131
+ "description": "220 dereplicated cyanobacterial genomes, which CoverM "
132
+ "maps the reads against. Any set of FASTA files is a "
133
+ "valid answer, so TIMON_GENOMES_DB can name a smaller one.",
134
+ },
135
+ }
136
+
137
+ # What the **install databases** button fetches: installed once on a machine
138
+ # and read by every run after. A run that reads one of them cannot start until
139
+ # it is there, so the button is the first thing a fresh install offers and is
140
+ # frozen once there is nothing left for it to do.
141
+ #
142
+ # GTDB-Tk is deliberately not in it. At ~100 GB it would make every install
143
+ # pay for a step only mag-ont's bin QA takes, and a run that skips that step
144
+ # never reads it; it is offered on its own, and only to a run that needs it.
145
+ DATABASE_BUNDLE = ["kraken_db", "genomes_db"]
146
+
147
+ # ── linked pipelines ─────────────────────────────────────────────────────────
148
+ #
149
+ # Each entry links a Nextflow pipeline to timon. Required keys:
150
+ #
151
+ # name, description, icon what the workflow card shows
152
+ # pipeline GitHub "owner/repo", passed to `nextflow run`
153
+ # revision the release tag the run is pinned to — a tag, never
154
+ # a branch or a bare commit, so what a run used has a
155
+ # name upstream (tests/test_config.py holds that).
156
+ # None refuses to run (see model.nextflow.build_command)
157
+ # profiles container engines the pipeline supports, of those
158
+ # timon does (model.nextflow.CONTAINER_BINARIES): a
159
+ # pipeline declaring more is listed with the
160
+ # intersection, and a profile absent here is refused
161
+ # columns, file_column sample-sheet columns, and which one holds reads
162
+ # reference_data the databases the pipeline reads, by the keys of
163
+ # timon.paths.REFERENCE_DATA, each passed under its
164
+ # flag in model.nextflow.REFERENCE_FLAGS. Never
165
+ # asked for: timon knows where they are, and a run
166
+ # that reads a missing one cannot start. So they
167
+ # are never declared in "params" either — that
168
+ # would be a second source
169
+ # params the run-configuration form
170
+ #
171
+ # "params" is written by hand, and is a *selection*: a pipeline accepts far
172
+ # more than a user should have to look at, and everything left out simply
173
+ # keeps the pipeline's own default. Add a parameter here when a run actually
174
+ # turns on it. Each entry is:
175
+ #
176
+ # id the Nextflow parameter, passed as --<id>. Required.
177
+ # type "bool" | "number" | "text" | "select" (default: text)
178
+ # label what the form shows; defaults to the id
179
+ # default must match the pipeline's own default at the pinned
180
+ # revision, or the form will lie about what a run does.
181
+ # None, for a number, is a pipeline that sets no value of its
182
+ # own: the field may be left empty and nothing is passed
183
+ # description one line under the field
184
+ # help_text the part that actually decides a value; disclosed on demand
185
+ # min/max/step numbers only. step "any" means the value is a float, not an
186
+ # integer, and is cast as one (see model.params.coerce)
187
+ # enum selects only: the whole set of accepted values
188
+ # options_from selects only, instead of enum: the options are what an
189
+ # installed database holds, read by the reader of that name in
190
+ # timon.paths.DATABASE_OPTIONS. Until the database can be
191
+ # looked into, the default is the only option
192
+ # placeholder text fields only: hint shown while the field is empty
193
+ # path text fields only: the value is a filesystem path, so the
194
+ # field gets the same browse button the sample sheet's path
195
+ # columns have. "dir" picks a folder, "file" picks a file,
196
+ # "any" takes either — a database shipped as a directory or
197
+ # as a tarball
198
+ # required refuse to save the configuration without a value
199
+ # required_when {other param id: [values]} — required only while that other
200
+ # parameter holds one of those values (a checkbox is True)
201
+ # active_when {other param id: [values]} — part of the run only while
202
+ # *every* one of those parameters holds one of the listed
203
+ # values. Anything else drops out of the form, out of
204
+ # validation and off the command line: a QC threshold under
205
+ # "skip QC", the assembly options of a route this mode does
206
+ # not take. Conditions chain — a parameter whose controller
207
+ # has itself dropped out goes with it
208
+ # group section heading; blurbs come from "param_groups" below
209
+ #
210
+ # Optional keys:
211
+ #
212
+ # required_columns sample-sheet columns that must be filled on every row.
213
+ # Defaults to *all* of "columns". Declared only where the
214
+ # pipeline's own assets/schema_input.json asks for less:
215
+ # a column timon insists on that the pipeline treats as
216
+ # optional locks the user out of a sheet the pipeline
217
+ # would have accepted — a nanopore run with no short reads
218
+ # beside it, say
219
+ # one_of_columns groups of columns of which at least one must be filled on
220
+ # each row, for a schema's "anyOf" — a row needing either
221
+ # reads to assemble or an assembly already made
222
+ # param_groups {group: blurb} for the section headings used by "params"
223
+ # db_optional_when {db key: [flags]} — a database under "reference_data"
224
+ # that is only read while none of those flags is set: not
225
+ # passed, and not missing, for a run that skips its step
226
+ # test_profile the pipeline's own smoke-test profile, launched as
227
+ # `-profile <test_profile>,<engine>` by the quick test
228
+ # button. Declared only for a pipeline whose test profile
229
+ # fetches its own reads and its own databases: timon hands
230
+ # it no sample sheet, no parameters and no database, so a
231
+ # profile expecting any of them would fail at once. A
232
+ # pipeline without this key is offered no quick test
233
+ # selectable False leaves a declared pipeline out of the workflow
234
+ # picker, and refuses a switch to it. For one written down
235
+ # before it can be run — the entry stays, so nothing has to
236
+ # be reconstructed when it can
237
+ PIPELINES = {
238
+ "roshab-cli": {
239
+ "name": "roshab-cli",
240
+ "description": "Taxonomic classification and evaluation of cyanotoxin biosynthesis potential from nanopore reads",
241
+ "icon": "img/bloom_orig.png",
242
+ "pipeline": "dsamoht/roshab-cli",
243
+ # The first release. Bumping it is a timon release, and the defaults
244
+ # below have to be rechecked against that tag's nextflow_schema.json.
245
+ "revision": "v0.1.0",
246
+ "profiles": ["docker", "singularity", "apptainer"],
247
+ # conf/test.config at this revision names its own sample sheet and
248
+ # every database it reads — the viral Kraken2 index by URL, the CoverM
249
+ # genomes and the gene database from the pipeline itself — so
250
+ # the quick test needs nothing from the form. It checks the install
251
+ # rather than a configuration, and what it downloads is the same
252
+ # 569 MB index the roshab-cli example run asks the user for.
253
+ "test_profile": "test",
254
+ "file_column": "reads",
255
+ # Found by timon and installed by the install databases button — see
256
+ # timon.paths.REFERENCE_DATA. Not the gene database:
257
+ # from v0.1.0 the pipeline ships its own (assets/cyanotoxin_genes_
258
+ # mibig-4.0_two-class_v2.faa), whose sixth header field tells toxin
259
+ # genes from the rest, and the reports are written against that panel.
260
+ "reference_data": ["kraken_db", "genomes_db"],
261
+ # Matches assets/schema_input.json at this revision, which requires
262
+ # every one of them — so no "required_columns" here.
263
+ "columns": ["sample_id", "group", "info", "date", "reads"],
264
+ # The `--db_dir` install route is deliberately absent: setting it runs
265
+ # the database download *instead of* an analysis, which is not what
266
+ # this form builds.
267
+ "param_groups": {
268
+ "workflow": "Which screening route the run takes. Taxonomic profiling runs in every mode.",
269
+ "read QC": "Chopper trimming and filtering thresholds.",
270
+ "taxonomic profiling": "Kraken2, Bracken and CoverM options.",
271
+ "read screening": "The `reads` route: cyanotoxin genes called straight off the QC'd reads with DIAMOND.",
272
+ "assembly": "The `assembly` and `both` routes: how the reads are assembled before they are screened.",
273
+ "BGC screening": "Detection of biosynthetic gene clusters in the assembled contigs.",
274
+ "reference databases": "Needed only by the steps named. Kraken2 and the genome database are "
275
+ "timon's own — installed once, from the databases card — and the gene "
276
+ "database ships with the pipeline, so none of them is asked for here.",
277
+ },
278
+ "params": [
279
+ {"id": "mode", "label": "screening mode", "type": "select", "default": "reads",
280
+ "enum": ["reads", "assembly", "both"], "group": "workflow",
281
+ "description": "Cyanotoxin screening route.",
282
+ "help_text": "`reads` runs `diamond blastx` on the QC reads (minutes). `assembly` assembles the reads "
283
+ "and screens the contigs for biosynthetic gene clusters (hours, high memory). `both` runs "
284
+ "the two routes on the same reads."},
285
+ {"id": "skip_qc", "label": "skip QC", "type": "bool", "default": False, "group": "workflow",
286
+ "description": "Skip the read QC steps (NanoPlot and Chopper)."},
287
+ {"id": "skip_nanoplot", "label": "skip Nanoplot", "type": "bool", "default": False, "group": "workflow",
288
+ "active_when": {"skip_qc": [False]},
289
+ "description": "Skip the NanoPlot quality assessment but still run Chopper."},
290
+
291
+ # Chopper is the whole of this group, so skipping QC empties it.
292
+ #
293
+ # Every number declares both bounds (tests/test_config.py insists).
294
+ # Where the pipeline sets none, "max" is timon's own ceiling on what
295
+ # a typo can send: generous past any real run, short of absurd.
296
+ {"id": "chopper_headcrop", "label": "headcrop", "type": "number", "default": 80, "min": 0, "max": 10000,
297
+ "group": "read QC", "active_when": {"skip_qc": [False]},
298
+ "description": "Bases trimmed from the start of each read."},
299
+ {"id": "chopper_tailcrop", "label": "tailcrop", "type": "number", "default": 50, "min": 0, "max": 10000,
300
+ "group": "read QC", "active_when": {"skip_qc": [False]},
301
+ "description": "Bases trimmed from the end of each read."},
302
+ {"id": "chopper_minlength", "label": "min. length", "type": "number", "default": 500, "min": 1, "max": 1000000,
303
+ "group": "read QC", "active_when": {"skip_qc": [False]},
304
+ "description": "Minimum read length kept after trimming."},
305
+ {"id": "chopper_minq", "label": "min. Q-score (Phred)", "type": "number", "default": 9, "min": 0, "max": 60,
306
+ "group": "read QC", "active_when": {"skip_qc": [False]},
307
+ "description": "Minimum average read quality kept."},
308
+
309
+ # A fraction the pipeline declares as a float: step "any" keeps
310
+ # 0.05 from being saved back as 0.
311
+ {"id": "kraken_confidence", "label": "Kraken2 confidence", "type": "number", "default": 0.0,
312
+ "min": 0, "max": 1, "step": "any", "group": "taxonomic profiling",
313
+ "description": "Kraken2 confidence score threshold, between 0 and 1.",
314
+ "help_text": "Fraction of a read's k-mers that must map to a taxon's clade for the read to be assigned "
315
+ "to it. `0` (the default) keeps Kraken2's own behaviour; higher values make classification "
316
+ "more conservative at the cost of sensitivity."},
317
+ # Only a length the installed Kraken2 index has a distribution
318
+ # for is a length Bracken can run with, so the options are read
319
+ # off the index (timon.paths.DATABASE_OPTIONS) and not written here.
320
+ {"id": "bracken_length", "label": "Bracken read length", "type": "select", "default": "300",
321
+ "options_from": "bracken_lengths", "group": "taxonomic profiling",
322
+ "description": "Read length Bracken was built for, also used as the SeqKit window and step size.",
323
+ "help_text": "Long reads are cut into non-overlapping windows of this length before Kraken2 "
324
+ "classification, so that the read lengths match the Bracken k-mer distribution."},
325
+
326
+ {"id": "diamond_blastx_id", "label": "blastx min. identity (%)", "type": "number", "default": 70,
327
+ "min": 0, "max": 100, "group": "read screening",
328
+ "active_when": {"mode": ["reads", "both"]},
329
+ "description": "Minimum percentage identity of the read-level `diamond blastx` alignments."},
330
+ # Both read by PLOT_GENE_DIAMOND_READS alone (conf/modules.config),
331
+ # so they go with the read route.
332
+ {"id": "diamond_min_aln_length", "label": "min. alignment length (aa)", "type": "number", "default": 25,
333
+ "min": 1, "max": 10000, "group": "read screening", "active_when": {"mode": ["reads", "both"]},
334
+ "description": "Minimum alignment length kept when summarising the DIAMOND hits.",
335
+ "help_text": "A short alignment over a conserved NRPS/PKS domain is not diagnostic on its own: 25 aa "
336
+ "is roughly the span of a single adenylation-domain core motif, shared across "
337
+ "essentially every NRPS."},
338
+ # A fraction the pipeline declares as a float, as kraken_confidence.
339
+ {"id": "diamond_range_overlap_frac", "label": "range overlap fraction", "type": "number", "default": 0.5,
340
+ "min": 0, "max": 1, "step": "any", "group": "read screening",
341
+ "active_when": {"mode": ["reads", "both"]},
342
+ "description": "Overlap at which two alignments on one read count as the same gene.",
343
+ "help_text": "One long read yields several alignments along different segments. Alignments "
344
+ "overlapping by at least this fraction of the shorter one compete for the same range "
345
+ "and only the best-scoring is counted; the rest count as separate gene occurrences."},
346
+
347
+ # The assembly and BGC groups belong to the contig route: the
348
+ # `reads` mode never assembles, so nothing in them is asked for.
349
+ {"id": "assembler", "label": "assembler", "type": "select", "default": "flye",
350
+ "enum": ["flye", "metamdbg"], "group": "assembly",
351
+ "active_when": {"mode": ["assembly", "both"]},
352
+ "description": "Assembler used for the contig-level route."},
353
+ # Each passed to its own assembler only (conf/modules.config), and
354
+ # typically starts with a dash — build_command joins such a value
355
+ # to its flag so nextflow does not read it as a flag of its own.
356
+ {"id": "flye_args", "label": "extra Flye arguments", "type": "text", "default": "",
357
+ "placeholder": "e.g. --iterations 2", "group": "assembly",
358
+ "active_when": {"mode": ["assembly", "both"], "assembler": ["flye"]},
359
+ "description": "Added to `flye`, after the `--meta` the pipeline always passes."},
360
+ {"id": "metamdbg_args", "label": "extra metaMDBG arguments", "type": "text", "default": "",
361
+ "placeholder": "e.g. --min-read-quality 12", "group": "assembly",
362
+ "active_when": {"mode": ["assembly", "both"], "assembler": ["metamdbg"]},
363
+ "description": "Added to `metaMDBG asm`."},
364
+ # No default of its own: unset, the heavy steps keep the
365
+ # process_high label's CPUs.
366
+ {"id": "assembly_cpus", "label": "assembly CPUs", "type": "number", "default": None, "min": 1, "max": 1024,
367
+ "group": "assembly", "active_when": {"mode": ["assembly", "both"]},
368
+ "description": "CPUs given to the assembly and antiSMASH steps. Leave empty for the pipeline's own.",
369
+ "help_text": "Overrides the `process_high` CPU default for Flye, metaMDBG and antiSMASH. Memory and "
370
+ "time still come from that label."},
371
+ {"id": "min_contig_length", "label": "min. contig length", "type": "number", "default": 1000, "min": 1, "max": 10000000,
372
+ "group": "assembly", "active_when": {"mode": ["assembly", "both"]},
373
+ "description": "Minimum contig length kept for screening, also passed to antiSMASH as `--minlength`."},
374
+ {"id": "coassemble_by_group", "label": "co-assemble by group", "type": "bool", "default": False,
375
+ "group": "assembly", "active_when": {"mode": ["assembly", "both"]},
376
+ "description": "Co-assemble all the samples of a group instead of one assembly per sample."},
377
+
378
+ {"id": "antismash_genefinding", "label": "antiSMASH gene finding", "type": "select", "default": "prodigal-m",
379
+ "enum": ["glimmerhmm", "prodigal", "prodigal-m", "none", "error"], "group": "BGC screening",
380
+ "active_when": {"mode": ["assembly", "both"]},
381
+ "description": "antiSMASH `--genefinding-tool`."},
382
+ # v0.1.0 took the BGC route down to antiSMASH alone: DIAMOND
383
+ # blastp, GECCO, DeepBGC, BiG-SCAPE and the merge between them are
384
+ # gone, and so are their parameters and databases.
385
+
386
+ # A database only the contig route reads. Asked for here rather
387
+ # than through reference_data because whether a run needs it follows
388
+ # from the mode, not from the pipeline choice, and an empty value
389
+ # is simply not passed on the command line. active_when and
390
+ # required_when say the same thing: it is asked for exactly when
391
+ # it is read.
392
+ {"id": "antismash_db", "label": "antiSMASH databases", "type": "text", "default": "",
393
+ "placeholder": "/path/to/antismash_db", "path": "any", "group": "reference databases",
394
+ "active_when": {"mode": ["assembly", "both"]},
395
+ "required_when": {"mode": ["assembly", "both"]},
396
+ "description": "Required with screening mode `assembly` or `both`.",
397
+ "help_text": "A directory or a `.tar.gz` tarball, created with `download-antismash-databases` "
398
+ "from the antiSMASH distribution."},
399
+ ]
400
+ },
401
+ "mag-ont": {
402
+ "name": "mag-ont",
403
+ "description": "Automation of metagenome assembly and binning with support for nanopore reads",
404
+ "icon": "img/mag-icon.png",
405
+ "pipeline": "dsamoht/mag-ont",
406
+ # The nf-core template release. Bumping it is a timon release, and the
407
+ # defaults below have to be rechecked against that tag's
408
+ # nextflow_schema.json. The pipeline refuses a run that names
409
+ # --medaka_model or --skip_medaka with any assembler but flye, which
410
+ # is what the active_when on those two keeps off the command line.
411
+ "revision": "v1.4.0",
412
+ # Those of the tag's nextflow.config that timon supports (see
413
+ # model.nextflow.CONTAINER_BINARIES). The pipeline declares more —
414
+ # podman, shifter, charliecloud, wave, mamba — and "debug", "gpu" and
415
+ # "drac" besides, which are not container engines at all.
416
+ "profiles": ["docker", "singularity", "apptainer", "conda"],
417
+ "file_column": "long_reads",
418
+ "reference_data": ["gtdbtk_db"],
419
+ # The pipeline's own rule for gtdbtk_db: required unless --skip_gtdbtk
420
+ # is set. Demanding it from a run that skips the step it feeds would
421
+ # lock the user out of a configuration the pipeline accepts.
422
+ # --only_qc stops before binning, so nothing reads it then either.
423
+ "db_optional_when": {"gtdbtk_db": ["skip_gtdbtk", "skip_bin_qa", "only_qc"]},
424
+ # Matches assets/schema_input.json at this revision, which requires
425
+ # only the two identifiers and then either something to assemble or
426
+ # something already assembled. Short reads are an optional extra used
427
+ # for binning coverage, so a long-read-only sheet — the ordinary
428
+ # nanopore case — has to be accepted.
429
+ "columns": ["sample_id", "group", "assembly_fasta", "long_reads", "short_reads_1", "short_reads_2"],
430
+ "required_columns": ["sample_id", "group"],
431
+ "one_of_columns": [["assembly_fasta", "long_reads"]],
432
+ "param_groups": {
433
+ "read QC": "Quality control and filtering of the long reads.",
434
+ "assembly": "How the long reads are assembled, and — with Flye — polished.",
435
+ "binning": "Which binners run, and how they are configured.",
436
+ "bin QC & reporting": "Quality assessment and classification of the recovered bins, and the run report. "
437
+ "GTDB-Tk reads timon's own copy of its database; skipping it, or bin QA, "
438
+ "means the database is not needed at all.",
439
+ },
440
+ "params": [
441
+ # "skip QC" is the switch for the whole group: with it set, the
442
+ # thresholds and the per-tool skips below are all moot.
443
+ {"id": "chopper_minlength", "label": "min. length", "type": "number", "default": 1000, "min": 0, "max": 1000000,
444
+ "group": "read QC", "active_when": {"skip_qc": [False]},
445
+ "description": "Minimum read length kept by Chopper."},
446
+ {"id": "chopper_minq", "label": "min. Q-score (Phred)", "type": "number", "default": 10, "min": 0, "max": 60,
447
+ "group": "read QC", "active_when": {"skip_qc": [False]},
448
+ "description": "Minimum average read quality kept by Chopper."},
449
+ {"id": "skip_qc", "label": "skip QC", "type": "bool", "default": False, "group": "read QC",
450
+ "description": "Skip long read quality control entirely."},
451
+ # The switch for everything *after* QC. Not made to hang off
452
+ # "skip QC", though the pipeline refuses the two together: a
453
+ # condition that drops takes whatever hangs off it along, and
454
+ # every assembly and binning field hangs off this one — so
455
+ # ticking "skip QC" would silently take the rest of the run with
456
+ # it. The pipeline says so at startup instead.
457
+ {"id": "only_qc", "label": "only QC", "type": "bool", "default": False, "group": "read QC",
458
+ "description": "Stop after read QC: no assembly, binning or bin QA. Cannot be combined with skip QC.",
459
+ "help_text": "Runs NanoPlot, Porechop_ABI and Chopper, and publishes the QC'd reads and the MultiQC "
460
+ "report. Samples that come with their own assembly skip read QC, so nothing runs for them."},
461
+ {"id": "skip_nanoplot", "label": "skip Nanoplot", "type": "bool", "default": False, "group": "read QC",
462
+ "active_when": {"skip_qc": [False]},
463
+ "description": "Skip NanoPlot read quality reports."},
464
+ {"id": "skip_porechop", "label": "skip Porechop", "type": "bool", "default": False, "group": "read QC",
465
+ "active_when": {"skip_qc": [False]},
466
+ "description": "Skip Porechop_ABI adapter removal."},
467
+
468
+ {"id": "assembler", "label": "assembler", "type": "select", "default": "flye",
469
+ "enum": ["flye", "metamdbg"], "active_when": {"only_qc": [False]}, "group": "assembly",
470
+ "description": "Long read assembler to use."},
471
+ # Medaka polishes Flye assemblies only — metaMDBG produces a
472
+ # consensus of its own — and the pipeline refuses a run that names
473
+ # either Medaka parameter with any other assembler rather than
474
+ # ignoring it. So both leave the form with `flye`, which is also
475
+ # what keeps them off the command line. The model is asked for on
476
+ # top of that only by a run that actually polishes.
477
+ {"id": "medaka_model", "label": "Medaka model", "type": "text", "default": "r1041_e82_400bps_hac_v5.2.0",
478
+ "group": "assembly", "required": True,
479
+ "active_when": {"only_qc": [False], "assembler": ["flye"], "skip_medaka": [False]},
480
+ "description": "Medaka model used to polish the assembly.",
481
+ "help_text": "Only used once 'skip Medaka' is unticked. Must match the flow cell, kit and basecaller used to produce the reads. "
482
+ "Run `medaka tools list_models` to see the models available in the container."},
483
+ # Skipped by default from v1.4.0: Medaka is a haploid consensus
484
+ # model, and on a co-assembly it can rewrite a low-coverage genome
485
+ # with the sequence of a high-coverage relative.
486
+ {"id": "skip_medaka", "label": "skip Medaka", "type": "bool", "default": True, "group": "assembly",
487
+ "active_when": {"only_qc": [False], "assembler": ["flye"]},
488
+ "description": "Skip Medaka polishing of the assembly.",
489
+ "help_text": "On by default: Medaka is a haploid consensus model, and on a co-assembly it can "
490
+ "rewrite a low-coverage genome with the sequence of a high-coverage relative. Untick "
491
+ "to polish anyway, with a Medaka model that matches the basecaller."},
492
+
493
+ # Read by the MAXBIN process alone (conf/modules.config), so it
494
+ # goes with the binner.
495
+ {"id": "maxbin_minlen", "label": "MaxBin2 min. contig length", "type": "number", "default": 2500, "min": 0, "max": 10000000,
496
+ "group": "binning", "active_when": {"only_qc": [False], "skip_maxbin": [False]},
497
+ "description": "Minimum contig length considered by MaxBin2."},
498
+ # The single-contig MAG hold-out is decided by CheckM2, so the
499
+ # whole of it sits under "skip bin QA" in the pipeline (main.nf).
500
+ {"id": "sc_mag_minimum", "label": "single-contig MAG length", "type": "number", "default": 500000, "min": 0, "max": 100000000,
501
+ "group": "binning", "active_when": {"only_qc": [False], "skip_bin_qa": [False]},
502
+ "description": "Contigs at least this long are assessed on their own as candidate single-contig MAGs."},
503
+ # A percentage the pipeline declares as a float: step "any" keeps
504
+ # 92.5 from being saved back as 92.
505
+ {"id": "sc_mag_min_completeness", "label": "single-contig MAG min. completeness (%)", "type": "number",
506
+ "default": 90, "min": 0, "max": 100, "step": "any", "group": "binning",
507
+ "active_when": {"only_qc": [False], "skip_bin_qa": [False]},
508
+ "description": "CheckM2 completeness a long contig must reach to be held out of binning "
509
+ "and kept as a single-contig MAG."},
510
+ # Co-binning across groups. Its rules — one read type for the
511
+ # whole run, sample ids unique across it — are the pipeline's to
512
+ # enforce, and it does so before anything is submitted.
513
+ {"id": "binning_map_mode", "label": "coverage mapping", "type": "select", "default": "group",
514
+ "enum": ["group", "all"], "group": "binning", "active_when": {"only_qc": [False]},
515
+ "description": "`group` maps a group's own samples against its assembly; `all` maps every sample "
516
+ "of the run against every assembly.",
517
+ "help_text": "With `all`, each binning run sees one coverage column per sample of the run instead of "
518
+ "one per sample of the group, which gives the binners a differential coverage signal even "
519
+ "for a group holding a single sample. It costs one mapping job per assembly and per "
520
+ "sample, and every group of the run has to use the same read type."},
521
+ {"id": "skip_maxbin", "label": "skip MaxBin2", "type": "bool", "default": False, "active_when": {"only_qc": [False]}, "group": "binning",
522
+ "description": "Skip binning with MaxBin2."},
523
+ {"id": "skip_concoct", "label": "skip CONCOCT", "type": "bool", "default": False, "active_when": {"only_qc": [False]}, "group": "binning",
524
+ "description": "Skip binning with CONCOCT."},
525
+ {"id": "skip_semibin", "label": "skip SemiBin2", "type": "bool", "default": False, "active_when": {"only_qc": [False]}, "group": "binning",
526
+ "description": "Skip binning with SemiBin2."},
527
+
528
+ {"id": "skip_bin_qa", "label": "skip bin QA", "type": "bool", "default": False, "active_when": {"only_qc": [False]}, "group": "bin QC & reporting",
529
+ "description": "Skip CheckM2, GTDB-Tk and the MAG summary."},
530
+ # Under "skip bin QA", which already takes GTDB-Tk with it — and
531
+ # with it goes gtdbtk_db, through "db_optional_when" above.
532
+ {"id": "skip_gtdbtk", "label": "skip GTDB-Tk", "type": "bool", "default": False, "group": "bin QC & reporting",
533
+ "active_when": {"only_qc": [False], "skip_bin_qa": [False]},
534
+ "description": "Skip taxonomic classification with GTDB-Tk."},
535
+ {"id": "skip_multiqc", "label": "skip MultiQC", "type": "bool", "default": False, "group": "bin QC & reporting",
536
+ "description": "Skip the MultiQC report."},
537
+ ]
538
+ },
539
+ "isolate-wf": {
540
+ "name": "isolate-wf",
541
+ "description": "Workflow for consensus isolate genome assembly",
542
+ "icon": "img/isolate-icon.png",
543
+ "pipeline": "dsamoht/isolate-wf",
544
+ # The repository is not published yet; runs are refused until it exists
545
+ # and a revision is pinned here. Until then it is not offered either:
546
+ # a card whose only answer is a refusal is worse than no card, and the
547
+ # entry is kept so that publishing the pipeline is a revision and a
548
+ # flag rather than a rewrite.
549
+ "revision": None,
550
+ "selectable": False,
551
+ "profiles": ["docker", "singularity", "apptainer"],
552
+ "file_column": "long_reads",
553
+ "reference_data": [],
554
+ "columns": ["sample_id", "long_reads", "short_reads_1", "short_reads_2"],
555
+ # Short reads are what makes the assembly a consensus one, but the
556
+ # pipeline also runs without them.
557
+ "required_columns": ["sample_id", "long_reads"],
558
+ "params": [
559
+ {"id": "skip_qc", "label": "skip QC", "type": "bool", "default": False},
560
+ {"id": "skip_porechop", "label": "skip Porechop", "type": "bool", "default": False,
561
+ "active_when": {"skip_qc": [False]}},
562
+ {"id": "chopper_minq", "label": "min. Q-score (Phred)", "type": "number", "default": 10, "min": 1, "max": 60,
563
+ "active_when": {"skip_qc": [False]}},
564
+ {"id": "chopper_minlength","label": "min. length", "type": "number", "default": 1000, "min": 0, "max": 10000,
565
+ "active_when": {"skip_qc": [False]}},
566
+ ]
567
+ }
568
+ }