salmopredict 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {salmopredict-0.2.0 → salmopredict-0.3.0}/MANIFEST.in +1 -0
- {salmopredict-0.2.0/salmopredict.egg-info → salmopredict-0.3.0}/PKG-INFO +100 -14
- {salmopredict-0.2.0 → salmopredict-0.3.0}/README.md +97 -12
- {salmopredict-0.2.0 → salmopredict-0.3.0}/environment.yml +1 -5
- salmopredict-0.3.0/examples/README.md +83 -0
- salmopredict-0.3.0/examples/example_gene_frequencies.csv +3 -0
- salmopredict-0.3.0/examples/example_samples.csv +11 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/pyproject.toml +4 -1
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/__init__.py +1 -1
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/cli.py +15 -4
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/config.py +12 -0
- salmopredict-0.3.0/salmopredict/core/frequencies.py +106 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/gui/app.py +67 -14
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/pipeline.py +31 -5
- {salmopredict-0.2.0 → salmopredict-0.3.0/salmopredict.egg-info}/PKG-INFO +100 -14
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict.egg-info/SOURCES.txt +6 -1
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict.egg-info/requires.txt +2 -1
- salmopredict-0.3.0/tests/test_two_table_input.py +144 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/LICENSE +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/examples/example_features.csv +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/examples/example_meta.csv +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/examples/example_with_sample.csv +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/core/__init__.py +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/core/align.py +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/core/io_tables.py +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/core/modelinfo.py +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/core/predict.py +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/gui/__init__.py +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/gui/assets/cfsa_logo.png +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/gui/assets/salmopredict_icon.png +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/gui/assets/vphs_logo.png +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/.DS_Store +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/learner.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/metadata.json +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F1/model-internals.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F1/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F2/model-internals.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F2/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F3/model-internals.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F3/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F4/model-internals.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F4/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F5/model-internals.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F5/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F1/model-internals.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F1/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F2/model-internals.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F2/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F3/model-internals.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F3/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F4/model-internals.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F4/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F5/model-internals.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F5/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F1/model-internals.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F1/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F2/model-internals.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F2/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F3/model-internals.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F3/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F4/model-internals.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F4/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F5/model-internals.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F5/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r121_BAG_L1/S1F1/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r121_BAG_L1/S1F2/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r121_BAG_L1/S1F3/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r121_BAG_L1/S1F4/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r121_BAG_L1/S1F5/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r121_BAG_L1/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r1_BAG_L1/S1F1/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r1_BAG_L1/S1F2/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r1_BAG_L1/S1F3/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r1_BAG_L1/S1F4/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r1_BAG_L1/S1F5/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r1_BAG_L1/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r30_BAG_L1/S1F1/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r30_BAG_L1/S1F2/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r30_BAG_L1/S1F3/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r30_BAG_L1/S1F4/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r30_BAG_L1/S1F5/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r30_BAG_L1/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r79_BAG_L1/S1F1/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r79_BAG_L1/S1F2/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r79_BAG_L1/S1F3/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r79_BAG_L1/S1F4/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r79_BAG_L1/S1F5/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r79_BAG_L1/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/WeightedEnsemble_L2/model.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/trainer.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/predictor.pkl +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/version.txt +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict.egg-info/dependency_links.txt +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict.egg-info/entry_points.txt +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict.egg-info/top_level.txt +0 -0
- {salmopredict-0.2.0 → salmopredict-0.3.0}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: salmopredict
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: AutoGluon-based Incidence predictor for Salmonella virulence-factor gene-frequency features
|
|
5
5
|
Author-email: Dongyan Shao <563608176@qq.com>
|
|
6
6
|
License: PolyForm-Noncommercial-1.0.0
|
|
@@ -14,7 +14,8 @@ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
|
14
14
|
Requires-Python: <3.11,>=3.10
|
|
15
15
|
Description-Content-Type: text/markdown
|
|
16
16
|
License-File: LICENSE
|
|
17
|
-
Requires-Dist: autogluon.tabular==1.1.1
|
|
17
|
+
Requires-Dist: autogluon.tabular[fastai]==1.1.1
|
|
18
|
+
Requires-Dist: setuptools<81
|
|
18
19
|
Requires-Dist: pandas>=2.0
|
|
19
20
|
Requires-Dist: openpyxl>=3.0
|
|
20
21
|
Requires-Dist: rich-argparse>=1.4
|
|
@@ -40,8 +41,8 @@ single prediction file. It reproduces the alignment used by the original
|
|
|
40
41
|
(`/` and `-` become `.`), genes the model expects but the input lacks are filled
|
|
41
42
|
with `0` (a missing gene means frequency 0), and extra input columns are ignored.
|
|
42
43
|
|
|
43
|
-
**
|
|
44
|
-
|
|
44
|
+
**Feature CSV input/output** — the output columns depend on whether the input
|
|
45
|
+
has a `Sample` column:
|
|
45
46
|
|
|
46
47
|
| Input | Output columns |
|
|
47
48
|
|-------|----------------|
|
|
@@ -58,36 +59,78 @@ salmopredict runs on **Python 3.10** and loads its model with **AutoGluon
|
|
|
58
59
|
1.1.1** — both are hard requirements, because the model is pickled with that
|
|
59
60
|
exact stack.
|
|
60
61
|
|
|
61
|
-
**
|
|
62
|
+
**New installation from this source directory.** Run these commands from the
|
|
63
|
+
directory containing `pyproject.toml`:
|
|
62
64
|
|
|
63
65
|
```bash
|
|
64
|
-
|
|
66
|
+
conda create -n salmopredict python=3.10
|
|
67
|
+
conda activate salmopredict
|
|
68
|
+
python -m pip install .
|
|
69
|
+
salmopredict check
|
|
65
70
|
```
|
|
66
71
|
|
|
67
|
-
This
|
|
68
|
-
|
|
72
|
+
This installs AutoGluon 1.1.1 with its **Torch and FastAI backends**, compatible
|
|
73
|
+
`setuptools<81`, the Streamlit GUI, and the bundled prediction model. The
|
|
74
|
+
backends are required by the bundled ensemble; base `autogluon.tabular` alone
|
|
75
|
+
does not install them. AutoGluon 1.1.1 also needs `pkg_resources`, which newer
|
|
76
|
+
setuptools releases no longer provide.
|
|
77
|
+
|
|
78
|
+
Both interfaces are then available:
|
|
69
79
|
|
|
70
80
|
```bash
|
|
71
81
|
salmopredict run -i features.csv -o results/ # command line
|
|
72
82
|
salmopredict gui # browser GUI
|
|
73
83
|
```
|
|
74
84
|
|
|
75
|
-
|
|
76
|
-
`conda create -n salmopredict python=3.10 && conda activate salmopredict`.
|
|
77
|
-
|
|
78
|
-
**Reproducible environment (from a clone).** Pins Python 3.10 and installs
|
|
85
|
+
**Conda environment (alternative, from this source directory).** Pins Python 3.10 and installs
|
|
79
86
|
AutoGluon via pip inside the env (conda-installed AutoGluon does not resolve
|
|
80
87
|
cleanly for this project):
|
|
81
88
|
|
|
82
89
|
```bash
|
|
83
90
|
conda env create -f environment.yml
|
|
84
91
|
conda activate salmopredict
|
|
92
|
+
salmopredict check
|
|
85
93
|
```
|
|
86
94
|
|
|
87
95
|
**Editable / development install (from a clone).**
|
|
88
96
|
|
|
89
97
|
```bash
|
|
90
|
-
pip install -e . # installs the CLI and the Streamlit GUI
|
|
98
|
+
python -m pip install -e . # installs the CLI and the Streamlit GUI
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
**Install the updated local wheel.** In a Python 3.10 environment:
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
python -m pip install --upgrade dist/salmopredict-0.3.0-py3-none-any.whl
|
|
105
|
+
salmopredict check
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
**Update an existing source installation.** Stop a running GUI with `Ctrl+C`,
|
|
109
|
+
then run from this source directory:
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
conda activate salmopredict
|
|
113
|
+
python -m pip install --upgrade -e .
|
|
114
|
+
salmopredict check
|
|
115
|
+
salmopredict gui
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
**PyPI installation / upgrade.** In a Python 3.10 environment:
|
|
119
|
+
|
|
120
|
+
```bash
|
|
121
|
+
python -m pip install --upgrade "salmopredict>=0.3.0"
|
|
122
|
+
salmopredict check
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
Version 0.3.0 includes two-table prediction and installs the required Torch,
|
|
126
|
+
FastAI and compatible setuptools dependencies automatically.
|
|
127
|
+
|
|
128
|
+
If the page opens but prediction reports `No module named 'pkg_resources'`,
|
|
129
|
+
`torch`, or `fastai`, use the update/repair command above and restart the GUI.
|
|
130
|
+
To verify actual prediction from a source checkout (use a new output folder):
|
|
131
|
+
|
|
132
|
+
```bash
|
|
133
|
+
salmopredict run -i examples/example_features.csv -o results_install_check/
|
|
91
134
|
```
|
|
92
135
|
|
|
93
136
|
## The model
|
|
@@ -130,12 +173,55 @@ salmopredict gui
|
|
|
130
173
|
salmopredict check --model /path/to/model
|
|
131
174
|
```
|
|
132
175
|
|
|
133
|
-
Each run writes one `pred_<input-stem>.csv` to the output directory; the
|
|
176
|
+
Each feature-input run writes one `pred_<input-stem>.csv` to the output directory; the
|
|
134
177
|
prediction column is `Incidence(%)`. Features filled with `0` (genes the model
|
|
135
178
|
expects but the input lacks) are always reported, and a prominent warning
|
|
136
179
|
appears when more than `--missing-warn-frac` (default 0.3) of the model's
|
|
137
180
|
features are missing.
|
|
138
181
|
|
|
182
|
+
## Predict from samples and gene frequencies
|
|
183
|
+
|
|
184
|
+
Supply two CSV files instead of calculating features yourself:
|
|
185
|
+
|
|
186
|
+
* **Samples**: `Sample,dose_cfu,serotype`. `dose_cfu` contains raw CFU, e.g.
|
|
187
|
+
`1000`, not `3`. Each Sample must be nonblank and unique.
|
|
188
|
+
* **Gene frequencies**: `Serotype` plus one column per gene, one row per
|
|
189
|
+
serotype, with numeric frequencies from 0 to 1. This accepts the layout of
|
|
190
|
+
`02_gene_frequencies.csv` directly.
|
|
191
|
+
|
|
192
|
+
```bash
|
|
193
|
+
salmopredict run \
|
|
194
|
+
--samples examples/example_samples.csv \
|
|
195
|
+
--gene-frequencies examples/example_gene_frequencies.csv \
|
|
196
|
+
-o results_two_tables/
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
For each sample, the program looks up its serotype and calculates
|
|
200
|
+
`gene_frequency × log10(dose_cfu)`, then predicts incidence. It writes:
|
|
201
|
+
|
|
202
|
+
* `features_example_samples.csv`: `Sample` and the calculated gene features.
|
|
203
|
+
* `pred_example_samples.csv`: `Sample,dose_cfu,serotype,Incidence(%)`.
|
|
204
|
+
|
|
205
|
+
Sample order and identifier strings (including leading zeros) are preserved.
|
|
206
|
+
Required header names are case-insensitive. Serotype values match exactly after
|
|
207
|
+
trimming outer spaces; synonyms and spelling differences are not guessed.
|
|
208
|
+
Missing serotypes stop the run and list affected samples. Duplicate sample IDs
|
|
209
|
+
or serotypes, blank required values, nonfinite/nonpositive doses, and frequencies
|
|
210
|
+
outside [0, 1] also stop the run. Additional sample columns are ignored. Missing
|
|
211
|
+
model genes use the existing fill-and-warning behavior.
|
|
212
|
+
|
|
213
|
+
`--samples` and `--gene-frequencies` must be used together and cannot be combined
|
|
214
|
+
with `-i` or `--attach`. Use `--force` to replace existing output files.
|
|
215
|
+
|
|
216
|
+
In the **GUI**, choose **Samples + gene frequencies** under **Input mode**,
|
|
217
|
+
upload both CSVs, inspect their previews, choose an output folder and click
|
|
218
|
+
**Run prediction**. The result table and both CSV download buttons appear after
|
|
219
|
+
success. **Feature CSV** selects the existing single-table workflow.
|
|
220
|
+
|
|
221
|
+
The two example inputs were reconstructed from the matching sample/dose metadata
|
|
222
|
+
and dose-weighted features. See [examples/README.md](examples/README.md) for their
|
|
223
|
+
provenance and rounding tolerance.
|
|
224
|
+
|
|
139
225
|
## License
|
|
140
226
|
|
|
141
227
|
Licensed under the [PolyForm Noncommercial License 1.0.0](LICENSE): free to use,
|
|
@@ -15,8 +15,8 @@ single prediction file. It reproduces the alignment used by the original
|
|
|
15
15
|
(`/` and `-` become `.`), genes the model expects but the input lacks are filled
|
|
16
16
|
with `0` (a missing gene means frequency 0), and extra input columns are ignored.
|
|
17
17
|
|
|
18
|
-
**
|
|
19
|
-
|
|
18
|
+
**Feature CSV input/output** — the output columns depend on whether the input
|
|
19
|
+
has a `Sample` column:
|
|
20
20
|
|
|
21
21
|
| Input | Output columns |
|
|
22
22
|
|-------|----------------|
|
|
@@ -33,36 +33,78 @@ salmopredict runs on **Python 3.10** and loads its model with **AutoGluon
|
|
|
33
33
|
1.1.1** — both are hard requirements, because the model is pickled with that
|
|
34
34
|
exact stack.
|
|
35
35
|
|
|
36
|
-
**
|
|
36
|
+
**New installation from this source directory.** Run these commands from the
|
|
37
|
+
directory containing `pyproject.toml`:
|
|
37
38
|
|
|
38
39
|
```bash
|
|
39
|
-
|
|
40
|
+
conda create -n salmopredict python=3.10
|
|
41
|
+
conda activate salmopredict
|
|
42
|
+
python -m pip install .
|
|
43
|
+
salmopredict check
|
|
40
44
|
```
|
|
41
45
|
|
|
42
|
-
This
|
|
43
|
-
|
|
46
|
+
This installs AutoGluon 1.1.1 with its **Torch and FastAI backends**, compatible
|
|
47
|
+
`setuptools<81`, the Streamlit GUI, and the bundled prediction model. The
|
|
48
|
+
backends are required by the bundled ensemble; base `autogluon.tabular` alone
|
|
49
|
+
does not install them. AutoGluon 1.1.1 also needs `pkg_resources`, which newer
|
|
50
|
+
setuptools releases no longer provide.
|
|
51
|
+
|
|
52
|
+
Both interfaces are then available:
|
|
44
53
|
|
|
45
54
|
```bash
|
|
46
55
|
salmopredict run -i features.csv -o results/ # command line
|
|
47
56
|
salmopredict gui # browser GUI
|
|
48
57
|
```
|
|
49
58
|
|
|
50
|
-
|
|
51
|
-
`conda create -n salmopredict python=3.10 && conda activate salmopredict`.
|
|
52
|
-
|
|
53
|
-
**Reproducible environment (from a clone).** Pins Python 3.10 and installs
|
|
59
|
+
**Conda environment (alternative, from this source directory).** Pins Python 3.10 and installs
|
|
54
60
|
AutoGluon via pip inside the env (conda-installed AutoGluon does not resolve
|
|
55
61
|
cleanly for this project):
|
|
56
62
|
|
|
57
63
|
```bash
|
|
58
64
|
conda env create -f environment.yml
|
|
59
65
|
conda activate salmopredict
|
|
66
|
+
salmopredict check
|
|
60
67
|
```
|
|
61
68
|
|
|
62
69
|
**Editable / development install (from a clone).**
|
|
63
70
|
|
|
64
71
|
```bash
|
|
65
|
-
pip install -e . # installs the CLI and the Streamlit GUI
|
|
72
|
+
python -m pip install -e . # installs the CLI and the Streamlit GUI
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
**Install the updated local wheel.** In a Python 3.10 environment:
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
python -m pip install --upgrade dist/salmopredict-0.3.0-py3-none-any.whl
|
|
79
|
+
salmopredict check
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
**Update an existing source installation.** Stop a running GUI with `Ctrl+C`,
|
|
83
|
+
then run from this source directory:
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
conda activate salmopredict
|
|
87
|
+
python -m pip install --upgrade -e .
|
|
88
|
+
salmopredict check
|
|
89
|
+
salmopredict gui
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
**PyPI installation / upgrade.** In a Python 3.10 environment:
|
|
93
|
+
|
|
94
|
+
```bash
|
|
95
|
+
python -m pip install --upgrade "salmopredict>=0.3.0"
|
|
96
|
+
salmopredict check
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
Version 0.3.0 includes two-table prediction and installs the required Torch,
|
|
100
|
+
FastAI and compatible setuptools dependencies automatically.
|
|
101
|
+
|
|
102
|
+
If the page opens but prediction reports `No module named 'pkg_resources'`,
|
|
103
|
+
`torch`, or `fastai`, use the update/repair command above and restart the GUI.
|
|
104
|
+
To verify actual prediction from a source checkout (use a new output folder):
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
salmopredict run -i examples/example_features.csv -o results_install_check/
|
|
66
108
|
```
|
|
67
109
|
|
|
68
110
|
## The model
|
|
@@ -105,12 +147,55 @@ salmopredict gui
|
|
|
105
147
|
salmopredict check --model /path/to/model
|
|
106
148
|
```
|
|
107
149
|
|
|
108
|
-
Each run writes one `pred_<input-stem>.csv` to the output directory; the
|
|
150
|
+
Each feature-input run writes one `pred_<input-stem>.csv` to the output directory; the
|
|
109
151
|
prediction column is `Incidence(%)`. Features filled with `0` (genes the model
|
|
110
152
|
expects but the input lacks) are always reported, and a prominent warning
|
|
111
153
|
appears when more than `--missing-warn-frac` (default 0.3) of the model's
|
|
112
154
|
features are missing.
|
|
113
155
|
|
|
156
|
+
## Predict from samples and gene frequencies
|
|
157
|
+
|
|
158
|
+
Supply two CSV files instead of calculating features yourself:
|
|
159
|
+
|
|
160
|
+
* **Samples**: `Sample,dose_cfu,serotype`. `dose_cfu` contains raw CFU, e.g.
|
|
161
|
+
`1000`, not `3`. Each Sample must be nonblank and unique.
|
|
162
|
+
* **Gene frequencies**: `Serotype` plus one column per gene, one row per
|
|
163
|
+
serotype, with numeric frequencies from 0 to 1. This accepts the layout of
|
|
164
|
+
`02_gene_frequencies.csv` directly.
|
|
165
|
+
|
|
166
|
+
```bash
|
|
167
|
+
salmopredict run \
|
|
168
|
+
--samples examples/example_samples.csv \
|
|
169
|
+
--gene-frequencies examples/example_gene_frequencies.csv \
|
|
170
|
+
-o results_two_tables/
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
For each sample, the program looks up its serotype and calculates
|
|
174
|
+
`gene_frequency × log10(dose_cfu)`, then predicts incidence. It writes:
|
|
175
|
+
|
|
176
|
+
* `features_example_samples.csv`: `Sample` and the calculated gene features.
|
|
177
|
+
* `pred_example_samples.csv`: `Sample,dose_cfu,serotype,Incidence(%)`.
|
|
178
|
+
|
|
179
|
+
Sample order and identifier strings (including leading zeros) are preserved.
|
|
180
|
+
Required header names are case-insensitive. Serotype values match exactly after
|
|
181
|
+
trimming outer spaces; synonyms and spelling differences are not guessed.
|
|
182
|
+
Missing serotypes stop the run and list affected samples. Duplicate sample IDs
|
|
183
|
+
or serotypes, blank required values, nonfinite/nonpositive doses, and frequencies
|
|
184
|
+
outside [0, 1] also stop the run. Additional sample columns are ignored. Missing
|
|
185
|
+
model genes use the existing fill-and-warning behavior.
|
|
186
|
+
|
|
187
|
+
`--samples` and `--gene-frequencies` must be used together and cannot be combined
|
|
188
|
+
with `-i` or `--attach`. Use `--force` to replace existing output files.
|
|
189
|
+
|
|
190
|
+
In the **GUI**, choose **Samples + gene frequencies** under **Input mode**,
|
|
191
|
+
upload both CSVs, inspect their previews, choose an output folder and click
|
|
192
|
+
**Run prediction**. The result table and both CSV download buttons appear after
|
|
193
|
+
success. **Feature CSV** selects the existing single-table workflow.
|
|
194
|
+
|
|
195
|
+
The two example inputs were reconstructed from the matching sample/dose metadata
|
|
196
|
+
and dose-weighted features. See [examples/README.md](examples/README.md) for their
|
|
197
|
+
provenance and rounding tolerance.
|
|
198
|
+
|
|
114
199
|
## License
|
|
115
200
|
|
|
116
201
|
Licensed under the [PolyForm Noncommercial License 1.0.0](LICENSE): free to use,
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# Example inputs
|
|
2
|
+
|
|
3
|
+
Three ready-to-run files that cover the two input types and the metadata attach.
|
|
4
|
+
Feature values are built the way the model was trained:
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
feature = gene_frequency(serotype, gene) × log10(CFU dose)
|
|
8
|
+
```
|
|
9
|
+
|
|
10
|
+
so the same serotype at a higher CFU has larger feature values and a higher
|
|
11
|
+
predicted incidence. All files share the same 10 samples — two serotypes
|
|
12
|
+
(Enteritidis, Typhimurium) each at CFU = 500 / 1 000 / 2 000 / 10 000 / 100 000.
|
|
13
|
+
Gene column names keep the original biological form (`mig-5`, `spiC/ssaB`, …);
|
|
14
|
+
salmopredict normalises them to the model's names (`mig-5` → `mig.5`).
|
|
15
|
+
|
|
16
|
+
| File | Rows × cols | Type | Output when run |
|
|
17
|
+
|------|-------------|------|-----------------|
|
|
18
|
+
| `example_features.csv` | 10 × 123 | **Type 1** — features only, **no `Sample`** column (just the 123 genes the model uses) | `Incidence(%)` |
|
|
19
|
+
| `example_with_sample.csv` | 10 × 348 | **Type 2** — a `Sample` column + all 347 genes (the ~224 extra genes are ignored) | `Sample`, `Incidence(%)` |
|
|
20
|
+
| `example_meta.csv` | 10 × 5 | Metadata to **attach** — `Sample` + `serotype`, `dose_cfu`, `source`, `region` | joined onto a Type-2 run by the `Sample` key |
|
|
21
|
+
|
|
22
|
+
## Run them
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
# Type 1: features only -> a single Incidence(%) column
|
|
26
|
+
salmopredict run -i examples/example_features.csv -o results/
|
|
27
|
+
|
|
28
|
+
# Type 2: a Sample column -> Sample, Incidence(%)
|
|
29
|
+
salmopredict run -i examples/example_with_sample.csv -o results/
|
|
30
|
+
|
|
31
|
+
# Type 2 + attach: metadata joined on Sample -> Sample, Incidence(%), + meta columns
|
|
32
|
+
salmopredict run -i examples/example_with_sample.csv -o results/ \
|
|
33
|
+
--attach examples/example_meta.csv
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
The attached run shows the dose–response, since `example_meta.csv` carries the
|
|
37
|
+
`dose_cfu` alongside each `Sample`:
|
|
38
|
+
|
|
39
|
+
```
|
|
40
|
+
Sample,Incidence(%),serotype,dose_cfu,source,region
|
|
41
|
+
S001,12.39,Enteritidis,500,retail chicken,North
|
|
42
|
+
S002,14.15,Enteritidis,1000,retail pork,East
|
|
43
|
+
...
|
|
44
|
+
S005,37.43,Enteritidis,100000,retail pork,West
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
Rules for attaching metadata:
|
|
48
|
+
|
|
49
|
+
* the **input** must have a `Sample` column (Type 2), and so must the
|
|
50
|
+
**metadata** file — the two are joined on `Sample`;
|
|
51
|
+
* the metadata's `Sample` values should be unique (duplicates are rejected);
|
|
52
|
+
* every metadata column except `Sample` is appended to the output.
|
|
53
|
+
|
|
54
|
+
In the GUI (`salmopredict gui`), the **Attach metadata** box only appears once
|
|
55
|
+
the chosen input is detected to have a `Sample` column.
|
|
56
|
+
|
|
57
|
+
## Rebuilding these files
|
|
58
|
+
|
|
59
|
+
The values come from the per-serotype gene-frequency table used to train the
|
|
60
|
+
model (`results/02_gene_frequencies.csv` in the assembly project) multiplied by
|
|
61
|
+
`log10(dose)`, reproducing `multiply_CFU_geneFreq.R`. To change the serotypes or
|
|
62
|
+
CFU values, edit that grid and recompute `gene_frequency × log10(CFU)` for every
|
|
63
|
+
gene column — do not edit a dose value alone, or the features and the dose will
|
|
64
|
+
disagree.
|
|
65
|
+
|
|
66
|
+
## Two-table input examples
|
|
67
|
+
|
|
68
|
+
`example_samples.csv` contains 10 samples with raw `dose_cfu` and `serotype`.
|
|
69
|
+
`example_gene_frequencies.csv` has 2 serotypes and 123 model gene columns.
|
|
70
|
+
Both are also saved in the sibling `../test/` folder requested for verification.
|
|
71
|
+
|
|
72
|
+
These files were reconstructed from `../test/pred_example_with_sample.csv`
|
|
73
|
+
(sample IDs, serotypes and doses only) and `../test/example_features.csv`
|
|
74
|
+
(weighted features), pairing rows in their original order. The prediction column
|
|
75
|
+
is not used to derive frequencies. For each serotype, the 1000-CFU row gives
|
|
76
|
+
`frequency = feature / log10(1000) = feature / 3`. All five dose rows per serotype
|
|
77
|
+
were checked against that frequency. The maximum reconstructed feature error
|
|
78
|
+
is less than 5e-7, consistent with six-decimal rounding in the original features.
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
salmopredict run --samples examples/example_samples.csv \
|
|
82
|
+
--gene-frequencies examples/example_gene_frequencies.csv -o results_two_tables/
|
|
83
|
+
```
|
|
@@ -0,0 +1,3 @@
|
|
|
1
|
+
Serotype,SEAG_RS23305,SEAG_RS23320,SEN_RS22090,SG_RS05215,SG_RS05220,SG_RS05300,SG_RS24060,STM0266,STM0267,STM0268,STM0270,STM0271,STM0272,STM0273,STM0274,STM0275,STM0276,STM0278,STM0279,STM0280,STM0281,STM0282,STM0283,STM0284,STM0285,STM0286,STM0287,STM0289,STM0290,STM0306,STM3026,STM4261,STY_RS21720,STY_RS21730,apeE,avrA,bcfB,cheA,cheB,cheM,cheW,csgB,csgD,fimC,fimY,flgD,flgG,fliC,fliD,fliF,fliG,fliI,fliK,fliL,fliM,fliP,fljB,gogB,hilC,hilD,invA,iroB,iroC,iroD,iroN,lpfA,lpfD,mig-5,motB,nmpC,pefA,pefB,pefC,pefD,pegA,pegB,pegC,pipB,pltA,ratB,rck,rpoS,safA,safB,safC,safD,sefA,sefB,sefC,sefD,shdA,sifB,sinH,sipD,sodCI,sopA,sopD2,sopE,spiC/ssaB,spvB,spvC,spvD,ssaT,ssaU,sseI/srfH,sseK1,sseK2,sseL,sspH1,sspH2,staA,staB,stcA,stcD,stdB,steC,steD,steF,stkA,tae4,tcfA,tcfD,tlde1
|
|
2
|
+
Enteritidis,0,0,0.9,0.9,1,0,0.9,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0.9,1,0,0,1,1,1,1,1,1,1,1,1,1,1,1,1,0,0,1,1,1,1,1,1,1,0,0,0.9,1,1,1,1,1,1,1,1,0.7,1,1,0,0.6,0.6,0.6,1,1,1,1,0,1,0.5,1,1,1,1,1,1,0.9,0.9,0.9,0,0.9,0,1,1,1,0.9,1,1,0.7,0.7,0.6,1,1,1,0.6,0.9,1,0,0.9,0,0,0,0,1,1,1,1,0,0,0,0,1
|
|
3
|
+
Typhimurium,1,1,0,0,0,0,0,1,1,1,1,1,1,1,1,0.9848,0.8333,0.9848,0.8333,1,1,1,0.9848,0.9545,1,1,1,1,1,0.9848,0.9848,0.8939,0,0,1,1,1,1,1,1,1,1,0.9848,1,0.9848,1,1,0.9848,1,1,1,1,1,1,1,1,0.9242,0.1515,0.9848,0.9848,1,1,1,1,1,1,0.9848,0.1061,1,1,0.0909,0.0909,0.0909,0.0909,0,0,0,0.9848,0,1,0.0455,1,0.9697,1,1,1,0,0,0,0,0.1212,0.9394,1,1,0.9697,0.9848,0.9394,0,0.9848,0.1061,0.0909,0.1061,0.9848,0.9848,0.9242,0.9394,0.9848,0.9697,0.0152,0.9697,0,0,0.9848,0.9848,1,0.9848,0,0,0,0.9848,0,0,1
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
Sample,dose_cfu,serotype
|
|
2
|
+
S001,500,Enteritidis
|
|
3
|
+
S002,1000,Enteritidis
|
|
4
|
+
S003,2000,Enteritidis
|
|
5
|
+
S004,10000,Enteritidis
|
|
6
|
+
S005,100000,Enteritidis
|
|
7
|
+
S006,500,Typhimurium
|
|
8
|
+
S007,1000,Typhimurium
|
|
9
|
+
S008,2000,Typhimurium
|
|
10
|
+
S009,10000,Typhimurium
|
|
11
|
+
S010,100000,Typhimurium
|
|
@@ -18,7 +18,10 @@ classifiers = [
|
|
|
18
18
|
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
19
19
|
]
|
|
20
20
|
dependencies = [
|
|
21
|
-
|
|
21
|
+
# The bundled ensemble uses both Torch and FastAI neural networks.
|
|
22
|
+
"autogluon.tabular[fastai]==1.1.1",
|
|
23
|
+
# AutoGluon 1.1.1 imports pkg_resources, removed in newer setuptools.
|
|
24
|
+
"setuptools<81",
|
|
22
25
|
"pandas>=2.0",
|
|
23
26
|
"openpyxl>=3.0",
|
|
24
27
|
"rich-argparse>=1.4",
|
|
@@ -13,7 +13,7 @@ salmopredict:
|
|
|
13
13
|
The same core is shared by a command-line interface and a Streamlit GUI.
|
|
14
14
|
"""
|
|
15
15
|
|
|
16
|
-
__version__ = "0.
|
|
16
|
+
__version__ = "0.3.0"
|
|
17
17
|
|
|
18
18
|
# One-paragraph summary shown in both the CLI help and the GUI, so the two
|
|
19
19
|
# interfaces describe the tool with identical wording.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
"""Command-line interface for salmopredict.
|
|
2
2
|
|
|
3
3
|
Commands:
|
|
4
|
-
salmopredict run -- predict
|
|
4
|
+
salmopredict run -- predict from features or samples + gene frequencies
|
|
5
5
|
salmopredict gui -- launch the Streamlit graphical interface
|
|
6
6
|
salmopredict check -- verify dependencies and the model
|
|
7
7
|
salmopredict build-model -- clone a full AutoGluon model into a slim deploy copy
|
|
@@ -73,12 +73,18 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
73
73
|
|
|
74
74
|
default_model = default_model_path()
|
|
75
75
|
|
|
76
|
-
run = sub.add_parser("run", help="Predict
|
|
76
|
+
run = sub.add_parser("run", help="Predict from features or samples + gene frequencies.",
|
|
77
77
|
formatter_class=_HelpFormatter)
|
|
78
|
-
run.
|
|
78
|
+
inputs = run.add_mutually_exclusive_group(required=True)
|
|
79
|
+
inputs.add_argument("-i", "--input",
|
|
79
80
|
help="One feature CSV file. If it has a 'Sample' column, the "
|
|
80
81
|
"output carries Sample + the prediction; otherwise it is "
|
|
81
82
|
"the prediction column only.")
|
|
83
|
+
inputs.add_argument("--samples", help="Sample CSV with Sample, dose_cfu (raw CFU), "
|
|
84
|
+
"and serotype. Requires --gene-frequencies.")
|
|
85
|
+
run.add_argument("--gene-frequencies", help="CSV with Serotype and gene columns "
|
|
86
|
+
"containing frequencies in [0, 1]. Used only with --samples; "
|
|
87
|
+
"also writes features_<samples-stem>.csv.")
|
|
82
88
|
run.add_argument("-o", "--output", required=True,
|
|
83
89
|
help="Output directory (created if missing); the result is "
|
|
84
90
|
"written there as pred_<input-stem>.csv.")
|
|
@@ -128,8 +134,10 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
128
134
|
|
|
129
135
|
|
|
130
136
|
def _cmd_run(args: argparse.Namespace) -> int:
|
|
137
|
+
if bool(args.samples) != bool(args.gene_frequencies):
|
|
138
|
+
raise ValueError("--samples and --gene-frequencies must be supplied together.")
|
|
131
139
|
config = PredictConfig(
|
|
132
|
-
input_path=Path(args.input),
|
|
140
|
+
input_path=Path(args.samples or args.input),
|
|
133
141
|
output_dir=Path(args.output),
|
|
134
142
|
model_path=Path(args.model) if args.model else None,
|
|
135
143
|
ensemble_model=args.model_name,
|
|
@@ -137,6 +145,7 @@ def _cmd_run(args: argparse.Namespace) -> int:
|
|
|
137
145
|
missing_warn_fraction=args.missing_warn_frac,
|
|
138
146
|
attach_metadata=Path(args.attach) if args.attach else None,
|
|
139
147
|
force=args.force,
|
|
148
|
+
gene_frequencies=Path(args.gene_frequencies) if args.gene_frequencies else None,
|
|
140
149
|
)
|
|
141
150
|
|
|
142
151
|
# Imported here so 'check'/'gui' work even if autogluon is absent.
|
|
@@ -147,6 +156,8 @@ def _cmd_run(args: argparse.Namespace) -> int:
|
|
|
147
156
|
cols = ", ".join(result.predictions.columns)
|
|
148
157
|
print(f"\nDone. label={result.label} model={result.model_used}")
|
|
149
158
|
print(f" {result.input_path.name}: {result.n_rows} rows -> {result.output_path}")
|
|
159
|
+
if result.features_path is not None:
|
|
160
|
+
print(f" weighted features: {result.features_path}")
|
|
150
161
|
print(f" Sample column: {'yes' if result.has_sample else 'no'}"
|
|
151
162
|
+ (" (metadata attached)" if result.attached else ""))
|
|
152
163
|
print(f" output columns: {cols}")
|
|
@@ -71,6 +71,10 @@ class PredictConfig:
|
|
|
71
71
|
One input feature CSV is aligned to the model and predicted, producing one
|
|
72
72
|
output file ``pred_<input-stem>.csv`` in ``output_dir``.
|
|
73
73
|
|
|
74
|
+
With ``gene_frequencies``, input_path instead contains Sample, dose_cfu and
|
|
75
|
+
serotype. The run also exports ``features_<input-stem>.csv`` and includes
|
|
76
|
+
those three sample columns in the prediction output.
|
|
77
|
+
|
|
74
78
|
Output columns depend on the input:
|
|
75
79
|
* no ``Sample`` column -> just the prediction (``Incidence(%)``);
|
|
76
80
|
* a ``Sample`` column -> ``Sample`` + ``Incidence(%)``, and, if
|
|
@@ -90,6 +94,9 @@ class PredictConfig:
|
|
|
90
94
|
|
|
91
95
|
force: bool = False
|
|
92
96
|
|
|
97
|
+
# When provided, input_path is a Sample/dose_cfu/serotype table.
|
|
98
|
+
gene_frequencies: Optional[Path] = None
|
|
99
|
+
|
|
93
100
|
def __post_init__(self) -> None:
|
|
94
101
|
self.input_path = Path(self.input_path)
|
|
95
102
|
self.output_dir = Path(self.output_dir)
|
|
@@ -97,6 +104,11 @@ class PredictConfig:
|
|
|
97
104
|
self.model_path = Path(self.model_path)
|
|
98
105
|
if self.attach_metadata is not None:
|
|
99
106
|
self.attach_metadata = Path(self.attach_metadata)
|
|
107
|
+
if self.gene_frequencies is not None:
|
|
108
|
+
self.gene_frequencies = Path(self.gene_frequencies)
|
|
109
|
+
if self.attach_metadata is not None:
|
|
110
|
+
raise ValueError("--attach is only supported with feature CSV input; "
|
|
111
|
+
"two-table input already includes sample metadata.")
|
|
100
112
|
if not 0.0 <= float(self.missing_warn_fraction) <= 1.0:
|
|
101
113
|
raise ValueError("missing_warn_fraction must be between 0 and 1")
|
|
102
114
|
self.missing_warn_fraction = float(self.missing_warn_fraction)
|