salmopredict 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. {salmopredict-0.2.0 → salmopredict-0.3.0}/MANIFEST.in +1 -0
  2. {salmopredict-0.2.0/salmopredict.egg-info → salmopredict-0.3.0}/PKG-INFO +100 -14
  3. {salmopredict-0.2.0 → salmopredict-0.3.0}/README.md +97 -12
  4. {salmopredict-0.2.0 → salmopredict-0.3.0}/environment.yml +1 -5
  5. salmopredict-0.3.0/examples/README.md +83 -0
  6. salmopredict-0.3.0/examples/example_gene_frequencies.csv +3 -0
  7. salmopredict-0.3.0/examples/example_samples.csv +11 -0
  8. {salmopredict-0.2.0 → salmopredict-0.3.0}/pyproject.toml +4 -1
  9. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/__init__.py +1 -1
  10. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/cli.py +15 -4
  11. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/config.py +12 -0
  12. salmopredict-0.3.0/salmopredict/core/frequencies.py +106 -0
  13. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/gui/app.py +67 -14
  14. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/pipeline.py +31 -5
  15. {salmopredict-0.2.0 → salmopredict-0.3.0/salmopredict.egg-info}/PKG-INFO +100 -14
  16. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict.egg-info/SOURCES.txt +6 -1
  17. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict.egg-info/requires.txt +2 -1
  18. salmopredict-0.3.0/tests/test_two_table_input.py +144 -0
  19. {salmopredict-0.2.0 → salmopredict-0.3.0}/LICENSE +0 -0
  20. {salmopredict-0.2.0 → salmopredict-0.3.0}/examples/example_features.csv +0 -0
  21. {salmopredict-0.2.0 → salmopredict-0.3.0}/examples/example_meta.csv +0 -0
  22. {salmopredict-0.2.0 → salmopredict-0.3.0}/examples/example_with_sample.csv +0 -0
  23. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/core/__init__.py +0 -0
  24. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/core/align.py +0 -0
  25. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/core/io_tables.py +0 -0
  26. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/core/modelinfo.py +0 -0
  27. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/core/predict.py +0 -0
  28. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/gui/__init__.py +0 -0
  29. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/gui/assets/cfsa_logo.png +0 -0
  30. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/gui/assets/salmopredict_icon.png +0 -0
  31. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/gui/assets/vphs_logo.png +0 -0
  32. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/.DS_Store +0 -0
  33. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/learner.pkl +0 -0
  34. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/metadata.json +0 -0
  35. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F1/model-internals.pkl +0 -0
  36. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F1/model.pkl +0 -0
  37. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F2/model-internals.pkl +0 -0
  38. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F2/model.pkl +0 -0
  39. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F3/model-internals.pkl +0 -0
  40. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F3/model.pkl +0 -0
  41. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F4/model-internals.pkl +0 -0
  42. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F4/model.pkl +0 -0
  43. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F5/model-internals.pkl +0 -0
  44. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/S1F5/model.pkl +0 -0
  45. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r100_BAG_L1/model.pkl +0 -0
  46. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F1/model-internals.pkl +0 -0
  47. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F1/model.pkl +0 -0
  48. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F2/model-internals.pkl +0 -0
  49. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F2/model.pkl +0 -0
  50. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F3/model-internals.pkl +0 -0
  51. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F3/model.pkl +0 -0
  52. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F4/model-internals.pkl +0 -0
  53. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F4/model.pkl +0 -0
  54. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F5/model-internals.pkl +0 -0
  55. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/S1F5/model.pkl +0 -0
  56. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r134_BAG_L1/model.pkl +0 -0
  57. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F1/model-internals.pkl +0 -0
  58. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F1/model.pkl +0 -0
  59. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F2/model-internals.pkl +0 -0
  60. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F2/model.pkl +0 -0
  61. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F3/model-internals.pkl +0 -0
  62. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F3/model.pkl +0 -0
  63. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F4/model-internals.pkl +0 -0
  64. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F4/model.pkl +0 -0
  65. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F5/model-internals.pkl +0 -0
  66. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/S1F5/model.pkl +0 -0
  67. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetFastAI_r156_BAG_L1/model.pkl +0 -0
  68. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r121_BAG_L1/S1F1/model.pkl +0 -0
  69. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r121_BAG_L1/S1F2/model.pkl +0 -0
  70. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r121_BAG_L1/S1F3/model.pkl +0 -0
  71. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r121_BAG_L1/S1F4/model.pkl +0 -0
  72. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r121_BAG_L1/S1F5/model.pkl +0 -0
  73. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r121_BAG_L1/model.pkl +0 -0
  74. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r1_BAG_L1/S1F1/model.pkl +0 -0
  75. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r1_BAG_L1/S1F2/model.pkl +0 -0
  76. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r1_BAG_L1/S1F3/model.pkl +0 -0
  77. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r1_BAG_L1/S1F4/model.pkl +0 -0
  78. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r1_BAG_L1/S1F5/model.pkl +0 -0
  79. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r1_BAG_L1/model.pkl +0 -0
  80. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r30_BAG_L1/S1F1/model.pkl +0 -0
  81. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r30_BAG_L1/S1F2/model.pkl +0 -0
  82. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r30_BAG_L1/S1F3/model.pkl +0 -0
  83. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r30_BAG_L1/S1F4/model.pkl +0 -0
  84. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r30_BAG_L1/S1F5/model.pkl +0 -0
  85. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r30_BAG_L1/model.pkl +0 -0
  86. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r79_BAG_L1/S1F1/model.pkl +0 -0
  87. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r79_BAG_L1/S1F2/model.pkl +0 -0
  88. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r79_BAG_L1/S1F3/model.pkl +0 -0
  89. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r79_BAG_L1/S1F4/model.pkl +0 -0
  90. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r79_BAG_L1/S1F5/model.pkl +0 -0
  91. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/NeuralNetTorch_r79_BAG_L1/model.pkl +0 -0
  92. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/WeightedEnsemble_L2/model.pkl +0 -0
  93. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/models/trainer.pkl +0 -0
  94. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/predictor.pkl +0 -0
  95. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict/models/model_default/version.txt +0 -0
  96. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict.egg-info/dependency_links.txt +0 -0
  97. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict.egg-info/entry_points.txt +0 -0
  98. {salmopredict-0.2.0 → salmopredict-0.3.0}/salmopredict.egg-info/top_level.txt +0 -0
  99. {salmopredict-0.2.0 → salmopredict-0.3.0}/setup.cfg +0 -0
@@ -6,3 +6,4 @@ recursive-include salmopredict/gui/assets *.png
6
6
  include LICENSE
7
7
  include environment.yml
8
8
  include examples/*.csv
9
+ include examples/README.md
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: salmopredict
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: AutoGluon-based Incidence predictor for Salmonella virulence-factor gene-frequency features
5
5
  Author-email: Dongyan Shao <563608176@qq.com>
6
6
  License: PolyForm-Noncommercial-1.0.0
@@ -14,7 +14,8 @@ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
14
14
  Requires-Python: <3.11,>=3.10
15
15
  Description-Content-Type: text/markdown
16
16
  License-File: LICENSE
17
- Requires-Dist: autogluon.tabular==1.1.1
17
+ Requires-Dist: autogluon.tabular[fastai]==1.1.1
18
+ Requires-Dist: setuptools<81
18
19
  Requires-Dist: pandas>=2.0
19
20
  Requires-Dist: openpyxl>=3.0
20
21
  Requires-Dist: rich-argparse>=1.4
@@ -40,8 +41,8 @@ single prediction file. It reproduces the alignment used by the original
40
41
  (`/` and `-` become `.`), genes the model expects but the input lacks are filled
41
42
  with `0` (a missing gene means frequency 0), and extra input columns are ignored.
42
43
 
43
- **Input/output contract** — one input CSV in, one output CSV out. The output
44
- columns depend on whether the input has a `Sample` column:
44
+ **Feature CSV input/output** — the output columns depend on whether the input
45
+ has a `Sample` column:
45
46
 
46
47
  | Input | Output columns |
47
48
  |-------|----------------|
@@ -58,36 +59,78 @@ salmopredict runs on **Python 3.10** and loads its model with **AutoGluon
58
59
  1.1.1** — both are hard requirements, because the model is pickled with that
59
60
  exact stack.
60
61
 
61
- **From PyPI (recommended).** In a Python 3.10 environment:
62
+ **New installation from this source directory.** Run these commands from the
63
+ directory containing `pyproject.toml`:
62
64
 
63
65
  ```bash
64
- pip install salmopredict
66
+ conda create -n salmopredict python=3.10
67
+ conda activate salmopredict
68
+ python -m pip install .
69
+ salmopredict check
65
70
  ```
66
71
 
67
- This pulls in AutoGluon 1.1.1, the Streamlit GUI, and the bundled prediction
68
- model, so both interfaces work out of the box:
72
+ This installs AutoGluon 1.1.1 with its **Torch and FastAI backends**, compatible
73
+ `setuptools<81`, the Streamlit GUI, and the bundled prediction model. The
74
+ backends are required by the bundled ensemble; base `autogluon.tabular` alone
75
+ does not install them. AutoGluon 1.1.1 also needs `pkg_resources`, which newer
76
+ setuptools releases no longer provide.
77
+
78
+ Both interfaces are then available:
69
79
 
70
80
  ```bash
71
81
  salmopredict run -i features.csv -o results/ # command line
72
82
  salmopredict gui # browser GUI
73
83
  ```
74
84
 
75
- No Python 3.10 environment yet? Create one first, e.g.
76
- `conda create -n salmopredict python=3.10 && conda activate salmopredict`.
77
-
78
- **Reproducible environment (from a clone).** Pins Python 3.10 and installs
85
+ **Conda environment (alternative, from this source directory).** Pins Python 3.10 and installs
79
86
  AutoGluon via pip inside the env (conda-installed AutoGluon does not resolve
80
87
  cleanly for this project):
81
88
 
82
89
  ```bash
83
90
  conda env create -f environment.yml
84
91
  conda activate salmopredict
92
+ salmopredict check
85
93
  ```
86
94
 
87
95
  **Editable / development install (from a clone).**
88
96
 
89
97
  ```bash
90
- pip install -e . # installs the CLI and the Streamlit GUI
98
+ python -m pip install -e . # installs the CLI and the Streamlit GUI
99
+ ```
100
+
101
+ **Install the updated local wheel.** In a Python 3.10 environment:
102
+
103
+ ```bash
104
+ python -m pip install --upgrade dist/salmopredict-0.3.0-py3-none-any.whl
105
+ salmopredict check
106
+ ```
107
+
108
+ **Update an existing source installation.** Stop a running GUI with `Ctrl+C`,
109
+ then run from this source directory:
110
+
111
+ ```bash
112
+ conda activate salmopredict
113
+ python -m pip install --upgrade -e .
114
+ salmopredict check
115
+ salmopredict gui
116
+ ```
117
+
118
+ **PyPI installation / upgrade.** In a Python 3.10 environment:
119
+
120
+ ```bash
121
+ python -m pip install --upgrade "salmopredict>=0.3.0"
122
+ salmopredict check
123
+ ```
124
+
125
+ Version 0.3.0 includes two-table prediction and installs the required Torch,
126
+ FastAI and compatible setuptools dependencies automatically.
127
+
128
+ If the page opens but prediction reports `No module named 'pkg_resources'`,
129
+ `torch`, or `fastai`, use the update/repair command above and restart the GUI.
130
+ To verify actual prediction from a source checkout (use a new output folder):
131
+
132
+ ```bash
133
+ salmopredict run -i examples/example_features.csv -o results_install_check/
91
134
  ```
92
135
 
93
136
  ## The model
@@ -130,12 +173,55 @@ salmopredict gui
130
173
  salmopredict check --model /path/to/model
131
174
  ```
132
175
 
133
- Each run writes one `pred_<input-stem>.csv` to the output directory; the
176
+ Each feature-input run writes one `pred_<input-stem>.csv` to the output directory; the
134
177
  prediction column is `Incidence(%)`. Features filled with `0` (genes the model
135
178
  expects but the input lacks) are always reported, and a prominent warning
136
179
  appears when more than `--missing-warn-frac` (default 0.3) of the model's
137
180
  features are missing.
138
181
 
182
+ ## Predict from samples and gene frequencies
183
+
184
+ Supply two CSV files instead of calculating features yourself:
185
+
186
+ * **Samples**: `Sample,dose_cfu,serotype`. `dose_cfu` contains raw CFU, e.g.
187
+ `1000`, not `3`. Each Sample must be nonblank and unique.
188
+ * **Gene frequencies**: `Serotype` plus one column per gene, one row per
189
+ serotype, with numeric frequencies from 0 to 1. This accepts the layout of
190
+ `02_gene_frequencies.csv` directly.
191
+
192
+ ```bash
193
+ salmopredict run \
194
+ --samples examples/example_samples.csv \
195
+ --gene-frequencies examples/example_gene_frequencies.csv \
196
+ -o results_two_tables/
197
+ ```
198
+
199
+ For each sample, the program looks up its serotype and calculates
200
+ `gene_frequency × log10(dose_cfu)`, then predicts incidence. It writes:
201
+
202
+ * `features_example_samples.csv`: `Sample` and the calculated gene features.
203
+ * `pred_example_samples.csv`: `Sample,dose_cfu,serotype,Incidence(%)`.
204
+
205
+ Sample order and identifier strings (including leading zeros) are preserved.
206
+ Required header names are case-insensitive. Serotype values match exactly after
207
+ trimming outer spaces; synonyms and spelling differences are not guessed.
208
+ Missing serotypes stop the run and list affected samples. Duplicate sample IDs
209
+ or serotypes, blank required values, nonfinite/nonpositive doses, and frequencies
210
+ outside [0, 1] also stop the run. Additional sample columns are ignored. Missing
211
+ model genes use the existing fill-and-warning behavior.
212
+
213
+ `--samples` and `--gene-frequencies` must be used together and cannot be combined
214
+ with `-i` or `--attach`. Use `--force` to replace existing output files.
215
+
216
+ In the **GUI**, choose **Samples + gene frequencies** under **Input mode**,
217
+ upload both CSVs, inspect their previews, choose an output folder and click
218
+ **Run prediction**. The result table and both CSV download buttons appear after
219
+ success. **Feature CSV** selects the existing single-table workflow.
220
+
221
+ The two example inputs were reconstructed from the matching sample/dose metadata
222
+ and dose-weighted features. See [examples/README.md](examples/README.md) for their
223
+ provenance and rounding tolerance.
224
+
139
225
  ## License
140
226
 
141
227
  Licensed under the [PolyForm Noncommercial License 1.0.0](LICENSE): free to use,
@@ -15,8 +15,8 @@ single prediction file. It reproduces the alignment used by the original
15
15
  (`/` and `-` become `.`), genes the model expects but the input lacks are filled
16
16
  with `0` (a missing gene means frequency 0), and extra input columns are ignored.
17
17
 
18
- **Input/output contract** — one input CSV in, one output CSV out. The output
19
- columns depend on whether the input has a `Sample` column:
18
+ **Feature CSV input/output** — the output columns depend on whether the input
19
+ has a `Sample` column:
20
20
 
21
21
  | Input | Output columns |
22
22
  |-------|----------------|
@@ -33,36 +33,78 @@ salmopredict runs on **Python 3.10** and loads its model with **AutoGluon
33
33
  1.1.1** — both are hard requirements, because the model is pickled with that
34
34
  exact stack.
35
35
 
36
- **From PyPI (recommended).** In a Python 3.10 environment:
36
+ **New installation from this source directory.** Run these commands from the
37
+ directory containing `pyproject.toml`:
37
38
 
38
39
  ```bash
39
- pip install salmopredict
40
+ conda create -n salmopredict python=3.10
41
+ conda activate salmopredict
42
+ python -m pip install .
43
+ salmopredict check
40
44
  ```
41
45
 
42
- This pulls in AutoGluon 1.1.1, the Streamlit GUI, and the bundled prediction
43
- model, so both interfaces work out of the box:
46
+ This installs AutoGluon 1.1.1 with its **Torch and FastAI backends**, compatible
47
+ `setuptools<81`, the Streamlit GUI, and the bundled prediction model. The
48
+ backends are required by the bundled ensemble; base `autogluon.tabular` alone
49
+ does not install them. AutoGluon 1.1.1 also needs `pkg_resources`, which newer
50
+ setuptools releases no longer provide.
51
+
52
+ Both interfaces are then available:
44
53
 
45
54
  ```bash
46
55
  salmopredict run -i features.csv -o results/ # command line
47
56
  salmopredict gui # browser GUI
48
57
  ```
49
58
 
50
- No Python 3.10 environment yet? Create one first, e.g.
51
- `conda create -n salmopredict python=3.10 && conda activate salmopredict`.
52
-
53
- **Reproducible environment (from a clone).** Pins Python 3.10 and installs
59
+ **Conda environment (alternative, from this source directory).** Pins Python 3.10 and installs
54
60
  AutoGluon via pip inside the env (conda-installed AutoGluon does not resolve
55
61
  cleanly for this project):
56
62
 
57
63
  ```bash
58
64
  conda env create -f environment.yml
59
65
  conda activate salmopredict
66
+ salmopredict check
60
67
  ```
61
68
 
62
69
  **Editable / development install (from a clone).**
63
70
 
64
71
  ```bash
65
- pip install -e . # installs the CLI and the Streamlit GUI
72
+ python -m pip install -e . # installs the CLI and the Streamlit GUI
73
+ ```
74
+
75
+ **Install the updated local wheel.** In a Python 3.10 environment:
76
+
77
+ ```bash
78
+ python -m pip install --upgrade dist/salmopredict-0.3.0-py3-none-any.whl
79
+ salmopredict check
80
+ ```
81
+
82
+ **Update an existing source installation.** Stop a running GUI with `Ctrl+C`,
83
+ then run from this source directory:
84
+
85
+ ```bash
86
+ conda activate salmopredict
87
+ python -m pip install --upgrade -e .
88
+ salmopredict check
89
+ salmopredict gui
90
+ ```
91
+
92
+ **PyPI installation / upgrade.** In a Python 3.10 environment:
93
+
94
+ ```bash
95
+ python -m pip install --upgrade "salmopredict>=0.3.0"
96
+ salmopredict check
97
+ ```
98
+
99
+ Version 0.3.0 includes two-table prediction and installs the required Torch,
100
+ FastAI and compatible setuptools dependencies automatically.
101
+
102
+ If the page opens but prediction reports `No module named 'pkg_resources'`,
103
+ `torch`, or `fastai`, use the update/repair command above and restart the GUI.
104
+ To verify actual prediction from a source checkout (use a new output folder):
105
+
106
+ ```bash
107
+ salmopredict run -i examples/example_features.csv -o results_install_check/
66
108
  ```
67
109
 
68
110
  ## The model
@@ -105,12 +147,55 @@ salmopredict gui
105
147
  salmopredict check --model /path/to/model
106
148
  ```
107
149
 
108
- Each run writes one `pred_<input-stem>.csv` to the output directory; the
150
+ Each feature-input run writes one `pred_<input-stem>.csv` to the output directory; the
109
151
  prediction column is `Incidence(%)`. Features filled with `0` (genes the model
110
152
  expects but the input lacks) are always reported, and a prominent warning
111
153
  appears when more than `--missing-warn-frac` (default 0.3) of the model's
112
154
  features are missing.
113
155
 
156
+ ## Predict from samples and gene frequencies
157
+
158
+ Supply two CSV files instead of calculating features yourself:
159
+
160
+ * **Samples**: `Sample,dose_cfu,serotype`. `dose_cfu` contains raw CFU, e.g.
161
+ `1000`, not `3`. Each Sample must be nonblank and unique.
162
+ * **Gene frequencies**: `Serotype` plus one column per gene, one row per
163
+ serotype, with numeric frequencies from 0 to 1. This accepts the layout of
164
+ `02_gene_frequencies.csv` directly.
165
+
166
+ ```bash
167
+ salmopredict run \
168
+ --samples examples/example_samples.csv \
169
+ --gene-frequencies examples/example_gene_frequencies.csv \
170
+ -o results_two_tables/
171
+ ```
172
+
173
+ For each sample, the program looks up its serotype and calculates
174
+ `gene_frequency × log10(dose_cfu)`, then predicts incidence. It writes:
175
+
176
+ * `features_example_samples.csv`: `Sample` and the calculated gene features.
177
+ * `pred_example_samples.csv`: `Sample,dose_cfu,serotype,Incidence(%)`.
178
+
179
+ Sample order and identifier strings (including leading zeros) are preserved.
180
+ Required header names are case-insensitive. Serotype values match exactly after
181
+ trimming outer spaces; synonyms and spelling differences are not guessed.
182
+ Missing serotypes stop the run and list affected samples. Duplicate sample IDs
183
+ or serotypes, blank required values, nonfinite/nonpositive doses, and frequencies
184
+ outside [0, 1] also stop the run. Additional sample columns are ignored. Missing
185
+ model genes use the existing fill-and-warning behavior.
186
+
187
+ `--samples` and `--gene-frequencies` must be used together and cannot be combined
188
+ with `-i` or `--attach`. Use `--force` to replace existing output files.
189
+
190
+ In the **GUI**, choose **Samples + gene frequencies** under **Input mode**,
191
+ upload both CSVs, inspect their previews, choose an output folder and click
192
+ **Run prediction**. The result table and both CSV download buttons appear after
193
+ success. **Feature CSV** selects the existing single-table workflow.
194
+
195
+ The two example inputs were reconstructed from the matching sample/dose metadata
196
+ and dose-weighted features. See [examples/README.md](examples/README.md) for their
197
+ provenance and rounding tolerance.
198
+
114
199
  ## License
115
200
 
116
201
  Licensed under the [PolyForm Noncommercial License 1.0.0](LICENSE): free to use,
@@ -8,9 +8,5 @@ dependencies:
8
8
  - python=3.10
9
9
  - pip
10
10
  - pip:
11
- - autogluon.tabular==1.1.1
12
- - pandas>=2.0
13
- - openpyxl>=3.0
14
- - rich-argparse>=1.4
15
- - streamlit>=1.30
11
+ # pyproject.toml includes the Torch/FastAI backends and setuptools<81.
16
12
  - -e .
@@ -0,0 +1,83 @@
1
+ # Example inputs
2
+
3
+ Three ready-to-run files that cover the two input types and the metadata attach.
4
+ Feature values are built the way the model was trained:
5
+
6
+ ```
7
+ feature = gene_frequency(serotype, gene) × log10(CFU dose)
8
+ ```
9
+
10
+ so the same serotype at a higher CFU has larger feature values and a higher
11
+ predicted incidence. All files share the same 10 samples — two serotypes
12
+ (Enteritidis, Typhimurium) each at CFU = 500 / 1 000 / 2 000 / 10 000 / 100 000.
13
+ Gene column names keep the original biological form (`mig-5`, `spiC/ssaB`, …);
14
+ salmopredict normalises them to the model's names (`mig-5` → `mig.5`).
15
+
16
+ | File | Rows × cols | Type | Output when run |
17
+ |------|-------------|------|-----------------|
18
+ | `example_features.csv` | 10 × 123 | **Type 1** — features only, **no `Sample`** column (just the 123 genes the model uses) | `Incidence(%)` |
19
+ | `example_with_sample.csv` | 10 × 348 | **Type 2** — a `Sample` column + all 347 genes (the ~224 extra genes are ignored) | `Sample`, `Incidence(%)` |
20
+ | `example_meta.csv` | 10 × 5 | Metadata to **attach** — `Sample` + `serotype`, `dose_cfu`, `source`, `region` | joined onto a Type-2 run by the `Sample` key |
21
+
22
+ ## Run them
23
+
24
+ ```bash
25
+ # Type 1: features only -> a single Incidence(%) column
26
+ salmopredict run -i examples/example_features.csv -o results/
27
+
28
+ # Type 2: a Sample column -> Sample, Incidence(%)
29
+ salmopredict run -i examples/example_with_sample.csv -o results/
30
+
31
+ # Type 2 + attach: metadata joined on Sample -> Sample, Incidence(%), + meta columns
32
+ salmopredict run -i examples/example_with_sample.csv -o results/ \
33
+ --attach examples/example_meta.csv
34
+ ```
35
+
36
+ The attached run shows the dose–response, since `example_meta.csv` carries the
37
+ `dose_cfu` alongside each `Sample`:
38
+
39
+ ```
40
+ Sample,Incidence(%),serotype,dose_cfu,source,region
41
+ S001,12.39,Enteritidis,500,retail chicken,North
42
+ S002,14.15,Enteritidis,1000,retail pork,East
43
+ ...
44
+ S005,37.43,Enteritidis,100000,retail pork,West
45
+ ```
46
+
47
+ Rules for attaching metadata:
48
+
49
+ * the **input** must have a `Sample` column (Type 2), and so must the
50
+ **metadata** file — the two are joined on `Sample`;
51
+ * the metadata's `Sample` values should be unique (duplicates are rejected);
52
+ * every metadata column except `Sample` is appended to the output.
53
+
54
+ In the GUI (`salmopredict gui`), the **Attach metadata** box only appears once
55
+ the chosen input is detected to have a `Sample` column.
56
+
57
+ ## Rebuilding these files
58
+
59
+ The values come from the per-serotype gene-frequency table used to train the
60
+ model (`results/02_gene_frequencies.csv` in the assembly project) multiplied by
61
+ `log10(dose)`, reproducing `multiply_CFU_geneFreq.R`. To change the serotypes or
62
+ CFU values, edit that grid and recompute `gene_frequency × log10(CFU)` for every
63
+ gene column — do not edit a dose value alone, or the features and the dose will
64
+ disagree.
65
+
66
+ ## Two-table input examples
67
+
68
+ `example_samples.csv` contains 10 samples with raw `dose_cfu` and `serotype`.
69
+ `example_gene_frequencies.csv` has 2 serotypes and 123 model gene columns.
70
+ Both are also saved in the sibling `../test/` folder requested for verification.
71
+
72
+ These files were reconstructed from `../test/pred_example_with_sample.csv`
73
+ (sample IDs, serotypes and doses only) and `../test/example_features.csv`
74
+ (weighted features), pairing rows in their original order. The prediction column
75
+ is not used to derive frequencies. For each serotype, the 1000-CFU row gives
76
+ `frequency = feature / log10(1000) = feature / 3`. All five dose rows per serotype
77
+ were checked against that frequency. The maximum reconstructed feature error
78
+ is less than 5e-7, consistent with six-decimal rounding in the original features.
79
+
80
+ ```bash
81
+ salmopredict run --samples examples/example_samples.csv \
82
+ --gene-frequencies examples/example_gene_frequencies.csv -o results_two_tables/
83
+ ```
@@ -0,0 +1,3 @@
1
+ Serotype,SEAG_RS23305,SEAG_RS23320,SEN_RS22090,SG_RS05215,SG_RS05220,SG_RS05300,SG_RS24060,STM0266,STM0267,STM0268,STM0270,STM0271,STM0272,STM0273,STM0274,STM0275,STM0276,STM0278,STM0279,STM0280,STM0281,STM0282,STM0283,STM0284,STM0285,STM0286,STM0287,STM0289,STM0290,STM0306,STM3026,STM4261,STY_RS21720,STY_RS21730,apeE,avrA,bcfB,cheA,cheB,cheM,cheW,csgB,csgD,fimC,fimY,flgD,flgG,fliC,fliD,fliF,fliG,fliI,fliK,fliL,fliM,fliP,fljB,gogB,hilC,hilD,invA,iroB,iroC,iroD,iroN,lpfA,lpfD,mig-5,motB,nmpC,pefA,pefB,pefC,pefD,pegA,pegB,pegC,pipB,pltA,ratB,rck,rpoS,safA,safB,safC,safD,sefA,sefB,sefC,sefD,shdA,sifB,sinH,sipD,sodCI,sopA,sopD2,sopE,spiC/ssaB,spvB,spvC,spvD,ssaT,ssaU,sseI/srfH,sseK1,sseK2,sseL,sspH1,sspH2,staA,staB,stcA,stcD,stdB,steC,steD,steF,stkA,tae4,tcfA,tcfD,tlde1
2
+ Enteritidis,0,0,0.9,0.9,1,0,0.9,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0.9,1,0,0,1,1,1,1,1,1,1,1,1,1,1,1,1,0,0,1,1,1,1,1,1,1,0,0,0.9,1,1,1,1,1,1,1,1,0.7,1,1,0,0.6,0.6,0.6,1,1,1,1,0,1,0.5,1,1,1,1,1,1,0.9,0.9,0.9,0,0.9,0,1,1,1,0.9,1,1,0.7,0.7,0.6,1,1,1,0.6,0.9,1,0,0.9,0,0,0,0,1,1,1,1,0,0,0,0,1
3
+ Typhimurium,1,1,0,0,0,0,0,1,1,1,1,1,1,1,1,0.9848,0.8333,0.9848,0.8333,1,1,1,0.9848,0.9545,1,1,1,1,1,0.9848,0.9848,0.8939,0,0,1,1,1,1,1,1,1,1,0.9848,1,0.9848,1,1,0.9848,1,1,1,1,1,1,1,1,0.9242,0.1515,0.9848,0.9848,1,1,1,1,1,1,0.9848,0.1061,1,1,0.0909,0.0909,0.0909,0.0909,0,0,0,0.9848,0,1,0.0455,1,0.9697,1,1,1,0,0,0,0,0.1212,0.9394,1,1,0.9697,0.9848,0.9394,0,0.9848,0.1061,0.0909,0.1061,0.9848,0.9848,0.9242,0.9394,0.9848,0.9697,0.0152,0.9697,0,0,0.9848,0.9848,1,0.9848,0,0,0,0.9848,0,0,1
@@ -0,0 +1,11 @@
1
+ Sample,dose_cfu,serotype
2
+ S001,500,Enteritidis
3
+ S002,1000,Enteritidis
4
+ S003,2000,Enteritidis
5
+ S004,10000,Enteritidis
6
+ S005,100000,Enteritidis
7
+ S006,500,Typhimurium
8
+ S007,1000,Typhimurium
9
+ S008,2000,Typhimurium
10
+ S009,10000,Typhimurium
11
+ S010,100000,Typhimurium
@@ -18,7 +18,10 @@ classifiers = [
18
18
  "Topic :: Scientific/Engineering :: Bio-Informatics",
19
19
  ]
20
20
  dependencies = [
21
- "autogluon.tabular==1.1.1",
21
+ # The bundled ensemble uses both Torch and FastAI neural networks.
22
+ "autogluon.tabular[fastai]==1.1.1",
23
+ # AutoGluon 1.1.1 imports pkg_resources, removed in newer setuptools.
24
+ "setuptools<81",
22
25
  "pandas>=2.0",
23
26
  "openpyxl>=3.0",
24
27
  "rich-argparse>=1.4",
@@ -13,7 +13,7 @@ salmopredict:
13
13
  The same core is shared by a command-line interface and a Streamlit GUI.
14
14
  """
15
15
 
16
- __version__ = "0.2.0"
16
+ __version__ = "0.3.0"
17
17
 
18
18
  # One-paragraph summary shown in both the CLI help and the GUI, so the two
19
19
  # interfaces describe the tool with identical wording.
@@ -1,7 +1,7 @@
1
1
  """Command-line interface for salmopredict.
2
2
 
3
3
  Commands:
4
- salmopredict run -- predict Incidence for one or more feature CSVs
4
+ salmopredict run -- predict from features or samples + gene frequencies
5
5
  salmopredict gui -- launch the Streamlit graphical interface
6
6
  salmopredict check -- verify dependencies and the model
7
7
  salmopredict build-model -- clone a full AutoGluon model into a slim deploy copy
@@ -73,12 +73,18 @@ def build_parser() -> argparse.ArgumentParser:
73
73
 
74
74
  default_model = default_model_path()
75
75
 
76
- run = sub.add_parser("run", help="Predict Incidence for one feature CSV.",
76
+ run = sub.add_parser("run", help="Predict from features or samples + gene frequencies.",
77
77
  formatter_class=_HelpFormatter)
78
- run.add_argument("-i", "--input", required=True,
78
+ inputs = run.add_mutually_exclusive_group(required=True)
79
+ inputs.add_argument("-i", "--input",
79
80
  help="One feature CSV file. If it has a 'Sample' column, the "
80
81
  "output carries Sample + the prediction; otherwise it is "
81
82
  "the prediction column only.")
83
+ inputs.add_argument("--samples", help="Sample CSV with Sample, dose_cfu (raw CFU), "
84
+ "and serotype. Requires --gene-frequencies.")
85
+ run.add_argument("--gene-frequencies", help="CSV with Serotype and gene columns "
86
+ "containing frequencies in [0, 1]. Used only with --samples; "
87
+ "also writes features_<samples-stem>.csv.")
82
88
  run.add_argument("-o", "--output", required=True,
83
89
  help="Output directory (created if missing); the result is "
84
90
  "written there as pred_<input-stem>.csv.")
@@ -128,8 +134,10 @@ def build_parser() -> argparse.ArgumentParser:
128
134
 
129
135
 
130
136
  def _cmd_run(args: argparse.Namespace) -> int:
137
+ if bool(args.samples) != bool(args.gene_frequencies):
138
+ raise ValueError("--samples and --gene-frequencies must be supplied together.")
131
139
  config = PredictConfig(
132
- input_path=Path(args.input),
140
+ input_path=Path(args.samples or args.input),
133
141
  output_dir=Path(args.output),
134
142
  model_path=Path(args.model) if args.model else None,
135
143
  ensemble_model=args.model_name,
@@ -137,6 +145,7 @@ def _cmd_run(args: argparse.Namespace) -> int:
137
145
  missing_warn_fraction=args.missing_warn_frac,
138
146
  attach_metadata=Path(args.attach) if args.attach else None,
139
147
  force=args.force,
148
+ gene_frequencies=Path(args.gene_frequencies) if args.gene_frequencies else None,
140
149
  )
141
150
 
142
151
  # Imported here so 'check'/'gui' work even if autogluon is absent.
@@ -147,6 +156,8 @@ def _cmd_run(args: argparse.Namespace) -> int:
147
156
  cols = ", ".join(result.predictions.columns)
148
157
  print(f"\nDone. label={result.label} model={result.model_used}")
149
158
  print(f" {result.input_path.name}: {result.n_rows} rows -> {result.output_path}")
159
+ if result.features_path is not None:
160
+ print(f" weighted features: {result.features_path}")
150
161
  print(f" Sample column: {'yes' if result.has_sample else 'no'}"
151
162
  + (" (metadata attached)" if result.attached else ""))
152
163
  print(f" output columns: {cols}")
@@ -71,6 +71,10 @@ class PredictConfig:
71
71
  One input feature CSV is aligned to the model and predicted, producing one
72
72
  output file ``pred_<input-stem>.csv`` in ``output_dir``.
73
73
 
74
+ With ``gene_frequencies``, input_path instead contains Sample, dose_cfu and
75
+ serotype. The run also exports ``features_<input-stem>.csv`` and includes
76
+ those three sample columns in the prediction output.
77
+
74
78
  Output columns depend on the input:
75
79
  * no ``Sample`` column -> just the prediction (``Incidence(%)``);
76
80
  * a ``Sample`` column -> ``Sample`` + ``Incidence(%)``, and, if
@@ -90,6 +94,9 @@ class PredictConfig:
90
94
 
91
95
  force: bool = False
92
96
 
97
+ # When provided, input_path is a Sample/dose_cfu/serotype table.
98
+ gene_frequencies: Optional[Path] = None
99
+
93
100
  def __post_init__(self) -> None:
94
101
  self.input_path = Path(self.input_path)
95
102
  self.output_dir = Path(self.output_dir)
@@ -97,6 +104,11 @@ class PredictConfig:
97
104
  self.model_path = Path(self.model_path)
98
105
  if self.attach_metadata is not None:
99
106
  self.attach_metadata = Path(self.attach_metadata)
107
+ if self.gene_frequencies is not None:
108
+ self.gene_frequencies = Path(self.gene_frequencies)
109
+ if self.attach_metadata is not None:
110
+ raise ValueError("--attach is only supported with feature CSV input; "
111
+ "two-table input already includes sample metadata.")
100
112
  if not 0.0 <= float(self.missing_warn_fraction) <= 1.0:
101
113
  raise ValueError("missing_warn_fraction must be between 0 and 1")
102
114
  self.missing_warn_fraction = float(self.missing_warn_fraction)