prform 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- prform-0.2.0/LICENSE +21 -0
- prform-0.2.0/MANIFEST.in +6 -0
- prform-0.2.0/PKG-INFO +222 -0
- prform-0.2.0/README.md +195 -0
- prform-0.2.0/prform/__init__.py +14 -0
- prform-0.2.0/prform/__main__.py +6 -0
- prform-0.2.0/prform/_calibrator.py +75 -0
- prform-0.2.0/prform/_data/calibrator_components.joblib +0 -0
- prform-0.2.0/prform/_data/examples/minus1_PRF/GCA_031253115.1_ASM3125311v1_genomic.gff +24 -0
- prform-0.2.0/prform/_data/examples/minus1_PRF/coronavirus_khosta2.fasta +488 -0
- prform-0.2.0/prform/_data/examples/minus1_PRF/phage_tyson.fasta +1242 -0
- prform-0.2.0/prform/_data/examples/no_PRF/ledantevirus.fasta +198 -0
- prform-0.2.0/prform/_data/examples/no_PRF/monkeypox.fasta +3292 -0
- prform-0.2.0/prform/_data/examples/no_PRF/nanovirus.fasta +91 -0
- prform-0.2.0/prform/_data/examples/no_PRF/orbivirus.fasta +48 -0
- prform-0.2.0/prform/_data/examples/no_PRF/ourmiavirus.fasta +84 -0
- prform-0.2.0/prform/_data/examples/plus1_PRF/influenza_pax.fasta +38 -0
- prform-0.2.0/prform/_data/model_best.pth +0 -0
- prform-0.2.0/prform/_io.py +97 -0
- prform-0.2.0/prform/_model.py +75 -0
- prform-0.2.0/prform/cli.py +281 -0
- prform-0.2.0/prform/predictor.py +265 -0
- prform-0.2.0/prform.egg-info/PKG-INFO +222 -0
- prform-0.2.0/prform.egg-info/SOURCES.txt +29 -0
- prform-0.2.0/prform.egg-info/dependency_links.txt +1 -0
- prform-0.2.0/prform.egg-info/entry_points.txt +2 -0
- prform-0.2.0/prform.egg-info/requires.txt +9 -0
- prform-0.2.0/prform.egg-info/top_level.txt +1 -0
- prform-0.2.0/pyproject.toml +55 -0
- prform-0.2.0/setup.cfg +4 -0
- prform-0.2.0/tests/test_smoke.py +150 -0
prform-0.2.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Khoa Hoang
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
prform-0.2.0/MANIFEST.in
ADDED
prform-0.2.0/PKG-INFO
ADDED
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: prform
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Predict programmed ribosomal frameshift (PRF) sites in RNA sequences.
|
|
5
|
+
Author-email: Khoa Hoang <khoang99@stanford.edu>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/khoa-yelo/PRForm
|
|
8
|
+
Project-URL: Issues, https://github.com/khoa-yelo/PRForm/issues
|
|
9
|
+
Keywords: bioinformatics,rna,ribosomal-frameshift,deep-learning
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
15
|
+
Requires-Python: >=3.9
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Requires-Dist: torch>=2.0
|
|
19
|
+
Requires-Dist: numpy>=1.22
|
|
20
|
+
Requires-Dist: scipy>=1.9
|
|
21
|
+
Requires-Dist: scikit-learn>=1.1
|
|
22
|
+
Requires-Dist: joblib>=1.2
|
|
23
|
+
Requires-Dist: h5py>=3.7
|
|
24
|
+
Provides-Extra: dev
|
|
25
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
26
|
+
Dynamic: license-file
|
|
27
|
+
|
|
28
|
+
# PRForm
|
|
29
|
+
|
|
30
|
+
Predict programmed ribosomal frameshift (PRF) sites in Viral genomes.
|
|
31
|
+
|
|
32
|
+
`prform` is a deep-learning tool that scores every nucleotide in viral genome
|
|
33
|
+
sequence for the probability that a -1 or +1 frameshift event occurs there.
|
|
34
|
+
Predictions come from a 1D residual network with a 10 kb receptive
|
|
35
|
+
field.
|
|
36
|
+
|
|
37
|
+
The model was trained on NCBI ~50,000 deduplicated viral genomes (as of Feb 2026). Trained model checkpoint and calibrator are bundled with the package and ready to run.
|
|
38
|
+
|
|
39
|
+
## Install
|
|
40
|
+
|
|
41
|
+
From PyPI:
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
pip install prform
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
From bioconda:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
conda install -c bioconda prform
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
From source (development):
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
git clone https://github.com/khoa-yelo/PRForm.git
|
|
57
|
+
cd PRForm/prform_pkg
|
|
58
|
+
pip install -e .[dev]
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
## Usage
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
prform predict input.fasta --outdir results/
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
(equivalent: `python -m prform predict ...` — useful when `~/.local/bin`
|
|
68
|
+
isn't on your `PATH`.)
|
|
69
|
+
|
|
70
|
+
Writes two files into `results/`:
|
|
71
|
+
|
|
72
|
+
- `predictions.csv` — one row per FASTA record: the top-k predicted peak
|
|
73
|
+
positions with their calibrated PRF probability and frameshift type
|
|
74
|
+
(-1 PRF, +1 PRF, or no frameshift detected).
|
|
75
|
+
- `predictions.h5` — full per-nucleotide calibrated probability tracks.
|
|
76
|
+
|
|
77
|
+
`predictions.h5` layout:
|
|
78
|
+
|
|
79
|
+
```
|
|
80
|
+
records/
|
|
81
|
+
accession_id (R,) str
|
|
82
|
+
length (R,) int
|
|
83
|
+
top{k}_pos (R,) int # k = 1..top_k, 1-based; -1 if no peak
|
|
84
|
+
top{k}_prob (R,) float32
|
|
85
|
+
top{k}_type (R,) int8 # -1 PRF, +1 PRF, or 0 = no frameshift detected
|
|
86
|
+
per_nucleotide/
|
|
87
|
+
{accession_id}/ # one group per FASTA record
|
|
88
|
+
no_PRF_prob (L,) float32
|
|
89
|
+
minus1_PRF_prob (L,) float32
|
|
90
|
+
plus1_PRF_prob (L,) float32
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
With `--both-strands` each orientation gets its own block — `fwd_`/`rev_`
|
|
94
|
+
column and dataset prefixes, and `fwd`/`rev` per-nucleotide subgroups — see
|
|
95
|
+
[Scoring both strands](#scoring-both-strands).
|
|
96
|
+
|
|
97
|
+
Read a record's per-nucleotide tracks in Python:
|
|
98
|
+
|
|
99
|
+
```python
|
|
100
|
+
import h5py
|
|
101
|
+
with h5py.File("results/predictions.h5") as f:
|
|
102
|
+
for acc in f["records/accession_id"].asstr()[:]:
|
|
103
|
+
g = f[f"per_nucleotide/{acc}"]
|
|
104
|
+
minus1 = g["minus1_PRF_prob"][:]
|
|
105
|
+
plus1 = g["plus1_PRF_prob"][:]
|
|
106
|
+
prf_track = minus1 + plus1
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
## Scoring both strands
|
|
110
|
+
|
|
111
|
+
PRForm is a **directional** detector: it only sees a frameshift in the
|
|
112
|
+
orientation it is given. A PRF encoded on the minus strand of your input is
|
|
113
|
+
invisible on a forward-only pass. Pass `--both-strands` to score each record in
|
|
114
|
+
both orientations:
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
prform predict genomes.fasta --outdir results/ --both-strands
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
**Both orientations are reported.** The peaks and tracks are duplicated into a
|
|
121
|
+
forward block and a reverse block:
|
|
122
|
+
|
|
123
|
+
```
|
|
124
|
+
predictions.csv
|
|
125
|
+
accession_id, length,
|
|
126
|
+
fwd_top{k}_pos, fwd_top{k}_prob, fwd_top{k}_type,
|
|
127
|
+
rev_top{k}_pos, rev_top{k}_prob, rev_top{k}_type
|
|
128
|
+
|
|
129
|
+
predictions.h5
|
|
130
|
+
records/ # same fwd_/rev_ prefixes as the CSV
|
|
131
|
+
accession_id, length,
|
|
132
|
+
fwd_top{k}_{pos,prob,type}, rev_top{k}_{pos,prob,type}
|
|
133
|
+
per_nucleotide/
|
|
134
|
+
{accession_id}/
|
|
135
|
+
fwd/ no_PRF_prob, minus1_PRF_prob, plus1_PRF_prob (L,) each
|
|
136
|
+
rev/ no_PRF_prob, minus1_PRF_prob, plus1_PRF_prob (L,) each
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
The reverse block is mirrored back into **forward-genome coordinates**, so both
|
|
140
|
+
blocks index the same nucleotide: `fwd_top1_pos`, `rev_top1_pos` and index `i`
|
|
141
|
+
of either track all refer to position `i+1` of the sequence as you supplied it.
|
|
142
|
+
Class channels are *not* swapped — a -1 PRF read on the reverse complement
|
|
143
|
+
stays in `minus1_PRF_prob`, because the shift direction is defined relative to
|
|
144
|
+
the strand being read.
|
|
145
|
+
|
|
146
|
+
## Test the install
|
|
147
|
+
|
|
148
|
+
`prform` ships with small FASTA fixtures (one folder per labeled
|
|
149
|
+
frameshift class) so you can verify the install end-to-end without any
|
|
150
|
+
external data. Copy them into your working directory:
|
|
151
|
+
|
|
152
|
+
```bash
|
|
153
|
+
prform download-examples ./prform_examples
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
Layout after download:
|
|
157
|
+
|
|
158
|
+
```
|
|
159
|
+
prform_examples/
|
|
160
|
+
plus1_PRF/ influenza_pax.fasta
|
|
161
|
+
minus1_PRF/ coronavirus_khosta2.fasta, phage_tyson.fasta
|
|
162
|
+
no_PRF/ monkeypox, ledantevirus, nanovirus, ourmiavirus, orbivirus
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
Run `predict` on each class:
|
|
166
|
+
|
|
167
|
+
```bash
|
|
168
|
+
# +1 PRF: peak should land near the labeled site
|
|
169
|
+
prform predict prform_examples/plus1_PRF/influenza_pax.fasta \
|
|
170
|
+
--outdir prform_test/plus1
|
|
171
|
+
|
|
172
|
+
# -1 PRF: peak should land near the labeled site
|
|
173
|
+
prform predict prform_examples/minus1_PRF/coronavirus_khosta2.fasta \
|
|
174
|
+
--outdir prform_test/minus1
|
|
175
|
+
|
|
176
|
+
# no PRF: no frameshift should be called
|
|
177
|
+
prform predict prform_examples/no_PRF/nanovirus.fasta \
|
|
178
|
+
--outdir prform_test/no_prf
|
|
179
|
+
|
|
180
|
+
cat prform_test/plus1/predictions.csv
|
|
181
|
+
cat prform_test/minus1/predictions.csv
|
|
182
|
+
cat prform_test/no_prf/predictions.csv
|
|
183
|
+
```
|
|
184
|
+
|
|
185
|
+
Expected behavior:
|
|
186
|
+
|
|
187
|
+
| fixture | labeled site | what to look for |
|
|
188
|
+
| --- | --- | --- |
|
|
189
|
+
| `plus1_PRF/influenza_pax.fasta` | +1 PRF @ pos 582 | `top1_pos` within a few nt of 582, `top1_type == 1` |
|
|
190
|
+
| `minus1_PRF/coronavirus_khosta2.fasta` | -1 PRF @ pos 13248 | `top1_pos` near 13248, `top1_type == -1` |
|
|
191
|
+
| `minus1_PRF/phage_tyson.fasta` | -1 PRF @ pos 10279 | `top1_pos` near 10279, `top1_type == -1` |
|
|
192
|
+
| `no_PRF/*.fasta` | none | `top1_type == 0` (no frameshift called) |
|
|
193
|
+
|
|
194
|
+
Each labeled-positive FASTA header carries the labeled site for easy comparison.
|
|
195
|
+
|
|
196
|
+
## Python API
|
|
197
|
+
|
|
198
|
+
```python
|
|
199
|
+
from prform import PRFormPredictor
|
|
200
|
+
|
|
201
|
+
predictor = PRFormPredictor() # loads bundled weights
|
|
202
|
+
results = predictor.predict_fasta("input.fasta") # list of dicts
|
|
203
|
+
# or
|
|
204
|
+
results = predictor.predict_sequences([("rec1", "ACGT" * 1000)])
|
|
205
|
+
# score both orientations, keep the better strand per record
|
|
206
|
+
results = predictor.predict_fasta("input.fasta", both_strands=True)
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
Each result entry has `accession_id`, `length`, `strand`, `cal_probs_3`
|
|
210
|
+
(3, L) and a `top` list of peak summaries (`pos`, `prob`, `type`) — all from
|
|
211
|
+
the winning strand. With `both_strands=True`, `rec["fwd"]` and `rec["rev"]`
|
|
212
|
+
each hold that orientation's own `cal_probs_3` and `top`, both in
|
|
213
|
+
forward-genome coordinates; otherwise they are `None`.
|
|
214
|
+
|
|
215
|
+
## Hardware
|
|
216
|
+
|
|
217
|
+
CPU works. GPU is faster — `prform` automatically uses CUDA when available.
|
|
218
|
+
For long sequences (>1 Mb total input) a GPU is recommended.
|
|
219
|
+
|
|
220
|
+
## License
|
|
221
|
+
|
|
222
|
+
MIT
|
prform-0.2.0/README.md
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
# PRForm
|
|
2
|
+
|
|
3
|
+
Predict programmed ribosomal frameshift (PRF) sites in Viral genomes.
|
|
4
|
+
|
|
5
|
+
`prform` is a deep-learning tool that scores every nucleotide in viral genome
|
|
6
|
+
sequence for the probability that a -1 or +1 frameshift event occurs there.
|
|
7
|
+
Predictions come from a 1D residual network with a 10 kb receptive
|
|
8
|
+
field.
|
|
9
|
+
|
|
10
|
+
The model was trained on NCBI ~50,000 deduplicated viral genomes (as of Feb 2026). Trained model checkpoint and calibrator are bundled with the package and ready to run.
|
|
11
|
+
|
|
12
|
+
## Install
|
|
13
|
+
|
|
14
|
+
From PyPI:
|
|
15
|
+
|
|
16
|
+
```bash
|
|
17
|
+
pip install prform
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
From bioconda:
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
conda install -c bioconda prform
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
From source (development):
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
git clone https://github.com/khoa-yelo/PRForm.git
|
|
30
|
+
cd PRForm/prform_pkg
|
|
31
|
+
pip install -e .[dev]
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## Usage
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
prform predict input.fasta --outdir results/
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
(equivalent: `python -m prform predict ...` — useful when `~/.local/bin`
|
|
41
|
+
isn't on your `PATH`.)
|
|
42
|
+
|
|
43
|
+
Writes two files into `results/`:
|
|
44
|
+
|
|
45
|
+
- `predictions.csv` — one row per FASTA record: the top-k predicted peak
|
|
46
|
+
positions with their calibrated PRF probability and frameshift type
|
|
47
|
+
(-1 PRF, +1 PRF, or no frameshift detected).
|
|
48
|
+
- `predictions.h5` — full per-nucleotide calibrated probability tracks.
|
|
49
|
+
|
|
50
|
+
`predictions.h5` layout:
|
|
51
|
+
|
|
52
|
+
```
|
|
53
|
+
records/
|
|
54
|
+
accession_id (R,) str
|
|
55
|
+
length (R,) int
|
|
56
|
+
top{k}_pos (R,) int # k = 1..top_k, 1-based; -1 if no peak
|
|
57
|
+
top{k}_prob (R,) float32
|
|
58
|
+
top{k}_type (R,) int8 # -1 PRF, +1 PRF, or 0 = no frameshift detected
|
|
59
|
+
per_nucleotide/
|
|
60
|
+
{accession_id}/ # one group per FASTA record
|
|
61
|
+
no_PRF_prob (L,) float32
|
|
62
|
+
minus1_PRF_prob (L,) float32
|
|
63
|
+
plus1_PRF_prob (L,) float32
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
With `--both-strands` each orientation gets its own block — `fwd_`/`rev_`
|
|
67
|
+
column and dataset prefixes, and `fwd`/`rev` per-nucleotide subgroups — see
|
|
68
|
+
[Scoring both strands](#scoring-both-strands).
|
|
69
|
+
|
|
70
|
+
Read a record's per-nucleotide tracks in Python:
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
import h5py
|
|
74
|
+
with h5py.File("results/predictions.h5") as f:
|
|
75
|
+
for acc in f["records/accession_id"].asstr()[:]:
|
|
76
|
+
g = f[f"per_nucleotide/{acc}"]
|
|
77
|
+
minus1 = g["minus1_PRF_prob"][:]
|
|
78
|
+
plus1 = g["plus1_PRF_prob"][:]
|
|
79
|
+
prf_track = minus1 + plus1
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
## Scoring both strands
|
|
83
|
+
|
|
84
|
+
PRForm is a **directional** detector: it only sees a frameshift in the
|
|
85
|
+
orientation it is given. A PRF encoded on the minus strand of your input is
|
|
86
|
+
invisible on a forward-only pass. Pass `--both-strands` to score each record in
|
|
87
|
+
both orientations:
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
prform predict genomes.fasta --outdir results/ --both-strands
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
**Both orientations are reported.** The peaks and tracks are duplicated into a
|
|
94
|
+
forward block and a reverse block:
|
|
95
|
+
|
|
96
|
+
```
|
|
97
|
+
predictions.csv
|
|
98
|
+
accession_id, length,
|
|
99
|
+
fwd_top{k}_pos, fwd_top{k}_prob, fwd_top{k}_type,
|
|
100
|
+
rev_top{k}_pos, rev_top{k}_prob, rev_top{k}_type
|
|
101
|
+
|
|
102
|
+
predictions.h5
|
|
103
|
+
records/ # same fwd_/rev_ prefixes as the CSV
|
|
104
|
+
accession_id, length,
|
|
105
|
+
fwd_top{k}_{pos,prob,type}, rev_top{k}_{pos,prob,type}
|
|
106
|
+
per_nucleotide/
|
|
107
|
+
{accession_id}/
|
|
108
|
+
fwd/ no_PRF_prob, minus1_PRF_prob, plus1_PRF_prob (L,) each
|
|
109
|
+
rev/ no_PRF_prob, minus1_PRF_prob, plus1_PRF_prob (L,) each
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
The reverse block is mirrored back into **forward-genome coordinates**, so both
|
|
113
|
+
blocks index the same nucleotide: `fwd_top1_pos`, `rev_top1_pos` and index `i`
|
|
114
|
+
of either track all refer to position `i+1` of the sequence as you supplied it.
|
|
115
|
+
Class channels are *not* swapped — a -1 PRF read on the reverse complement
|
|
116
|
+
stays in `minus1_PRF_prob`, because the shift direction is defined relative to
|
|
117
|
+
the strand being read.
|
|
118
|
+
|
|
119
|
+
## Test the install
|
|
120
|
+
|
|
121
|
+
`prform` ships with small FASTA fixtures (one folder per labeled
|
|
122
|
+
frameshift class) so you can verify the install end-to-end without any
|
|
123
|
+
external data. Copy them into your working directory:
|
|
124
|
+
|
|
125
|
+
```bash
|
|
126
|
+
prform download-examples ./prform_examples
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
Layout after download:
|
|
130
|
+
|
|
131
|
+
```
|
|
132
|
+
prform_examples/
|
|
133
|
+
plus1_PRF/ influenza_pax.fasta
|
|
134
|
+
minus1_PRF/ coronavirus_khosta2.fasta, phage_tyson.fasta
|
|
135
|
+
no_PRF/ monkeypox, ledantevirus, nanovirus, ourmiavirus, orbivirus
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
Run `predict` on each class:
|
|
139
|
+
|
|
140
|
+
```bash
|
|
141
|
+
# +1 PRF: peak should land near the labeled site
|
|
142
|
+
prform predict prform_examples/plus1_PRF/influenza_pax.fasta \
|
|
143
|
+
--outdir prform_test/plus1
|
|
144
|
+
|
|
145
|
+
# -1 PRF: peak should land near the labeled site
|
|
146
|
+
prform predict prform_examples/minus1_PRF/coronavirus_khosta2.fasta \
|
|
147
|
+
--outdir prform_test/minus1
|
|
148
|
+
|
|
149
|
+
# no PRF: no frameshift should be called
|
|
150
|
+
prform predict prform_examples/no_PRF/nanovirus.fasta \
|
|
151
|
+
--outdir prform_test/no_prf
|
|
152
|
+
|
|
153
|
+
cat prform_test/plus1/predictions.csv
|
|
154
|
+
cat prform_test/minus1/predictions.csv
|
|
155
|
+
cat prform_test/no_prf/predictions.csv
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
Expected behavior:
|
|
159
|
+
|
|
160
|
+
| fixture | labeled site | what to look for |
|
|
161
|
+
| --- | --- | --- |
|
|
162
|
+
| `plus1_PRF/influenza_pax.fasta` | +1 PRF @ pos 582 | `top1_pos` within a few nt of 582, `top1_type == 1` |
|
|
163
|
+
| `minus1_PRF/coronavirus_khosta2.fasta` | -1 PRF @ pos 13248 | `top1_pos` near 13248, `top1_type == -1` |
|
|
164
|
+
| `minus1_PRF/phage_tyson.fasta` | -1 PRF @ pos 10279 | `top1_pos` near 10279, `top1_type == -1` |
|
|
165
|
+
| `no_PRF/*.fasta` | none | `top1_type == 0` (no frameshift called) |
|
|
166
|
+
|
|
167
|
+
Each labeled-positive FASTA header carries the labeled site for easy comparison.
|
|
168
|
+
|
|
169
|
+
## Python API
|
|
170
|
+
|
|
171
|
+
```python
|
|
172
|
+
from prform import PRFormPredictor
|
|
173
|
+
|
|
174
|
+
predictor = PRFormPredictor() # loads bundled weights
|
|
175
|
+
results = predictor.predict_fasta("input.fasta") # list of dicts
|
|
176
|
+
# or
|
|
177
|
+
results = predictor.predict_sequences([("rec1", "ACGT" * 1000)])
|
|
178
|
+
# score both orientations, keep the better strand per record
|
|
179
|
+
results = predictor.predict_fasta("input.fasta", both_strands=True)
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
Each result entry has `accession_id`, `length`, `strand`, `cal_probs_3`
|
|
183
|
+
(3, L) and a `top` list of peak summaries (`pos`, `prob`, `type`) — all from
|
|
184
|
+
the winning strand. With `both_strands=True`, `rec["fwd"]` and `rec["rev"]`
|
|
185
|
+
each hold that orientation's own `cal_probs_3` and `top`, both in
|
|
186
|
+
forward-genome coordinates; otherwise they are `None`.
|
|
187
|
+
|
|
188
|
+
## Hardware
|
|
189
|
+
|
|
190
|
+
CPU works. GPU is faster — `prform` automatically uses CUDA when available.
|
|
191
|
+
For long sequences (>1 Mb total input) a GPU is recommended.
|
|
192
|
+
|
|
193
|
+
## License
|
|
194
|
+
|
|
195
|
+
MIT
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""prform — predict programmed ribosomal frameshift (PRF) sites."""
|
|
2
|
+
|
|
3
|
+
from importlib.resources import files
|
|
4
|
+
|
|
5
|
+
from prform.predictor import PRFormPredictor
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def examples_dir():
|
|
9
|
+
"""Path to the bundled example FASTA fixtures (positive/ and negative/)."""
|
|
10
|
+
return files("prform") / "_data" / "examples"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
__version__ = "0.2.0"
|
|
14
|
+
__all__ = ["PRFormPredictor", "examples_dir", "__version__"]
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
"""
|
|
2
|
+
PRFormCalibrator — five isotonic regressors mapping raw 3-class softmax to
|
|
3
|
+
calibrated per-nucleotide probabilities, plus a record-level isotonic.
|
|
4
|
+
|
|
5
|
+
Loaded from a dict-of-components joblib file, so the saved data is
|
|
6
|
+
independent of any specific module path.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from importlib.resources import files
|
|
10
|
+
|
|
11
|
+
import joblib
|
|
12
|
+
import numpy as np
|
|
13
|
+
from scipy.ndimage import maximum_filter1d
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class PRFormCalibrator:
|
|
17
|
+
def __init__(self, iso_class0, iso_class1, iso_class2,
|
|
18
|
+
iso_minus1_flank, iso_plus1_flank, iso_seq, flank_k=3):
|
|
19
|
+
self.iso_class0 = iso_class0
|
|
20
|
+
self.iso_class1 = iso_class1
|
|
21
|
+
self.iso_class2 = iso_class2
|
|
22
|
+
self.iso_minus1_flank = iso_minus1_flank
|
|
23
|
+
self.iso_plus1_flank = iso_plus1_flank
|
|
24
|
+
self.iso_seq = iso_seq
|
|
25
|
+
self.flank_k = int(flank_k)
|
|
26
|
+
|
|
27
|
+
def predict(self, probs_3):
|
|
28
|
+
"""
|
|
29
|
+
probs_3 : (3, L) raw softmax — rows = [no_PRF, dir-1, dir+1].
|
|
30
|
+
|
|
31
|
+
Returns
|
|
32
|
+
-------
|
|
33
|
+
dict with
|
|
34
|
+
probs_3 : (3, L) float32, sharp per-class calibrated, renormalized
|
|
35
|
+
probs_minus1_flank : (L,) float32, P(any true -1 site within +/- k)
|
|
36
|
+
probs_plus1_flank : (L,) float32, P(any true +1 site within +/- k)
|
|
37
|
+
sequence : float, P(record contains any PRF site)
|
|
38
|
+
"""
|
|
39
|
+
probs_3 = np.asarray(probs_3, dtype=np.float32)
|
|
40
|
+
if probs_3.ndim != 2 or probs_3.shape[0] != 3:
|
|
41
|
+
raise ValueError(f"probs_3 must be (3, L); got {probs_3.shape}")
|
|
42
|
+
|
|
43
|
+
c0 = self.iso_class0.transform(probs_3[0])
|
|
44
|
+
c1 = self.iso_class1.transform(probs_3[1])
|
|
45
|
+
c2 = self.iso_class2.transform(probs_3[2])
|
|
46
|
+
s = c0 + c1 + c2
|
|
47
|
+
s = np.where(s == 0, 1.0, s)
|
|
48
|
+
sharp = np.stack([c0 / s, c1 / s, c2 / s], axis=0).astype(np.float32)
|
|
49
|
+
|
|
50
|
+
win = 2 * self.flank_k + 1
|
|
51
|
+
w1 = maximum_filter1d(probs_3[1], size=win, mode="nearest")
|
|
52
|
+
w2 = maximum_filter1d(probs_3[2], size=win, mode="nearest")
|
|
53
|
+
flank_minus1 = self.iso_minus1_flank.transform(w1).astype(np.float32)
|
|
54
|
+
flank_plus1 = self.iso_plus1_flank.transform(w2).astype(np.float32)
|
|
55
|
+
|
|
56
|
+
prf_max = float((probs_3[1] + probs_3[2]).max())
|
|
57
|
+
seq = float(self.iso_seq.transform([prf_max])[0])
|
|
58
|
+
|
|
59
|
+
return {
|
|
60
|
+
"probs_3": sharp,
|
|
61
|
+
"probs_minus1_flank": flank_minus1,
|
|
62
|
+
"probs_plus1_flank": flank_plus1,
|
|
63
|
+
"sequence": seq,
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
def transform_sequence(self, prf_max):
|
|
67
|
+
return float(self.iso_seq.transform([float(prf_max)])[0])
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def load_bundled_calibrator():
|
|
71
|
+
"""Load the calibrator shipped inside the package."""
|
|
72
|
+
path = files("prform._data").joinpath("calibrator_components.joblib")
|
|
73
|
+
with path.open("rb") as fh:
|
|
74
|
+
components = joblib.load(fh)
|
|
75
|
+
return PRFormCalibrator(**components)
|
|
Binary file
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
##gff-version 3
|
|
2
|
+
#!gff-spec-version 1.21
|
|
3
|
+
#!processor NCBI annotwriter
|
|
4
|
+
#!genome-build ASM3125311v1
|
|
5
|
+
#!genome-build-accession NCBI_Assembly:GCA_031253115.1
|
|
6
|
+
##sequence-region MZ190138.1 1 29220
|
|
7
|
+
##species https://www.ncbi.nlm.nih.gov/Taxonomy/Browser/wwwtax.cgi?id=2836124
|
|
8
|
+
MZ190138.1 Genbank region 1 29220 . + . ID=MZ190138.1:1..29220;Dbxref=taxon:2836124;collected-by=S. Lenshin%2C A. Romashin;collection-date=2020;country=Russia: Sochi;gbkey=Src;genome=genomic;isolation-source=feces pool;mol_type=genomic RNA;nat-host=Rhinolophus hipposideros (bat);strain=BtCoV/Khosta-2/Rh/Russia/2020
|
|
9
|
+
MZ190138.1 Genbank CDS 208 13248 . + 0 ID=cds-QVN46577.1;Dbxref=NCBI_GP:QVN46577.1;Name=QVN46577.1;exception=ribosomal slippage;gbkey=CDS;product=ORF1ab protein;protein_id=QVN46577.1
|
|
10
|
+
MZ190138.1 Genbank CDS 13248 21386 . + 0 ID=cds-QVN46577.1;Dbxref=NCBI_GP:QVN46577.1;Name=QVN46577.1;exception=ribosomal slippage;gbkey=CDS;product=ORF1ab protein;protein_id=QVN46577.1
|
|
11
|
+
MZ190138.1 Genbank CDS 208 13314 . + 0 ID=cds-QVN46568.1;Dbxref=NCBI_GP:QVN46568.1;Name=QVN46568.1;gbkey=CDS;product=ORF1a protein;protein_id=QVN46568.1
|
|
12
|
+
MZ190138.1 Genbank gene 21393 25154 . + . ID=gene-S;Name=S;gbkey=Gene;gene=S;gene_biotype=protein_coding
|
|
13
|
+
MZ190138.1 Genbank CDS 21393 25154 . + 0 ID=cds-QVN46569.1;Parent=gene-S;Dbxref=NCBI_GP:QVN46569.1;Name=QVN46569.1;gbkey=CDS;gene=S;product=spike glycoprotein;protein_id=QVN46569.1
|
|
14
|
+
MZ190138.1 Genbank CDS 25163 25975 . + 0 ID=cds-QVN46570.1;Dbxref=NCBI_GP:QVN46570.1;Name=QVN46570.1;gbkey=CDS;product=ORF3 protein;protein_id=QVN46570.1
|
|
15
|
+
MZ190138.1 Genbank gene 25999 26229 . + . ID=gene-E;Name=E;gbkey=Gene;gene=E;gene_biotype=protein_coding
|
|
16
|
+
MZ190138.1 Genbank CDS 25999 26229 . + 0 ID=cds-QVN46571.1;Parent=gene-E;Dbxref=NCBI_GP:QVN46571.1;Name=QVN46571.1;gbkey=CDS;gene=E;product=envelope protein;protein_id=QVN46571.1
|
|
17
|
+
MZ190138.1 Genbank gene 26277 26942 . + . ID=gene-M;Name=M;gbkey=Gene;gene=M;gene_biotype=protein_coding
|
|
18
|
+
MZ190138.1 Genbank CDS 26277 26942 . + 0 ID=cds-QVN46572.1;Parent=gene-M;Dbxref=NCBI_GP:QVN46572.1;Name=QVN46572.1;gbkey=CDS;gene=M;product=membrane protein;protein_id=QVN46572.1
|
|
19
|
+
MZ190138.1 Genbank CDS 26953 27141 . + 0 ID=cds-QVN46573.1;Dbxref=NCBI_GP:QVN46573.1;Name=QVN46573.1;gbkey=CDS;product=ORF6 protein;protein_id=QVN46573.1
|
|
20
|
+
MZ190138.1 Genbank CDS 27148 27504 . + 0 ID=cds-QVN46574.1;Dbxref=NCBI_GP:QVN46574.1;Name=QVN46574.1;gbkey=CDS;product=ORF7a protein;protein_id=QVN46574.1
|
|
21
|
+
MZ190138.1 Genbank CDS 27498 27626 . + 0 ID=cds-QVN46575.1;Dbxref=NCBI_GP:QVN46575.1;Name=QVN46575.1;gbkey=CDS;product=ORF7b protein;protein_id=QVN46575.1
|
|
22
|
+
MZ190138.1 Genbank gene 27623 28876 . + . ID=gene-N;Name=N;gbkey=Gene;gene=N;gene_biotype=protein_coding
|
|
23
|
+
MZ190138.1 Genbank CDS 27623 28876 . + 0 ID=cds-QVN46576.1;Parent=gene-N;Dbxref=NCBI_GP:QVN46576.1;Name=QVN46576.1;gbkey=CDS;gene=N;product=nucleocapsid protein;protein_id=QVN46576.1
|
|
24
|
+
###
|