prform 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. prform-0.2.0/LICENSE +21 -0
  2. prform-0.2.0/MANIFEST.in +6 -0
  3. prform-0.2.0/PKG-INFO +222 -0
  4. prform-0.2.0/README.md +195 -0
  5. prform-0.2.0/prform/__init__.py +14 -0
  6. prform-0.2.0/prform/__main__.py +6 -0
  7. prform-0.2.0/prform/_calibrator.py +75 -0
  8. prform-0.2.0/prform/_data/calibrator_components.joblib +0 -0
  9. prform-0.2.0/prform/_data/examples/minus1_PRF/GCA_031253115.1_ASM3125311v1_genomic.gff +24 -0
  10. prform-0.2.0/prform/_data/examples/minus1_PRF/coronavirus_khosta2.fasta +488 -0
  11. prform-0.2.0/prform/_data/examples/minus1_PRF/phage_tyson.fasta +1242 -0
  12. prform-0.2.0/prform/_data/examples/no_PRF/ledantevirus.fasta +198 -0
  13. prform-0.2.0/prform/_data/examples/no_PRF/monkeypox.fasta +3292 -0
  14. prform-0.2.0/prform/_data/examples/no_PRF/nanovirus.fasta +91 -0
  15. prform-0.2.0/prform/_data/examples/no_PRF/orbivirus.fasta +48 -0
  16. prform-0.2.0/prform/_data/examples/no_PRF/ourmiavirus.fasta +84 -0
  17. prform-0.2.0/prform/_data/examples/plus1_PRF/influenza_pax.fasta +38 -0
  18. prform-0.2.0/prform/_data/model_best.pth +0 -0
  19. prform-0.2.0/prform/_io.py +97 -0
  20. prform-0.2.0/prform/_model.py +75 -0
  21. prform-0.2.0/prform/cli.py +281 -0
  22. prform-0.2.0/prform/predictor.py +265 -0
  23. prform-0.2.0/prform.egg-info/PKG-INFO +222 -0
  24. prform-0.2.0/prform.egg-info/SOURCES.txt +29 -0
  25. prform-0.2.0/prform.egg-info/dependency_links.txt +1 -0
  26. prform-0.2.0/prform.egg-info/entry_points.txt +2 -0
  27. prform-0.2.0/prform.egg-info/requires.txt +9 -0
  28. prform-0.2.0/prform.egg-info/top_level.txt +1 -0
  29. prform-0.2.0/pyproject.toml +55 -0
  30. prform-0.2.0/setup.cfg +4 -0
  31. prform-0.2.0/tests/test_smoke.py +150 -0
prform-0.2.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Khoa Hoang
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,6 @@
1
+ include README.md
2
+ include LICENSE
3
+ include pyproject.toml
4
+ recursive-include prform/_data *
5
+ # Superseded artifacts kept in-repo for reproducibility, not shipped.
6
+ prune prform/_data/legacy
prform-0.2.0/PKG-INFO ADDED
@@ -0,0 +1,222 @@
1
+ Metadata-Version: 2.4
2
+ Name: prform
3
+ Version: 0.2.0
4
+ Summary: Predict programmed ribosomal frameshift (PRF) sites in RNA sequences.
5
+ Author-email: Khoa Hoang <khoang99@stanford.edu>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/khoa-yelo/PRForm
8
+ Project-URL: Issues, https://github.com/khoa-yelo/PRForm/issues
9
+ Keywords: bioinformatics,rna,ribosomal-frameshift,deep-learning
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Intended Audience :: Science/Research
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
15
+ Requires-Python: >=3.9
16
+ Description-Content-Type: text/markdown
17
+ License-File: LICENSE
18
+ Requires-Dist: torch>=2.0
19
+ Requires-Dist: numpy>=1.22
20
+ Requires-Dist: scipy>=1.9
21
+ Requires-Dist: scikit-learn>=1.1
22
+ Requires-Dist: joblib>=1.2
23
+ Requires-Dist: h5py>=3.7
24
+ Provides-Extra: dev
25
+ Requires-Dist: pytest>=7; extra == "dev"
26
+ Dynamic: license-file
27
+
28
+ # PRForm
29
+
30
+ Predict programmed ribosomal frameshift (PRF) sites in Viral genomes.
31
+
32
+ `prform` is a deep-learning tool that scores every nucleotide in viral genome
33
+ sequence for the probability that a -1 or +1 frameshift event occurs there.
34
+ Predictions come from a 1D residual network with a 10 kb receptive
35
+ field.
36
+
37
+ The model was trained on NCBI ~50,000 deduplicated viral genomes (as of Feb 2026). Trained model checkpoint and calibrator are bundled with the package and ready to run.
38
+
39
+ ## Install
40
+
41
+ From PyPI:
42
+
43
+ ```bash
44
+ pip install prform
45
+ ```
46
+
47
+ From bioconda:
48
+
49
+ ```bash
50
+ conda install -c bioconda prform
51
+ ```
52
+
53
+ From source (development):
54
+
55
+ ```bash
56
+ git clone https://github.com/khoa-yelo/PRForm.git
57
+ cd PRForm/prform_pkg
58
+ pip install -e .[dev]
59
+ ```
60
+
61
+ ## Usage
62
+
63
+ ```bash
64
+ prform predict input.fasta --outdir results/
65
+ ```
66
+
67
+ (equivalent: `python -m prform predict ...` — useful when `~/.local/bin`
68
+ isn't on your `PATH`.)
69
+
70
+ Writes two files into `results/`:
71
+
72
+ - `predictions.csv` — one row per FASTA record: the top-k predicted peak
73
+ positions with their calibrated PRF probability and frameshift type
74
+ (-1 PRF, +1 PRF, or no frameshift detected).
75
+ - `predictions.h5` — full per-nucleotide calibrated probability tracks.
76
+
77
+ `predictions.h5` layout:
78
+
79
+ ```
80
+ records/
81
+ accession_id (R,) str
82
+ length (R,) int
83
+ top{k}_pos (R,) int # k = 1..top_k, 1-based; -1 if no peak
84
+ top{k}_prob (R,) float32
85
+ top{k}_type (R,) int8 # -1 PRF, +1 PRF, or 0 = no frameshift detected
86
+ per_nucleotide/
87
+ {accession_id}/ # one group per FASTA record
88
+ no_PRF_prob (L,) float32
89
+ minus1_PRF_prob (L,) float32
90
+ plus1_PRF_prob (L,) float32
91
+ ```
92
+
93
+ With `--both-strands` each orientation gets its own block — `fwd_`/`rev_`
94
+ column and dataset prefixes, and `fwd`/`rev` per-nucleotide subgroups — see
95
+ [Scoring both strands](#scoring-both-strands).
96
+
97
+ Read a record's per-nucleotide tracks in Python:
98
+
99
+ ```python
100
+ import h5py
101
+ with h5py.File("results/predictions.h5") as f:
102
+ for acc in f["records/accession_id"].asstr()[:]:
103
+ g = f[f"per_nucleotide/{acc}"]
104
+ minus1 = g["minus1_PRF_prob"][:]
105
+ plus1 = g["plus1_PRF_prob"][:]
106
+ prf_track = minus1 + plus1
107
+ ```
108
+
109
+ ## Scoring both strands
110
+
111
+ PRForm is a **directional** detector: it only sees a frameshift in the
112
+ orientation it is given. A PRF encoded on the minus strand of your input is
113
+ invisible on a forward-only pass. Pass `--both-strands` to score each record in
114
+ both orientations:
115
+
116
+ ```bash
117
+ prform predict genomes.fasta --outdir results/ --both-strands
118
+ ```
119
+
120
+ **Both orientations are reported.** The peaks and tracks are duplicated into a
121
+ forward block and a reverse block:
122
+
123
+ ```
124
+ predictions.csv
125
+ accession_id, length,
126
+ fwd_top{k}_pos, fwd_top{k}_prob, fwd_top{k}_type,
127
+ rev_top{k}_pos, rev_top{k}_prob, rev_top{k}_type
128
+
129
+ predictions.h5
130
+ records/ # same fwd_/rev_ prefixes as the CSV
131
+ accession_id, length,
132
+ fwd_top{k}_{pos,prob,type}, rev_top{k}_{pos,prob,type}
133
+ per_nucleotide/
134
+ {accession_id}/
135
+ fwd/ no_PRF_prob, minus1_PRF_prob, plus1_PRF_prob (L,) each
136
+ rev/ no_PRF_prob, minus1_PRF_prob, plus1_PRF_prob (L,) each
137
+ ```
138
+
139
+ The reverse block is mirrored back into **forward-genome coordinates**, so both
140
+ blocks index the same nucleotide: `fwd_top1_pos`, `rev_top1_pos` and index `i`
141
+ of either track all refer to position `i+1` of the sequence as you supplied it.
142
+ Class channels are *not* swapped — a -1 PRF read on the reverse complement
143
+ stays in `minus1_PRF_prob`, because the shift direction is defined relative to
144
+ the strand being read.
145
+
146
+ ## Test the install
147
+
148
+ `prform` ships with small FASTA fixtures (one folder per labeled
149
+ frameshift class) so you can verify the install end-to-end without any
150
+ external data. Copy them into your working directory:
151
+
152
+ ```bash
153
+ prform download-examples ./prform_examples
154
+ ```
155
+
156
+ Layout after download:
157
+
158
+ ```
159
+ prform_examples/
160
+ plus1_PRF/ influenza_pax.fasta
161
+ minus1_PRF/ coronavirus_khosta2.fasta, phage_tyson.fasta
162
+ no_PRF/ monkeypox, ledantevirus, nanovirus, ourmiavirus, orbivirus
163
+ ```
164
+
165
+ Run `predict` on each class:
166
+
167
+ ```bash
168
+ # +1 PRF: peak should land near the labeled site
169
+ prform predict prform_examples/plus1_PRF/influenza_pax.fasta \
170
+ --outdir prform_test/plus1
171
+
172
+ # -1 PRF: peak should land near the labeled site
173
+ prform predict prform_examples/minus1_PRF/coronavirus_khosta2.fasta \
174
+ --outdir prform_test/minus1
175
+
176
+ # no PRF: no frameshift should be called
177
+ prform predict prform_examples/no_PRF/nanovirus.fasta \
178
+ --outdir prform_test/no_prf
179
+
180
+ cat prform_test/plus1/predictions.csv
181
+ cat prform_test/minus1/predictions.csv
182
+ cat prform_test/no_prf/predictions.csv
183
+ ```
184
+
185
+ Expected behavior:
186
+
187
+ | fixture | labeled site | what to look for |
188
+ | --- | --- | --- |
189
+ | `plus1_PRF/influenza_pax.fasta` | +1 PRF @ pos 582 | `top1_pos` within a few nt of 582, `top1_type == 1` |
190
+ | `minus1_PRF/coronavirus_khosta2.fasta` | -1 PRF @ pos 13248 | `top1_pos` near 13248, `top1_type == -1` |
191
+ | `minus1_PRF/phage_tyson.fasta` | -1 PRF @ pos 10279 | `top1_pos` near 10279, `top1_type == -1` |
192
+ | `no_PRF/*.fasta` | none | `top1_type == 0` (no frameshift called) |
193
+
194
+ Each labeled-positive FASTA header carries the labeled site for easy comparison.
195
+
196
+ ## Python API
197
+
198
+ ```python
199
+ from prform import PRFormPredictor
200
+
201
+ predictor = PRFormPredictor() # loads bundled weights
202
+ results = predictor.predict_fasta("input.fasta") # list of dicts
203
+ # or
204
+ results = predictor.predict_sequences([("rec1", "ACGT" * 1000)])
205
+ # score both orientations, keep the better strand per record
206
+ results = predictor.predict_fasta("input.fasta", both_strands=True)
207
+ ```
208
+
209
+ Each result entry has `accession_id`, `length`, `strand`, `cal_probs_3`
210
+ (3, L) and a `top` list of peak summaries (`pos`, `prob`, `type`) — all from
211
+ the winning strand. With `both_strands=True`, `rec["fwd"]` and `rec["rev"]`
212
+ each hold that orientation's own `cal_probs_3` and `top`, both in
213
+ forward-genome coordinates; otherwise they are `None`.
214
+
215
+ ## Hardware
216
+
217
+ CPU works. GPU is faster — `prform` automatically uses CUDA when available.
218
+ For long sequences (>1 Mb total input) a GPU is recommended.
219
+
220
+ ## License
221
+
222
+ MIT
prform-0.2.0/README.md ADDED
@@ -0,0 +1,195 @@
1
+ # PRForm
2
+
3
+ Predict programmed ribosomal frameshift (PRF) sites in Viral genomes.
4
+
5
+ `prform` is a deep-learning tool that scores every nucleotide in viral genome
6
+ sequence for the probability that a -1 or +1 frameshift event occurs there.
7
+ Predictions come from a 1D residual network with a 10 kb receptive
8
+ field.
9
+
10
+ The model was trained on NCBI ~50,000 deduplicated viral genomes (as of Feb 2026). Trained model checkpoint and calibrator are bundled with the package and ready to run.
11
+
12
+ ## Install
13
+
14
+ From PyPI:
15
+
16
+ ```bash
17
+ pip install prform
18
+ ```
19
+
20
+ From bioconda:
21
+
22
+ ```bash
23
+ conda install -c bioconda prform
24
+ ```
25
+
26
+ From source (development):
27
+
28
+ ```bash
29
+ git clone https://github.com/khoa-yelo/PRForm.git
30
+ cd PRForm/prform_pkg
31
+ pip install -e .[dev]
32
+ ```
33
+
34
+ ## Usage
35
+
36
+ ```bash
37
+ prform predict input.fasta --outdir results/
38
+ ```
39
+
40
+ (equivalent: `python -m prform predict ...` — useful when `~/.local/bin`
41
+ isn't on your `PATH`.)
42
+
43
+ Writes two files into `results/`:
44
+
45
+ - `predictions.csv` — one row per FASTA record: the top-k predicted peak
46
+ positions with their calibrated PRF probability and frameshift type
47
+ (-1 PRF, +1 PRF, or no frameshift detected).
48
+ - `predictions.h5` — full per-nucleotide calibrated probability tracks.
49
+
50
+ `predictions.h5` layout:
51
+
52
+ ```
53
+ records/
54
+ accession_id (R,) str
55
+ length (R,) int
56
+ top{k}_pos (R,) int # k = 1..top_k, 1-based; -1 if no peak
57
+ top{k}_prob (R,) float32
58
+ top{k}_type (R,) int8 # -1 PRF, +1 PRF, or 0 = no frameshift detected
59
+ per_nucleotide/
60
+ {accession_id}/ # one group per FASTA record
61
+ no_PRF_prob (L,) float32
62
+ minus1_PRF_prob (L,) float32
63
+ plus1_PRF_prob (L,) float32
64
+ ```
65
+
66
+ With `--both-strands` each orientation gets its own block — `fwd_`/`rev_`
67
+ column and dataset prefixes, and `fwd`/`rev` per-nucleotide subgroups — see
68
+ [Scoring both strands](#scoring-both-strands).
69
+
70
+ Read a record's per-nucleotide tracks in Python:
71
+
72
+ ```python
73
+ import h5py
74
+ with h5py.File("results/predictions.h5") as f:
75
+ for acc in f["records/accession_id"].asstr()[:]:
76
+ g = f[f"per_nucleotide/{acc}"]
77
+ minus1 = g["minus1_PRF_prob"][:]
78
+ plus1 = g["plus1_PRF_prob"][:]
79
+ prf_track = minus1 + plus1
80
+ ```
81
+
82
+ ## Scoring both strands
83
+
84
+ PRForm is a **directional** detector: it only sees a frameshift in the
85
+ orientation it is given. A PRF encoded on the minus strand of your input is
86
+ invisible on a forward-only pass. Pass `--both-strands` to score each record in
87
+ both orientations:
88
+
89
+ ```bash
90
+ prform predict genomes.fasta --outdir results/ --both-strands
91
+ ```
92
+
93
+ **Both orientations are reported.** The peaks and tracks are duplicated into a
94
+ forward block and a reverse block:
95
+
96
+ ```
97
+ predictions.csv
98
+ accession_id, length,
99
+ fwd_top{k}_pos, fwd_top{k}_prob, fwd_top{k}_type,
100
+ rev_top{k}_pos, rev_top{k}_prob, rev_top{k}_type
101
+
102
+ predictions.h5
103
+ records/ # same fwd_/rev_ prefixes as the CSV
104
+ accession_id, length,
105
+ fwd_top{k}_{pos,prob,type}, rev_top{k}_{pos,prob,type}
106
+ per_nucleotide/
107
+ {accession_id}/
108
+ fwd/ no_PRF_prob, minus1_PRF_prob, plus1_PRF_prob (L,) each
109
+ rev/ no_PRF_prob, minus1_PRF_prob, plus1_PRF_prob (L,) each
110
+ ```
111
+
112
+ The reverse block is mirrored back into **forward-genome coordinates**, so both
113
+ blocks index the same nucleotide: `fwd_top1_pos`, `rev_top1_pos` and index `i`
114
+ of either track all refer to position `i+1` of the sequence as you supplied it.
115
+ Class channels are *not* swapped — a -1 PRF read on the reverse complement
116
+ stays in `minus1_PRF_prob`, because the shift direction is defined relative to
117
+ the strand being read.
118
+
119
+ ## Test the install
120
+
121
+ `prform` ships with small FASTA fixtures (one folder per labeled
122
+ frameshift class) so you can verify the install end-to-end without any
123
+ external data. Copy them into your working directory:
124
+
125
+ ```bash
126
+ prform download-examples ./prform_examples
127
+ ```
128
+
129
+ Layout after download:
130
+
131
+ ```
132
+ prform_examples/
133
+ plus1_PRF/ influenza_pax.fasta
134
+ minus1_PRF/ coronavirus_khosta2.fasta, phage_tyson.fasta
135
+ no_PRF/ monkeypox, ledantevirus, nanovirus, ourmiavirus, orbivirus
136
+ ```
137
+
138
+ Run `predict` on each class:
139
+
140
+ ```bash
141
+ # +1 PRF: peak should land near the labeled site
142
+ prform predict prform_examples/plus1_PRF/influenza_pax.fasta \
143
+ --outdir prform_test/plus1
144
+
145
+ # -1 PRF: peak should land near the labeled site
146
+ prform predict prform_examples/minus1_PRF/coronavirus_khosta2.fasta \
147
+ --outdir prform_test/minus1
148
+
149
+ # no PRF: no frameshift should be called
150
+ prform predict prform_examples/no_PRF/nanovirus.fasta \
151
+ --outdir prform_test/no_prf
152
+
153
+ cat prform_test/plus1/predictions.csv
154
+ cat prform_test/minus1/predictions.csv
155
+ cat prform_test/no_prf/predictions.csv
156
+ ```
157
+
158
+ Expected behavior:
159
+
160
+ | fixture | labeled site | what to look for |
161
+ | --- | --- | --- |
162
+ | `plus1_PRF/influenza_pax.fasta` | +1 PRF @ pos 582 | `top1_pos` within a few nt of 582, `top1_type == 1` |
163
+ | `minus1_PRF/coronavirus_khosta2.fasta` | -1 PRF @ pos 13248 | `top1_pos` near 13248, `top1_type == -1` |
164
+ | `minus1_PRF/phage_tyson.fasta` | -1 PRF @ pos 10279 | `top1_pos` near 10279, `top1_type == -1` |
165
+ | `no_PRF/*.fasta` | none | `top1_type == 0` (no frameshift called) |
166
+
167
+ Each labeled-positive FASTA header carries the labeled site for easy comparison.
168
+
169
+ ## Python API
170
+
171
+ ```python
172
+ from prform import PRFormPredictor
173
+
174
+ predictor = PRFormPredictor() # loads bundled weights
175
+ results = predictor.predict_fasta("input.fasta") # list of dicts
176
+ # or
177
+ results = predictor.predict_sequences([("rec1", "ACGT" * 1000)])
178
+ # score both orientations, keep the better strand per record
179
+ results = predictor.predict_fasta("input.fasta", both_strands=True)
180
+ ```
181
+
182
+ Each result entry has `accession_id`, `length`, `strand`, `cal_probs_3`
183
+ (3, L) and a `top` list of peak summaries (`pos`, `prob`, `type`) — all from
184
+ the winning strand. With `both_strands=True`, `rec["fwd"]` and `rec["rev"]`
185
+ each hold that orientation's own `cal_probs_3` and `top`, both in
186
+ forward-genome coordinates; otherwise they are `None`.
187
+
188
+ ## Hardware
189
+
190
+ CPU works. GPU is faster — `prform` automatically uses CUDA when available.
191
+ For long sequences (>1 Mb total input) a GPU is recommended.
192
+
193
+ ## License
194
+
195
+ MIT
@@ -0,0 +1,14 @@
1
+ """prform — predict programmed ribosomal frameshift (PRF) sites."""
2
+
3
+ from importlib.resources import files
4
+
5
+ from prform.predictor import PRFormPredictor
6
+
7
+
8
+ def examples_dir():
9
+ """Path to the bundled example FASTA fixtures (positive/ and negative/)."""
10
+ return files("prform") / "_data" / "examples"
11
+
12
+
13
+ __version__ = "0.2.0"
14
+ __all__ = ["PRFormPredictor", "examples_dir", "__version__"]
@@ -0,0 +1,6 @@
1
+ """Allow `python -m prform ...` regardless of PATH."""
2
+
3
+ from prform.cli import main
4
+
5
+ if __name__ == "__main__":
6
+ main()
@@ -0,0 +1,75 @@
1
+ """
2
+ PRFormCalibrator — five isotonic regressors mapping raw 3-class softmax to
3
+ calibrated per-nucleotide probabilities, plus a record-level isotonic.
4
+
5
+ Loaded from a dict-of-components joblib file, so the saved data is
6
+ independent of any specific module path.
7
+ """
8
+
9
+ from importlib.resources import files
10
+
11
+ import joblib
12
+ import numpy as np
13
+ from scipy.ndimage import maximum_filter1d
14
+
15
+
16
+ class PRFormCalibrator:
17
+ def __init__(self, iso_class0, iso_class1, iso_class2,
18
+ iso_minus1_flank, iso_plus1_flank, iso_seq, flank_k=3):
19
+ self.iso_class0 = iso_class0
20
+ self.iso_class1 = iso_class1
21
+ self.iso_class2 = iso_class2
22
+ self.iso_minus1_flank = iso_minus1_flank
23
+ self.iso_plus1_flank = iso_plus1_flank
24
+ self.iso_seq = iso_seq
25
+ self.flank_k = int(flank_k)
26
+
27
+ def predict(self, probs_3):
28
+ """
29
+ probs_3 : (3, L) raw softmax — rows = [no_PRF, dir-1, dir+1].
30
+
31
+ Returns
32
+ -------
33
+ dict with
34
+ probs_3 : (3, L) float32, sharp per-class calibrated, renormalized
35
+ probs_minus1_flank : (L,) float32, P(any true -1 site within +/- k)
36
+ probs_plus1_flank : (L,) float32, P(any true +1 site within +/- k)
37
+ sequence : float, P(record contains any PRF site)
38
+ """
39
+ probs_3 = np.asarray(probs_3, dtype=np.float32)
40
+ if probs_3.ndim != 2 or probs_3.shape[0] != 3:
41
+ raise ValueError(f"probs_3 must be (3, L); got {probs_3.shape}")
42
+
43
+ c0 = self.iso_class0.transform(probs_3[0])
44
+ c1 = self.iso_class1.transform(probs_3[1])
45
+ c2 = self.iso_class2.transform(probs_3[2])
46
+ s = c0 + c1 + c2
47
+ s = np.where(s == 0, 1.0, s)
48
+ sharp = np.stack([c0 / s, c1 / s, c2 / s], axis=0).astype(np.float32)
49
+
50
+ win = 2 * self.flank_k + 1
51
+ w1 = maximum_filter1d(probs_3[1], size=win, mode="nearest")
52
+ w2 = maximum_filter1d(probs_3[2], size=win, mode="nearest")
53
+ flank_minus1 = self.iso_minus1_flank.transform(w1).astype(np.float32)
54
+ flank_plus1 = self.iso_plus1_flank.transform(w2).astype(np.float32)
55
+
56
+ prf_max = float((probs_3[1] + probs_3[2]).max())
57
+ seq = float(self.iso_seq.transform([prf_max])[0])
58
+
59
+ return {
60
+ "probs_3": sharp,
61
+ "probs_minus1_flank": flank_minus1,
62
+ "probs_plus1_flank": flank_plus1,
63
+ "sequence": seq,
64
+ }
65
+
66
+ def transform_sequence(self, prf_max):
67
+ return float(self.iso_seq.transform([float(prf_max)])[0])
68
+
69
+
70
+ def load_bundled_calibrator():
71
+ """Load the calibrator shipped inside the package."""
72
+ path = files("prform._data").joinpath("calibrator_components.joblib")
73
+ with path.open("rb") as fh:
74
+ components = joblib.load(fh)
75
+ return PRFormCalibrator(**components)
@@ -0,0 +1,24 @@
1
+ ##gff-version 3
2
+ #!gff-spec-version 1.21
3
+ #!processor NCBI annotwriter
4
+ #!genome-build ASM3125311v1
5
+ #!genome-build-accession NCBI_Assembly:GCA_031253115.1
6
+ ##sequence-region MZ190138.1 1 29220
7
+ ##species https://www.ncbi.nlm.nih.gov/Taxonomy/Browser/wwwtax.cgi?id=2836124
8
+ MZ190138.1 Genbank region 1 29220 . + . ID=MZ190138.1:1..29220;Dbxref=taxon:2836124;collected-by=S. Lenshin%2C A. Romashin;collection-date=2020;country=Russia: Sochi;gbkey=Src;genome=genomic;isolation-source=feces pool;mol_type=genomic RNA;nat-host=Rhinolophus hipposideros (bat);strain=BtCoV/Khosta-2/Rh/Russia/2020
9
+ MZ190138.1 Genbank CDS 208 13248 . + 0 ID=cds-QVN46577.1;Dbxref=NCBI_GP:QVN46577.1;Name=QVN46577.1;exception=ribosomal slippage;gbkey=CDS;product=ORF1ab protein;protein_id=QVN46577.1
10
+ MZ190138.1 Genbank CDS 13248 21386 . + 0 ID=cds-QVN46577.1;Dbxref=NCBI_GP:QVN46577.1;Name=QVN46577.1;exception=ribosomal slippage;gbkey=CDS;product=ORF1ab protein;protein_id=QVN46577.1
11
+ MZ190138.1 Genbank CDS 208 13314 . + 0 ID=cds-QVN46568.1;Dbxref=NCBI_GP:QVN46568.1;Name=QVN46568.1;gbkey=CDS;product=ORF1a protein;protein_id=QVN46568.1
12
+ MZ190138.1 Genbank gene 21393 25154 . + . ID=gene-S;Name=S;gbkey=Gene;gene=S;gene_biotype=protein_coding
13
+ MZ190138.1 Genbank CDS 21393 25154 . + 0 ID=cds-QVN46569.1;Parent=gene-S;Dbxref=NCBI_GP:QVN46569.1;Name=QVN46569.1;gbkey=CDS;gene=S;product=spike glycoprotein;protein_id=QVN46569.1
14
+ MZ190138.1 Genbank CDS 25163 25975 . + 0 ID=cds-QVN46570.1;Dbxref=NCBI_GP:QVN46570.1;Name=QVN46570.1;gbkey=CDS;product=ORF3 protein;protein_id=QVN46570.1
15
+ MZ190138.1 Genbank gene 25999 26229 . + . ID=gene-E;Name=E;gbkey=Gene;gene=E;gene_biotype=protein_coding
16
+ MZ190138.1 Genbank CDS 25999 26229 . + 0 ID=cds-QVN46571.1;Parent=gene-E;Dbxref=NCBI_GP:QVN46571.1;Name=QVN46571.1;gbkey=CDS;gene=E;product=envelope protein;protein_id=QVN46571.1
17
+ MZ190138.1 Genbank gene 26277 26942 . + . ID=gene-M;Name=M;gbkey=Gene;gene=M;gene_biotype=protein_coding
18
+ MZ190138.1 Genbank CDS 26277 26942 . + 0 ID=cds-QVN46572.1;Parent=gene-M;Dbxref=NCBI_GP:QVN46572.1;Name=QVN46572.1;gbkey=CDS;gene=M;product=membrane protein;protein_id=QVN46572.1
19
+ MZ190138.1 Genbank CDS 26953 27141 . + 0 ID=cds-QVN46573.1;Dbxref=NCBI_GP:QVN46573.1;Name=QVN46573.1;gbkey=CDS;product=ORF6 protein;protein_id=QVN46573.1
20
+ MZ190138.1 Genbank CDS 27148 27504 . + 0 ID=cds-QVN46574.1;Dbxref=NCBI_GP:QVN46574.1;Name=QVN46574.1;gbkey=CDS;product=ORF7a protein;protein_id=QVN46574.1
21
+ MZ190138.1 Genbank CDS 27498 27626 . + 0 ID=cds-QVN46575.1;Dbxref=NCBI_GP:QVN46575.1;Name=QVN46575.1;gbkey=CDS;product=ORF7b protein;protein_id=QVN46575.1
22
+ MZ190138.1 Genbank gene 27623 28876 . + . ID=gene-N;Name=N;gbkey=Gene;gene=N;gene_biotype=protein_coding
23
+ MZ190138.1 Genbank CDS 27623 28876 . + 0 ID=cds-QVN46576.1;Parent=gene-N;Dbxref=NCBI_GP:QVN46576.1;Name=QVN46576.1;gbkey=CDS;gene=N;product=nucleocapsid protein;protein_id=QVN46576.1
24
+ ###