immunoHERD 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- immunoherd-0.1.1/LICENSE +21 -0
- immunoherd-0.1.1/MANIFEST.in +4 -0
- immunoherd-0.1.1/PKG-INFO +131 -0
- immunoherd-0.1.1/README.md +103 -0
- immunoherd-0.1.1/pyproject.toml +75 -0
- immunoherd-0.1.1/setup.cfg +4 -0
- immunoherd-0.1.1/src/herd/__init__.py +17 -0
- immunoherd-0.1.1/src/herd/cli.py +124 -0
- immunoherd-0.1.1/src/herd/config.py +56 -0
- immunoherd-0.1.1/src/herd/esm.py +85 -0
- immunoherd-0.1.1/src/herd/fasta.py +51 -0
- immunoherd-0.1.1/src/herd/inference.py +85 -0
- immunoherd-0.1.1/src/herd/models/individual/Bos_taurus.keras +0 -0
- immunoherd-0.1.1/src/herd/models/individual/Canis_sp.keras +0 -0
- immunoherd-0.1.1/src/herd/models/individual/Equus_caballus.keras +0 -0
- immunoherd-0.1.1/src/herd/models/individual/Gallus_gallus.keras +0 -0
- immunoherd-0.1.1/src/herd/models/individual/Homo_sapiens.keras +0 -0
- immunoherd-0.1.1/src/herd/models/individual/Sus_scrofa.keras +0 -0
- immunoherd-0.1.1/src/herd/models/integrated.keras +0 -0
- immunoherd-0.1.1/src/herd/models.py +94 -0
- immunoherd-0.1.1/src/herd/validation.py +78 -0
- immunoherd-0.1.1/src/immunoHERD.egg-info/PKG-INFO +131 -0
- immunoherd-0.1.1/src/immunoHERD.egg-info/SOURCES.txt +26 -0
- immunoherd-0.1.1/src/immunoHERD.egg-info/dependency_links.txt +1 -0
- immunoherd-0.1.1/src/immunoHERD.egg-info/entry_points.txt +2 -0
- immunoherd-0.1.1/src/immunoHERD.egg-info/requires.txt +6 -0
- immunoherd-0.1.1/src/immunoHERD.egg-info/top_level.txt +1 -0
- immunoherd-0.1.1/tests/test_basics.py +56 -0
immunoherd-0.1.1/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Isidro Sobrino
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: immunoHERD
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: HERD — host-aware prediction of protein immunogenicity
|
|
5
|
+
Author: Isidro Sobrino
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Keywords: immunogenicity,antigen,vaccine,protein,ESM-2,protein-language-model,veterinary,bioinformatics,deep-learning
|
|
8
|
+
Classifier: Development Status :: 4 - Beta
|
|
9
|
+
Classifier: Intended Audience :: Science/Research
|
|
10
|
+
Classifier: Operating System :: OS Independent
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
18
|
+
Requires-Python: >=3.11
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: fair-esm
|
|
22
|
+
Requires-Dist: torch
|
|
23
|
+
Requires-Dist: tensorflow>=2.16
|
|
24
|
+
Requires-Dist: keras<4,>=3.13
|
|
25
|
+
Requires-Dist: numpy
|
|
26
|
+
Requires-Dist: pandas
|
|
27
|
+
Dynamic: license-file
|
|
28
|
+
|
|
29
|
+
# HERD
|
|
30
|
+
|
|
31
|
+
**H**ost-aware **E**stimation of antigen immunogenicity and **R**anking through
|
|
32
|
+
**D**eep-learning.
|
|
33
|
+
|
|
34
|
+
HERD predicts the immunogenicity of proteins taking the host species into
|
|
35
|
+
account. This package is the reusable core shared by the different surfaces of
|
|
36
|
+
the project (web Space, Colab and command line).
|
|
37
|
+
|
|
38
|
+
## Pipeline
|
|
39
|
+
|
|
40
|
+
1. Sequence cleaning: any character outside `ACDEFGHIKLMNPQRSTVWY` becomes `X`,
|
|
41
|
+
and the sequence is truncated to 1022 residues.
|
|
42
|
+
2. Embedding with ESM-2 (`esm2_t33_650M_UR50D`, layer 33) via `fair-esm`,
|
|
43
|
+
mean-pooled over the real residues: a 1280-dimensional vector.
|
|
44
|
+
3. A host-aware head on top of that embedding. Normalization lives inside the
|
|
45
|
+
`.keras` file itself, so the raw embedding is fed in unscaled.
|
|
46
|
+
4. Decision threshold: 0.5.
|
|
47
|
+
|
|
48
|
+
## Installation
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
pip install immunoHERD
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
The package is installed as `immunoHERD` but imported as `herd`, and the
|
|
55
|
+
console command is `herd`.
|
|
56
|
+
|
|
57
|
+
The model weights ship inside the package. The ESM-2 weights, by contrast, are
|
|
58
|
+
downloaded automatically the first time an embedding is computed.
|
|
59
|
+
|
|
60
|
+
## Usage
|
|
61
|
+
|
|
62
|
+
### From Python
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
from herd import predict_fasta
|
|
66
|
+
|
|
67
|
+
df = predict_fasta("proteins.fasta", host="Bos_taurus", model="integrated")
|
|
68
|
+
print(df)
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
`predict` (a list of `(id, sequence)` tuples) and `predict_text` (pasted FASTA
|
|
72
|
+
text, or a bare sequence) are also available. All three return a `DataFrame`
|
|
73
|
+
sorted by descending score with the columns `id`, `length`, `host`, `model`,
|
|
74
|
+
`score`, `prediction` and `warnings`. Sequences that fail validation are
|
|
75
|
+
returned at the end, with `score` set to `NaN`.
|
|
76
|
+
|
|
77
|
+
The `warnings` column also records when a sequence was modified: non-canonical
|
|
78
|
+
residues replaced by `X`, or truncation to 1022 residues.
|
|
79
|
+
|
|
80
|
+
### From the command line
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
herd predict proteins.fasta --host Bos_taurus --out results.csv
|
|
84
|
+
herd predict proteins.fasta --host Canis_sp --model individual --out dog.csv
|
|
85
|
+
herd hosts
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
`--out` is required, so a long run cannot be lost by forgetting to save it.
|
|
89
|
+
Other options: `--model`, `--threshold`, `--batch-seqs`, `--device`.
|
|
90
|
+
Warnings go to standard error, so the CSV stays clean when redirected.
|
|
91
|
+
|
|
92
|
+
## Models and hosts
|
|
93
|
+
|
|
94
|
+
Two models, selected with the `model` argument:
|
|
95
|
+
|
|
96
|
+
- `"integrated"` — a pan-species model with 15 heads, one per host.
|
|
97
|
+
This is the recommended default.
|
|
98
|
+
- `"individual"` — one model trained per species. Only available for the six
|
|
99
|
+
core hosts: `Homo_sapiens`, `Bos_taurus`, `Sus_scrofa`, `Gallus_gallus`,
|
|
100
|
+
`Canis_sp` and `Equus_caballus`.
|
|
101
|
+
|
|
102
|
+
The integrated model also covers nine auxiliary hosts: `Camelidae`,
|
|
103
|
+
`Capra_hircus`, `Cavia_porcellus`, `Macaca_sp_`, `Mus_musculus`,
|
|
104
|
+
`Non_human_primate`, `Oryctolagus_cuniculus`, `Ovis_aries` and `Rattus_sp_`.
|
|
105
|
+
**These nine heads are not validated to the same level as the six core hosts,
|
|
106
|
+
and their scores should be interpreted with more caution.**
|
|
107
|
+
|
|
108
|
+
Host names accept some common aliases (`Canis_familiaris`, `Homo`, `Rattus`…)
|
|
109
|
+
and a loose genus match.
|
|
110
|
+
|
|
111
|
+
## Where the models are looked up
|
|
112
|
+
|
|
113
|
+
By default, in the `models/` directory that ships inside the installed package.
|
|
114
|
+
A different location can be given through the `HERD_MODELS_DIR` environment
|
|
115
|
+
variable, which takes precedence:
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
export HERD_MODELS_DIR=/path/to/my/models
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
The expected layout is `<HERD_MODELS_DIR>/integrated.keras` and
|
|
122
|
+
`<HERD_MODELS_DIR>/individual/<host>.keras`.
|
|
123
|
+
|
|
124
|
+
## Performance notes
|
|
125
|
+
|
|
126
|
+
Sequences are grouped into batches **sorted by length**, which removes almost
|
|
127
|
+
all padding: on a real proteome (4,570 sequences) this halves the GPU work.
|
|
128
|
+
Results are unaffected, since each mean is computed only over the real residues
|
|
129
|
+
of its own sequence.
|
|
130
|
+
|
|
131
|
+
A GPU is not required, but embedding on CPU is considerably slower.
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
# HERD
|
|
2
|
+
|
|
3
|
+
**H**ost-aware **E**stimation of antigen immunogenicity and **R**anking through
|
|
4
|
+
**D**eep-learning.
|
|
5
|
+
|
|
6
|
+
HERD predicts the immunogenicity of proteins taking the host species into
|
|
7
|
+
account. This package is the reusable core shared by the different surfaces of
|
|
8
|
+
the project (web Space, Colab and command line).
|
|
9
|
+
|
|
10
|
+
## Pipeline
|
|
11
|
+
|
|
12
|
+
1. Sequence cleaning: any character outside `ACDEFGHIKLMNPQRSTVWY` becomes `X`,
|
|
13
|
+
and the sequence is truncated to 1022 residues.
|
|
14
|
+
2. Embedding with ESM-2 (`esm2_t33_650M_UR50D`, layer 33) via `fair-esm`,
|
|
15
|
+
mean-pooled over the real residues: a 1280-dimensional vector.
|
|
16
|
+
3. A host-aware head on top of that embedding. Normalization lives inside the
|
|
17
|
+
`.keras` file itself, so the raw embedding is fed in unscaled.
|
|
18
|
+
4. Decision threshold: 0.5.
|
|
19
|
+
|
|
20
|
+
## Installation
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
pip install immunoHERD
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
The package is installed as `immunoHERD` but imported as `herd`, and the
|
|
27
|
+
console command is `herd`.
|
|
28
|
+
|
|
29
|
+
The model weights ship inside the package. The ESM-2 weights, by contrast, are
|
|
30
|
+
downloaded automatically the first time an embedding is computed.
|
|
31
|
+
|
|
32
|
+
## Usage
|
|
33
|
+
|
|
34
|
+
### From Python
|
|
35
|
+
|
|
36
|
+
```python
|
|
37
|
+
from herd import predict_fasta
|
|
38
|
+
|
|
39
|
+
df = predict_fasta("proteins.fasta", host="Bos_taurus", model="integrated")
|
|
40
|
+
print(df)
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
`predict` (a list of `(id, sequence)` tuples) and `predict_text` (pasted FASTA
|
|
44
|
+
text, or a bare sequence) are also available. All three return a `DataFrame`
|
|
45
|
+
sorted by descending score with the columns `id`, `length`, `host`, `model`,
|
|
46
|
+
`score`, `prediction` and `warnings`. Sequences that fail validation are
|
|
47
|
+
returned at the end, with `score` set to `NaN`.
|
|
48
|
+
|
|
49
|
+
The `warnings` column also records when a sequence was modified: non-canonical
|
|
50
|
+
residues replaced by `X`, or truncation to 1022 residues.
|
|
51
|
+
|
|
52
|
+
### From the command line
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
herd predict proteins.fasta --host Bos_taurus --out results.csv
|
|
56
|
+
herd predict proteins.fasta --host Canis_sp --model individual --out dog.csv
|
|
57
|
+
herd hosts
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
`--out` is required, so a long run cannot be lost by forgetting to save it.
|
|
61
|
+
Other options: `--model`, `--threshold`, `--batch-seqs`, `--device`.
|
|
62
|
+
Warnings go to standard error, so the CSV stays clean when redirected.
|
|
63
|
+
|
|
64
|
+
## Models and hosts
|
|
65
|
+
|
|
66
|
+
Two models, selected with the `model` argument:
|
|
67
|
+
|
|
68
|
+
- `"integrated"` — a pan-species model with 15 heads, one per host.
|
|
69
|
+
This is the recommended default.
|
|
70
|
+
- `"individual"` — one model trained per species. Only available for the six
|
|
71
|
+
core hosts: `Homo_sapiens`, `Bos_taurus`, `Sus_scrofa`, `Gallus_gallus`,
|
|
72
|
+
`Canis_sp` and `Equus_caballus`.
|
|
73
|
+
|
|
74
|
+
The integrated model also covers nine auxiliary hosts: `Camelidae`,
|
|
75
|
+
`Capra_hircus`, `Cavia_porcellus`, `Macaca_sp_`, `Mus_musculus`,
|
|
76
|
+
`Non_human_primate`, `Oryctolagus_cuniculus`, `Ovis_aries` and `Rattus_sp_`.
|
|
77
|
+
**These nine heads are not validated to the same level as the six core hosts,
|
|
78
|
+
and their scores should be interpreted with more caution.**
|
|
79
|
+
|
|
80
|
+
Host names accept some common aliases (`Canis_familiaris`, `Homo`, `Rattus`…)
|
|
81
|
+
and a loose genus match.
|
|
82
|
+
|
|
83
|
+
## Where the models are looked up
|
|
84
|
+
|
|
85
|
+
By default, in the `models/` directory that ships inside the installed package.
|
|
86
|
+
A different location can be given through the `HERD_MODELS_DIR` environment
|
|
87
|
+
variable, which takes precedence:
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
export HERD_MODELS_DIR=/path/to/my/models
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
The expected layout is `<HERD_MODELS_DIR>/integrated.keras` and
|
|
94
|
+
`<HERD_MODELS_DIR>/individual/<host>.keras`.
|
|
95
|
+
|
|
96
|
+
## Performance notes
|
|
97
|
+
|
|
98
|
+
Sequences are grouped into batches **sorted by length**, which removes almost
|
|
99
|
+
all padding: on a real proteome (4,570 sequences) this halves the GPU work.
|
|
100
|
+
Results are unaffected, since each mean is computed only over the real residues
|
|
101
|
+
of its own sequence.
|
|
102
|
+
|
|
103
|
+
A GPU is not required, but embedding on CPU is considerably slower.
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
# Distribution name (pip install immunoHERD). The IMPORT name and the
|
|
7
|
+
# console command are both `herd`.
|
|
8
|
+
name = "immunoHERD"
|
|
9
|
+
# Única fuente de la versión: herd/__init__.py (__version__).
|
|
10
|
+
dynamic = ["version"]
|
|
11
|
+
description = "HERD — host-aware prediction of protein immunogenicity"
|
|
12
|
+
readme = "README.md"
|
|
13
|
+
# >=3.11 porque los modelos necesitan keras>=3.13 (ver dependencies),
|
|
14
|
+
# y Keras 3.13 dejó de publicar para Python 3.10 y anteriores.
|
|
15
|
+
requires-python = ">=3.11"
|
|
16
|
+
authors = [{ name = "Isidro Sobrino" }]
|
|
17
|
+
license = "MIT"
|
|
18
|
+
license-files = ["LICENSE"]
|
|
19
|
+
keywords = [
|
|
20
|
+
"immunogenicity", "antigen", "vaccine", "protein", "ESM-2",
|
|
21
|
+
"protein-language-model", "veterinary", "bioinformatics", "deep-learning",
|
|
22
|
+
]
|
|
23
|
+
classifiers = [
|
|
24
|
+
"Development Status :: 4 - Beta",
|
|
25
|
+
"Intended Audience :: Science/Research",
|
|
26
|
+
"Operating System :: OS Independent",
|
|
27
|
+
"Programming Language :: Python :: 3",
|
|
28
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
29
|
+
"Programming Language :: Python :: 3.11",
|
|
30
|
+
"Programming Language :: Python :: 3.12",
|
|
31
|
+
"Programming Language :: Python :: 3.13",
|
|
32
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
33
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
# HERD needs torch (for ESM-2) and TensorFlow (for the .keras heads).
|
|
37
|
+
#
|
|
38
|
+
# keras>=3.13 is NOT optional: the bundled .keras files were written by
|
|
39
|
+
# Keras 3.13.2 and their layer configs carry `quantization_config`, a field
|
|
40
|
+
# Keras 3.12 does not know. TensorFlow only requires keras>=3.12, so without
|
|
41
|
+
# this floor pip can resolve a Keras that installs fine but fails at load
|
|
42
|
+
# time with "Unrecognized keyword arguments passed to Dense".
|
|
43
|
+
dependencies = [
|
|
44
|
+
"fair-esm",
|
|
45
|
+
"torch",
|
|
46
|
+
# First TensorFlow release that requires Keras 3 (2.15 pins keras<2.16,
|
|
47
|
+
# i.e. Keras 2, which cannot load these models at all).
|
|
48
|
+
"tensorflow>=2.16",
|
|
49
|
+
# Floor: the bundled .keras files were written by Keras 3.13.2 and their
|
|
50
|
+
# layer configs carry `quantization_config`, unknown to Keras 3.12.
|
|
51
|
+
# Ceiling: the saved-model format has already broken twice across Keras
|
|
52
|
+
# minor versions, so a major bump is not assumed to be safe.
|
|
53
|
+
"keras>=3.13,<4",
|
|
54
|
+
"numpy",
|
|
55
|
+
"pandas",
|
|
56
|
+
]
|
|
57
|
+
|
|
58
|
+
# Console command: `herd predict ...` / `herd hosts`
|
|
59
|
+
[project.scripts]
|
|
60
|
+
herd = "herd.cli:main"
|
|
61
|
+
|
|
62
|
+
[tool.setuptools.dynamic]
|
|
63
|
+
version = { attr = "herd.__version__" }
|
|
64
|
+
|
|
65
|
+
[tool.setuptools]
|
|
66
|
+
# src-layout: el paquete no es importable desde la raíz del repositorio,
|
|
67
|
+
# así que cualquier prueba usa por fuerza la versión instalada.
|
|
68
|
+
package-dir = { "" = "src" }
|
|
69
|
+
packages = ["herd"]
|
|
70
|
+
include-package-data = true
|
|
71
|
+
|
|
72
|
+
# The .keras weights are data, not code: without this declaration the wheel
|
|
73
|
+
# is built without them and nothing fails until prediction time.
|
|
74
|
+
[tool.setuptools.package-data]
|
|
75
|
+
herd = ["models/*.keras", "models/individual/*.keras"]
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""
|
|
2
|
+
HERD — Host-aware Estimation of antigen immunogenicity and Ranking
|
|
3
|
+
through Deep-learning. Núcleo reutilizable (Space / Colab / CLI).
|
|
4
|
+
|
|
5
|
+
from herd import predict_fasta
|
|
6
|
+
df = predict_fasta("proteinas.fasta", host="Bos_taurus", model="integrated")
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from .inference import predict, predict_fasta, predict_text
|
|
10
|
+
from .config import SPECIES_ORDER, CORE_HOSTS, THRESHOLD
|
|
11
|
+
|
|
12
|
+
__all__ = [
|
|
13
|
+
"predict", "predict_fasta", "predict_text",
|
|
14
|
+
"SPECIES_ORDER", "CORE_HOSTS", "THRESHOLD",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
__version__ = "0.1.1"
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""
|
|
2
|
+
herd/cli.py — Interfaz de línea de comandos.
|
|
3
|
+
|
|
4
|
+
herd predict proteins.fasta --host Bos_taurus --out results.csv
|
|
5
|
+
herd hosts
|
|
6
|
+
|
|
7
|
+
No contiene ciencia: traduce argumentos a llamadas de `herd.inference`.
|
|
8
|
+
Los textos visibles para el usuario van en inglés, como el resto del proyecto.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import argparse
|
|
12
|
+
import sys
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from .config import SPECIES_ORDER, CORE_HOSTS, THRESHOLD, LABEL_POS
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
# --- herd hosts --------------------------------------------------------------
|
|
19
|
+
def _cmd_hosts(args) -> int:
|
|
20
|
+
core = set(CORE_HOSTS)
|
|
21
|
+
print("Hosts supported by the integrated model (15 heads):\n")
|
|
22
|
+
for sp in SPECIES_ORDER:
|
|
23
|
+
print(f" {'core ' if sp in core else 'auxiliary'} {sp}")
|
|
24
|
+
print(
|
|
25
|
+
"\ncore also has its own individual model (--model individual)."
|
|
26
|
+
"\nauxiliary exists only as a head of the integrated model. These nine"
|
|
27
|
+
"\n heads are not validated to the same level as the core"
|
|
28
|
+
"\n hosts: interpret their scores with caution."
|
|
29
|
+
)
|
|
30
|
+
return 0
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
# --- herd predict ------------------------------------------------------------
|
|
34
|
+
def _cmd_predict(args) -> int:
|
|
35
|
+
fasta = Path(args.fasta)
|
|
36
|
+
if not fasta.is_file():
|
|
37
|
+
print(f"herd: no such file '{fasta}'", file=sys.stderr)
|
|
38
|
+
return 2
|
|
39
|
+
|
|
40
|
+
out = Path(args.out)
|
|
41
|
+
if out.parent and not out.parent.exists():
|
|
42
|
+
print(f"herd: output directory '{out.parent}' does not exist", file=sys.stderr)
|
|
43
|
+
return 2
|
|
44
|
+
|
|
45
|
+
# Importación diferida: arrastra torch y TensorFlow, que tardan en cargar.
|
|
46
|
+
from .models import resolve_host
|
|
47
|
+
from .inference import predict_fasta
|
|
48
|
+
|
|
49
|
+
sp = resolve_host(args.host)
|
|
50
|
+
if sp is None:
|
|
51
|
+
print(f"herd: unknown host '{args.host}'. "
|
|
52
|
+
f"Run 'herd hosts' to see the list.", file=sys.stderr)
|
|
53
|
+
return 2
|
|
54
|
+
if args.model == "individual" and sp not in CORE_HOSTS:
|
|
55
|
+
print(f"herd: no individual model for '{sp}'. "
|
|
56
|
+
f"Only available for: {', '.join(CORE_HOSTS)}.", file=sys.stderr)
|
|
57
|
+
return 2
|
|
58
|
+
if sp != args.host:
|
|
59
|
+
print(f"herd: '{args.host}' interpreted as '{sp}'.", file=sys.stderr)
|
|
60
|
+
if sp not in CORE_HOSTS:
|
|
61
|
+
print(f"herd: warning — '{sp}' is an auxiliary head, not validated to "
|
|
62
|
+
f"the same level as the core hosts.", file=sys.stderr)
|
|
63
|
+
|
|
64
|
+
df = predict_fasta(
|
|
65
|
+
str(fasta),
|
|
66
|
+
host=sp,
|
|
67
|
+
model=args.model,
|
|
68
|
+
threshold=args.threshold,
|
|
69
|
+
device=args.device,
|
|
70
|
+
batch_seqs=args.batch_seqs,
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
df.to_csv(out, index=False)
|
|
74
|
+
|
|
75
|
+
total = len(df)
|
|
76
|
+
errors = int((df["prediction"] == "ERROR").sum()) if total else 0
|
|
77
|
+
positives = int((df["prediction"] == LABEL_POS).sum()) if total else 0
|
|
78
|
+
print(f"herd: {total} sequences | {positives} above threshold "
|
|
79
|
+
f"{args.threshold} | {errors} with errors", file=sys.stderr)
|
|
80
|
+
print(f"herd: results written to {out}", file=sys.stderr)
|
|
81
|
+
return 0
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
# --- Punto de entrada --------------------------------------------------------
|
|
85
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
86
|
+
from . import __version__
|
|
87
|
+
|
|
88
|
+
p = argparse.ArgumentParser(
|
|
89
|
+
prog="herd",
|
|
90
|
+
description="HERD — host-aware prediction of protein immunogenicity.",
|
|
91
|
+
)
|
|
92
|
+
p.add_argument("--version", action="version", version=f"herd {__version__}")
|
|
93
|
+
sub = p.add_subparsers(dest="command", required=True, metavar="COMMAND")
|
|
94
|
+
|
|
95
|
+
pr = sub.add_parser("predict", help="Predict from a FASTA file.")
|
|
96
|
+
pr.add_argument("fasta", help="Input FASTA file.")
|
|
97
|
+
pr.add_argument("--host", required=True,
|
|
98
|
+
help="Target host (see 'herd hosts').")
|
|
99
|
+
pr.add_argument("--out", required=True,
|
|
100
|
+
help="Output CSV file (required).")
|
|
101
|
+
pr.add_argument("--model", choices=("integrated", "individual"),
|
|
102
|
+
default="integrated",
|
|
103
|
+
help="Model to use (default: integrated).")
|
|
104
|
+
pr.add_argument("--threshold", type=float, default=THRESHOLD,
|
|
105
|
+
help=f"Decision threshold (default: {THRESHOLD}).")
|
|
106
|
+
pr.add_argument("--batch-seqs", type=int, default=8, dest="batch_seqs",
|
|
107
|
+
help="Sequences per batch when embedding (default: 8).")
|
|
108
|
+
pr.add_argument("--device", default=None,
|
|
109
|
+
help="Torch device: cpu or cuda (default: automatic).")
|
|
110
|
+
pr.set_defaults(func=_cmd_predict)
|
|
111
|
+
|
|
112
|
+
ho = sub.add_parser("hosts", help="List the available hosts.")
|
|
113
|
+
ho.set_defaults(func=_cmd_hosts)
|
|
114
|
+
|
|
115
|
+
return p
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def main(argv=None) -> int:
|
|
119
|
+
args = build_parser().parse_args(argv)
|
|
120
|
+
return args.func(args)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
if __name__ == "__main__":
|
|
124
|
+
sys.exit(main())
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""
|
|
2
|
+
herd/config.py — Constantes inmutables de HERD (sin lógica).
|
|
3
|
+
Verificado desde el código original. Si es un número o nombre fijo, va aquí.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import os
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
# --- Pipeline ESM-2 (idéntico en generación e inferencia) -------------------
|
|
10
|
+
ESM_MODEL = "esm2_t33_650M_UR50D" # fair-esm, NO HuggingFace
|
|
11
|
+
REPR_LAYER = 33
|
|
12
|
+
EMBED_DIM = 1280
|
|
13
|
+
SEQ_LENGTH = 1022
|
|
14
|
+
VALID_AA = "ACDEFGHIKLMNPQRSTVWY"
|
|
15
|
+
UNKNOWN_AA = "X"
|
|
16
|
+
THRESHOLD = 0.5
|
|
17
|
+
|
|
18
|
+
# --- Modelo integrado: orden de las 15 cabezas (config_modelo_final.json) ----
|
|
19
|
+
SPECIES_ORDER = [
|
|
20
|
+
"Bos_taurus", "Gallus_gallus", "Homo_sapiens", "Equus_caballus",
|
|
21
|
+
"Sus_scrofa", "Canis_sp", "Camelidae", "Capra_hircus", "Cavia_porcellus",
|
|
22
|
+
"Macaca_sp_", "Mus_musculus", "Non_human_primate", "Oryctolagus_cuniculus",
|
|
23
|
+
"Ovis_aries", "Rattus_sp_",
|
|
24
|
+
]
|
|
25
|
+
SP2ID = {sp: i for i, sp in enumerate(SPECIES_ORDER)}
|
|
26
|
+
|
|
27
|
+
# Hosts con modelo INDIVIDUAL propio (solo los 6 core)
|
|
28
|
+
CORE_HOSTS = [
|
|
29
|
+
"Homo_sapiens", "Bos_taurus", "Sus_scrofa",
|
|
30
|
+
"Gallus_gallus", "Canis_sp", "Equus_caballus",
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
# Alias de nombres de host (de app_integrado_porhost.ipynb)
|
|
34
|
+
HOST_ALIASES = {
|
|
35
|
+
"Canis": "Canis_sp", "Canis_familiaris": "Canis_sp",
|
|
36
|
+
"Canis_lupus_familiaris": "Canis_sp", "Cavia": "Cavia_porcellus",
|
|
37
|
+
"Rattus": "Rattus_sp_", "Macaca": "Macaca_sp_", "Homo": "Homo_sapiens",
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
# --- Localización de los pesos (layout limpio) ------------------------------
|
|
41
|
+
# Estructura esperada:
|
|
42
|
+
# <MODELS_ROOT>/individual/<host>.keras (host = clave de especie)
|
|
43
|
+
# <MODELS_ROOT>/integrated.keras
|
|
44
|
+
# Resolución en dos escalones:
|
|
45
|
+
# 1) HERD_MODELS_DIR si está definida (Space, pruebas, modelos alternativos).
|
|
46
|
+
# 2) la carpeta empaquetada junto a este módulo (paquete instalado).
|
|
47
|
+
# Se resuelve desde __file__, NO desde el directorio de trabajo: tras un
|
|
48
|
+
# `pip install` el cwd del usuario no tiene ninguna carpeta 'models'.
|
|
49
|
+
_ENV_MODELS = os.environ.get("HERD_MODELS_DIR")
|
|
50
|
+
MODELS_ROOT = Path(_ENV_MODELS) if _ENV_MODELS else Path(__file__).resolve().parent / "models"
|
|
51
|
+
INDIVIDUAL_DIR = "individual"
|
|
52
|
+
INTEGRATED_FILE = "integrated.keras"
|
|
53
|
+
|
|
54
|
+
# Etiquetas de salida
|
|
55
|
+
LABEL_POS = "probable immunogen"
|
|
56
|
+
LABEL_NEG = "probable non immunogen"
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""
|
|
2
|
+
herd/esm.py — ESM-2: carga y embedding (parte de GPU).
|
|
3
|
+
|
|
4
|
+
Reglas heredadas del código original que NO se tocan:
|
|
5
|
+
- el modelo se carga UNA sola vez (singleton perezoso),
|
|
6
|
+
- .eval() + torch.no_grad(),
|
|
7
|
+
- se devuelve numpy (CPU), nunca tensores CUDA (crítico en ZeroGPU),
|
|
8
|
+
- la receta de pooling `reps[k, 1:tr+1].mean(0)` es intocable.
|
|
9
|
+
|
|
10
|
+
Recibe secuencias YA LIMPIAS (ver validation.py).
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from typing import List, Optional
|
|
14
|
+
|
|
15
|
+
import numpy as np
|
|
16
|
+
import torch
|
|
17
|
+
from esm import pretrained
|
|
18
|
+
|
|
19
|
+
from .config import ESM_MODEL, REPR_LAYER, SEQ_LENGTH, EMBED_DIM
|
|
20
|
+
|
|
21
|
+
_esm = None # caché a nivel de módulo: (model, alphabet, batch_converter, device)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def get_esm(device: Optional[str] = None):
|
|
25
|
+
"""Carga ESM-2 la primera vez y lo reutiliza en llamadas posteriores."""
|
|
26
|
+
global _esm
|
|
27
|
+
if _esm is None:
|
|
28
|
+
model, alphabet = pretrained.load_model_and_alphabet(ESM_MODEL)
|
|
29
|
+
model.eval()
|
|
30
|
+
device = device or ("cuda" if torch.cuda.is_available() else "cpu")
|
|
31
|
+
model = model.to(device)
|
|
32
|
+
batch_converter = alphabet.get_batch_converter(SEQ_LENGTH)
|
|
33
|
+
_esm = (model, alphabet, batch_converter, device)
|
|
34
|
+
return _esm
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@torch.no_grad()
|
|
38
|
+
def embed_sequences(
|
|
39
|
+
seqs: List[str],
|
|
40
|
+
ids: Optional[List[str]] = None,
|
|
41
|
+
device: Optional[str] = None,
|
|
42
|
+
batch_seqs: int = 8,
|
|
43
|
+
) -> np.ndarray:
|
|
44
|
+
"""
|
|
45
|
+
Secuencias limpias -> matriz (N, 1280) float32, en el orden de entrada.
|
|
46
|
+
|
|
47
|
+
`batch_seqs` es el nº máximo de secuencias por lote (como en las apps).
|
|
48
|
+
|
|
49
|
+
Las secuencias se agrupan ORDENADAS POR LONGITUD y el resultado se
|
|
50
|
+
devuelve al orden original. Cada lote se rellena hasta su secuencia más
|
|
51
|
+
larga, así que agrupar longitudes parecidas elimina casi todo el relleno:
|
|
52
|
+
sobre un proteoma real (4.570 secuencias, media 386 residuos, con un 5%
|
|
53
|
+
en el tope de 1022) esto reduce a la mitad el cómputo de GPU.
|
|
54
|
+
|
|
55
|
+
NO cambia el resultado numérico: la media de cada secuencia se hace solo
|
|
56
|
+
sobre sus residuos reales, y el relleno nunca entra en ella. Tampoco sube
|
|
57
|
+
el pico de memoria: el lote más caro sigue siendo `batch_seqs` secuencias
|
|
58
|
+
de longitud máxima, igual que antes.
|
|
59
|
+
"""
|
|
60
|
+
model, alphabet, batch_converter, dev = get_esm(device)
|
|
61
|
+
n = len(seqs)
|
|
62
|
+
X = np.zeros((n, EMBED_DIM), dtype="float32")
|
|
63
|
+
if n == 0:
|
|
64
|
+
return X
|
|
65
|
+
|
|
66
|
+
if ids is None:
|
|
67
|
+
ids = [str(i) for i in range(n)]
|
|
68
|
+
|
|
69
|
+
# Índices ordenados de más larga a más corta. Descendente a propósito:
|
|
70
|
+
# si algo va a quedarse sin memoria, falla en el primer lote y no a mitad.
|
|
71
|
+
orden = sorted(range(n), key=lambda i: len(seqs[i]), reverse=True)
|
|
72
|
+
|
|
73
|
+
for st in range(0, n, batch_seqs):
|
|
74
|
+
idx = orden[st:st + batch_seqs]
|
|
75
|
+
chunk = [seqs[i] for i in idx]
|
|
76
|
+
labels = [ids[i] for i in idx]
|
|
77
|
+
_, _, toks = batch_converter(list(zip(labels, chunk)))
|
|
78
|
+
toks = toks.to(dev)
|
|
79
|
+
reps = model(toks, repr_layers=[REPR_LAYER])["representations"][REPR_LAYER].to("cpu")
|
|
80
|
+
for k, (i, s) in enumerate(zip(idx, chunk)):
|
|
81
|
+
tr = min(SEQ_LENGTH, len(s))
|
|
82
|
+
# ---- receta EXACTA: descarta BOS, excluye EOS y padding ----
|
|
83
|
+
# Se escribe en la posición ORIGINAL i, no en la del lote.
|
|
84
|
+
X[i] = reps[k, 1:tr + 1].mean(0).numpy().astype("float32")
|
|
85
|
+
return X
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""
|
|
2
|
+
herd/fasta.py — Lectura de secuencias (CPU).
|
|
3
|
+
parse_fasta es la función original de las apps; parse_fasta_text es para la web.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from typing import List, Tuple
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def parse_fasta(path: str) -> Tuple[List[str], List[str]]:
|
|
10
|
+
"""Lee un fichero FASTA -> (ids, seqs). Copiado de las apps."""
|
|
11
|
+
ids, seqs, name, chunks = [], [], None, []
|
|
12
|
+
with open(path) as f:
|
|
13
|
+
for line in f:
|
|
14
|
+
line = line.rstrip()
|
|
15
|
+
if line.startswith(">"):
|
|
16
|
+
if name is not None:
|
|
17
|
+
ids.append(name)
|
|
18
|
+
seqs.append("".join(chunks))
|
|
19
|
+
name = line[1:].split()[0]
|
|
20
|
+
chunks = []
|
|
21
|
+
else:
|
|
22
|
+
chunks.append(line.strip())
|
|
23
|
+
if name is not None:
|
|
24
|
+
ids.append(name)
|
|
25
|
+
seqs.append("".join(chunks))
|
|
26
|
+
return ids, seqs
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def parse_fasta_text(text: str) -> Tuple[List[str], List[str]]:
|
|
30
|
+
"""Igual pero desde un string. Sin '>' -> una sola secuencia 'sequence_1'."""
|
|
31
|
+
text = text.strip()
|
|
32
|
+
if not text:
|
|
33
|
+
return [], []
|
|
34
|
+
if not text.lstrip().startswith(">"):
|
|
35
|
+
return ["sequence_1"], ["".join(text.split())]
|
|
36
|
+
|
|
37
|
+
ids, seqs, name, chunks = [], [], None, []
|
|
38
|
+
for line in text.splitlines():
|
|
39
|
+
line = line.rstrip()
|
|
40
|
+
if line.startswith(">"):
|
|
41
|
+
if name is not None:
|
|
42
|
+
ids.append(name)
|
|
43
|
+
seqs.append("".join(chunks))
|
|
44
|
+
name = line[1:].split()[0]
|
|
45
|
+
chunks = []
|
|
46
|
+
else:
|
|
47
|
+
chunks.append(line.strip())
|
|
48
|
+
if name is not None:
|
|
49
|
+
ids.append(name)
|
|
50
|
+
seqs.append("".join(chunks))
|
|
51
|
+
return ids, seqs
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""
|
|
2
|
+
herd/inference.py — Orquestación: secuencias -> embedding -> modelo -> scores.
|
|
3
|
+
|
|
4
|
+
Es el "pegamento" que une CPU (validación) y GPU (embedding) y produce la tabla
|
|
5
|
+
de resultados. No contiene ciencia nueva: solo encadena los módulos.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from typing import List, Optional, Tuple
|
|
9
|
+
|
|
10
|
+
import numpy as np
|
|
11
|
+
import pandas as pd
|
|
12
|
+
|
|
13
|
+
from . import esm, models
|
|
14
|
+
from .config import THRESHOLD, LABEL_POS, LABEL_NEG, SP2ID, CORE_HOSTS
|
|
15
|
+
from .validation import validate_records
|
|
16
|
+
from .fasta import parse_fasta, parse_fasta_text
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def predict(
|
|
20
|
+
records: List[Tuple[str, str]],
|
|
21
|
+
host: str,
|
|
22
|
+
model: str = "integrated",
|
|
23
|
+
threshold: float = THRESHOLD,
|
|
24
|
+
device: Optional[str] = None,
|
|
25
|
+
batch_seqs: int = 8,
|
|
26
|
+
) -> pd.DataFrame:
|
|
27
|
+
"""
|
|
28
|
+
records : lista de (id, seq).
|
|
29
|
+
host : especie objetivo (clave o alias).
|
|
30
|
+
model : 'integrated' (recomendado, 15 cabezas) o 'individual' (6 core).
|
|
31
|
+
|
|
32
|
+
Devuelve un DataFrame ordenado por score descendente con columnas:
|
|
33
|
+
id, length, host, model, score, prediction, warnings.
|
|
34
|
+
Las secuencias inválidas se devuelven al final con score NaN.
|
|
35
|
+
"""
|
|
36
|
+
if model not in ("integrated", "individual"):
|
|
37
|
+
raise ValueError("model must be 'integrated' or 'individual'.")
|
|
38
|
+
if model == "individual" and host not in CORE_HOSTS:
|
|
39
|
+
raise ValueError(f"'individual' is only available for the 6 core hosts: {CORE_HOSTS}")
|
|
40
|
+
|
|
41
|
+
valid, invalid = validate_records(records)
|
|
42
|
+
|
|
43
|
+
rows = []
|
|
44
|
+
if valid:
|
|
45
|
+
seqs = [r.seq for r in valid]
|
|
46
|
+
ids = [r.id for r in valid]
|
|
47
|
+
X = esm.embed_sequences(seqs, ids=ids, device=device, batch_seqs=batch_seqs)
|
|
48
|
+
if model == "integrated":
|
|
49
|
+
scores = models.score_integrated(X, host)
|
|
50
|
+
else:
|
|
51
|
+
scores = models.score_individual(X, host)
|
|
52
|
+
for r, sc in zip(valid, scores):
|
|
53
|
+
sc = float(sc)
|
|
54
|
+
rows.append({
|
|
55
|
+
"id": r.id,
|
|
56
|
+
"length": r.orig_length,
|
|
57
|
+
"host": host,
|
|
58
|
+
"model": model,
|
|
59
|
+
"score": round(sc, 4),
|
|
60
|
+
"prediction": LABEL_POS if sc >= threshold else LABEL_NEG,
|
|
61
|
+
"warnings": "; ".join(r.warnings),
|
|
62
|
+
})
|
|
63
|
+
|
|
64
|
+
for inv in invalid:
|
|
65
|
+
rows.append({
|
|
66
|
+
"id": inv.id, "length": 0, "host": host, "model": model,
|
|
67
|
+
"score": np.nan, "prediction": "ERROR", "warnings": inv.reason,
|
|
68
|
+
})
|
|
69
|
+
|
|
70
|
+
df = pd.DataFrame(rows)
|
|
71
|
+
if not df.empty:
|
|
72
|
+
df = df.sort_values("score", ascending=False, na_position="last").reset_index(drop=True)
|
|
73
|
+
return df
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def predict_fasta(path: str, host: str, **kwargs) -> pd.DataFrame:
|
|
77
|
+
"""Atajo: predice directamente desde un fichero FASTA."""
|
|
78
|
+
ids, seqs = parse_fasta(path)
|
|
79
|
+
return predict(list(zip(ids, seqs)), host, **kwargs)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def predict_text(text: str, host: str, **kwargs) -> pd.DataFrame:
|
|
83
|
+
"""Atajo: predice desde texto pegado (FASTA o una secuencia cruda)."""
|
|
84
|
+
ids, seqs = parse_fasta_text(text)
|
|
85
|
+
return predict(list(zip(ids, seqs)), host, **kwargs)
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""
|
|
2
|
+
herd/models.py — Carga de los modelos Keras y puntuación (CPU).
|
|
3
|
+
|
|
4
|
+
Individual: `model.predict(X)` -> sigmoide única.
|
|
5
|
+
Integrado: se reconstruye `heads_model` apuntando a la capa 'species_heads'
|
|
6
|
+
y se toma la columna del host. Esto reproduce EXACTAMENTE el truco
|
|
7
|
+
del código original, que evita las capas Lambda ('mask'/'pick')
|
|
8
|
+
que se rompen al recargar el modelo.
|
|
9
|
+
|
|
10
|
+
La normalización va DENTRO del .keras (capa Normalization adaptada en
|
|
11
|
+
entrenamiento), así que se alimenta el embedding crudo (N, 1280) sin escalar.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Optional
|
|
16
|
+
|
|
17
|
+
import numpy as np
|
|
18
|
+
import tensorflow as tf
|
|
19
|
+
|
|
20
|
+
from .config import (
|
|
21
|
+
MODELS_ROOT, INDIVIDUAL_DIR, INTEGRATED_FILE,
|
|
22
|
+
SP2ID, CORE_HOSTS, HOST_ALIASES, SPECIES_ORDER,
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
_individual_cache = {} # host -> keras.Model
|
|
26
|
+
_integrated_cache = {} # 'heads' -> keras.Model (salida species_heads)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# --- Resolución de nombres de host ------------------------------------------
|
|
30
|
+
def resolve_host(host: str) -> Optional[str]:
|
|
31
|
+
"""Normaliza un nombre de host al de una cabeza del modelo, o None."""
|
|
32
|
+
if host is None:
|
|
33
|
+
return None
|
|
34
|
+
h = str(host).strip()
|
|
35
|
+
if h in SP2ID:
|
|
36
|
+
return h
|
|
37
|
+
if h in HOST_ALIASES and HOST_ALIASES[h] in SP2ID:
|
|
38
|
+
return HOST_ALIASES[h]
|
|
39
|
+
genus = h.lower().split("_")[0] # match laxo por género
|
|
40
|
+
for sp in SP2ID:
|
|
41
|
+
if sp.lower().startswith(genus):
|
|
42
|
+
return sp
|
|
43
|
+
return None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
# --- Rutas de los pesos ------------------------------------------------------
|
|
47
|
+
def individual_path(host: str) -> Path:
|
|
48
|
+
if host not in CORE_HOSTS:
|
|
49
|
+
raise ValueError(f"No individual model for '{host}'. Core hosts: {CORE_HOSTS}")
|
|
50
|
+
return MODELS_ROOT / INDIVIDUAL_DIR / f"{host}.keras"
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def integrated_path() -> Path:
|
|
54
|
+
return MODELS_ROOT / INTEGRATED_FILE
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
# --- Carga -------------------------------------------------------------------
|
|
58
|
+
def load_individual(host: str):
|
|
59
|
+
"""Carga (y cachea) el modelo individual de un host core."""
|
|
60
|
+
if host not in _individual_cache:
|
|
61
|
+
path = individual_path(host)
|
|
62
|
+
_individual_cache[host] = tf.keras.models.load_model(str(path), compile=False)
|
|
63
|
+
return _individual_cache[host]
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def load_integrated_heads():
|
|
67
|
+
"""
|
|
68
|
+
Carga el modelo integrado y devuelve un sub-modelo cuya salida es la capa
|
|
69
|
+
'species_heads' (N, 15 sigmoides). Cachea el resultado.
|
|
70
|
+
"""
|
|
71
|
+
if "heads" not in _integrated_cache:
|
|
72
|
+
path = str(integrated_path())
|
|
73
|
+
try:
|
|
74
|
+
integ = tf.keras.models.load_model(path, compile=False, safe_mode=False)
|
|
75
|
+
except TypeError: # versiones de Keras sin 'safe_mode'
|
|
76
|
+
integ = tf.keras.models.load_model(path, compile=False)
|
|
77
|
+
heads = tf.keras.Model(integ.inputs[0], integ.get_layer("species_heads").output)
|
|
78
|
+
_integrated_cache["heads"] = heads
|
|
79
|
+
return _integrated_cache["heads"]
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
# --- Puntuación --------------------------------------------------------------
|
|
83
|
+
def score_individual(X: np.ndarray, host: str) -> np.ndarray:
|
|
84
|
+
model = load_individual(host)
|
|
85
|
+
return model.predict(X, verbose=0).ravel()
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def score_integrated(X: np.ndarray, host: str) -> np.ndarray:
|
|
89
|
+
sp = resolve_host(host)
|
|
90
|
+
if sp is None:
|
|
91
|
+
raise ValueError(f"Host '{host}' is not in the model. Available: {SPECIES_ORDER}")
|
|
92
|
+
heads = load_integrated_heads()
|
|
93
|
+
H = heads.predict(X, verbose=0) # (N, 15)
|
|
94
|
+
return H[:, SP2ID[sp]]
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""
|
|
2
|
+
herd/validation.py — Limpieza y validación de secuencias (CPU).
|
|
3
|
+
|
|
4
|
+
Se ejecuta ANTES de tocar la GPU. Saca a su propia función la limpieza que
|
|
5
|
+
en el código original vivía dentro de `embed_sequences`, para poder:
|
|
6
|
+
- reportar problemas por secuencia sin reventar el lote entero (guía §23),
|
|
7
|
+
- dejar que `esm.embed_sequences` reciba secuencias ya limpias (guía §5, §13).
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import List, Tuple
|
|
13
|
+
|
|
14
|
+
from .config import VALID_AA, UNKNOWN_AA, SEQ_LENGTH
|
|
15
|
+
|
|
16
|
+
# Precompila la regex una vez: cualquier carácter fuera de los 20 canónicos.
|
|
17
|
+
_NON_CANONICAL = re.compile(f"[^{VALID_AA}]")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def clean_sequence(seq: str) -> Tuple[str, List[str]]:
|
|
21
|
+
"""
|
|
22
|
+
Aplica la receta EXACTA del código original:
|
|
23
|
+
residuos no canónicos -> 'X', y truncado a SEQ_LENGTH.
|
|
24
|
+
Devuelve (secuencia_limpia, avisos).
|
|
25
|
+
"""
|
|
26
|
+
warnings: List[str] = []
|
|
27
|
+
up = seq.upper()
|
|
28
|
+
|
|
29
|
+
n_bad = len(_NON_CANONICAL.findall(up))
|
|
30
|
+
if n_bad:
|
|
31
|
+
warnings.append(f"{n_bad} non-canonical residue(s) -> '{UNKNOWN_AA}'")
|
|
32
|
+
clean = _NON_CANONICAL.sub(UNKNOWN_AA, up)
|
|
33
|
+
|
|
34
|
+
if len(clean) > SEQ_LENGTH:
|
|
35
|
+
warnings.append(f"truncated from {len(clean)} to {SEQ_LENGTH} residues")
|
|
36
|
+
clean = clean[:SEQ_LENGTH]
|
|
37
|
+
|
|
38
|
+
return clean, warnings
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass
|
|
42
|
+
class ValidatedRecord:
|
|
43
|
+
id: str
|
|
44
|
+
seq: str # secuencia limpia (lista para ESM)
|
|
45
|
+
orig_length: int
|
|
46
|
+
warnings: List[str] = field(default_factory=list)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass
|
|
50
|
+
class InvalidRecord:
|
|
51
|
+
id: str
|
|
52
|
+
reason: str
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def validate_records(
|
|
56
|
+
records: List[Tuple[str, str]]
|
|
57
|
+
) -> Tuple[List[ValidatedRecord], List[InvalidRecord]]:
|
|
58
|
+
"""
|
|
59
|
+
Valida y limpia una lista de (id, seq).
|
|
60
|
+
Devuelve (validas, invalidas). Las inválidas se reportan aparte en vez
|
|
61
|
+
de abortar todo el trabajo (guía §23). No se impone longitud mínima:
|
|
62
|
+
en el paper el mínimo de 40 era un filtro de dataset, no de inferencia.
|
|
63
|
+
"""
|
|
64
|
+
valid: List[ValidatedRecord] = []
|
|
65
|
+
invalid: List[InvalidRecord] = []
|
|
66
|
+
|
|
67
|
+
for sid, seq in records:
|
|
68
|
+
raw = (seq or "").strip()
|
|
69
|
+
if not raw:
|
|
70
|
+
invalid.append(InvalidRecord(sid, "empty sequence"))
|
|
71
|
+
continue
|
|
72
|
+
clean, warns = clean_sequence(raw)
|
|
73
|
+
if not clean:
|
|
74
|
+
invalid.append(InvalidRecord(sid, "no valid residues"))
|
|
75
|
+
continue
|
|
76
|
+
valid.append(ValidatedRecord(id=sid, seq=clean, orig_length=len(raw), warnings=warns))
|
|
77
|
+
|
|
78
|
+
return valid, invalid
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: immunoHERD
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: HERD — host-aware prediction of protein immunogenicity
|
|
5
|
+
Author: Isidro Sobrino
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Keywords: immunogenicity,antigen,vaccine,protein,ESM-2,protein-language-model,veterinary,bioinformatics,deep-learning
|
|
8
|
+
Classifier: Development Status :: 4 - Beta
|
|
9
|
+
Classifier: Intended Audience :: Science/Research
|
|
10
|
+
Classifier: Operating System :: OS Independent
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
18
|
+
Requires-Python: >=3.11
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: fair-esm
|
|
22
|
+
Requires-Dist: torch
|
|
23
|
+
Requires-Dist: tensorflow>=2.16
|
|
24
|
+
Requires-Dist: keras<4,>=3.13
|
|
25
|
+
Requires-Dist: numpy
|
|
26
|
+
Requires-Dist: pandas
|
|
27
|
+
Dynamic: license-file
|
|
28
|
+
|
|
29
|
+
# HERD
|
|
30
|
+
|
|
31
|
+
**H**ost-aware **E**stimation of antigen immunogenicity and **R**anking through
|
|
32
|
+
**D**eep-learning.
|
|
33
|
+
|
|
34
|
+
HERD predicts the immunogenicity of proteins taking the host species into
|
|
35
|
+
account. This package is the reusable core shared by the different surfaces of
|
|
36
|
+
the project (web Space, Colab and command line).
|
|
37
|
+
|
|
38
|
+
## Pipeline
|
|
39
|
+
|
|
40
|
+
1. Sequence cleaning: any character outside `ACDEFGHIKLMNPQRSTVWY` becomes `X`,
|
|
41
|
+
and the sequence is truncated to 1022 residues.
|
|
42
|
+
2. Embedding with ESM-2 (`esm2_t33_650M_UR50D`, layer 33) via `fair-esm`,
|
|
43
|
+
mean-pooled over the real residues: a 1280-dimensional vector.
|
|
44
|
+
3. A host-aware head on top of that embedding. Normalization lives inside the
|
|
45
|
+
`.keras` file itself, so the raw embedding is fed in unscaled.
|
|
46
|
+
4. Decision threshold: 0.5.
|
|
47
|
+
|
|
48
|
+
## Installation
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
pip install immunoHERD
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
The package is installed as `immunoHERD` but imported as `herd`, and the
|
|
55
|
+
console command is `herd`.
|
|
56
|
+
|
|
57
|
+
The model weights ship inside the package. The ESM-2 weights, by contrast, are
|
|
58
|
+
downloaded automatically the first time an embedding is computed.
|
|
59
|
+
|
|
60
|
+
## Usage
|
|
61
|
+
|
|
62
|
+
### From Python
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
from herd import predict_fasta
|
|
66
|
+
|
|
67
|
+
df = predict_fasta("proteins.fasta", host="Bos_taurus", model="integrated")
|
|
68
|
+
print(df)
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
`predict` (a list of `(id, sequence)` tuples) and `predict_text` (pasted FASTA
|
|
72
|
+
text, or a bare sequence) are also available. All three return a `DataFrame`
|
|
73
|
+
sorted by descending score with the columns `id`, `length`, `host`, `model`,
|
|
74
|
+
`score`, `prediction` and `warnings`. Sequences that fail validation are
|
|
75
|
+
returned at the end, with `score` set to `NaN`.
|
|
76
|
+
|
|
77
|
+
The `warnings` column also records when a sequence was modified: non-canonical
|
|
78
|
+
residues replaced by `X`, or truncation to 1022 residues.
|
|
79
|
+
|
|
80
|
+
### From the command line
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
herd predict proteins.fasta --host Bos_taurus --out results.csv
|
|
84
|
+
herd predict proteins.fasta --host Canis_sp --model individual --out dog.csv
|
|
85
|
+
herd hosts
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
`--out` is required, so a long run cannot be lost by forgetting to save it.
|
|
89
|
+
Other options: `--model`, `--threshold`, `--batch-seqs`, `--device`.
|
|
90
|
+
Warnings go to standard error, so the CSV stays clean when redirected.
|
|
91
|
+
|
|
92
|
+
## Models and hosts
|
|
93
|
+
|
|
94
|
+
Two models, selected with the `model` argument:
|
|
95
|
+
|
|
96
|
+
- `"integrated"` — a pan-species model with 15 heads, one per host.
|
|
97
|
+
This is the recommended default.
|
|
98
|
+
- `"individual"` — one model trained per species. Only available for the six
|
|
99
|
+
core hosts: `Homo_sapiens`, `Bos_taurus`, `Sus_scrofa`, `Gallus_gallus`,
|
|
100
|
+
`Canis_sp` and `Equus_caballus`.
|
|
101
|
+
|
|
102
|
+
The integrated model also covers nine auxiliary hosts: `Camelidae`,
|
|
103
|
+
`Capra_hircus`, `Cavia_porcellus`, `Macaca_sp_`, `Mus_musculus`,
|
|
104
|
+
`Non_human_primate`, `Oryctolagus_cuniculus`, `Ovis_aries` and `Rattus_sp_`.
|
|
105
|
+
**These nine heads are not validated to the same level as the six core hosts,
|
|
106
|
+
and their scores should be interpreted with more caution.**
|
|
107
|
+
|
|
108
|
+
Host names accept some common aliases (`Canis_familiaris`, `Homo`, `Rattus`…)
|
|
109
|
+
and a loose genus match.
|
|
110
|
+
|
|
111
|
+
## Where the models are looked up
|
|
112
|
+
|
|
113
|
+
By default, in the `models/` directory that ships inside the installed package.
|
|
114
|
+
A different location can be given through the `HERD_MODELS_DIR` environment
|
|
115
|
+
variable, which takes precedence:
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
export HERD_MODELS_DIR=/path/to/my/models
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
The expected layout is `<HERD_MODELS_DIR>/integrated.keras` and
|
|
122
|
+
`<HERD_MODELS_DIR>/individual/<host>.keras`.
|
|
123
|
+
|
|
124
|
+
## Performance notes
|
|
125
|
+
|
|
126
|
+
Sequences are grouped into batches **sorted by length**, which removes almost
|
|
127
|
+
all padding: on a real proteome (4,570 sequences) this halves the GPU work.
|
|
128
|
+
Results are unaffected, since each mean is computed only over the real residues
|
|
129
|
+
of its own sequence.
|
|
130
|
+
|
|
131
|
+
A GPU is not required, but embedding on CPU is considerably slower.
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
MANIFEST.in
|
|
3
|
+
README.md
|
|
4
|
+
pyproject.toml
|
|
5
|
+
src/herd/__init__.py
|
|
6
|
+
src/herd/cli.py
|
|
7
|
+
src/herd/config.py
|
|
8
|
+
src/herd/esm.py
|
|
9
|
+
src/herd/fasta.py
|
|
10
|
+
src/herd/inference.py
|
|
11
|
+
src/herd/models.py
|
|
12
|
+
src/herd/validation.py
|
|
13
|
+
src/herd/models/integrated.keras
|
|
14
|
+
src/herd/models/individual/Bos_taurus.keras
|
|
15
|
+
src/herd/models/individual/Canis_sp.keras
|
|
16
|
+
src/herd/models/individual/Equus_caballus.keras
|
|
17
|
+
src/herd/models/individual/Gallus_gallus.keras
|
|
18
|
+
src/herd/models/individual/Homo_sapiens.keras
|
|
19
|
+
src/herd/models/individual/Sus_scrofa.keras
|
|
20
|
+
src/immunoHERD.egg-info/PKG-INFO
|
|
21
|
+
src/immunoHERD.egg-info/SOURCES.txt
|
|
22
|
+
src/immunoHERD.egg-info/dependency_links.txt
|
|
23
|
+
src/immunoHERD.egg-info/entry_points.txt
|
|
24
|
+
src/immunoHERD.egg-info/requires.txt
|
|
25
|
+
src/immunoHERD.egg-info/top_level.txt
|
|
26
|
+
tests/test_basics.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
herd
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Pruebas mínimas del paquete instalado.
|
|
3
|
+
|
|
4
|
+
Con el layout src/ estas pruebas NO pueden importar la carpeta del repositorio:
|
|
5
|
+
importan siempre `herd` instalado, que es justo lo que se quiere comprobar.
|
|
6
|
+
|
|
7
|
+
pip install -e .
|
|
8
|
+
pytest
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from herd.validation import clean_sequence, validate_records
|
|
12
|
+
from herd.fasta import parse_fasta_text
|
|
13
|
+
from herd.config import SEQ_LENGTH, CORE_HOSTS, SPECIES_ORDER
|
|
14
|
+
from herd.models import resolve_host
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def test_non_canonical_residues_become_x():
|
|
18
|
+
clean, warnings = clean_sequence("MKVBBBZZZ")
|
|
19
|
+
assert "B" not in clean and "Z" not in clean
|
|
20
|
+
assert clean.count("X") == 6
|
|
21
|
+
assert "non-canonical" in warnings[0]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def test_truncation_to_1022():
|
|
25
|
+
clean, warnings = clean_sequence("M" * 1315)
|
|
26
|
+
assert len(clean) == SEQ_LENGTH
|
|
27
|
+
assert warnings == [f"truncated from 1315 to {SEQ_LENGTH} residues"]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def test_empty_sequence_is_invalid():
|
|
31
|
+
valid, invalid = validate_records([("a", "")])
|
|
32
|
+
assert valid == []
|
|
33
|
+
assert invalid[0].reason == "empty sequence"
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def test_fasta_text_multiline_is_joined():
|
|
37
|
+
ids, seqs = parse_fasta_text(">p1\nMKV\nLAA\n>p2\nMSE")
|
|
38
|
+
assert ids == ["p1", "p2"]
|
|
39
|
+
assert seqs == ["MKVLAA", "MSE"]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def test_bare_sequence_without_header():
|
|
43
|
+
ids, seqs = parse_fasta_text("MKVLAA")
|
|
44
|
+
assert ids == ["sequence_1"]
|
|
45
|
+
assert seqs == ["MKVLAA"]
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def test_host_aliases_resolve():
|
|
49
|
+
assert resolve_host("Canis_familiaris") == "Canis_sp"
|
|
50
|
+
assert resolve_host("Homo") == "Homo_sapiens"
|
|
51
|
+
assert resolve_host("Tyrannosaurus") is None
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def test_core_hosts_are_a_subset_of_the_heads():
|
|
55
|
+
assert set(CORE_HOSTS) <= set(SPECIES_ORDER)
|
|
56
|
+
assert len(SPECIES_ORDER) == 15
|