whisper-ppi 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- whisper_ppi-0.1.0/PKG-INFO +125 -0
- whisper_ppi-0.1.0/README.md +103 -0
- whisper_ppi-0.1.0/pyproject.toml +3 -0
- whisper_ppi-0.1.0/setup.cfg +40 -0
- whisper_ppi-0.1.0/setup.py +2 -0
- whisper_ppi-0.1.0/whisper/__init__.py +14 -0
- whisper_ppi-0.1.0/whisper/fragment_features.py +143 -0
- whisper_ppi-0.1.0/whisper/fragment_train.py +175 -0
- whisper_ppi-0.1.0/whisper/peptide_features.py +137 -0
- whisper_ppi-0.1.0/whisper/peptide_train.py +174 -0
- whisper_ppi-0.1.0/whisper/protein_features.py +164 -0
- whisper_ppi-0.1.0/whisper/protein_train.py +145 -0
- whisper_ppi-0.1.0/whisper_ppi.egg-info/PKG-INFO +125 -0
- whisper_ppi-0.1.0/whisper_ppi.egg-info/SOURCES.txt +23 -0
- whisper_ppi-0.1.0/whisper_ppi.egg-info/dependency_links.txt +1 -0
- whisper_ppi-0.1.0/whisper_ppi.egg-info/requires.txt +4 -0
- whisper_ppi-0.1.0/whisper_ppi.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: whisper-ppi
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Weak Heuristic Inference for Supervisory Protein intERaction mapping for PDB and AP-MS datasets
|
|
5
|
+
Home-page: https://github.com/camlab-bioml/whisper
|
|
6
|
+
Author: Vesal Kasmaeifar
|
|
7
|
+
Author-email: vesal.kasmaeifar@mail.utoronto.com
|
|
8
|
+
License: MIT
|
|
9
|
+
Project-URL: Documentation, https://whisper.readthedocs.io/en/latest/
|
|
10
|
+
Project-URL: Source, https://github.com/camlab-bioml/whisper
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
16
|
+
Requires-Python: >=3.10
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
Requires-Dist: numpy
|
|
19
|
+
Requires-Dist: pandas
|
|
20
|
+
Requires-Dist: scikit-learn
|
|
21
|
+
Requires-Dist: scipy
|
|
22
|
+
|
|
23
|
+
# whisper
|
|
24
|
+
|
|
25
|
+
`whisper` is a Python package for scoring protein–protein interactions from proximity labeling and affinity purification mass spectrometry datasets. It uses interpretable features, programmatic weak supervision, and decoy-based false discovery rate (FDR) estimation to identify high-confidence interactors.
|
|
26
|
+
|
|
27
|
+
## Installation
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
git clone https://github.com/camlab-bioml/whisper
|
|
31
|
+
cd whisper
|
|
32
|
+
pip install .
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
## Input Format
|
|
36
|
+
|
|
37
|
+
- A CSV file with:
|
|
38
|
+
- One column named `Protein`
|
|
39
|
+
- Other columns representing bait replicate intensities, named as `BAIT_1`, `BAIT_2`, etc.
|
|
40
|
+
- Control samples must be identifiable via substrings in their column names (e.g., `"EGFP"` or `"Empty"`).
|
|
41
|
+
|
|
42
|
+
## Usage
|
|
43
|
+
|
|
44
|
+
```python
|
|
45
|
+
#protein-level
|
|
46
|
+
from whisper.protein_features import feature_engineering_protein
|
|
47
|
+
from whisper.protein_train import train_and_score_protein
|
|
48
|
+
import pandas as pd
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
# Load intensity table
|
|
52
|
+
intensity_df = pd.read_csv("input_intensity_dataset.tsv", sep="\t")
|
|
53
|
+
|
|
54
|
+
controls = ['EGFP', 'Empty', 'NminiTurbo']
|
|
55
|
+
|
|
56
|
+
# Run feature engineering
|
|
57
|
+
features_df = feature_engineering_protein(intensity_df, controls)
|
|
58
|
+
|
|
59
|
+
# You can save the features to use in the next step with different settings without generating them again.
|
|
60
|
+
features_df = pd.read_csv("features.csv")
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
# Run scoring and FDR estimation
|
|
64
|
+
scored_df = train_and_score_protein(features_df, initial_positives=15, initial_negatives=200)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
#peptide-level
|
|
68
|
+
from whisper.peptide_features import feature_engineering_peptide
|
|
69
|
+
from whisper.peptide_train import train_and_score_peptide
|
|
70
|
+
import pandas as pd
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
# Load intensity table
|
|
74
|
+
intensity_df = pd.read_csv("input_intensity_dataset.tsv", sep="\t")
|
|
75
|
+
|
|
76
|
+
controls = ['EGFP', 'Empty', 'NminiTurbo']
|
|
77
|
+
|
|
78
|
+
# Run feature engineering
|
|
79
|
+
features_df = feature_engineering_peptide(intensity_df, controls)
|
|
80
|
+
|
|
81
|
+
# features_df = pd.read_csv("features.csv")
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
# Run scoring and FDR estimation
|
|
85
|
+
scored_df = train_and_score_peptide(features_df, initial_positives=15, initial_negatives=200)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
#fragment-level
|
|
89
|
+
from whisper.fragment_features import feature_engineering_fragment
|
|
90
|
+
from whisper.fragment_train import train_and_score_fragment
|
|
91
|
+
import pandas as pd
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
# Load intensity table
|
|
95
|
+
intensity_df = pd.read_csv("input_intensity_dataset.tsv", sep="\t")
|
|
96
|
+
|
|
97
|
+
controls = ['EGFP', 'Empty', 'NminiTurbo']
|
|
98
|
+
|
|
99
|
+
# Run feature engineering
|
|
100
|
+
features_df = feature_engineering_fragment(intensity_df, controls)
|
|
101
|
+
|
|
102
|
+
# features_df = pd.read_csv("features.csv")
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
# Run scoring and FDR estimation
|
|
106
|
+
scored_df = train_and_score_fragment(features_df, initial_positives=15, initial_negatives=200)
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
## Output
|
|
110
|
+
|
|
111
|
+
The final output includes:
|
|
112
|
+
- `predicted_probability`: Probability of each bait–prey interaction being real
|
|
113
|
+
- `FDR`: Estimated false discovery rate
|
|
114
|
+
- `global_cv_flag`: Flag for likely background preys based on variability across all samples
|
|
115
|
+
|
|
116
|
+
## Tutorial
|
|
117
|
+
|
|
118
|
+
[Read the full documentation](https://whisper.readthedocs.io/en/latest/)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
## Citation
|
|
122
|
+
|
|
123
|
+
This software is authored by: Vesal Kasmaeifar, Kieran R Campbell
|
|
124
|
+
|
|
125
|
+
Lunenfeld-Tanenbaum Research Institute & University of Toronto
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
# whisper
|
|
2
|
+
|
|
3
|
+
`whisper` is a Python package for scoring protein–protein interactions from proximity labeling and affinity purification mass spectrometry datasets. It uses interpretable features, programmatic weak supervision, and decoy-based false discovery rate (FDR) estimation to identify high-confidence interactors.
|
|
4
|
+
|
|
5
|
+
## Installation
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
git clone https://github.com/camlab-bioml/whisper
|
|
9
|
+
cd whisper
|
|
10
|
+
pip install .
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Input Format
|
|
14
|
+
|
|
15
|
+
- A CSV file with:
|
|
16
|
+
- One column named `Protein`
|
|
17
|
+
- Other columns representing bait replicate intensities, named as `BAIT_1`, `BAIT_2`, etc.
|
|
18
|
+
- Control samples must be identifiable via substrings in their column names (e.g., `"EGFP"` or `"Empty"`).
|
|
19
|
+
|
|
20
|
+
## Usage
|
|
21
|
+
|
|
22
|
+
```python
|
|
23
|
+
#protein-level
|
|
24
|
+
from whisper.protein_features import feature_engineering_protein
|
|
25
|
+
from whisper.protein_train import train_and_score_protein
|
|
26
|
+
import pandas as pd
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# Load intensity table
|
|
30
|
+
intensity_df = pd.read_csv("input_intensity_dataset.tsv", sep="\t")
|
|
31
|
+
|
|
32
|
+
controls = ['EGFP', 'Empty', 'NminiTurbo']
|
|
33
|
+
|
|
34
|
+
# Run feature engineering
|
|
35
|
+
features_df = feature_engineering_protein(intensity_df, controls)
|
|
36
|
+
|
|
37
|
+
# You can save the features to use in the next step with different settings without generating them again.
|
|
38
|
+
features_df = pd.read_csv("features.csv")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
# Run scoring and FDR estimation
|
|
42
|
+
scored_df = train_and_score_protein(features_df, initial_positives=15, initial_negatives=200)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
#peptide-level
|
|
46
|
+
from whisper.peptide_features import feature_engineering_peptide
|
|
47
|
+
from whisper.peptide_train import train_and_score_peptide
|
|
48
|
+
import pandas as pd
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
# Load intensity table
|
|
52
|
+
intensity_df = pd.read_csv("input_intensity_dataset.tsv", sep="\t")
|
|
53
|
+
|
|
54
|
+
controls = ['EGFP', 'Empty', 'NminiTurbo']
|
|
55
|
+
|
|
56
|
+
# Run feature engineering
|
|
57
|
+
features_df = feature_engineering_peptide(intensity_df, controls)
|
|
58
|
+
|
|
59
|
+
# features_df = pd.read_csv("features.csv")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
# Run scoring and FDR estimation
|
|
63
|
+
scored_df = train_and_score_peptide(features_df, initial_positives=15, initial_negatives=200)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
#fragment-level
|
|
67
|
+
from whisper.fragment_features import feature_engineering_fragment
|
|
68
|
+
from whisper.fragment_train import train_and_score_fragment
|
|
69
|
+
import pandas as pd
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# Load intensity table
|
|
73
|
+
intensity_df = pd.read_csv("input_intensity_dataset.tsv", sep="\t")
|
|
74
|
+
|
|
75
|
+
controls = ['EGFP', 'Empty', 'NminiTurbo']
|
|
76
|
+
|
|
77
|
+
# Run feature engineering
|
|
78
|
+
features_df = feature_engineering_fragment(intensity_df, controls)
|
|
79
|
+
|
|
80
|
+
# features_df = pd.read_csv("features.csv")
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
# Run scoring and FDR estimation
|
|
84
|
+
scored_df = train_and_score_fragment(features_df, initial_positives=15, initial_negatives=200)
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
## Output
|
|
88
|
+
|
|
89
|
+
The final output includes:
|
|
90
|
+
- `predicted_probability`: Probability of each bait–prey interaction being real
|
|
91
|
+
- `FDR`: Estimated false discovery rate
|
|
92
|
+
- `global_cv_flag`: Flag for likely background preys based on variability across all samples
|
|
93
|
+
|
|
94
|
+
## Tutorial
|
|
95
|
+
|
|
96
|
+
[Read the full documentation](https://whisper.readthedocs.io/en/latest/)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
## Citation
|
|
100
|
+
|
|
101
|
+
This software is authored by: Vesal Kasmaeifar, Kieran R Campbell
|
|
102
|
+
|
|
103
|
+
Lunenfeld-Tanenbaum Research Institute & University of Toronto
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
[metadata]
|
|
2
|
+
name = whisper-ppi
|
|
3
|
+
version = 0.1.0
|
|
4
|
+
author = Vesal Kasmaeifar
|
|
5
|
+
author_email = vesal.kasmaeifar@mail.utoronto.com
|
|
6
|
+
description = Weak Heuristic Inference for Supervisory Protein intERaction mapping for PDB and AP-MS datasets
|
|
7
|
+
long_description = file: README.md
|
|
8
|
+
long_description_content_type = text/markdown
|
|
9
|
+
url = https://github.com/camlab-bioml/whisper
|
|
10
|
+
project_urls =
|
|
11
|
+
Documentation = https://whisper.readthedocs.io/en/latest/
|
|
12
|
+
Source = https://github.com/camlab-bioml/whisper
|
|
13
|
+
license = MIT
|
|
14
|
+
license_files = LICENSE
|
|
15
|
+
classifiers =
|
|
16
|
+
Programming Language :: Python :: 3
|
|
17
|
+
License :: OSI Approved :: MIT License
|
|
18
|
+
Operating System :: OS Independent
|
|
19
|
+
Intended Audience :: Science/Research
|
|
20
|
+
Topic :: Scientific/Engineering :: Bio-Informatics
|
|
21
|
+
|
|
22
|
+
[options]
|
|
23
|
+
package_dir =
|
|
24
|
+
= .
|
|
25
|
+
packages = find:
|
|
26
|
+
python_requires = >=3.10
|
|
27
|
+
include_package_data = True
|
|
28
|
+
install_requires =
|
|
29
|
+
numpy
|
|
30
|
+
pandas
|
|
31
|
+
scikit-learn
|
|
32
|
+
scipy
|
|
33
|
+
|
|
34
|
+
[options.packages.find]
|
|
35
|
+
where = .
|
|
36
|
+
|
|
37
|
+
[egg_info]
|
|
38
|
+
tag_build =
|
|
39
|
+
tag_date = 0
|
|
40
|
+
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
from .protein_features import feature_engineering_protein as feature_engineering_protein
|
|
2
|
+
from .protein_train import train_and_score_protein as train_and_score_protein
|
|
3
|
+
|
|
4
|
+
from .peptide_features import feature_engineering_peptide
|
|
5
|
+
from .peptide_train import train_and_score_peptide
|
|
6
|
+
|
|
7
|
+
from .fragment_features import feature_engineering_fragment
|
|
8
|
+
from .fragment_train import train_and_score_fragment
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
"feature_engineering_protein", "train_and_score_protein",
|
|
12
|
+
"feature_engineering_peptide", "train_and_score_peptide",
|
|
13
|
+
"feature_engineering_fragment", "train_and_score_fragment",
|
|
14
|
+
]
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
# whisper/fragment_features.py
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
import numpy as np
|
|
5
|
+
import re
|
|
6
|
+
from sklearn.preprocessing import StandardScaler
|
|
7
|
+
import warnings
|
|
8
|
+
warnings.filterwarnings("ignore")
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def feature_engineering_fragment(intensity_df: pd.DataFrame, controls: list) -> pd.DataFrame:
|
|
12
|
+
"""
|
|
13
|
+
Compute fragment-level features.
|
|
14
|
+
Mirrors protein/peptide feature engineering but operates on (Protein, Peptide, Fragment).
|
|
15
|
+
|
|
16
|
+
Parameters
|
|
17
|
+
----------
|
|
18
|
+
intensity_df : pd.DataFrame
|
|
19
|
+
Fragment-level intensity matrix with columns:
|
|
20
|
+
['Protein', 'Peptide', 'Fragment', <bait/replicate and control columns>]
|
|
21
|
+
controls : list
|
|
22
|
+
List of control identifiers (e.g., ["EGFP", "Empty", "NminiTurbo"])
|
|
23
|
+
|
|
24
|
+
Returns
|
|
25
|
+
-------
|
|
26
|
+
pd.DataFrame
|
|
27
|
+
Aggregated feature table per bait–(protein, peptide, fragment) with computed metrics.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
# --- Identify columns ---
|
|
31
|
+
id_cols = ["Protein", "Peptide", "Fragment"]
|
|
32
|
+
control_columns = [c for c in intensity_df.columns if any(ctrl in c for ctrl in controls)]
|
|
33
|
+
sample_columns = [c for c in intensity_df.columns if c not in id_cols]
|
|
34
|
+
bait_like_columns = [c for c in sample_columns if c not in control_columns]
|
|
35
|
+
baits = sorted(list(set(col.split("_")[0] for col in bait_like_columns)))
|
|
36
|
+
intensity_columns = control_columns + [c for c in bait_like_columns if any(b in c for b in baits)]
|
|
37
|
+
|
|
38
|
+
# --- Global CV across ALL samples for each fragment (exclude BirA and bait proteins for map) ---
|
|
39
|
+
global_cv = {}
|
|
40
|
+
for _, row in intensity_df.iterrows():
|
|
41
|
+
prot = row["Protein"]
|
|
42
|
+
vals = row[intensity_columns].astype(float).values
|
|
43
|
+
mean_all = np.mean(vals)
|
|
44
|
+
sd_all = np.std(vals)
|
|
45
|
+
global_cv[(row["Protein"], row["Peptide"], row["Fragment"])] = sd_all / mean_all if mean_all > 0 else 0
|
|
46
|
+
|
|
47
|
+
all_bait_features = []
|
|
48
|
+
|
|
49
|
+
for bait in baits:
|
|
50
|
+
bait_columns = [c for c in intensity_df.columns if re.fullmatch(fr"{bait}_\d+", c)]
|
|
51
|
+
if len(bait_columns) == 0:
|
|
52
|
+
# skip baits without replicate columns that match the "<bait>_#" pattern
|
|
53
|
+
continue
|
|
54
|
+
|
|
55
|
+
# Exclude the bait's own protein & BirA
|
|
56
|
+
filtered_df = intensity_df[~intensity_df["Protein"].isin([bait, "birA"])].copy()
|
|
57
|
+
|
|
58
|
+
# --- Control summary stats ---
|
|
59
|
+
ctrl_vals = filtered_df[control_columns].astype(float).values
|
|
60
|
+
ctrl_means = np.mean(ctrl_vals, axis=1)
|
|
61
|
+
ctrl_sds = np.std(ctrl_vals, axis=1)
|
|
62
|
+
|
|
63
|
+
min_mean_ctrl = np.min(ctrl_means[ctrl_means > 0]) if np.any(ctrl_means > 0) else 1.0
|
|
64
|
+
min_sd_ctrl = np.min(ctrl_sds[ctrl_sds > 0]) if np.any(ctrl_sds > 0) else 1.0
|
|
65
|
+
|
|
66
|
+
features = []
|
|
67
|
+
for _, row in filtered_df.iterrows():
|
|
68
|
+
prey_prot = row["Protein"]
|
|
69
|
+
prey_pep = row["Peptide"]
|
|
70
|
+
prey_frag = row["Fragment"]
|
|
71
|
+
|
|
72
|
+
bait_int = row[bait_columns].astype(float).values
|
|
73
|
+
ctrl_int = row[control_columns].astype(float).values
|
|
74
|
+
|
|
75
|
+
mean_bait = np.mean(bait_int)
|
|
76
|
+
median_bait = np.median(bait_int)
|
|
77
|
+
sd_bait = np.std(bait_int)
|
|
78
|
+
|
|
79
|
+
mean_ctrl = np.mean(ctrl_int)
|
|
80
|
+
sd_ctrl = np.std(ctrl_int)
|
|
81
|
+
mean_ctrl = mean_ctrl if mean_ctrl > 0 else min_mean_ctrl
|
|
82
|
+
sd_ctrl = sd_ctrl if sd_ctrl > 0 else min_sd_ctrl
|
|
83
|
+
|
|
84
|
+
zero_count = np.sum(bait_int == 0)
|
|
85
|
+
fold_change = mean_bait / mean_ctrl
|
|
86
|
+
log_fc = np.log2(fold_change + 1e-5)
|
|
87
|
+
penalized_log_fc = log_fc / max(1, zero_count)
|
|
88
|
+
snr = mean_bait / sd_ctrl
|
|
89
|
+
penalized_snr = snr / max(1, zero_count)
|
|
90
|
+
|
|
91
|
+
replicate_fc_sd = np.std(bait_int / mean_ctrl)
|
|
92
|
+
bait_cv = sd_bait / mean_bait if mean_bait != 0 else 0
|
|
93
|
+
bait_ctrl_sd_ratio = sd_bait / sd_ctrl
|
|
94
|
+
|
|
95
|
+
nonzero_reps = int(np.sum(bait_int > 0))
|
|
96
|
+
reps_above_ctrl_med = int(np.sum(bait_int > np.median(ctrl_int)))
|
|
97
|
+
single_rep_flag = 1 if nonzero_reps == 1 else 0
|
|
98
|
+
|
|
99
|
+
features.append({
|
|
100
|
+
"Bait": bait,
|
|
101
|
+
"Protein": prey_prot,
|
|
102
|
+
"Peptide": prey_pep,
|
|
103
|
+
"Fragment": prey_frag,
|
|
104
|
+
"log_fold_change": penalized_log_fc,
|
|
105
|
+
"snr": penalized_snr,
|
|
106
|
+
"mean_diff": mean_bait - mean_ctrl,
|
|
107
|
+
"median_diff": median_bait - np.median(ctrl_int),
|
|
108
|
+
"replicate_fold_change_sd": replicate_fc_sd,
|
|
109
|
+
"bait_cv": bait_cv,
|
|
110
|
+
"bait_control_sd_ratio": bait_ctrl_sd_ratio,
|
|
111
|
+
"zero_or_neg_fc": 0 if penalized_log_fc <= 0 else 1,
|
|
112
|
+
"nonzero_reps": nonzero_reps,
|
|
113
|
+
"reps_above_ctrl_med": reps_above_ctrl_med,
|
|
114
|
+
"single_rep_flag": single_rep_flag,
|
|
115
|
+
})
|
|
116
|
+
|
|
117
|
+
bait_features = pd.DataFrame(features)
|
|
118
|
+
|
|
119
|
+
# --- Scale and composite score (consistent with protein/peptide) ---
|
|
120
|
+
scale_cols = [
|
|
121
|
+
"log_fold_change", "snr", "mean_diff", "median_diff",
|
|
122
|
+
"replicate_fold_change_sd", "bait_cv", "bait_control_sd_ratio",
|
|
123
|
+
"zero_or_neg_fc",
|
|
124
|
+
]
|
|
125
|
+
scaler = StandardScaler()
|
|
126
|
+
scaled_df = pd.DataFrame(
|
|
127
|
+
scaler.fit_transform(bait_features[scale_cols]),
|
|
128
|
+
columns=scale_cols, index=bait_features.index,
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
bait_features["composite_score"] = scaled_df[
|
|
132
|
+
["log_fold_change", "snr", "mean_diff", "median_diff"]
|
|
133
|
+
].mean(axis=1)
|
|
134
|
+
|
|
135
|
+
bait_features["global_cv"] = bait_features.apply(
|
|
136
|
+
lambda r: global_cv.get((r["Protein"], r["Peptide"], r["Fragment"]), np.nan), axis=1
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
all_bait_features.append(bait_features.sort_values("composite_score", ascending=False))
|
|
140
|
+
|
|
141
|
+
aggregated_features_df = pd.concat(all_bait_features, ignore_index=True)
|
|
142
|
+
aggregated_features_df.to_csv("features_fragment.csv", index=False)
|
|
143
|
+
return aggregated_features_df
|
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
# whsiper/fragment_train.py
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import pandas as pd
|
|
5
|
+
import numpy as np
|
|
6
|
+
from sklearn.ensemble import RandomForestClassifier, BaggingClassifier
|
|
7
|
+
from sklearn.preprocessing import StandardScaler
|
|
8
|
+
from scipy.cluster.hierarchy import linkage, fcluster
|
|
9
|
+
import warnings
|
|
10
|
+
warnings.filterwarnings("ignore")
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def train_and_score_fragment(
|
|
14
|
+
features_df: pd.DataFrame,
|
|
15
|
+
initial_positives: int = 15,
|
|
16
|
+
initial_negatives: int = 200,
|
|
17
|
+
random_state: int = 42,
|
|
18
|
+
save_dir: str = ".",
|
|
19
|
+
fragment_out: str = "whisper_fragment_scores.csv",
|
|
20
|
+
protein_out: str = "whisper_protein_scores_from_fragments.csv",
|
|
21
|
+
aggregate_strategy: str = "max", # "max" or "mean" for fragment->protein prob aggregation
|
|
22
|
+
):
|
|
23
|
+
"""
|
|
24
|
+
Train a model on FRAGMENT-level features, compute bait-specific decoy FDR,
|
|
25
|
+
and aggregate to PROTEIN-level scores per bait.
|
|
26
|
+
|
|
27
|
+
Expected columns in `features_df`:
|
|
28
|
+
- Bait, Protein, Peptide, Fragment
|
|
29
|
+
- composite_score, global_cv (optional), single_rep_flag (optional)
|
|
30
|
+
- Feature columns:
|
|
31
|
+
['log_fold_change','snr','mean_diff','median_diff',
|
|
32
|
+
'replicate_fold_change_sd','bait_cv','bait_control_sd_ratio','zero_or_neg_fc']
|
|
33
|
+
|
|
34
|
+
Saves:
|
|
35
|
+
- <save_dir>/<fragment_out>: fragment-level scores with FDR
|
|
36
|
+
- <save_dir>/<protein_out>: protein-level aggregation per bait
|
|
37
|
+
|
|
38
|
+
Returns:
|
|
39
|
+
(fragment_df, protein_df)
|
|
40
|
+
"""
|
|
41
|
+
rng = np.random.RandomState(random_state)
|
|
42
|
+
|
|
43
|
+
# Stable order
|
|
44
|
+
df = (
|
|
45
|
+
features_df.copy()
|
|
46
|
+
.sort_values(["Bait", "Protein", "Peptide", "Fragment"])
|
|
47
|
+
.reset_index(drop=True)
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
feature_columns = [
|
|
51
|
+
"log_fold_change", "snr", "mean_diff", "median_diff",
|
|
52
|
+
"replicate_fold_change_sd", "bait_cv", "bait_control_sd_ratio",
|
|
53
|
+
"zero_or_neg_fc",
|
|
54
|
+
]
|
|
55
|
+
X = df[feature_columns].values
|
|
56
|
+
|
|
57
|
+
# ---------- Cluster baits to identify "strong" set ----------
|
|
58
|
+
bait_top50_stds = {
|
|
59
|
+
b: df[df["Bait"] == b]["composite_score"].nlargest(50).std()
|
|
60
|
+
for b in df["Bait"].unique()
|
|
61
|
+
}
|
|
62
|
+
bait_names = np.array(list(bait_top50_stds.keys()))
|
|
63
|
+
bait_scores = np.array(list(bait_top50_stds.values()), dtype=float).reshape(-1, 1)
|
|
64
|
+
|
|
65
|
+
if len(bait_names) > 2:
|
|
66
|
+
Z = linkage(bait_scores, method="ward")
|
|
67
|
+
clusters = fcluster(Z, t=2, criterion="maxclust")
|
|
68
|
+
else:
|
|
69
|
+
clusters = np.ones(len(bait_names), dtype=int)
|
|
70
|
+
|
|
71
|
+
cluster_sizes = {c: int(np.sum(clusters == c)) for c in np.unique(clusters)}
|
|
72
|
+
cluster_means = {c: float(bait_scores[clusters == c].mean()) for c in np.unique(clusters)}
|
|
73
|
+
max_size = max(cluster_sizes.values())
|
|
74
|
+
cands = [c for c, n in cluster_sizes.items() if n == max_size]
|
|
75
|
+
strong_cluster_id = cands[0] if len(cands) == 1 else max(cands, key=lambda c: cluster_means[c])
|
|
76
|
+
strong_baits = [b for b, c in zip(bait_names, clusters) if c == strong_cluster_id]
|
|
77
|
+
|
|
78
|
+
# ---------- Pseudo-labels ----------
|
|
79
|
+
y = pd.Series(0, index=df.index) # 0=unlabeled, 1=pos, -1=neg
|
|
80
|
+
bait_pos_quota = {b: (initial_positives if b in strong_baits else 0) for b in df["Bait"].unique()}
|
|
81
|
+
|
|
82
|
+
for bait in df["Bait"].unique():
|
|
83
|
+
sub = df[df["Bait"] == bait].copy()
|
|
84
|
+
n_pos = bait_pos_quota[bait]
|
|
85
|
+
if n_pos > 0:
|
|
86
|
+
ranked = sub.sort_values("composite_score", ascending=False)
|
|
87
|
+
elig = ranked[ranked.get("single_rep_flag", 0) != 1] # exclude single-rep spikes if present
|
|
88
|
+
pos_idx = elig.index[:n_pos]
|
|
89
|
+
y.loc[pos_idx] = 1
|
|
90
|
+
|
|
91
|
+
remaining = sub.drop(index=pos_idx, errors="ignore")
|
|
92
|
+
neg_idx = remaining["composite_score"].nsmallest(initial_negatives).index
|
|
93
|
+
y.loc[neg_idx] = -1
|
|
94
|
+
|
|
95
|
+
labeled_idx = y[y != 0].index
|
|
96
|
+
X_tr = X[labeled_idx]
|
|
97
|
+
y_tr = y.loc[labeled_idx].values
|
|
98
|
+
|
|
99
|
+
# ---------- Train bagged RF ----------
|
|
100
|
+
scaler = StandardScaler().fit(X_tr)
|
|
101
|
+
X_tr_std = scaler.transform(X_tr)
|
|
102
|
+
|
|
103
|
+
base = RandomForestClassifier(n_estimators=100, random_state=random_state)
|
|
104
|
+
clf = BaggingClassifier(estimator=base, n_estimators=100, random_state=random_state)
|
|
105
|
+
clf.fit(X_tr_std, y_tr)
|
|
106
|
+
|
|
107
|
+
X_std = scaler.transform(X)
|
|
108
|
+
df["predicted_probability"] = clf.predict_proba(X_std)[:, 1]
|
|
109
|
+
|
|
110
|
+
# ---------- Bait-specific decoy shuffles for FDR ----------
|
|
111
|
+
decoys = []
|
|
112
|
+
for i, bait in enumerate(df["Bait"].unique()):
|
|
113
|
+
rng_i = np.random.RandomState(random_state + i)
|
|
114
|
+
sub = df[df["Bait"] == bait].copy()
|
|
115
|
+
dec = sub[feature_columns].apply(lambda col: rng_i.permutation(col.values))
|
|
116
|
+
X_dec = scaler.transform(dec.values)
|
|
117
|
+
decoys.append(clf.predict_proba(X_dec)[:, 1])
|
|
118
|
+
decoy_probs = np.concatenate(decoys) if len(decoys) else np.array([])
|
|
119
|
+
|
|
120
|
+
real_probs = df["predicted_probability"].values
|
|
121
|
+
unique_p = np.unique(real_probs)
|
|
122
|
+
|
|
123
|
+
# raw FDR
|
|
124
|
+
raw_fdr = {}
|
|
125
|
+
for p in unique_p:
|
|
126
|
+
n_real = np.sum(real_probs >= p)
|
|
127
|
+
n_dec = np.sum(decoy_probs >= p) if decoy_probs.size else 0
|
|
128
|
+
raw_fdr[p] = min(n_dec / n_real if n_real > 0 else 1.0, 1.0)
|
|
129
|
+
|
|
130
|
+
# monotone FDR (non-increasing with prob)
|
|
131
|
+
sorted_p = np.sort(unique_p)
|
|
132
|
+
mono_fdr = {sorted_p[0]: raw_fdr[sorted_p[0]]}
|
|
133
|
+
for i in range(1, len(sorted_p)):
|
|
134
|
+
p = sorted_p[i]
|
|
135
|
+
prev = sorted_p[i - 1]
|
|
136
|
+
mono_fdr[p] = min(raw_fdr[p], mono_fdr[prev])
|
|
137
|
+
|
|
138
|
+
df["FDR"] = df["predicted_probability"].map(mono_fdr)
|
|
139
|
+
|
|
140
|
+
# ---------- Background flag by global CV (optional) ----------
|
|
141
|
+
if "global_cv" in df.columns:
|
|
142
|
+
cv_thresh = np.nanpercentile(df["global_cv"], 25)
|
|
143
|
+
df["global_cv_flag"] = df["global_cv"].apply(
|
|
144
|
+
lambda v: "likely background" if pd.notna(v) and v <= cv_thresh else ""
|
|
145
|
+
)
|
|
146
|
+
else:
|
|
147
|
+
df["global_cv_flag"] = ""
|
|
148
|
+
|
|
149
|
+
# ===== AGGREGATE TO PROTEIN-LEVEL (per bait) =====
|
|
150
|
+
prob_agg = "max" if aggregate_strategy.lower() == "max" else "mean"
|
|
151
|
+
|
|
152
|
+
grp = df.groupby(["Bait", "Protein"])
|
|
153
|
+
protein_df = grp.agg(
|
|
154
|
+
predicted_probability=("predicted_probability", prob_agg),
|
|
155
|
+
FDR=("FDR", "min"),
|
|
156
|
+
n_fragments=("Fragment", "count"),
|
|
157
|
+
n_background=("global_cv_flag", lambda x: (x == "likely background").sum()),
|
|
158
|
+
mean_cv=("global_cv", "mean"),
|
|
159
|
+
).reset_index()
|
|
160
|
+
|
|
161
|
+
protein_df["background_flag_protein"] = np.where(
|
|
162
|
+
protein_df["n_background"] >= 0.5 * protein_df["n_fragments"],
|
|
163
|
+
"likely background",
|
|
164
|
+
"",
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
# ----- Save both -----
|
|
168
|
+
os.makedirs(save_dir, exist_ok=True)
|
|
169
|
+
fragment_path = os.path.join(save_dir, fragment_out)
|
|
170
|
+
protein_path = os.path.join(save_dir, protein_out)
|
|
171
|
+
|
|
172
|
+
df.to_csv(fragment_path, index=False)
|
|
173
|
+
protein_df.to_csv(protein_path, index=False)
|
|
174
|
+
|
|
175
|
+
return protein_df
|