whisper-ppi 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,125 @@
1
+ Metadata-Version: 2.4
2
+ Name: whisper-ppi
3
+ Version: 0.1.0
4
+ Summary: Weak Heuristic Inference for Supervisory Protein intERaction mapping for PDB and AP-MS datasets
5
+ Home-page: https://github.com/camlab-bioml/whisper
6
+ Author: Vesal Kasmaeifar
7
+ Author-email: vesal.kasmaeifar@mail.utoronto.com
8
+ License: MIT
9
+ Project-URL: Documentation, https://whisper.readthedocs.io/en/latest/
10
+ Project-URL: Source, https://github.com/camlab-bioml/whisper
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Operating System :: OS Independent
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
16
+ Requires-Python: >=3.10
17
+ Description-Content-Type: text/markdown
18
+ Requires-Dist: numpy
19
+ Requires-Dist: pandas
20
+ Requires-Dist: scikit-learn
21
+ Requires-Dist: scipy
22
+
23
+ # whisper
24
+
25
+ `whisper` is a Python package for scoring protein–protein interactions from proximity labeling and affinity purification mass spectrometry datasets. It uses interpretable features, programmatic weak supervision, and decoy-based false discovery rate (FDR) estimation to identify high-confidence interactors.
26
+
27
+ ## Installation
28
+
29
+ ```bash
30
+ git clone https://github.com/camlab-bioml/whisper
31
+ cd whisper
32
+ pip install .
33
+ ```
34
+
35
+ ## Input Format
36
+
37
+ - A CSV file with:
38
+ - One column named `Protein`
39
+ - Other columns representing bait replicate intensities, named as `BAIT_1`, `BAIT_2`, etc.
40
+ - Control samples must be identifiable via substrings in their column names (e.g., `"EGFP"` or `"Empty"`).
41
+
42
+ ## Usage
43
+
44
+ ```python
45
+ #protein-level
46
+ from whisper.protein_features import feature_engineering_protein
47
+ from whisper.protein_train import train_and_score_protein
48
+ import pandas as pd
49
+
50
+
51
+ # Load intensity table
52
+ intensity_df = pd.read_csv("input_intensity_dataset.tsv", sep="\t")
53
+
54
+ controls = ['EGFP', 'Empty', 'NminiTurbo']
55
+
56
+ # Run feature engineering
57
+ features_df = feature_engineering_protein(intensity_df, controls)
58
+
59
+ # You can save the features to use in the next step with different settings without generating them again.
60
+ features_df = pd.read_csv("features.csv")
61
+
62
+
63
+ # Run scoring and FDR estimation
64
+ scored_df = train_and_score_protein(features_df, initial_positives=15, initial_negatives=200)
65
+
66
+
67
+ #peptide-level
68
+ from whisper.peptide_features import feature_engineering_peptide
69
+ from whisper.peptide_train import train_and_score_peptide
70
+ import pandas as pd
71
+
72
+
73
+ # Load intensity table
74
+ intensity_df = pd.read_csv("input_intensity_dataset.tsv", sep="\t")
75
+
76
+ controls = ['EGFP', 'Empty', 'NminiTurbo']
77
+
78
+ # Run feature engineering
79
+ features_df = feature_engineering_peptide(intensity_df, controls)
80
+
81
+ # features_df = pd.read_csv("features.csv")
82
+
83
+
84
+ # Run scoring and FDR estimation
85
+ scored_df = train_and_score_peptide(features_df, initial_positives=15, initial_negatives=200)
86
+
87
+
88
+ #fragment-level
89
+ from whisper.fragment_features import feature_engineering_fragment
90
+ from whisper.fragment_train import train_and_score_fragment
91
+ import pandas as pd
92
+
93
+
94
+ # Load intensity table
95
+ intensity_df = pd.read_csv("input_intensity_dataset.tsv", sep="\t")
96
+
97
+ controls = ['EGFP', 'Empty', 'NminiTurbo']
98
+
99
+ # Run feature engineering
100
+ features_df = feature_engineering_fragment(intensity_df, controls)
101
+
102
+ # features_df = pd.read_csv("features.csv")
103
+
104
+
105
+ # Run scoring and FDR estimation
106
+ scored_df = train_and_score_fragment(features_df, initial_positives=15, initial_negatives=200)
107
+ ```
108
+
109
+ ## Output
110
+
111
+ The final output includes:
112
+ - `predicted_probability`: Probability of each bait–prey interaction being real
113
+ - `FDR`: Estimated false discovery rate
114
+ - `global_cv_flag`: Flag for likely background preys based on variability across all samples
115
+
116
+ ## Tutorial
117
+
118
+ [Read the full documentation](https://whisper.readthedocs.io/en/latest/)
119
+
120
+
121
+ ## Citation
122
+
123
+ This software is authored by: Vesal Kasmaeifar, Kieran R Campbell
124
+
125
+ Lunenfeld-Tanenbaum Research Institute & University of Toronto
@@ -0,0 +1,103 @@
1
+ # whisper
2
+
3
+ `whisper` is a Python package for scoring protein–protein interactions from proximity labeling and affinity purification mass spectrometry datasets. It uses interpretable features, programmatic weak supervision, and decoy-based false discovery rate (FDR) estimation to identify high-confidence interactors.
4
+
5
+ ## Installation
6
+
7
+ ```bash
8
+ git clone https://github.com/camlab-bioml/whisper
9
+ cd whisper
10
+ pip install .
11
+ ```
12
+
13
+ ## Input Format
14
+
15
+ - A CSV file with:
16
+ - One column named `Protein`
17
+ - Other columns representing bait replicate intensities, named as `BAIT_1`, `BAIT_2`, etc.
18
+ - Control samples must be identifiable via substrings in their column names (e.g., `"EGFP"` or `"Empty"`).
19
+
20
+ ## Usage
21
+
22
+ ```python
23
+ #protein-level
24
+ from whisper.protein_features import feature_engineering_protein
25
+ from whisper.protein_train import train_and_score_protein
26
+ import pandas as pd
27
+
28
+
29
+ # Load intensity table
30
+ intensity_df = pd.read_csv("input_intensity_dataset.tsv", sep="\t")
31
+
32
+ controls = ['EGFP', 'Empty', 'NminiTurbo']
33
+
34
+ # Run feature engineering
35
+ features_df = feature_engineering_protein(intensity_df, controls)
36
+
37
+ # You can save the features to use in the next step with different settings without generating them again.
38
+ features_df = pd.read_csv("features.csv")
39
+
40
+
41
+ # Run scoring and FDR estimation
42
+ scored_df = train_and_score_protein(features_df, initial_positives=15, initial_negatives=200)
43
+
44
+
45
+ #peptide-level
46
+ from whisper.peptide_features import feature_engineering_peptide
47
+ from whisper.peptide_train import train_and_score_peptide
48
+ import pandas as pd
49
+
50
+
51
+ # Load intensity table
52
+ intensity_df = pd.read_csv("input_intensity_dataset.tsv", sep="\t")
53
+
54
+ controls = ['EGFP', 'Empty', 'NminiTurbo']
55
+
56
+ # Run feature engineering
57
+ features_df = feature_engineering_peptide(intensity_df, controls)
58
+
59
+ # features_df = pd.read_csv("features.csv")
60
+
61
+
62
+ # Run scoring and FDR estimation
63
+ scored_df = train_and_score_peptide(features_df, initial_positives=15, initial_negatives=200)
64
+
65
+
66
+ #fragment-level
67
+ from whisper.fragment_features import feature_engineering_fragment
68
+ from whisper.fragment_train import train_and_score_fragment
69
+ import pandas as pd
70
+
71
+
72
+ # Load intensity table
73
+ intensity_df = pd.read_csv("input_intensity_dataset.tsv", sep="\t")
74
+
75
+ controls = ['EGFP', 'Empty', 'NminiTurbo']
76
+
77
+ # Run feature engineering
78
+ features_df = feature_engineering_fragment(intensity_df, controls)
79
+
80
+ # features_df = pd.read_csv("features.csv")
81
+
82
+
83
+ # Run scoring and FDR estimation
84
+ scored_df = train_and_score_fragment(features_df, initial_positives=15, initial_negatives=200)
85
+ ```
86
+
87
+ ## Output
88
+
89
+ The final output includes:
90
+ - `predicted_probability`: Probability of each bait–prey interaction being real
91
+ - `FDR`: Estimated false discovery rate
92
+ - `global_cv_flag`: Flag for likely background preys based on variability across all samples
93
+
94
+ ## Tutorial
95
+
96
+ [Read the full documentation](https://whisper.readthedocs.io/en/latest/)
97
+
98
+
99
+ ## Citation
100
+
101
+ This software is authored by: Vesal Kasmaeifar, Kieran R Campbell
102
+
103
+ Lunenfeld-Tanenbaum Research Institute & University of Toronto
@@ -0,0 +1,3 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61", "wheel"]
3
+ build-backend = "setuptools.build_meta"
@@ -0,0 +1,40 @@
1
+ [metadata]
2
+ name = whisper-ppi
3
+ version = 0.1.0
4
+ author = Vesal Kasmaeifar
5
+ author_email = vesal.kasmaeifar@mail.utoronto.com
6
+ description = Weak Heuristic Inference for Supervisory Protein intERaction mapping for PDB and AP-MS datasets
7
+ long_description = file: README.md
8
+ long_description_content_type = text/markdown
9
+ url = https://github.com/camlab-bioml/whisper
10
+ project_urls =
11
+ Documentation = https://whisper.readthedocs.io/en/latest/
12
+ Source = https://github.com/camlab-bioml/whisper
13
+ license = MIT
14
+ license_files = LICENSE
15
+ classifiers =
16
+ Programming Language :: Python :: 3
17
+ License :: OSI Approved :: MIT License
18
+ Operating System :: OS Independent
19
+ Intended Audience :: Science/Research
20
+ Topic :: Scientific/Engineering :: Bio-Informatics
21
+
22
+ [options]
23
+ package_dir =
24
+ = .
25
+ packages = find:
26
+ python_requires = >=3.10
27
+ include_package_data = True
28
+ install_requires =
29
+ numpy
30
+ pandas
31
+ scikit-learn
32
+ scipy
33
+
34
+ [options.packages.find]
35
+ where = .
36
+
37
+ [egg_info]
38
+ tag_build =
39
+ tag_date = 0
40
+
@@ -0,0 +1,2 @@
1
+ from setuptools import setup
2
+ setup()
@@ -0,0 +1,14 @@
1
+ from .protein_features import feature_engineering_protein as feature_engineering_protein
2
+ from .protein_train import train_and_score_protein as train_and_score_protein
3
+
4
+ from .peptide_features import feature_engineering_peptide
5
+ from .peptide_train import train_and_score_peptide
6
+
7
+ from .fragment_features import feature_engineering_fragment
8
+ from .fragment_train import train_and_score_fragment
9
+
10
+ __all__ = [
11
+ "feature_engineering_protein", "train_and_score_protein",
12
+ "feature_engineering_peptide", "train_and_score_peptide",
13
+ "feature_engineering_fragment", "train_and_score_fragment",
14
+ ]
@@ -0,0 +1,143 @@
1
+ # whisper/fragment_features.py
2
+
3
+ import pandas as pd
4
+ import numpy as np
5
+ import re
6
+ from sklearn.preprocessing import StandardScaler
7
+ import warnings
8
+ warnings.filterwarnings("ignore")
9
+
10
+
11
+ def feature_engineering_fragment(intensity_df: pd.DataFrame, controls: list) -> pd.DataFrame:
12
+ """
13
+ Compute fragment-level features.
14
+ Mirrors protein/peptide feature engineering but operates on (Protein, Peptide, Fragment).
15
+
16
+ Parameters
17
+ ----------
18
+ intensity_df : pd.DataFrame
19
+ Fragment-level intensity matrix with columns:
20
+ ['Protein', 'Peptide', 'Fragment', <bait/replicate and control columns>]
21
+ controls : list
22
+ List of control identifiers (e.g., ["EGFP", "Empty", "NminiTurbo"])
23
+
24
+ Returns
25
+ -------
26
+ pd.DataFrame
27
+ Aggregated feature table per bait–(protein, peptide, fragment) with computed metrics.
28
+ """
29
+
30
+ # --- Identify columns ---
31
+ id_cols = ["Protein", "Peptide", "Fragment"]
32
+ control_columns = [c for c in intensity_df.columns if any(ctrl in c for ctrl in controls)]
33
+ sample_columns = [c for c in intensity_df.columns if c not in id_cols]
34
+ bait_like_columns = [c for c in sample_columns if c not in control_columns]
35
+ baits = sorted(list(set(col.split("_")[0] for col in bait_like_columns)))
36
+ intensity_columns = control_columns + [c for c in bait_like_columns if any(b in c for b in baits)]
37
+
38
+ # --- Global CV across ALL samples for each fragment (exclude BirA and bait proteins for map) ---
39
+ global_cv = {}
40
+ for _, row in intensity_df.iterrows():
41
+ prot = row["Protein"]
42
+ vals = row[intensity_columns].astype(float).values
43
+ mean_all = np.mean(vals)
44
+ sd_all = np.std(vals)
45
+ global_cv[(row["Protein"], row["Peptide"], row["Fragment"])] = sd_all / mean_all if mean_all > 0 else 0
46
+
47
+ all_bait_features = []
48
+
49
+ for bait in baits:
50
+ bait_columns = [c for c in intensity_df.columns if re.fullmatch(fr"{bait}_\d+", c)]
51
+ if len(bait_columns) == 0:
52
+ # skip baits without replicate columns that match the "<bait>_#" pattern
53
+ continue
54
+
55
+ # Exclude the bait's own protein & BirA
56
+ filtered_df = intensity_df[~intensity_df["Protein"].isin([bait, "birA"])].copy()
57
+
58
+ # --- Control summary stats ---
59
+ ctrl_vals = filtered_df[control_columns].astype(float).values
60
+ ctrl_means = np.mean(ctrl_vals, axis=1)
61
+ ctrl_sds = np.std(ctrl_vals, axis=1)
62
+
63
+ min_mean_ctrl = np.min(ctrl_means[ctrl_means > 0]) if np.any(ctrl_means > 0) else 1.0
64
+ min_sd_ctrl = np.min(ctrl_sds[ctrl_sds > 0]) if np.any(ctrl_sds > 0) else 1.0
65
+
66
+ features = []
67
+ for _, row in filtered_df.iterrows():
68
+ prey_prot = row["Protein"]
69
+ prey_pep = row["Peptide"]
70
+ prey_frag = row["Fragment"]
71
+
72
+ bait_int = row[bait_columns].astype(float).values
73
+ ctrl_int = row[control_columns].astype(float).values
74
+
75
+ mean_bait = np.mean(bait_int)
76
+ median_bait = np.median(bait_int)
77
+ sd_bait = np.std(bait_int)
78
+
79
+ mean_ctrl = np.mean(ctrl_int)
80
+ sd_ctrl = np.std(ctrl_int)
81
+ mean_ctrl = mean_ctrl if mean_ctrl > 0 else min_mean_ctrl
82
+ sd_ctrl = sd_ctrl if sd_ctrl > 0 else min_sd_ctrl
83
+
84
+ zero_count = np.sum(bait_int == 0)
85
+ fold_change = mean_bait / mean_ctrl
86
+ log_fc = np.log2(fold_change + 1e-5)
87
+ penalized_log_fc = log_fc / max(1, zero_count)
88
+ snr = mean_bait / sd_ctrl
89
+ penalized_snr = snr / max(1, zero_count)
90
+
91
+ replicate_fc_sd = np.std(bait_int / mean_ctrl)
92
+ bait_cv = sd_bait / mean_bait if mean_bait != 0 else 0
93
+ bait_ctrl_sd_ratio = sd_bait / sd_ctrl
94
+
95
+ nonzero_reps = int(np.sum(bait_int > 0))
96
+ reps_above_ctrl_med = int(np.sum(bait_int > np.median(ctrl_int)))
97
+ single_rep_flag = 1 if nonzero_reps == 1 else 0
98
+
99
+ features.append({
100
+ "Bait": bait,
101
+ "Protein": prey_prot,
102
+ "Peptide": prey_pep,
103
+ "Fragment": prey_frag,
104
+ "log_fold_change": penalized_log_fc,
105
+ "snr": penalized_snr,
106
+ "mean_diff": mean_bait - mean_ctrl,
107
+ "median_diff": median_bait - np.median(ctrl_int),
108
+ "replicate_fold_change_sd": replicate_fc_sd,
109
+ "bait_cv": bait_cv,
110
+ "bait_control_sd_ratio": bait_ctrl_sd_ratio,
111
+ "zero_or_neg_fc": 0 if penalized_log_fc <= 0 else 1,
112
+ "nonzero_reps": nonzero_reps,
113
+ "reps_above_ctrl_med": reps_above_ctrl_med,
114
+ "single_rep_flag": single_rep_flag,
115
+ })
116
+
117
+ bait_features = pd.DataFrame(features)
118
+
119
+ # --- Scale and composite score (consistent with protein/peptide) ---
120
+ scale_cols = [
121
+ "log_fold_change", "snr", "mean_diff", "median_diff",
122
+ "replicate_fold_change_sd", "bait_cv", "bait_control_sd_ratio",
123
+ "zero_or_neg_fc",
124
+ ]
125
+ scaler = StandardScaler()
126
+ scaled_df = pd.DataFrame(
127
+ scaler.fit_transform(bait_features[scale_cols]),
128
+ columns=scale_cols, index=bait_features.index,
129
+ )
130
+
131
+ bait_features["composite_score"] = scaled_df[
132
+ ["log_fold_change", "snr", "mean_diff", "median_diff"]
133
+ ].mean(axis=1)
134
+
135
+ bait_features["global_cv"] = bait_features.apply(
136
+ lambda r: global_cv.get((r["Protein"], r["Peptide"], r["Fragment"]), np.nan), axis=1
137
+ )
138
+
139
+ all_bait_features.append(bait_features.sort_values("composite_score", ascending=False))
140
+
141
+ aggregated_features_df = pd.concat(all_bait_features, ignore_index=True)
142
+ aggregated_features_df.to_csv("features_fragment.csv", index=False)
143
+ return aggregated_features_df
@@ -0,0 +1,175 @@
1
+ # whsiper/fragment_train.py
2
+
3
+ import os
4
+ import pandas as pd
5
+ import numpy as np
6
+ from sklearn.ensemble import RandomForestClassifier, BaggingClassifier
7
+ from sklearn.preprocessing import StandardScaler
8
+ from scipy.cluster.hierarchy import linkage, fcluster
9
+ import warnings
10
+ warnings.filterwarnings("ignore")
11
+
12
+
13
+ def train_and_score_fragment(
14
+ features_df: pd.DataFrame,
15
+ initial_positives: int = 15,
16
+ initial_negatives: int = 200,
17
+ random_state: int = 42,
18
+ save_dir: str = ".",
19
+ fragment_out: str = "whisper_fragment_scores.csv",
20
+ protein_out: str = "whisper_protein_scores_from_fragments.csv",
21
+ aggregate_strategy: str = "max", # "max" or "mean" for fragment->protein prob aggregation
22
+ ):
23
+ """
24
+ Train a model on FRAGMENT-level features, compute bait-specific decoy FDR,
25
+ and aggregate to PROTEIN-level scores per bait.
26
+
27
+ Expected columns in `features_df`:
28
+ - Bait, Protein, Peptide, Fragment
29
+ - composite_score, global_cv (optional), single_rep_flag (optional)
30
+ - Feature columns:
31
+ ['log_fold_change','snr','mean_diff','median_diff',
32
+ 'replicate_fold_change_sd','bait_cv','bait_control_sd_ratio','zero_or_neg_fc']
33
+
34
+ Saves:
35
+ - <save_dir>/<fragment_out>: fragment-level scores with FDR
36
+ - <save_dir>/<protein_out>: protein-level aggregation per bait
37
+
38
+ Returns:
39
+ (fragment_df, protein_df)
40
+ """
41
+ rng = np.random.RandomState(random_state)
42
+
43
+ # Stable order
44
+ df = (
45
+ features_df.copy()
46
+ .sort_values(["Bait", "Protein", "Peptide", "Fragment"])
47
+ .reset_index(drop=True)
48
+ )
49
+
50
+ feature_columns = [
51
+ "log_fold_change", "snr", "mean_diff", "median_diff",
52
+ "replicate_fold_change_sd", "bait_cv", "bait_control_sd_ratio",
53
+ "zero_or_neg_fc",
54
+ ]
55
+ X = df[feature_columns].values
56
+
57
+ # ---------- Cluster baits to identify "strong" set ----------
58
+ bait_top50_stds = {
59
+ b: df[df["Bait"] == b]["composite_score"].nlargest(50).std()
60
+ for b in df["Bait"].unique()
61
+ }
62
+ bait_names = np.array(list(bait_top50_stds.keys()))
63
+ bait_scores = np.array(list(bait_top50_stds.values()), dtype=float).reshape(-1, 1)
64
+
65
+ if len(bait_names) > 2:
66
+ Z = linkage(bait_scores, method="ward")
67
+ clusters = fcluster(Z, t=2, criterion="maxclust")
68
+ else:
69
+ clusters = np.ones(len(bait_names), dtype=int)
70
+
71
+ cluster_sizes = {c: int(np.sum(clusters == c)) for c in np.unique(clusters)}
72
+ cluster_means = {c: float(bait_scores[clusters == c].mean()) for c in np.unique(clusters)}
73
+ max_size = max(cluster_sizes.values())
74
+ cands = [c for c, n in cluster_sizes.items() if n == max_size]
75
+ strong_cluster_id = cands[0] if len(cands) == 1 else max(cands, key=lambda c: cluster_means[c])
76
+ strong_baits = [b for b, c in zip(bait_names, clusters) if c == strong_cluster_id]
77
+
78
+ # ---------- Pseudo-labels ----------
79
+ y = pd.Series(0, index=df.index) # 0=unlabeled, 1=pos, -1=neg
80
+ bait_pos_quota = {b: (initial_positives if b in strong_baits else 0) for b in df["Bait"].unique()}
81
+
82
+ for bait in df["Bait"].unique():
83
+ sub = df[df["Bait"] == bait].copy()
84
+ n_pos = bait_pos_quota[bait]
85
+ if n_pos > 0:
86
+ ranked = sub.sort_values("composite_score", ascending=False)
87
+ elig = ranked[ranked.get("single_rep_flag", 0) != 1] # exclude single-rep spikes if present
88
+ pos_idx = elig.index[:n_pos]
89
+ y.loc[pos_idx] = 1
90
+
91
+ remaining = sub.drop(index=pos_idx, errors="ignore")
92
+ neg_idx = remaining["composite_score"].nsmallest(initial_negatives).index
93
+ y.loc[neg_idx] = -1
94
+
95
+ labeled_idx = y[y != 0].index
96
+ X_tr = X[labeled_idx]
97
+ y_tr = y.loc[labeled_idx].values
98
+
99
+ # ---------- Train bagged RF ----------
100
+ scaler = StandardScaler().fit(X_tr)
101
+ X_tr_std = scaler.transform(X_tr)
102
+
103
+ base = RandomForestClassifier(n_estimators=100, random_state=random_state)
104
+ clf = BaggingClassifier(estimator=base, n_estimators=100, random_state=random_state)
105
+ clf.fit(X_tr_std, y_tr)
106
+
107
+ X_std = scaler.transform(X)
108
+ df["predicted_probability"] = clf.predict_proba(X_std)[:, 1]
109
+
110
+ # ---------- Bait-specific decoy shuffles for FDR ----------
111
+ decoys = []
112
+ for i, bait in enumerate(df["Bait"].unique()):
113
+ rng_i = np.random.RandomState(random_state + i)
114
+ sub = df[df["Bait"] == bait].copy()
115
+ dec = sub[feature_columns].apply(lambda col: rng_i.permutation(col.values))
116
+ X_dec = scaler.transform(dec.values)
117
+ decoys.append(clf.predict_proba(X_dec)[:, 1])
118
+ decoy_probs = np.concatenate(decoys) if len(decoys) else np.array([])
119
+
120
+ real_probs = df["predicted_probability"].values
121
+ unique_p = np.unique(real_probs)
122
+
123
+ # raw FDR
124
+ raw_fdr = {}
125
+ for p in unique_p:
126
+ n_real = np.sum(real_probs >= p)
127
+ n_dec = np.sum(decoy_probs >= p) if decoy_probs.size else 0
128
+ raw_fdr[p] = min(n_dec / n_real if n_real > 0 else 1.0, 1.0)
129
+
130
+ # monotone FDR (non-increasing with prob)
131
+ sorted_p = np.sort(unique_p)
132
+ mono_fdr = {sorted_p[0]: raw_fdr[sorted_p[0]]}
133
+ for i in range(1, len(sorted_p)):
134
+ p = sorted_p[i]
135
+ prev = sorted_p[i - 1]
136
+ mono_fdr[p] = min(raw_fdr[p], mono_fdr[prev])
137
+
138
+ df["FDR"] = df["predicted_probability"].map(mono_fdr)
139
+
140
+ # ---------- Background flag by global CV (optional) ----------
141
+ if "global_cv" in df.columns:
142
+ cv_thresh = np.nanpercentile(df["global_cv"], 25)
143
+ df["global_cv_flag"] = df["global_cv"].apply(
144
+ lambda v: "likely background" if pd.notna(v) and v <= cv_thresh else ""
145
+ )
146
+ else:
147
+ df["global_cv_flag"] = ""
148
+
149
+ # ===== AGGREGATE TO PROTEIN-LEVEL (per bait) =====
150
+ prob_agg = "max" if aggregate_strategy.lower() == "max" else "mean"
151
+
152
+ grp = df.groupby(["Bait", "Protein"])
153
+ protein_df = grp.agg(
154
+ predicted_probability=("predicted_probability", prob_agg),
155
+ FDR=("FDR", "min"),
156
+ n_fragments=("Fragment", "count"),
157
+ n_background=("global_cv_flag", lambda x: (x == "likely background").sum()),
158
+ mean_cv=("global_cv", "mean"),
159
+ ).reset_index()
160
+
161
+ protein_df["background_flag_protein"] = np.where(
162
+ protein_df["n_background"] >= 0.5 * protein_df["n_fragments"],
163
+ "likely background",
164
+ "",
165
+ )
166
+
167
+ # ----- Save both -----
168
+ os.makedirs(save_dir, exist_ok=True)
169
+ fragment_path = os.path.join(save_dir, fragment_out)
170
+ protein_path = os.path.join(save_dir, protein_out)
171
+
172
+ df.to_csv(fragment_path, index=False)
173
+ protein_df.to_csv(protein_path, index=False)
174
+
175
+ return protein_df