sphot 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sphot-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Hoeyoung Kim and Donghee Kim
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
sphot-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,177 @@
1
+ Metadata-Version: 2.4
2
+ Name: sphot
3
+ Version: 0.1.0
4
+ Summary: Phenotype-associated spatial biomarker discovery in spatial transcriptomics with spHOT
5
+ Author: Hoeyoung Kim, Donghee Kim
6
+ License: MIT License
7
+
8
+ Copyright (c) 2026 Hoeyoung Kim and Donghee Kim
9
+
10
+ Permission is hereby granted, free of charge, to any person obtaining a copy
11
+ of this software and associated documentation files (the "Software"), to deal
12
+ in the Software without restriction, including without limitation the rights
13
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
14
+ copies of the Software, and to permit persons to whom the Software is
15
+ furnished to do so, subject to the following conditions:
16
+
17
+ The above copyright notice and this permission notice shall be included in all
18
+ copies or substantial portions of the Software.
19
+
20
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
21
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
22
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
23
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
24
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
25
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
26
+ SOFTWARE.
27
+
28
+ Project-URL: Homepage, https://github.com/DHKim327/spHOT
29
+ Project-URL: Repository, https://github.com/DHKim327/spHOT
30
+ Keywords: spatial transcriptomics,multiple instance learning,spatial biomarker,bioinformatics
31
+ Classifier: License :: OSI Approved :: MIT License
32
+ Classifier: Programming Language :: Python :: 3
33
+ Classifier: Programming Language :: Python :: 3.10
34
+ Classifier: Operating System :: POSIX :: Linux
35
+ Requires-Python: >=3.10
36
+ Description-Content-Type: text/markdown
37
+ License-File: LICENSE
38
+ Requires-Dist: numpy>=1.23
39
+ Requires-Dist: pandas>=1.5
40
+ Requires-Dist: scipy>=1.10
41
+ Requires-Dist: scikit-learn>=1.2
42
+ Requires-Dist: scanpy>=1.9
43
+ Requires-Dist: anndata>=0.9
44
+ Requires-Dist: torch>=2.0
45
+ Requires-Dist: novae>=1.0.0
46
+ Requires-Dist: networkx>=3.0
47
+ Requires-Dist: matplotlib>=3.6
48
+ Requires-Dist: seaborn>=0.12
49
+ Requires-Dist: termcolor>=2.0
50
+ Requires-Dist: pynvml>=11.0
51
+ Dynamic: license-file
52
+
53
+ # spHOT
54
+ ### Phenotype-associated spatial biomarker discovery in spatial transcriptomics with spHOT
55
+ [![License: MIT](https://img.shields.io/badge/license-MIT-pink.svg)](https://github.com/DHKim327/spHOT/blob/main/LICENSE)
56
+ [![Python](https://img.shields.io/badge/python-3.10-pink.svg)](https://www.python.org/)
57
+
58
+
59
+ ## 🧬 Description
60
+ **🔥spHOT🔥** is a framework for localizing phenotype-associated spatial biomarkers from multi-sample, multi-patient spatial transcriptomics datasets with sample-level case/control labels. spHOT integrates **spatial foundation model embeddings**, a **hierarchical domain tree**, and a dual-branch **teacher–student multiple instance learning architecture** to convert sample-level phenotype labels into spatially coherent cell-level biomarker scores.
61
+
62
+ <img src="https://raw.githubusercontent.com/DHKim327/spHOT/main/docs/spHOT_Figure1_V3.png" width="1000px" align="center" />
63
+
64
+
65
+ ## ⚙️ Installation
66
+
67
+ **Option 1 — PyPI**
68
+ ```bash
69
+ pip install sphot
70
+ ```
71
+
72
+ **Option 2 — GitHub**
73
+ ```bash
74
+ pip install git+https://github.com/DHKim327/spHOT.git
75
+ ```
76
+
77
+ **Option 3 — conda yml**
78
+ ```bash
79
+ git clone https://github.com/DHKim327/spHOT.git
80
+ cd spHOT
81
+ conda env create -f environment/env_spHOT.yml
82
+ conda activate env_spHOT
83
+ pip install -e .
84
+ ```
85
+
86
+ ## 📥 Inputs
87
+
88
+ **AnnData directory :**
89
+ <br>Path to `.h5ad` files, saved per sample :
90
+ ```
91
+ adatas/
92
+ ├── sample1/adata.h5ad
93
+ └── sample2/adata.h5ad
94
+ ```
95
+ - `.obs` key should contain sample phenotype label
96
+
97
+ **Split information :**
98
+
99
+ The `splits.csv` file defines train/valid/test splits for each sample across multiple folds.
100
+
101
+ | SID | 0 | 1 | 2 | 3 | 4 |
102
+ |--------------|-------|-------|-------|-------|-------|
103
+ | sample_1 | train | train | train | test | valid |
104
+ | sample_2 | train | test | valid | train | train |
105
+ | sample_3 | valid | train | train | train | test |
106
+
107
+ - **SID :** Sample identifier (e.g., slide or tissue ID)
108
+ - **0–4 :** Fold indices
109
+ - Each cell indicates the dataset role of that sample for a specific fold.
110
+
111
+
112
+ ## 📤 Outputs
113
+
114
+ After an spHOT run, the output directory structure will be organized as below :
115
+ ```
116
+ results/
117
+ └── {task_name}/ # Name of run
118
+ ├── de/ # Domain embedding module result
119
+ │ └── adata.h5ad # Merged AnnData with Novae embeddings & initial domain info
120
+ ├── dt/ # Domain tree module result
121
+ | ├── centroid_HC_results.pkl # Hierarchical domain tree
122
+ | └── adata.h5ad # Merged AnnData with metadomains association info
123
+ └── mil/ # MIL module result
124
+ ├── D{k}/ # k-level MIL result
125
+ │ ├── model_encoder_exp{exp}.pt # Model and outputs for each fold
126
+ │ ├── model_teacher_exp{exp}.pt
127
+ │ ├── model_student_exp{exp}.pt
128
+ │ ├── resource_{exp}.csv # Resource performance log
129
+ │ └── CELL_SCORE_test_{exp}.h5ad # spHOT cell scores (test) for each fold
130
+ ├── all_test_results.csv # Sample classification performance of test samples
131
+ ├── all_test_results.csv # Sample classification performance of validation samples
132
+ └── k_selection_results.csv # Spatial scores and k-selection result of spHOT run
133
+ ```
134
+
135
+ ## 📘 Tutorials
136
+
137
+ We provide the code and resources required to reproduce all main figures in the manuscript within the `./tutorial` directory. Also, see the `./tutorial` directory for step-by-step spHOT usage.
138
+
139
+ ### Figure-to-notebook map
140
+
141
+ | Figure | Dataset | Step | Notebook |
142
+ |---|---|---|---|
143
+ | **Fig. 2+@** | Simulation | Source data preparation | [`0.preproc_CosMx_Pouch_IBD.ipynb`](./tutorial/Fig2_Simulation/0.preproc_CosMx_Pouch_IBD.ipynb) |
144
+ | | | Simulation generation (V1) | [`1.simul_scCube_V1.ipynb`](./tutorial/Fig2_Simulation/1.simul_scCube_V1.ipynb) |
145
+ | | | Simulation generation (V2) | [`1.simul_scCube_V2.ipynb`](./tutorial/Fig2_Simulation/1.simul_scCube_V2.ipynb) |
146
+ | | | spHOT run (V1) | [`2.spHOT_simul_V1.ipynb`](./tutorial/Fig2_Simulation/2.spHOT_simul_V1.ipynb) |
147
+ | | | spHOT run (V2) | [`2.spHOT_simul_V2.ipynb`](./tutorial/Fig2_Simulation/2.spHOT_simul_V2.ipynb) |
148
+ | | | Evaluation (V1) | [`3.eval_simul_V1.ipynb`](./tutorial/Fig2_Simulation/3.eval_simul_V1.ipynb) |
149
+ | | | Evaluation (V2) | [`3.eval_simul_V2.ipynb`](./tutorial/Fig2_Simulation/3.eval_simul_V2.ipynb) |
150
+ | **Fig. 3–4** | Xenium IPF (GSE250346) | Preprocessing | [`0.preproc_Xenium_IPF.R`](./tutorial/Fig3_4_Xenium_IPF/0.preproc_Xenium_IPF.R) · [`0.preproc_Xenium_IPF.ipynb`](./tutorial/Fig3_4_Xenium_IPF/0.preproc_Xenium_IPF.ipynb) |
151
+ | | | spHOT run | [`1.spHOT_Xenium_IPF.ipynb`](./tutorial/Fig3_4_Xenium_IPF/1.spHOT_Xenium_IPF.ipynb) |
152
+ | | | Evaluation | [`2.eval_Xenium_IPF.ipynb`](./tutorial/Fig3_4_Xenium_IPF/2.eval_Xenium_IPF.ipynb) |
153
+ | | | Downstream analysis | [`3.downstream_Xenium_IPF.ipynb`](./tutorial/Fig3_4_Xenium_IPF/3.downstream_Xenium_IPF.ipynb) |
154
+ | **Fig. 5** | CosMx DKD | Preprocessing | [`0.preproc_CosMx_DKD.ipynb`](./tutorial/Fig5_CosMx_DKD/0.preproc_CosMx_DKD.ipynb) |
155
+ | | | spHOT run | [`1.spHOT_CosMx_DKD.ipynb`](./tutorial/Fig5_CosMx_DKD/1.spHOT_CosMx_DKD.ipynb) |
156
+ | | | Evaluation | [`2.eval_CosMx_DKD.ipynb`](./tutorial/Fig5_CosMx_DKD/2.eval_CosMx_DKD.ipynb) |
157
+ | | | Downstream analysis | [`3.downstream_CosMx_DKD.ipynb`](./tutorial/Fig5_CosMx_DKD/3.downstream_CosMx_DKD.ipynb) |
158
+ | **Fig. 6** | Xenium RPGN (GSE294965) | Preprocessing | [`0.preproc_Xenium_RPGN.ipynb`](./tutorial/Fig6_Xenium_RPGN/0.preproc_Xenium_RPGN.ipynb) |
159
+ | | | spHOT run | [`1.spHOT_Xenium_RPGN.ipynb`](./tutorial/Fig6_Xenium_RPGN/1.spHOT_Xenium_RPGN.ipynb) |
160
+ | | | Downstream analysis (1) | [`2.downstream_Xenium_RPGN_V1.ipynb`](./tutorial/Fig6_Xenium_RPGN/2.downstream_Xenium_RPGN_V1.ipynb) |
161
+ | | | Downstream analysis (2) | [`2.downstream_Xenium_RPGN_V2.ipynb`](./tutorial/Fig6_Xenium_RPGN/2.downstream_Xenium_RPGN_V2.ipynb) |
162
+ | **Fig. 7** | Xenium COPD (GSE313006) | Preprocessing & cell typing | [`0.preproc_Xenium_COPD.ipynb`](./tutorial/Fig7_Xenium_COPD/0.preproc_Xenium_COPD.ipynb) |
163
+ | | | Downstream analysis (1) | [`1.downstream_Xenium_COPD_V1.ipynb`](./tutorial/Fig7_Xenium_COPD/1.downstream_Xenium_COPD_V1.ipynb) |
164
+ | | | Downstream analysis (2) | [`1.downstream_Xenium_COPD_V2.ipynb`](./tutorial/Fig7_Xenium_COPD/1.downstream_Xenium_COPD_V2.ipynb) |
165
+ | **Fig. 2, 3, 5+@** | All datasets | Cross-dataset benchmarking & statistics | [`benchmark_stats_all_datasets.ipynb`](./tutorial/benchmark_stats_all_datasets.ipynb) |
166
+
167
+ > **Note :** Notebooks are numbered by execution order within each figure directory (`0.` → `3.`). Preprocessing notebooks start from public GEO accessions; the simulation dataset is downloaded from Zenodo as the version of record (see [Data availability](https://doi.org/10.5281/zenodo.21156231)).
168
+
169
+ ## 😊 Acknowledgements
170
+ spHOT is built upon codes from [scMILD: Single-cell multiple instance learning for sample classification and associated subpopulation discovery](https://github.com/Khreat0205/scMILD). We thank the authors for publicly releasing their codes.
171
+
172
+
173
+ ## 📚 References
174
+ Jeong, K., Choi, J. & Kim, K. scMILD: Single-cell multiple instance learning for sample classification and associated subpopulation discovery. iScience 29(2026).
175
+
176
+
177
+ <br><br>
sphot-0.1.0/README.md ADDED
@@ -0,0 +1,125 @@
1
+ # spHOT
2
+ ### Phenotype-associated spatial biomarker discovery in spatial transcriptomics with spHOT
3
+ [![License: MIT](https://img.shields.io/badge/license-MIT-pink.svg)](https://github.com/DHKim327/spHOT/blob/main/LICENSE)
4
+ [![Python](https://img.shields.io/badge/python-3.10-pink.svg)](https://www.python.org/)
5
+
6
+
7
+ ## 🧬 Description
8
+ **🔥spHOT🔥** is a framework for localizing phenotype-associated spatial biomarkers from multi-sample, multi-patient spatial transcriptomics datasets with sample-level case/control labels. spHOT integrates **spatial foundation model embeddings**, a **hierarchical domain tree**, and a dual-branch **teacher–student multiple instance learning architecture** to convert sample-level phenotype labels into spatially coherent cell-level biomarker scores.
9
+
10
+ <img src="https://raw.githubusercontent.com/DHKim327/spHOT/main/docs/spHOT_Figure1_V3.png" width="1000px" align="center" />
11
+
12
+
13
+ ## ⚙️ Installation
14
+
15
+ **Option 1 — PyPI**
16
+ ```bash
17
+ pip install sphot
18
+ ```
19
+
20
+ **Option 2 — GitHub**
21
+ ```bash
22
+ pip install git+https://github.com/DHKim327/spHOT.git
23
+ ```
24
+
25
+ **Option 3 — conda yml**
26
+ ```bash
27
+ git clone https://github.com/DHKim327/spHOT.git
28
+ cd spHOT
29
+ conda env create -f environment/env_spHOT.yml
30
+ conda activate env_spHOT
31
+ pip install -e .
32
+ ```
33
+
34
+ ## 📥 Inputs
35
+
36
+ **AnnData directory :**
37
+ <br>Path to `.h5ad` files, saved per sample :
38
+ ```
39
+ adatas/
40
+ ├── sample1/adata.h5ad
41
+ └── sample2/adata.h5ad
42
+ ```
43
+ - `.obs` key should contain sample phenotype label
44
+
45
+ **Split information :**
46
+
47
+ The `splits.csv` file defines train/valid/test splits for each sample across multiple folds.
48
+
49
+ | SID | 0 | 1 | 2 | 3 | 4 |
50
+ |--------------|-------|-------|-------|-------|-------|
51
+ | sample_1 | train | train | train | test | valid |
52
+ | sample_2 | train | test | valid | train | train |
53
+ | sample_3 | valid | train | train | train | test |
54
+
55
+ - **SID :** Sample identifier (e.g., slide or tissue ID)
56
+ - **0–4 :** Fold indices
57
+ - Each cell indicates the dataset role of that sample for a specific fold.
58
+
59
+
60
+ ## 📤 Outputs
61
+
62
+ After an spHOT run, the output directory structure will be organized as below :
63
+ ```
64
+ results/
65
+ └── {task_name}/ # Name of run
66
+ ├── de/ # Domain embedding module result
67
+ │ └── adata.h5ad # Merged AnnData with Novae embeddings & initial domain info
68
+ ├── dt/ # Domain tree module result
69
+ | ├── centroid_HC_results.pkl # Hierarchical domain tree
70
+ | └── adata.h5ad # Merged AnnData with metadomains association info
71
+ └── mil/ # MIL module result
72
+ ├── D{k}/ # k-level MIL result
73
+ │ ├── model_encoder_exp{exp}.pt # Model and outputs for each fold
74
+ │ ├── model_teacher_exp{exp}.pt
75
+ │ ├── model_student_exp{exp}.pt
76
+ │ ├── resource_{exp}.csv # Resource performance log
77
+ │ └── CELL_SCORE_test_{exp}.h5ad # spHOT cell scores (test) for each fold
78
+ ├── all_test_results.csv # Sample classification performance of test samples
79
+ ├── all_test_results.csv # Sample classification performance of validation samples
80
+ └── k_selection_results.csv # Spatial scores and k-selection result of spHOT run
81
+ ```
82
+
83
+ ## 📘 Tutorials
84
+
85
+ We provide the code and resources required to reproduce all main figures in the manuscript within the `./tutorial` directory. Also, see the `./tutorial` directory for step-by-step spHOT usage.
86
+
87
+ ### Figure-to-notebook map
88
+
89
+ | Figure | Dataset | Step | Notebook |
90
+ |---|---|---|---|
91
+ | **Fig. 2+@** | Simulation | Source data preparation | [`0.preproc_CosMx_Pouch_IBD.ipynb`](./tutorial/Fig2_Simulation/0.preproc_CosMx_Pouch_IBD.ipynb) |
92
+ | | | Simulation generation (V1) | [`1.simul_scCube_V1.ipynb`](./tutorial/Fig2_Simulation/1.simul_scCube_V1.ipynb) |
93
+ | | | Simulation generation (V2) | [`1.simul_scCube_V2.ipynb`](./tutorial/Fig2_Simulation/1.simul_scCube_V2.ipynb) |
94
+ | | | spHOT run (V1) | [`2.spHOT_simul_V1.ipynb`](./tutorial/Fig2_Simulation/2.spHOT_simul_V1.ipynb) |
95
+ | | | spHOT run (V2) | [`2.spHOT_simul_V2.ipynb`](./tutorial/Fig2_Simulation/2.spHOT_simul_V2.ipynb) |
96
+ | | | Evaluation (V1) | [`3.eval_simul_V1.ipynb`](./tutorial/Fig2_Simulation/3.eval_simul_V1.ipynb) |
97
+ | | | Evaluation (V2) | [`3.eval_simul_V2.ipynb`](./tutorial/Fig2_Simulation/3.eval_simul_V2.ipynb) |
98
+ | **Fig. 3–4** | Xenium IPF (GSE250346) | Preprocessing | [`0.preproc_Xenium_IPF.R`](./tutorial/Fig3_4_Xenium_IPF/0.preproc_Xenium_IPF.R) · [`0.preproc_Xenium_IPF.ipynb`](./tutorial/Fig3_4_Xenium_IPF/0.preproc_Xenium_IPF.ipynb) |
99
+ | | | spHOT run | [`1.spHOT_Xenium_IPF.ipynb`](./tutorial/Fig3_4_Xenium_IPF/1.spHOT_Xenium_IPF.ipynb) |
100
+ | | | Evaluation | [`2.eval_Xenium_IPF.ipynb`](./tutorial/Fig3_4_Xenium_IPF/2.eval_Xenium_IPF.ipynb) |
101
+ | | | Downstream analysis | [`3.downstream_Xenium_IPF.ipynb`](./tutorial/Fig3_4_Xenium_IPF/3.downstream_Xenium_IPF.ipynb) |
102
+ | **Fig. 5** | CosMx DKD | Preprocessing | [`0.preproc_CosMx_DKD.ipynb`](./tutorial/Fig5_CosMx_DKD/0.preproc_CosMx_DKD.ipynb) |
103
+ | | | spHOT run | [`1.spHOT_CosMx_DKD.ipynb`](./tutorial/Fig5_CosMx_DKD/1.spHOT_CosMx_DKD.ipynb) |
104
+ | | | Evaluation | [`2.eval_CosMx_DKD.ipynb`](./tutorial/Fig5_CosMx_DKD/2.eval_CosMx_DKD.ipynb) |
105
+ | | | Downstream analysis | [`3.downstream_CosMx_DKD.ipynb`](./tutorial/Fig5_CosMx_DKD/3.downstream_CosMx_DKD.ipynb) |
106
+ | **Fig. 6** | Xenium RPGN (GSE294965) | Preprocessing | [`0.preproc_Xenium_RPGN.ipynb`](./tutorial/Fig6_Xenium_RPGN/0.preproc_Xenium_RPGN.ipynb) |
107
+ | | | spHOT run | [`1.spHOT_Xenium_RPGN.ipynb`](./tutorial/Fig6_Xenium_RPGN/1.spHOT_Xenium_RPGN.ipynb) |
108
+ | | | Downstream analysis (1) | [`2.downstream_Xenium_RPGN_V1.ipynb`](./tutorial/Fig6_Xenium_RPGN/2.downstream_Xenium_RPGN_V1.ipynb) |
109
+ | | | Downstream analysis (2) | [`2.downstream_Xenium_RPGN_V2.ipynb`](./tutorial/Fig6_Xenium_RPGN/2.downstream_Xenium_RPGN_V2.ipynb) |
110
+ | **Fig. 7** | Xenium COPD (GSE313006) | Preprocessing & cell typing | [`0.preproc_Xenium_COPD.ipynb`](./tutorial/Fig7_Xenium_COPD/0.preproc_Xenium_COPD.ipynb) |
111
+ | | | Downstream analysis (1) | [`1.downstream_Xenium_COPD_V1.ipynb`](./tutorial/Fig7_Xenium_COPD/1.downstream_Xenium_COPD_V1.ipynb) |
112
+ | | | Downstream analysis (2) | [`1.downstream_Xenium_COPD_V2.ipynb`](./tutorial/Fig7_Xenium_COPD/1.downstream_Xenium_COPD_V2.ipynb) |
113
+ | **Fig. 2, 3, 5+@** | All datasets | Cross-dataset benchmarking & statistics | [`benchmark_stats_all_datasets.ipynb`](./tutorial/benchmark_stats_all_datasets.ipynb) |
114
+
115
+ > **Note :** Notebooks are numbered by execution order within each figure directory (`0.` → `3.`). Preprocessing notebooks start from public GEO accessions; the simulation dataset is downloaded from Zenodo as the version of record (see [Data availability](https://doi.org/10.5281/zenodo.21156231)).
116
+
117
+ ## 😊 Acknowledgements
118
+ spHOT is built upon codes from [scMILD: Single-cell multiple instance learning for sample classification and associated subpopulation discovery](https://github.com/Khreat0205/scMILD). We thank the authors for publicly releasing their codes.
119
+
120
+
121
+ ## 📚 References
122
+ Jeong, K., Choi, J. & Kim, K. scMILD: Single-cell multiple instance learning for sample classification and associated subpopulation discovery. iScience 29(2026).
123
+
124
+
125
+ <br><br>
@@ -0,0 +1,49 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "sphot"
7
+ version = "0.1.0"
8
+ description = "Phenotype-associated spatial biomarker discovery in spatial transcriptomics with spHOT"
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = { file = "LICENSE" }
12
+ authors = [
13
+ { name = "Hoeyoung Kim" },
14
+ { name = "Donghee Kim" },
15
+ ]
16
+ keywords = [
17
+ "spatial transcriptomics",
18
+ "multiple instance learning",
19
+ "spatial biomarker",
20
+ "bioinformatics",
21
+ ]
22
+ classifiers = [
23
+ "License :: OSI Approved :: MIT License",
24
+ "Programming Language :: Python :: 3",
25
+ "Programming Language :: Python :: 3.10",
26
+ "Operating System :: POSIX :: Linux",
27
+ ]
28
+ dependencies = [
29
+ "numpy>=1.23",
30
+ "pandas>=1.5",
31
+ "scipy>=1.10",
32
+ "scikit-learn>=1.2",
33
+ "scanpy>=1.9",
34
+ "anndata>=0.9",
35
+ "torch>=2.0",
36
+ "novae>=1.0.0",
37
+ "networkx>=3.0",
38
+ "matplotlib>=3.6",
39
+ "seaborn>=0.12",
40
+ "termcolor>=2.0",
41
+ "pynvml>=11.0",
42
+ ]
43
+
44
+ [project.urls]
45
+ Homepage = "https://github.com/DHKim327/spHOT"
46
+ Repository = "https://github.com/DHKim327/spHOT"
47
+
48
+ [tool.setuptools.packages.find]
49
+ include = ["sphot*"]
sphot-0.1.0/setup.cfg ADDED
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
File without changes
@@ -0,0 +1 @@
1
+ from .base import DomainEmbedder
@@ -0,0 +1,88 @@
1
+ import os
2
+ from pathlib import Path
3
+ from ..utils import load_adatas, set_seed
4
+
5
+ class DomainEmbedder:
6
+ def __init__(self, task_name,cfgs):
7
+ self.task_name = task_name
8
+ self.cfgs = cfgs
9
+ self.model_name = cfgs.get('model') # e.g., 'Novae'
10
+ self.adata_dir = cfgs.get('adata_dir')
11
+
12
+ self.save_dir = Path(cfgs.get('save_dir'))/ self.task_name / 'de'
13
+ self.save_dir.mkdir(parents=True, exist_ok=True)
14
+
15
+ # Set GPU device (applies to all embedders)
16
+ params = cfgs.get('params', {})
17
+ resources = params.get('resources', {})
18
+ gpu_id = resources.get('gpu_id', None)
19
+ if gpu_id is not None:
20
+ self._set_gpu_device(gpu_id)
21
+
22
+ # Set random seed for reproducibility (applies to all embedders)
23
+ self.random_seed = params.get('random_seed', 42)
24
+ set_seed(self.random_seed, deterministic=True)
25
+ print(f"[DomainEmbedder] Random seed set to {self.random_seed} for {self.model_name}")
26
+
27
+ def _set_gpu_device(self, gpu_id):
28
+ """
29
+ Set GPU device for all frameworks (PyTorch, TensorFlow, JAX, etc.)
30
+
31
+ Args:
32
+ gpu_id (int or str): GPU device ID (e.g., 0, 1, 2, or "0,1" for multiple GPUs)
33
+ """
34
+ gpu_id_str = str(gpu_id)
35
+
36
+ # 1. Set CUDA_VISIBLE_DEVICES (works for all frameworks)
37
+ os.environ['CUDA_VISIBLE_DEVICES'] = gpu_id_str
38
+ print(f"[DomainEmbedder] CUDA_VISIBLE_DEVICES set to: {gpu_id_str}")
39
+
40
+
41
+ def model_selection(self):
42
+ if self.model_name == 'Novae':
43
+ from .embedder_novae import NOVAEmbedder
44
+ return NOVAEmbedder
45
+ else:
46
+ raise KeyError("Invalid model name for DomainEmbedding: expected 'Novae'")
47
+
48
+ def run(self, adata_func=load_adatas):
49
+ embedder_cls = self.model_selection()
50
+ embedder = embedder_cls(self.cfgs.get('params'),self.save_dir)
51
+
52
+ adatas = adata_func(self.adata_dir)
53
+ # 각 embedder는 run(adatas) 수행 후 내부 상태에 결과가 있고,
54
+ # get_embed()로 {'obsm_key','obs_key','adata_integrated'} dict를 반환한다고 가정
55
+ embedder._run(adatas)
56
+ result = embedder._get_result()
57
+
58
+ adata = self._postprocessing(result)
59
+ adata.write_h5ad(self.save_dir / f'adata.h5ad',compression='gzip')
60
+
61
+
62
+ def _postprocessing(self, result):
63
+ """
64
+ result:
65
+ - 'adata_integrated': AnnData (통합)
66
+ - 'obsm_key': latent embedding key (ex: 'novae_latent')
67
+ - (optional) 'obs_key': cluster key (ex: 'novae_leaves')
68
+ """
69
+ adata = result['adata_integrated']
70
+ obsm_key = result['obsm_key']
71
+ adata = self._obsm_process(adata, obsm_key)
72
+
73
+ if 'obs_key' in result:
74
+ obs_key = result['obs_key']
75
+ adata = self._obs_process(adata, obs_key)
76
+
77
+ return adata
78
+
79
+ def _obsm_process(self, adata, obsm_key):
80
+ adata.obsm[f'X_{self.model_name}'] = adata.obsm[obsm_key]
81
+ return adata
82
+
83
+ def _obs_process(self, adata, obs_key):
84
+ # 클러스터 라벨을 0..C-1 정수로 맵핑해서 'Cluster_{model_name}'에 저장
85
+ cats = adata.obs[obs_key].astype('category').cat.categories.tolist()
86
+ mapping = {c: i for i, c in enumerate(sorted(cats))}
87
+ adata.obs[f'Cluster_{self.model_name}'] = adata.obs[obs_key].map(mapping)
88
+ return adata
@@ -0,0 +1 @@
1
+ from .embedder import NOVAEmbedder
@@ -0,0 +1,184 @@
1
+ import os
2
+ from pathlib import Path
3
+ import novae
4
+ import scanpy as sc
5
+
6
+ class NOVAEmbedder():
7
+ """
8
+ NOVAE-based Domain Embedding (Official Hugging Face Method)
9
+
10
+ Official usage:
11
+ https://huggingface.co/MICS-Lab/novae-human-0
12
+
13
+ Supports:
14
+ - Spatial Transcriptomics (pretrained & finetune modes)
15
+ - Spatial Proteomics (pretrained mode only)
16
+
17
+ Modes:
18
+ - 'pretrained': Zero-shot inference using pretrained model
19
+ - 'finetune': Fine-tune on user data then compute representations
20
+
21
+ Important: NOVAE creates spatial neighbor graphs (required for MIL k optimization)
22
+
23
+ Input: configuration files, adatas:list at runtime
24
+ Output: {
25
+ 'obsm_key': 'novae_latent',
26
+ 'obs_key': 'novae_leaves',
27
+ 'adata_integrated': concatenated AnnData
28
+ }
29
+ """
30
+ def __init__(self, params,save_dir):
31
+ self.params = params
32
+ self.save_dir = save_dir
33
+ self._setup()
34
+
35
+ def _setup(self):
36
+ """Initialize parameters and download model if needed"""
37
+ self.data_type = self.params.get('data_type', 'transcriptomics')
38
+ self.mode = self.params.get('mode', 'pretrained')
39
+
40
+ # Validate mode
41
+ if self.mode not in ['pretrained', 'finetune']:
42
+ raise ValueError(
43
+ f"\n{'='*80}\n"
44
+ f"Invalid mode: '{self.mode}'\n"
45
+ f"{'='*80}\n"
46
+ f"NOVAE supports:\n"
47
+ f" - 'pretrained' (zero-shot inference)\n"
48
+ f" - 'finetune' (fine-tune then compute representations)\n"
49
+ f"\nPlease check your configuration file.\n"
50
+ f"{'='*80}\n"
51
+ )
52
+
53
+ # NOVAE supports transcriptomics (and proteomics for pretrained mode only)
54
+ if self.data_type not in ['transcriptomics', 'proteomics']:
55
+ raise ValueError(
56
+ f"\n{'='*80}\n"
57
+ f"Invalid data_type: '{self.data_type}'\n"
58
+ f"{'='*80}\n"
59
+ f"NOVAE supports:\n"
60
+ f" - 'transcriptomics' (pretrained & finetune modes)\n"
61
+ f" - 'proteomics' (pretrained mode only)\n"
62
+ f"\nPlease check your configuration file.\n"
63
+ f"{'='*80}\n"
64
+ )
65
+
66
+ # Resources
67
+ self.resources = self.params.get('resources', {})
68
+
69
+ # Setup model directory
70
+ model_dir = self.params.get('model_dir')
71
+
72
+ if model_dir is None:
73
+ # Auto-download from Hugging Face Hub and save locally
74
+ model_dir = self.save_dir / 'novae_pretrained'
75
+ model_dir.mkdir(parents=True, exist_ok=True)
76
+
77
+ print(f"[NOVAE] Downloading from Hugging Face Hub (MICS-Lab/novae-human-0)...")
78
+ model = novae.Novae.from_pretrained("MICS-Lab/novae-human-0")
79
+ model.save_pretrained(model_dir.as_posix())
80
+ print(f"[NOVAE] Model saved to {model_dir}")
81
+
82
+ self.model_dir = model_dir
83
+ else:
84
+ # Use user-specified directory
85
+ self.model_dir = Path(model_dir)
86
+ print(f"[NOVAE] Using model from {self.model_dir}")
87
+
88
+ print(f"[NOVAE] Data type: {self.data_type}")
89
+ print(f"[NOVAE] Mode: {self.mode}")
90
+
91
+ def _inference(self):
92
+ """Zero-shot inference using pretrained model"""
93
+ accelerator = self.resources.get('accelerator', 'cpu')
94
+ num_workers = self.resources.get('num_workers', 4)
95
+
96
+ print(f"[NOVAE] Running zero-shot inference...")
97
+ print(f"[NOVAE] - accelerator: {accelerator}")
98
+ print(f"[NOVAE] - num_workers: {num_workers}")
99
+
100
+ model = novae.Novae.from_pretrained(self.model_dir.as_posix())
101
+ model.compute_representations(
102
+ self.adatas,
103
+ zero_shot=True,
104
+ accelerator=accelerator,
105
+ num_workers=num_workers,
106
+ )
107
+
108
+ print(f"[NOVAE] Inference completed")
109
+
110
+
111
+ def _finetune(self):
112
+ """Fine-tune pretrained model and compute representations"""
113
+ accelerator = self.resources.get('accelerator', 'cpu')
114
+ num_workers = self.resources.get('num_workers', 4)
115
+
116
+ fit_param = self.params.get('fit', {})
117
+
118
+ print(f"[NOVAE] Fine-tuning model...")
119
+ print(f"[NOVAE] - accelerator: {accelerator}")
120
+ print(f"[NOVAE] - num_workers: {num_workers}")
121
+ print(f"[NOVAE] - fit parameters: {fit_param}")
122
+
123
+ model = novae.Novae.from_pretrained(self.model_dir.as_posix())
124
+
125
+ # fine-tune
126
+ print(f"[NOVAE] Starting fine-tuning...")
127
+ model.fine_tune(
128
+ self.adatas,
129
+ accelerator=accelerator,
130
+ num_workers=num_workers,
131
+ **fit_param, # 여기에는 reference, max_epochs, patience, min_delta 등
132
+ )
133
+ print(f"[NOVAE] Fine-tuning completed")
134
+
135
+ # representation 계산
136
+ print(f"[NOVAE] Computing representations...")
137
+ model.compute_representations(
138
+ self.adatas,
139
+ accelerator=accelerator,
140
+ num_workers=num_workers,
141
+ )
142
+ print(f"[NOVAE] Representations computed")
143
+
144
+ if self.params.get('save', True):
145
+ save_path = self.save_dir / 'novae_finetuned'
146
+ model.save_pretrained(save_path.as_posix())
147
+ print(f"[NOVAE] Fine-tuned model saved to {save_path}")
148
+
149
+
150
+ def _construct_graph_per_adata(self):
151
+ """Build spatial neighbor graph for each AnnData (required for MIL k optimization)"""
152
+ spatial_params = self.params.get('spatial_neighbors', {})
153
+ radius = spatial_params.get('radius', 150) # Default radius for NOVAE
154
+
155
+ print(f"[NOVAE] Building spatial neighbor graphs...")
156
+ print(f"[NOVAE] - radius: {radius}")
157
+ novae.spatial_neighbors(self.adatas, radius=radius)
158
+ print(f"[NOVAE] Spatial graphs completed for {len(self.adatas)} samples")
159
+
160
+ def _get_result(self):
161
+ return {
162
+ 'obsm_key': 'novae_latent',
163
+ 'obs_key': 'novae_leaves',
164
+ 'adata_integrated': sc.concat(self.adatas)
165
+ }
166
+
167
+ def _run(self, adatas):
168
+ """Main execution function"""
169
+ self.adatas = adatas
170
+
171
+ print(f"[NOVAE] Starting domain embedding for {len(adatas)} samples")
172
+
173
+ # 1. Build spatial graphs (required for MIL k optimization)
174
+ self._construct_graph_per_adata()
175
+
176
+ # 2. Run model (pretrained or finetune)
177
+ if self.mode == 'finetune':
178
+ self._finetune()
179
+ elif self.mode == 'pretrained':
180
+ self._inference()
181
+ else:
182
+ raise ValueError(f"Invalid mode in NOVAEmbedder: expected 'finetune' or 'pretrained', got '{self.mode}'")
183
+
184
+ print(f"[NOVAE] Domain embedding completed")
@@ -0,0 +1 @@
1
+ from .base import MetaDomain