rectanglepy 1.4.2__tar.gz → 1.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.bumpversion.cfg +1 -1
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.readthedocs.yaml +1 -1
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/PKG-INFO +11 -3
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/README.md +9 -1
- rectanglepy-1.6.0/docs/_static/rectangle_workflow.png +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/api.md +1 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/index.md +1 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/references.bib +9 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/pyproject.toml +1 -1
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/__init__.py +2 -1
- rectanglepy-1.6.0/src/rectanglepy/parameters.py +11 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/pp/create_signature.py +75 -34
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/rectangle.py +8 -3
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/tl/deconvolution.py +217 -122
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/test_pp.py +51 -2
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/test_tl.py +1 -1
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.cruft.json +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.editorconfig +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/ISSUE_TEMPLATE/config.yml +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/workflows/build.yaml +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/workflows/release.yaml +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/workflows/release_testpypi.yaml +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/workflows/test.yaml +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.gitignore +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.pre-commit-config.yaml +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/CHANGELOG.md +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/CLA.md +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/LICENSE +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/LICENSE-Commercial.md +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/Makefile +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/_static/.gitkeep +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/_static/rec_logo.001.png +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/_templates/.gitkeep +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/_templates/autosummary/class.rst +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/changelog.md +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/conf.py +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/contributing.md +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/extensions/typed_returns.py +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/installation.md +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/make.bat +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/notebooks/example.ipynb +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/references.md +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/tutorials.md +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/data/hao1_annotations_small.zip +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/data/hao1_counts_small.zip +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/data/small_fino_bulks.zip +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/pp/__init__.py +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/pp/rectangle_signature.py +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/tl/__init__.py +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/data/TIL10_signature.txt +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/data/bulk_small.csv +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/data/cell_annotations_small.txt +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/data/quanTIseq_SimRNAseq_mixture_smaller.csv +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/data/quanTIseq_SimRNAseq_read_fractions_small.txt +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/data/sc_object_small.csv +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/data/signature_hao1.csv +0 -0
- {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/test_rectangle.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: rectanglepy
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.6.0
|
|
4
4
|
Summary: Hierarchical deconvolution of bulk transcriptomics
|
|
5
5
|
Project-URL: Documentation, https://rectanglepy.readthedocs.io/
|
|
6
6
|
Project-URL: Source, https://github.com/ComputationalBiomedicineGroup/Rectangle
|
|
@@ -72,6 +72,12 @@ Install the latest release of `Rectangle` from `PyPI` <https://pypi.org/project/
|
|
|
72
72
|
pip install rectanglepy
|
|
73
73
|
```
|
|
74
74
|
|
|
75
|
+
## How Rectangle works
|
|
76
|
+
|
|
77
|
+

|
|
78
|
+
|
|
79
|
+
Rectangle performs robust multiscale deconvolution informed by single-cell transcriptomics. Rectangle takes annotated scRNA-seq data and a bulk RNA-seq mixture as input. In the signature-building phase, it performs bootstrapped pseudobulking of the scRNA-seq data, followed by signature-gene selection based on log-fold-change (logFC) and p-value, and signature-matrix optimization (minimizing the condition number and correcting for cell-type-specific mRNA content bias). Based on this, it constructs two signature matrices: a “direct signature”, resolving individual cell types, and a “clustered signature” grouping transcriptionally similar cell types. In the deconvolution phase, Rectangle first uses the clustered signature to deconvolve the bulk data into coarse cell-type estimates. These are then used to constrain a second deconvolution step that leverages the direct signature to resolve individual cell-type fractions. Finally, the resulting estimates are scaled to account for unknown cellular content not represented in the reference.
|
|
80
|
+
|
|
75
81
|
## License
|
|
76
82
|
|
|
77
83
|
Rectangle is dual-licensed: BSD-3-Clause OR Commercial.
|
|
@@ -100,7 +106,9 @@ For commercial licensing: **innovation-psb@uibk.ac.at**
|
|
|
100
106
|
|
|
101
107
|
## Citation
|
|
102
108
|
|
|
103
|
-
|
|
109
|
+
If you use Rectangle in your project, please cite:
|
|
110
|
+
|
|
111
|
+
> Eder B, Rigato I, Dietrich A, Merotto L, Sturm G, Treis T, List M, Theis FJ, Finotello F. Rectangle: robust and scalable multiscale deconvolution informed by single-cell RNA sequencing data. bioRxiv. 2026. doi:[10.64898/2026.07.07.736950](https://doi.org/10.64898/2026.07.07.736950)
|
|
104
112
|
|
|
105
113
|
[scverse-discourse]: https://discourse.scverse.org/
|
|
106
114
|
[issue-tracker]: https://github.com/ComputationalBiomedicineGroup/Rectangle/issues
|
|
@@ -33,6 +33,12 @@ Install the latest release of `Rectangle` from `PyPI` <https://pypi.org/project/
|
|
|
33
33
|
pip install rectanglepy
|
|
34
34
|
```
|
|
35
35
|
|
|
36
|
+
## How Rectangle works
|
|
37
|
+
|
|
38
|
+

|
|
39
|
+
|
|
40
|
+
Rectangle performs robust multiscale deconvolution informed by single-cell transcriptomics. Rectangle takes annotated scRNA-seq data and a bulk RNA-seq mixture as input. In the signature-building phase, it performs bootstrapped pseudobulking of the scRNA-seq data, followed by signature-gene selection based on log-fold-change (logFC) and p-value, and signature-matrix optimization (minimizing the condition number and correcting for cell-type-specific mRNA content bias). Based on this, it constructs two signature matrices: a “direct signature”, resolving individual cell types, and a “clustered signature” grouping transcriptionally similar cell types. In the deconvolution phase, Rectangle first uses the clustered signature to deconvolve the bulk data into coarse cell-type estimates. These are then used to constrain a second deconvolution step that leverages the direct signature to resolve individual cell-type fractions. Finally, the resulting estimates are scaled to account for unknown cellular content not represented in the reference.
|
|
41
|
+
|
|
36
42
|
## License
|
|
37
43
|
|
|
38
44
|
Rectangle is dual-licensed: BSD-3-Clause OR Commercial.
|
|
@@ -61,7 +67,9 @@ For commercial licensing: **innovation-psb@uibk.ac.at**
|
|
|
61
67
|
|
|
62
68
|
## Citation
|
|
63
69
|
|
|
64
|
-
|
|
70
|
+
If you use Rectangle in your project, please cite:
|
|
71
|
+
|
|
72
|
+
> Eder B, Rigato I, Dietrich A, Merotto L, Sturm G, Treis T, List M, Theis FJ, Finotello F. Rectangle: robust and scalable multiscale deconvolution informed by single-cell RNA sequencing data. bioRxiv. 2026. doi:[10.64898/2026.07.07.736950](https://doi.org/10.64898/2026.07.07.736950)
|
|
65
73
|
|
|
66
74
|
[scverse-discourse]: https://discourse.scverse.org/
|
|
67
75
|
[issue-tracker]: https://github.com/ComputationalBiomedicineGroup/Rectangle/issues
|
|
Binary file
|
|
@@ -16,6 +16,15 @@
|
|
|
16
16
|
url = {https://doi.org/10.1186/s13059-017-1382-0}
|
|
17
17
|
}
|
|
18
18
|
|
|
19
|
+
@article{Eder2026,
|
|
20
|
+
author = {Bernhard Eder and Irene Rigato and Alexander Dietrich and Lorenzo Merotto and Gregor Sturm and Tim Treis and Markus List and Fabian J. Theis and Francesca Finotello},
|
|
21
|
+
title = {Rectangle: robust and scalable multiscale deconvolution informed by single-cell RNA sequencing data},
|
|
22
|
+
journal = {bioRxiv},
|
|
23
|
+
year = {2026},
|
|
24
|
+
doi = {10.64898/2026.07.07.736950},
|
|
25
|
+
url = {https://doi.org/10.64898/2026.07.07.736950}
|
|
26
|
+
}
|
|
27
|
+
|
|
19
28
|
@article{Finotello2019,
|
|
20
29
|
author = {Francesca Finotello and Gregor Mayer and Christian Plattner and Cornelia Laschober and Dietmar Rieder and Peter Hackl and Lukas Krogsdam and Michael L. T. Helmberg and Zlatko Trajanoski},
|
|
21
30
|
title = {Molecular and pharmacological modulators of the tumor immune contexture revealed by deconvolution of RNA-seq data},
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
from importlib.metadata import version
|
|
2
2
|
|
|
3
3
|
from . import pp, tl
|
|
4
|
+
from .parameters import RectangleAdvancedParameters
|
|
4
5
|
from .rectangle import load_tutorial_data, rectangle
|
|
5
6
|
|
|
6
|
-
__all__ = ["pp", "tl", "load_tutorial_data", "rectangle"]
|
|
7
|
+
__all__ = ["pp", "tl", "RectangleAdvancedParameters", "load_tutorial_data", "rectangle"]
|
|
7
8
|
|
|
8
9
|
__version__ = version("rectanglepy")
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
# SPDX-License-Identifier: BSD-3-Clause OR LicenseRef-Rectangle-Commercial
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
@dataclass(frozen=True)
|
|
7
|
+
class RectangleAdvancedParameters:
|
|
8
|
+
"""Advanced parameters for Rectangle internals."""
|
|
9
|
+
|
|
10
|
+
number_of_bootstraps: int = 20
|
|
11
|
+
grid_search_split_size: int = 50
|
|
@@ -13,6 +13,7 @@ from scipy.cluster.hierarchy import fcluster, linkage
|
|
|
13
13
|
from scipy.stats import pearsonr
|
|
14
14
|
from sklearn.metrics import silhouette_score
|
|
15
15
|
|
|
16
|
+
from rectanglepy.parameters import RectangleAdvancedParameters
|
|
16
17
|
from rectanglepy.tl.deconvolution import solve_qp
|
|
17
18
|
|
|
18
19
|
from .rectangle_signature import RectangleSignatureResult
|
|
@@ -129,24 +130,25 @@ def _filter_de_analysis_results(de_analysis_result, p, logfc):
|
|
|
129
130
|
|
|
130
131
|
|
|
131
132
|
def _run_deseq2(
|
|
132
|
-
countsig: pd.DataFrame,
|
|
133
|
+
countsig: pd.DataFrame,
|
|
134
|
+
sc_data,
|
|
135
|
+
annotations: pd.Series,
|
|
136
|
+
n_cpus: int = None,
|
|
137
|
+
gene_expression_threshold=0.4,
|
|
138
|
+
number_of_bootstraps: int = 20,
|
|
133
139
|
) -> dict[str | int, pd.DataFrame]:
|
|
134
140
|
results = {}
|
|
135
141
|
inference = DefaultInference(n_cpus=n_cpus)
|
|
136
|
-
bootstrapped_signature = _create_bootstrap_signature(countsig, sc_data, annotations)
|
|
142
|
+
bootstrapped_signature = _create_bootstrap_signature(countsig, sc_data, annotations, number_of_bootstraps)
|
|
137
143
|
np.random.seed(42)
|
|
138
144
|
for _i, cell_type in enumerate(countsig.columns):
|
|
139
|
-
bootstrapped_signature_copy = bootstrapped_signature.copy()
|
|
140
|
-
countsig_copy = countsig.copy()
|
|
141
145
|
sc_data_filtered = sc_data.T[annotations == cell_type]
|
|
142
146
|
expressed_cells = (sc_data_filtered > 0).sum(axis=0)
|
|
143
147
|
if expressed_cells.ndim > 1: # needed for sparse matrices
|
|
144
148
|
expressed_cells = np.squeeze(np.asarray(expressed_cells))
|
|
145
|
-
# make dense out of sparse
|
|
146
|
-
sc_data_filtered = sc_data_filtered.toarray()
|
|
147
149
|
threshold = gene_expression_threshold * sc_data_filtered.shape[0]
|
|
148
|
-
genes =
|
|
149
|
-
bootstrapped_signature_copy =
|
|
150
|
+
genes = countsig.index[expressed_cells > threshold].tolist()
|
|
151
|
+
bootstrapped_signature_copy = bootstrapped_signature.loc[genes].T
|
|
150
152
|
logger.info(f"Running DE analysis for {cell_type}")
|
|
151
153
|
condition = ["B" if (cell_type + "_") in x else "A" for x in bootstrapped_signature_copy.index]
|
|
152
154
|
clinical_df = pd.DataFrame({"condition": condition}, index=bootstrapped_signature_copy.index)
|
|
@@ -167,25 +169,29 @@ def _run_deseq2(
|
|
|
167
169
|
return results
|
|
168
170
|
|
|
169
171
|
|
|
170
|
-
def _create_bootstrap_signature(countsig, sc_data, annotations) -> pd.DataFrame:
|
|
171
|
-
|
|
172
|
-
|
|
172
|
+
def _create_bootstrap_signature(countsig, sc_data, annotations, number_of_bootstraps: int = 20) -> pd.DataFrame:
|
|
173
|
+
# Cells are only ever summed here, never read one by one, so sparse input is kept sparse:
|
|
174
|
+
# densifying the whole matrix costs genes * cells * 8 bytes, which is gigabytes on an
|
|
175
|
+
# atlas-sized reference. Dense input deliberately stays dense - gathering rows out of a
|
|
176
|
+
# dense array is faster than routing them through a sparse format.
|
|
177
|
+
cells_by_gene = sc_data.T.tocsr() if scipy.sparse.issparse(sc_data) else sc_data.T
|
|
173
178
|
celltypes = countsig.columns
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
samples_per_bootstrap = 500
|
|
179
|
+
columns = {}
|
|
180
|
+
samples_per_bootstrap = 1000
|
|
177
181
|
np.random.seed(42)
|
|
178
182
|
for celltype in celltypes:
|
|
179
|
-
sc_data_filtered =
|
|
183
|
+
sc_data_filtered = cells_by_gene[(annotations == celltype).to_numpy()]
|
|
184
|
+
# sparse matrices do not support len(); shape[0] is the cell count for both formats
|
|
185
|
+
number_of_cells = sc_data_filtered.shape[0]
|
|
180
186
|
for i in range(number_of_bootstraps):
|
|
181
|
-
selected_rows = np.random.choice(
|
|
187
|
+
selected_rows = np.random.choice(number_of_cells, samples_per_bootstrap, replace=True)
|
|
182
188
|
summed_rows = sc_data_filtered[selected_rows].sum(axis=0)
|
|
183
|
-
|
|
184
|
-
|
|
189
|
+
# a sparse sum is a (1, genes) matrix, a dense one a flat array
|
|
190
|
+
columns[f"{celltype}_{i}"] = np.asarray(summed_rows).ravel()
|
|
191
|
+
# built in one go: assigning column by column repeatedly reallocates the frame
|
|
192
|
+
bootstrapped_signature = pd.DataFrame(columns, index=countsig.index)
|
|
185
193
|
# to int
|
|
186
|
-
|
|
187
|
-
bootstrapped_signature = bootstrapped_signature.astype(int)
|
|
188
|
-
return bootstrapped_signature
|
|
194
|
+
return bootstrapped_signature.astype(int)
|
|
189
195
|
|
|
190
196
|
|
|
191
197
|
def _de_analysis(
|
|
@@ -197,15 +203,31 @@ def _de_analysis(
|
|
|
197
203
|
optimize_cutoffs: bool,
|
|
198
204
|
n_cpus: int = None,
|
|
199
205
|
genes=None,
|
|
200
|
-
gene_expression_threshold=0.
|
|
206
|
+
gene_expression_threshold=0.4,
|
|
207
|
+
advanced_parameters: RectangleAdvancedParameters = None,
|
|
201
208
|
) -> tuple[Series, dict[str, [str]] :, DataFrame | None]:
|
|
202
209
|
logger.info("Starting DE analysis")
|
|
203
|
-
|
|
210
|
+
advanced_parameters = advanced_parameters or RectangleAdvancedParameters()
|
|
211
|
+
deseq_results = _run_deseq2(
|
|
212
|
+
pseudo_count_sig,
|
|
213
|
+
sc_data,
|
|
214
|
+
annotations,
|
|
215
|
+
n_cpus,
|
|
216
|
+
gene_expression_threshold,
|
|
217
|
+
advanced_parameters.number_of_bootstraps,
|
|
218
|
+
)
|
|
204
219
|
optimization_results = None
|
|
205
220
|
|
|
206
221
|
if optimize_cutoffs:
|
|
207
222
|
logger.info("Optimizing cutoff parameters p and lfc")
|
|
208
|
-
optimization_results = _optimize_parameters(
|
|
223
|
+
optimization_results = _optimize_parameters(
|
|
224
|
+
sc_data,
|
|
225
|
+
annotations,
|
|
226
|
+
pseudo_count_sig,
|
|
227
|
+
deseq_results,
|
|
228
|
+
genes,
|
|
229
|
+
advanced_parameters.grid_search_split_size,
|
|
230
|
+
)
|
|
209
231
|
p, lfc = optimization_results.iloc[0, 0:2]
|
|
210
232
|
logger.info(f"Optimization done\n Best cutoffs p: {p} and lfc: {lfc}")
|
|
211
233
|
|
|
@@ -286,7 +308,8 @@ def build_rectangle_signatures(
|
|
|
286
308
|
p=0.015,
|
|
287
309
|
lfc=1.5,
|
|
288
310
|
n_cpus: int = None,
|
|
289
|
-
gene_expression_threshold=0.
|
|
311
|
+
gene_expression_threshold=0.4,
|
|
312
|
+
advanced_parameters: RectangleAdvancedParameters = None,
|
|
290
313
|
) -> RectangleSignatureResult:
|
|
291
314
|
r"""Builds rectangle signatures based on single-cell count data and annotations.
|
|
292
315
|
|
|
@@ -303,7 +326,7 @@ def build_rectangle_signatures(
|
|
|
303
326
|
raw
|
|
304
327
|
A flag indicating whether to use the raw Anndata data. Defaults to False.
|
|
305
328
|
optimize_cutoffs
|
|
306
|
-
Indicates whether to optimize the
|
|
329
|
+
Indicates whether to optimize the log fold change cutoffs using gridsearch. Defaults to True.
|
|
307
330
|
p
|
|
308
331
|
The p-value threshold for the DE analysis (only used if optimize_cutoffs is False).
|
|
309
332
|
lfc
|
|
@@ -311,12 +334,15 @@ def build_rectangle_signatures(
|
|
|
311
334
|
n_cpus
|
|
312
335
|
The number of cpus to use for the DE analysis. Defaults to the number of cpus available.
|
|
313
336
|
gene_expression_threshold
|
|
314
|
-
The gene expression threshold for the DE analysis.
|
|
337
|
+
The gene expression threshold for the DE analysis. The fraction of cells that must express a gene to be considered in DGE. Defaults to 0.4
|
|
338
|
+
advanced_parameters
|
|
339
|
+
Optional advanced Rectangle parameters. Defaults are used when not provided.
|
|
315
340
|
|
|
316
341
|
Returns
|
|
317
342
|
-------
|
|
318
343
|
The result of the rectangle signature analysis which is of type RectangleSignatureResult.
|
|
319
344
|
"""
|
|
345
|
+
advanced_parameters = advanced_parameters or RectangleAdvancedParameters()
|
|
320
346
|
annotations = adata.obs[cell_type_col]
|
|
321
347
|
adata = adata[:, adata.X.sum(axis=0) > len(annotations.value_counts())]
|
|
322
348
|
assert adata.var_names.is_unique, "Duplicate gene found in adata"
|
|
@@ -344,7 +370,16 @@ def build_rectangle_signatures(
|
|
|
344
370
|
m_rna_biasfactors = _create_bias_factors(pseudo_sig_counts, sc_counts, annotations)
|
|
345
371
|
|
|
346
372
|
marker_genes, marker_genes_per_cell_type, optimization_result = _de_analysis(
|
|
347
|
-
pseudo_sig_counts,
|
|
373
|
+
pseudo_sig_counts,
|
|
374
|
+
sc_counts,
|
|
375
|
+
annotations,
|
|
376
|
+
p,
|
|
377
|
+
lfc,
|
|
378
|
+
optimize_cutoffs,
|
|
379
|
+
n_cpus,
|
|
380
|
+
genes,
|
|
381
|
+
gene_expression_threshold,
|
|
382
|
+
advanced_parameters,
|
|
348
383
|
)
|
|
349
384
|
pseudo_sig_cpm = _convert_to_cpm(pseudo_sig_counts)
|
|
350
385
|
logger.info("Starting rectangle cluster analysis")
|
|
@@ -371,6 +406,7 @@ def build_rectangle_signatures(
|
|
|
371
406
|
lfc,
|
|
372
407
|
False,
|
|
373
408
|
gene_expression_threshold=gene_expression_threshold,
|
|
409
|
+
advanced_parameters=advanced_parameters,
|
|
374
410
|
)
|
|
375
411
|
clustered_signature = _convert_to_cpm(clustered_signature)
|
|
376
412
|
return RectangleSignatureResult(
|
|
@@ -390,9 +426,10 @@ def build_rectangle_signatures(
|
|
|
390
426
|
def _create_pseudo_count_sig(sc_counts: np.ndarray, annotations: pd.Series, var_names) -> pd.DataFrame:
|
|
391
427
|
unique_labels, label_indices = np.unique(annotations, return_inverse=True)
|
|
392
428
|
grouped_sum = np.zeros((len(unique_labels), sc_counts.shape[0]))
|
|
429
|
+
cells_by_gene = sc_counts.T
|
|
393
430
|
for i, _label in enumerate(unique_labels):
|
|
394
431
|
label_columns = label_indices == i
|
|
395
|
-
grouped_sum[i, :] = np.sum(
|
|
432
|
+
grouped_sum[i, :] = np.sum(cells_by_gene[label_columns, :], axis=0)
|
|
396
433
|
grouped_sum = grouped_sum.T
|
|
397
434
|
|
|
398
435
|
grouped_sum = pd.DataFrame(grouped_sum, index=var_names, columns=unique_labels).astype(int)
|
|
@@ -400,15 +437,20 @@ def _create_pseudo_count_sig(sc_counts: np.ndarray, annotations: pd.Series, var_
|
|
|
400
437
|
|
|
401
438
|
|
|
402
439
|
def _optimize_parameters(
|
|
403
|
-
sc_data: pd.DataFrame,
|
|
440
|
+
sc_data: pd.DataFrame,
|
|
441
|
+
annotations: pd.Series,
|
|
442
|
+
pseudo_signature_counts: pd.DataFrame,
|
|
443
|
+
de_results,
|
|
444
|
+
genes=None,
|
|
445
|
+
grid_search_split_size: int = 50,
|
|
404
446
|
) -> pd.DataFrame:
|
|
405
447
|
# search space for p and lfc
|
|
406
|
-
lfcs = [
|
|
448
|
+
lfcs = [1.5, 1.75, 2.0, 2.25, 2.5, 2.75, 3.0]
|
|
407
449
|
ps = [x / 1000 for x in range(50, 51, 1)]
|
|
408
450
|
|
|
409
451
|
results = []
|
|
410
452
|
logger.info("generating pseudo bulks")
|
|
411
|
-
bulks, real_fractions = _generate_pseudo_bulks(sc_data, annotations, genes)
|
|
453
|
+
bulks, real_fractions = _generate_pseudo_bulks(sc_data, annotations, genes, grid_search_split_size)
|
|
412
454
|
for p in ps:
|
|
413
455
|
for lfc in lfcs:
|
|
414
456
|
try:
|
|
@@ -441,9 +483,8 @@ def _assess_parameter_fit(
|
|
|
441
483
|
return rsme, pearson_r
|
|
442
484
|
|
|
443
485
|
|
|
444
|
-
def _generate_pseudo_bulks(sc_data, annotations, genes=None):
|
|
486
|
+
def _generate_pseudo_bulks(sc_data, annotations, genes=None, split_size: int = 50):
|
|
445
487
|
number_of_bulks = 50
|
|
446
|
-
split_size = 50
|
|
447
488
|
bulks = []
|
|
448
489
|
real_fractions = []
|
|
449
490
|
np.random.seed(42)
|
|
@@ -7,6 +7,7 @@ from anndata import AnnData
|
|
|
7
7
|
from loguru import logger
|
|
8
8
|
from pandas import DataFrame
|
|
9
9
|
|
|
10
|
+
from .parameters import RectangleAdvancedParameters
|
|
10
11
|
from .pp import RectangleSignatureResult, build_rectangle_signatures
|
|
11
12
|
from .tl import deconvolution
|
|
12
13
|
|
|
@@ -23,7 +24,8 @@ def rectangle(
|
|
|
23
24
|
p=0.015,
|
|
24
25
|
lfc=1.5,
|
|
25
26
|
n_cpus: int = None,
|
|
26
|
-
gene_expression_threshold=0.
|
|
27
|
+
gene_expression_threshold=0.4,
|
|
28
|
+
advanced_parameters: RectangleAdvancedParameters = None,
|
|
27
29
|
) -> tuple[DataFrame, RectangleSignatureResult]:
|
|
28
30
|
r"""All in one deconvolution method. Creates signatures and deconvolutes the bulk data. Has options for subsampling and consensus runs.
|
|
29
31
|
|
|
@@ -40,7 +42,7 @@ def rectangle(
|
|
|
40
42
|
raw
|
|
41
43
|
A flag indicating whether to use the raw Anndata data.
|
|
42
44
|
optimize_cutoffs
|
|
43
|
-
Indicates whether to optimize the
|
|
45
|
+
Indicates whether to optimize the log fold change cutoffs using gridsearch.
|
|
44
46
|
p
|
|
45
47
|
The p-value threshold for the DE analysis (only used if optimize_cutoffs is False).
|
|
46
48
|
lfc
|
|
@@ -50,7 +52,9 @@ def rectangle(
|
|
|
50
52
|
correct_mrna_bias : bool
|
|
51
53
|
A flag indicating whether to correct for mRNA bias. Defaults to True.
|
|
52
54
|
gene_expression_threshold : float
|
|
53
|
-
The threshold for gene expression. Genes
|
|
55
|
+
The threshold for gene expression. Genes must be expressed in at least this fraction of cells. Defaults to 0.4.
|
|
56
|
+
advanced_parameters
|
|
57
|
+
Optional advanced Rectangle parameters. Defaults are used when not provided.
|
|
54
58
|
|
|
55
59
|
Returns
|
|
56
60
|
-------
|
|
@@ -71,6 +75,7 @@ def rectangle(
|
|
|
71
75
|
lfc=lfc,
|
|
72
76
|
n_cpus=n_cpus,
|
|
73
77
|
gene_expression_threshold=gene_expression_threshold,
|
|
78
|
+
advanced_parameters=advanced_parameters,
|
|
74
79
|
)
|
|
75
80
|
|
|
76
81
|
estimations, bulk_err = deconvolution(signatures, bulks, correct_mrna_bias, n_cpus)
|
|
@@ -7,18 +7,166 @@ import numpy as np
|
|
|
7
7
|
import osqp
|
|
8
8
|
import pandas as pd
|
|
9
9
|
import scipy.sparse as sp
|
|
10
|
-
import statsmodels.api as sm
|
|
11
10
|
from joblib import Parallel, delayed, parallel_backend
|
|
12
11
|
from loguru import logger
|
|
13
12
|
|
|
14
13
|
from rectanglepy.pp.rectangle_signature import RectangleSignatureResult
|
|
15
14
|
|
|
15
|
+
# OSQP can return noticeably different solutions from active-set QP when P is singular/ill-conditioned.
|
|
16
|
+
# A tiny ridge makes the problem strictly convex and closer to quadprog behavior.
|
|
17
|
+
QP_RIDGE = 1e-8
|
|
18
|
+
|
|
16
19
|
|
|
17
20
|
def _scale_weights(weights: np.ndarray) -> np.ndarray:
|
|
18
21
|
min_weight = np.nextafter(min(weights), np.float64(1.0)) # prevent division by zero
|
|
19
22
|
return weights / min_weight
|
|
20
23
|
|
|
21
24
|
|
|
25
|
+
def _upper_triangular_csc_pattern(n_vars: int) -> tuple[np.ndarray, np.ndarray]:
|
|
26
|
+
"""Returns the (indices, indptr) of a fully populated upper triangular CSC matrix.
|
|
27
|
+
|
|
28
|
+
OSQP only stores the upper triangle of P, so handing it the triangle directly saves the
|
|
29
|
+
conversion of the full symmetric matrix on every solve.
|
|
30
|
+
"""
|
|
31
|
+
indices = np.concatenate([np.arange(column + 1) for column in range(n_vars)]).astype(np.int32)
|
|
32
|
+
indptr = np.concatenate(([0], np.cumsum(np.arange(1, n_vars + 1)))).astype(np.int32)
|
|
33
|
+
return indices, indptr
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class _QuadraticProgram:
|
|
37
|
+
"""A deconvolution QP for one (signature, bulk) pair, solvable at several dampening weights.
|
|
38
|
+
|
|
39
|
+
Everything that does not depend on the weights - the numpy views of the inputs and the constraint
|
|
40
|
+
matrix with its bounds - is built once and shared by all solves, which keeps it out of the
|
|
41
|
+
dampened least squares iteration.
|
|
42
|
+
|
|
43
|
+
Note that each solve still gets its own OSQP workspace: reusing one across solves makes OSQP
|
|
44
|
+
keep the problem scaling of the first objective, which perturbs the fixed point the iteration
|
|
45
|
+
converges to (and is not faster at these problem sizes).
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
def __init__(
|
|
49
|
+
self,
|
|
50
|
+
signature: pd.DataFrame,
|
|
51
|
+
bulk: pd.Series,
|
|
52
|
+
prev_assignments: list[int | str] = None,
|
|
53
|
+
prev_solution: pd.Series = None,
|
|
54
|
+
):
|
|
55
|
+
if not signature.index.equals(bulk.index):
|
|
56
|
+
bulk = bulk.reindex(signature.index)
|
|
57
|
+
self.signature = signature.to_numpy(dtype=np.float64)
|
|
58
|
+
self.bulk = bulk.to_numpy(dtype=np.float64)
|
|
59
|
+
self.n_vars = self.signature.shape[1] # number of cell types / fractions
|
|
60
|
+
|
|
61
|
+
self._triu_indices, self._triu_indptr = _upper_triangular_csc_pattern(self.n_vars)
|
|
62
|
+
self._triu_selection = np.tril_indices(self.n_vars) # reads P.T column-major upper triangle
|
|
63
|
+
self._constraints, self._lower_bounds = self._build_constraints(prev_assignments, prev_solution)
|
|
64
|
+
self._upper_bounds = np.full_like(self._lower_bounds, np.inf, dtype=np.float64)
|
|
65
|
+
|
|
66
|
+
def _build_constraints(
|
|
67
|
+
self, prev_assignments: list[int | str], prev_solution: pd.Series
|
|
68
|
+
) -> tuple[sp.csc_matrix, np.ndarray]:
|
|
69
|
+
# ----- Constraints in the original quadprog form: C.T x >= b
|
|
70
|
+
# We build C (n_vars, n_constraints) then map to OSQP: A = C.T, l=b, u=+inf
|
|
71
|
+
C_cols = []
|
|
72
|
+
b_list = []
|
|
73
|
+
|
|
74
|
+
# Constraint 1: sum(x) <= 1 -> -sum(x) >= -1
|
|
75
|
+
C_cols.append(-np.ones((self.n_vars, 1), dtype=np.float64))
|
|
76
|
+
b_list.append(np.array([-1.0], dtype=np.float64))
|
|
77
|
+
|
|
78
|
+
# Constraint 2: x >= 0 -> I x >= 0
|
|
79
|
+
C_cols.append(np.eye(self.n_vars, dtype=np.float64))
|
|
80
|
+
b_list.append(np.zeros(self.n_vars, dtype=np.float64))
|
|
81
|
+
|
|
82
|
+
# Constraint 3: keep close to prev_solution (this encoding already turns upper bounds into >= via negation)
|
|
83
|
+
if prev_solution is not None:
|
|
84
|
+
if prev_assignments is None:
|
|
85
|
+
raise ValueError("prev_assignments must be provided when prev_solution is provided.")
|
|
86
|
+
|
|
87
|
+
for cluster in prev_solution.index:
|
|
88
|
+
# x_cluster <= upper -> -x_cluster >= -upper
|
|
89
|
+
C_upper = np.array(
|
|
90
|
+
[-1.0 if str(x) == str(cluster) else 0.0 for x in prev_assignments],
|
|
91
|
+
dtype=np.float64,
|
|
92
|
+
).reshape(-1, 1)
|
|
93
|
+
|
|
94
|
+
# x_cluster >= lower -> +x_cluster >= +lower
|
|
95
|
+
C_lower = np.array(
|
|
96
|
+
[1.0 if str(x) == str(cluster) else 0.0 for x in prev_assignments],
|
|
97
|
+
dtype=np.float64,
|
|
98
|
+
).reshape(-1, 1)
|
|
99
|
+
|
|
100
|
+
prev_weight = float(prev_solution.loc[cluster])
|
|
101
|
+
upper = min(1.0, prev_weight + 0.03)
|
|
102
|
+
lower = max(0.0, prev_weight - 0.03)
|
|
103
|
+
|
|
104
|
+
C_cols.append(C_upper)
|
|
105
|
+
b_list.append(np.array([-upper], dtype=np.float64))
|
|
106
|
+
|
|
107
|
+
C_cols.append(C_lower)
|
|
108
|
+
b_list.append(np.array([lower], dtype=np.float64))
|
|
109
|
+
|
|
110
|
+
C = np.concatenate(C_cols, axis=1) # (n_vars, n_constraints)
|
|
111
|
+
b = np.concatenate(b_list, axis=0).astype(np.float64) # (n_constraints,)
|
|
112
|
+
|
|
113
|
+
# OSQP uses l <= A x <= u
|
|
114
|
+
return sp.csc_matrix(C.T), b
|
|
115
|
+
|
|
116
|
+
def _objective(self, gld: np.ndarray, multiplier: int) -> tuple[np.ndarray, np.ndarray]:
|
|
117
|
+
# Minimize 1/2 x^T G x - a^T x
|
|
118
|
+
if multiplier is None:
|
|
119
|
+
G = self.signature.T @ self.signature
|
|
120
|
+
a = self.signature.T @ self.bulk
|
|
121
|
+
else:
|
|
122
|
+
weights = np.square(1 / (self.signature @ gld))
|
|
123
|
+
weights_dampened = np.clip(_scale_weights(weights), None, multiplier)
|
|
124
|
+
# equivalent to signature.T @ diag(weights) @ signature, without the dense (genes, genes) matrix
|
|
125
|
+
G = self.signature.T @ (weights_dampened[:, None] * self.signature)
|
|
126
|
+
a = self.signature.T @ (weights_dampened * self.bulk)
|
|
127
|
+
|
|
128
|
+
# ----- Map to OSQP: minimize 1/2 x^T P x + q^T x -> q = -a
|
|
129
|
+
P = G
|
|
130
|
+
q = -a
|
|
131
|
+
|
|
132
|
+
# Optional scaling
|
|
133
|
+
scale = np.linalg.norm(P)
|
|
134
|
+
if scale > 0:
|
|
135
|
+
P = P / scale
|
|
136
|
+
q = q / scale
|
|
137
|
+
|
|
138
|
+
P = ((P + P.T) / 2.0) + QP_RIDGE * np.eye(self.n_vars, dtype=np.float64)
|
|
139
|
+
return P, q
|
|
140
|
+
|
|
141
|
+
def solve(self, gld: np.ndarray = None, multiplier: int = None) -> np.ndarray:
|
|
142
|
+
P, q = self._objective(gld, multiplier)
|
|
143
|
+
P_data = P.T[self._triu_selection] # upper triangle in CSC (column major) order
|
|
144
|
+
P_sp = sp.csc_matrix((P_data, self._triu_indices, self._triu_indptr), shape=(self.n_vars, self.n_vars))
|
|
145
|
+
|
|
146
|
+
solver = osqp.OSQP()
|
|
147
|
+
solver.setup(
|
|
148
|
+
P=P_sp,
|
|
149
|
+
q=q,
|
|
150
|
+
A=self._constraints,
|
|
151
|
+
l=self._lower_bounds,
|
|
152
|
+
u=self._upper_bounds,
|
|
153
|
+
verbose=False,
|
|
154
|
+
eps_abs=1e-7,
|
|
155
|
+
eps_rel=1e-7,
|
|
156
|
+
max_iter=50000,
|
|
157
|
+
polish=True,
|
|
158
|
+
warm_start=False,
|
|
159
|
+
scaled_termination=False,
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
res = solver.solve()
|
|
163
|
+
|
|
164
|
+
if res.info.status_val not in (1,): # 1 = solved
|
|
165
|
+
raise RuntimeError(f"OSQP did not solve the problem: {res.info.status}")
|
|
166
|
+
|
|
167
|
+
return res.x
|
|
168
|
+
|
|
169
|
+
|
|
22
170
|
def solve_qp(
|
|
23
171
|
signature: pd.DataFrame,
|
|
24
172
|
bulk: pd.Series,
|
|
@@ -52,143 +200,89 @@ def solve_qp(
|
|
|
52
200
|
Notes
|
|
53
201
|
-----
|
|
54
202
|
This function uses quadratic programming to solve the deconvolution problem. The objective is to minimize the difference between the observed bulk data and the data predicted by the signature matrix and the cell fractions. The function also includes constraints to ensure that the cell fractions are non-negative and sum to 1, and to make the solution similar to the previous assignments and weights if they are provided.
|
|
203
|
+
|
|
204
|
+
Callers that solve the same problem at several dampening weights (e.g. the dampened least squares
|
|
205
|
+
iteration) should use :class:`_QuadraticProgram` directly, which shares the weight independent
|
|
206
|
+
parts of the problem across solves.
|
|
55
207
|
"""
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
# We'll build C (n_vars, n_constraints) then map to OSQP: A = C.T, l=b, u=+inf
|
|
73
|
-
C_cols = []
|
|
74
|
-
b_list = []
|
|
75
|
-
|
|
76
|
-
# Constraint 1: sum(x) <= 1 -> -sum(x) >= -1
|
|
77
|
-
C_cols.append(-np.ones((n_vars, 1), dtype=np.float64))
|
|
78
|
-
b_list.append(np.array([-1.0], dtype=np.float64))
|
|
79
|
-
|
|
80
|
-
# Constraint 2: x >= 0 -> I x >= 0
|
|
81
|
-
C_cols.append(np.eye(n_vars, dtype=np.float64))
|
|
82
|
-
b_list.append(np.zeros(n_vars, dtype=np.float64))
|
|
83
|
-
|
|
84
|
-
# Constraint 3: keep close to prev_solution (your encoding already turns upper bounds into >= via negation)
|
|
85
|
-
if prev_solution is not None:
|
|
86
|
-
if prev_assignments is None:
|
|
87
|
-
raise ValueError("prev_assignments must be provided when prev_solution is provided.")
|
|
88
|
-
|
|
89
|
-
for cluster in prev_solution.index:
|
|
90
|
-
# x_cluster <= upper -> -x_cluster >= -upper
|
|
91
|
-
C_upper = np.array(
|
|
92
|
-
[-1.0 if str(x) == str(cluster) else 0.0 for x in prev_assignments],
|
|
93
|
-
dtype=np.float64,
|
|
94
|
-
).reshape(-1, 1)
|
|
95
|
-
|
|
96
|
-
# x_cluster >= lower -> +x_cluster >= +lower
|
|
97
|
-
C_lower = np.array(
|
|
98
|
-
[1.0 if str(x) == str(cluster) else 0.0 for x in prev_assignments],
|
|
99
|
-
dtype=np.float64,
|
|
100
|
-
).reshape(-1, 1)
|
|
101
|
-
|
|
102
|
-
prev_weight = float(prev_solution.loc[cluster])
|
|
103
|
-
upper = min(1.0, prev_weight + 0.03)
|
|
104
|
-
lower = max(0.0, prev_weight - 0.03)
|
|
105
|
-
|
|
106
|
-
C_cols.append(C_upper)
|
|
107
|
-
b_list.append(np.array([-upper], dtype=np.float64))
|
|
108
|
-
|
|
109
|
-
C_cols.append(C_lower)
|
|
110
|
-
b_list.append(np.array([lower], dtype=np.float64))
|
|
111
|
-
|
|
112
|
-
C = np.concatenate(C_cols, axis=1) # (n_vars, n_constraints)
|
|
113
|
-
b = np.concatenate(b_list, axis=0).astype(np.float64) # (n_constraints,)
|
|
114
|
-
|
|
115
|
-
# ----- Map to OSQP: minimize 1/2 x^T P x + q^T x
|
|
116
|
-
# Your objective: 1/2 x^T G x - a^T x -> q = -a
|
|
117
|
-
P = G
|
|
118
|
-
q = -a
|
|
119
|
-
|
|
120
|
-
# Optional scaling
|
|
121
|
-
scale = np.linalg.norm(P)
|
|
122
|
-
if scale > 0:
|
|
123
|
-
P = P / scale
|
|
124
|
-
q = q / scale
|
|
125
|
-
|
|
126
|
-
# OSQP can return noticeably different solutions from active-set QP when P is singular/ill-conditioned.
|
|
127
|
-
# Add tiny ridge to make the problem strictly convex and closer to quadprog behavior.
|
|
128
|
-
ridge = 1e-8
|
|
129
|
-
P = ((P + P.T) / 2.0) + ridge * np.eye(n_vars, dtype=np.float64)
|
|
130
|
-
|
|
131
|
-
# OSQP uses l <= A x <= u
|
|
132
|
-
A = C.T # (n_constraints, n_vars)
|
|
133
|
-
l = b
|
|
134
|
-
u = np.full_like(l, np.inf, dtype=np.float64)
|
|
135
|
-
|
|
136
|
-
# Sparse matrices (required/expected)
|
|
137
|
-
P_sp = sp.csc_matrix(P)
|
|
138
|
-
A_sp = sp.csc_matrix(A)
|
|
139
|
-
|
|
140
|
-
solver = osqp.OSQP()
|
|
141
|
-
solver.setup(
|
|
142
|
-
P=P_sp,
|
|
143
|
-
q=q,
|
|
144
|
-
A=A_sp,
|
|
145
|
-
l=l,
|
|
146
|
-
u=u,
|
|
147
|
-
verbose=False,
|
|
148
|
-
eps_abs=1e-7,
|
|
149
|
-
eps_rel=1e-7,
|
|
150
|
-
max_iter=50000,
|
|
151
|
-
polish=True,
|
|
152
|
-
warm_start=False,
|
|
153
|
-
scaled_termination=False,
|
|
154
|
-
)
|
|
208
|
+
problem = _QuadraticProgram(signature, bulk, prev_assignments, prev_solution)
|
|
209
|
+
return problem.solve(gld, multiplier)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _batched_wls(
|
|
213
|
+
design: np.ndarray,
|
|
214
|
+
response: np.ndarray,
|
|
215
|
+
weights: np.ndarray,
|
|
216
|
+
subsets: np.ndarray,
|
|
217
|
+
max_bytes: int = 64 * 1024**2,
|
|
218
|
+
) -> np.ndarray:
|
|
219
|
+
"""Fits one weighted least squares model per gene subset in batch.
|
|
220
|
+
|
|
221
|
+
Mirrors ``statsmodels.api.WLS(...).fit()``: the design and the response are whitened with the
|
|
222
|
+
square root of the weights and the parameters are read off the pseudo inverse of the whitened
|
|
223
|
+
design. Stacking the subsets lets numpy loop over them in C instead of Python.
|
|
155
224
|
|
|
156
|
-
|
|
225
|
+
Parameters
|
|
226
|
+
----------
|
|
227
|
+
design : np.ndarray
|
|
228
|
+
The (genes, cell types) design matrix.
|
|
229
|
+
response : np.ndarray
|
|
230
|
+
The (genes,) response vector.
|
|
231
|
+
weights : np.ndarray
|
|
232
|
+
The (genes,) regression weights.
|
|
233
|
+
subsets : np.ndarray
|
|
234
|
+
A (n_subsets, subset size) array of gene indices, one row per model.
|
|
235
|
+
max_bytes : int
|
|
236
|
+
Upper bound on the size of a single whitened design batch, used to chunk the subsets.
|
|
237
|
+
|
|
238
|
+
Returns
|
|
239
|
+
-------
|
|
240
|
+
np.ndarray
|
|
241
|
+
A (n_subsets, cell types) array of fitted parameters.
|
|
242
|
+
"""
|
|
243
|
+
n_subsets, subset_size = subsets.shape
|
|
244
|
+
n_params = design.shape[1]
|
|
245
|
+
params = np.empty((n_subsets, n_params), dtype=np.float64)
|
|
157
246
|
|
|
158
|
-
|
|
159
|
-
|
|
247
|
+
chunk_size = max(1, int(max_bytes // (subset_size * n_params * design.itemsize)))
|
|
248
|
+
for start in range(0, n_subsets, chunk_size):
|
|
249
|
+
chunk = subsets[start : start + chunk_size]
|
|
250
|
+
sqrt_weights = np.sqrt(weights[chunk]) # (chunk, subset size)
|
|
251
|
+
whitened_design = design[chunk] * sqrt_weights[:, :, None] # (chunk, subset size, cell types)
|
|
252
|
+
whitened_response = response[chunk] * sqrt_weights
|
|
253
|
+
params[start : start + chunk_size] = np.einsum("bpg,bg->bp", np.linalg.pinv(whitened_design), whitened_response)
|
|
160
254
|
|
|
161
|
-
return
|
|
255
|
+
return params
|
|
162
256
|
|
|
163
257
|
|
|
164
258
|
def _calculate_dampening_constant(signature: pd.DataFrame, bulk: pd.Series, qp_gld: np.ndarray) -> int:
|
|
165
259
|
solutions_std = []
|
|
166
260
|
np.random.seed(1)
|
|
167
|
-
|
|
261
|
+
signature_values = np.asarray(signature, dtype=np.float64)
|
|
262
|
+
bulk_values = np.asarray(bulk, dtype=np.float64)
|
|
263
|
+
n_genes = signature_values.shape[0]
|
|
264
|
+
weights = np.square(1 / (signature_values @ qp_gld))
|
|
168
265
|
weights_scaled = _scale_weights(weights)
|
|
169
266
|
weights_scaled_no_inf = weights_scaled[weights_scaled != np.inf]
|
|
170
267
|
qp_gld_sum = sum(qp_gld)
|
|
268
|
+
subset_size = n_genes // 2
|
|
269
|
+
n_subsets = 100
|
|
171
270
|
# try multiple values of the dampening constant (multiplier)
|
|
172
271
|
# for each, calculate the variance of the dampened weighted solution for a subset of genes
|
|
173
272
|
max_range = 40
|
|
174
273
|
multiplier_range = min(max_range, math.ceil(np.log2(max(weights_scaled_no_inf))))
|
|
175
274
|
for i in range(multiplier_range):
|
|
176
|
-
solutions = []
|
|
177
275
|
multiplier = 2**i
|
|
178
|
-
weights_dampened = np.
|
|
179
|
-
for _ in range(
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
solutions_std.append(solutions_df.std(axis=0))
|
|
189
|
-
solutions_std_df = pd.DataFrame(solutions_std)
|
|
190
|
-
means = solutions_std_df.apply(lambda x: np.mean(x**2), axis=1)
|
|
191
|
-
best_dampening_constant = means.idxmin()
|
|
276
|
+
weights_dampened = np.minimum(weights_scaled, multiplier)
|
|
277
|
+
subsets = np.array([np.random.choice(n_genes, size=subset_size, replace=False) for _ in range(n_subsets)])
|
|
278
|
+
# WLS uses the signature directly; no intercept column is added.
|
|
279
|
+
params = _batched_wls(signature_values, bulk_values, weights_dampened, subsets)
|
|
280
|
+
solutions = params * qp_gld_sum / params.sum(axis=1, keepdims=True)
|
|
281
|
+
|
|
282
|
+
solutions_std.append(np.nanstd(solutions, axis=0, ddof=1))
|
|
283
|
+
solutions_std_array = np.array(solutions_std)
|
|
284
|
+
means = np.nanmean(np.square(solutions_std_array), axis=1)
|
|
285
|
+
best_dampening_constant = int(np.nanargmin(means))
|
|
192
286
|
return best_dampening_constant
|
|
193
287
|
|
|
194
288
|
|
|
@@ -202,7 +296,8 @@ def _calculate_ls(
|
|
|
202
296
|
signature = signature.loc[genes].sort_index()
|
|
203
297
|
bulk = bulk.loc[genes].sort_index().astype("double")
|
|
204
298
|
|
|
205
|
-
|
|
299
|
+
problem = _QuadraticProgram(signature, bulk, prev_assignments, prev_weights)
|
|
300
|
+
approximate_solution = problem.solve()
|
|
206
301
|
dampening_constant = _calculate_dampening_constant(signature, bulk, approximate_solution)
|
|
207
302
|
multiplier = 2**dampening_constant
|
|
208
303
|
|
|
@@ -212,7 +307,7 @@ def _calculate_ls(
|
|
|
212
307
|
iterations = 2
|
|
213
308
|
solutions_sum = approximate_solution
|
|
214
309
|
while (change > convergence_threshold) and (iterations < max_iterations):
|
|
215
|
-
dampened_solution =
|
|
310
|
+
dampened_solution = problem.solve(approximate_solution, multiplier)
|
|
216
311
|
solutions_sum += dampened_solution
|
|
217
312
|
solution_averages = solutions_sum / iterations
|
|
218
313
|
change = np.linalg.norm(solution_averages - approximate_solution, 1)
|
|
@@ -6,6 +6,7 @@ import pytest
|
|
|
6
6
|
from anndata import AnnData
|
|
7
7
|
|
|
8
8
|
import rectanglepy as rectangle
|
|
9
|
+
from rectanglepy import RectangleAdvancedParameters
|
|
9
10
|
from rectanglepy.pp.create_signature import (
|
|
10
11
|
_assess_parameter_fit,
|
|
11
12
|
_calculate_cluster_range,
|
|
@@ -18,6 +19,7 @@ from rectanglepy.pp.create_signature import (
|
|
|
18
19
|
_de_analysis,
|
|
19
20
|
_generate_pseudo_bulks,
|
|
20
21
|
_get_fcluster_assignments,
|
|
22
|
+
_optimize_parameters,
|
|
21
23
|
_run_deseq2,
|
|
22
24
|
build_rectangle_signatures,
|
|
23
25
|
)
|
|
@@ -154,6 +156,7 @@ def test_generate_pseudo_bulks(small_data):
|
|
|
154
156
|
sc_counts = sc_counts.astype("int")
|
|
155
157
|
adata = AnnData(sc_counts.T, obs=annotations.to_frame(name="cell_type"))
|
|
156
158
|
result, _ = _generate_pseudo_bulks(adata.X.T, annotations, adata.var_names)
|
|
159
|
+
custom_result, _ = _generate_pseudo_bulks(adata.X.T, annotations, adata.var_names, split_size=10)
|
|
157
160
|
|
|
158
161
|
sc_data = sc_counts.astype(pd.SparseDtype("int"))
|
|
159
162
|
csr_sparse_matrix = sc_data.sparse.to_coo().tocsr()
|
|
@@ -162,11 +165,40 @@ def test_generate_pseudo_bulks(small_data):
|
|
|
162
165
|
result_sparse, _ = _generate_pseudo_bulks(adata_sparse.X.T, annotations, adata_sparse.var_names)
|
|
163
166
|
|
|
164
167
|
assert len(result) == 1000 and len(result.columns) == 50
|
|
168
|
+
assert custom_result.shape == result.shape
|
|
169
|
+
assert not np.allclose(result, custom_result)
|
|
165
170
|
# first gene should have all 0s
|
|
166
171
|
assert result.iloc[0, :].sum() == 0
|
|
167
172
|
assert np.allclose(result, result_sparse)
|
|
168
173
|
|
|
169
174
|
|
|
175
|
+
def test_optimize_parameters_uses_grid_search_split_size(monkeypatch):
|
|
176
|
+
seen = {}
|
|
177
|
+
|
|
178
|
+
def fake_generate_pseudo_bulks(sc_data, annotations, genes=None, split_size=50):
|
|
179
|
+
seen["split_size"] = split_size
|
|
180
|
+
bulks = pd.DataFrame([[1.0]], index=["gene"], columns=["bulk"])
|
|
181
|
+
real_fractions = pd.DataFrame([[1.0]], index=["cell_type"], columns=["bulk"])
|
|
182
|
+
return bulks, real_fractions
|
|
183
|
+
|
|
184
|
+
def fake_assess_parameter_fit(lfc, p, bulks, real_fractions, pseudo_signature_counts, de_results):
|
|
185
|
+
return 0.0, 1.0
|
|
186
|
+
|
|
187
|
+
monkeypatch.setattr(rectangle.pp.create_signature, "_generate_pseudo_bulks", fake_generate_pseudo_bulks)
|
|
188
|
+
monkeypatch.setattr(rectangle.pp.create_signature, "_assess_parameter_fit", fake_assess_parameter_fit)
|
|
189
|
+
|
|
190
|
+
results = _optimize_parameters(
|
|
191
|
+
pd.DataFrame(),
|
|
192
|
+
pd.Series(dtype=str),
|
|
193
|
+
pd.DataFrame(),
|
|
194
|
+
{},
|
|
195
|
+
grid_search_split_size=13,
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
assert seen["split_size"] == 13
|
|
199
|
+
assert results.iloc[0]["pearson_r"] == 1.0
|
|
200
|
+
|
|
201
|
+
|
|
170
202
|
def test_asses_fit(small_data):
|
|
171
203
|
sc_counts, annotations, bulk = small_data
|
|
172
204
|
sc_counts = sc_counts.astype("int")
|
|
@@ -204,12 +236,12 @@ def test_de_analysis(small_data):
|
|
|
204
236
|
# test with sparse matrix
|
|
205
237
|
_ = _de_analysis(sc_pseudo, adata_sparse.X.T, annotations, 0.4, 0.1, False, None, adata.var_names)
|
|
206
238
|
|
|
207
|
-
assert 5 < len(r1) <
|
|
239
|
+
assert 5 < len(r1) < 100
|
|
208
240
|
assert len(r2) == 3
|
|
209
241
|
|
|
210
242
|
|
|
211
243
|
def test_create_bootstrap_signature(small_data):
|
|
212
|
-
bootstraps_per_cell =
|
|
244
|
+
bootstraps_per_cell = RectangleAdvancedParameters().number_of_bootstraps
|
|
213
245
|
sc_counts, annotations, bulk = small_data
|
|
214
246
|
sc_counts = sc_counts.astype("int")
|
|
215
247
|
sc_pseudo = sc_counts.groupby(annotations.values, axis=1).sum()
|
|
@@ -217,3 +249,20 @@ def test_create_bootstrap_signature(small_data):
|
|
|
217
249
|
bootstrap = _create_bootstrap_signature(sc_pseudo, adata.X.T, annotations)
|
|
218
250
|
|
|
219
251
|
assert len(bootstrap.columns) == len(sc_pseudo.columns) * bootstraps_per_cell
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def test_create_bootstrap_signature_with_advanced_parameter(small_data):
|
|
255
|
+
bootstraps_per_cell = 3
|
|
256
|
+
advanced_parameters = RectangleAdvancedParameters(number_of_bootstraps=bootstraps_per_cell)
|
|
257
|
+
sc_counts, annotations, bulk = small_data
|
|
258
|
+
sc_counts = sc_counts.astype("int")
|
|
259
|
+
sc_pseudo = sc_counts.groupby(annotations.values, axis=1).sum()
|
|
260
|
+
adata = AnnData(sc_counts.T, obs=annotations.to_frame(name="cell_type"))
|
|
261
|
+
bootstrap = _create_bootstrap_signature(
|
|
262
|
+
sc_pseudo,
|
|
263
|
+
adata.X.T,
|
|
264
|
+
annotations,
|
|
265
|
+
advanced_parameters.number_of_bootstraps,
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
assert len(bootstrap.columns) == len(sc_pseudo.columns) * bootstraps_per_cell
|
|
@@ -62,7 +62,7 @@ def test_simple_weighted_dampened_deconvolution(quantiseq_data):
|
|
|
62
62
|
corr = np.corrcoef(result, expected)[0, 1]
|
|
63
63
|
rsme = np.sqrt(np.mean((result - expected) ** 2))
|
|
64
64
|
|
|
65
|
-
assert corr > 0.
|
|
65
|
+
assert corr > 0.81 and rsme < 0.01
|
|
66
66
|
|
|
67
67
|
|
|
68
68
|
def test_correct_for_unknown_cell_content(small_data, quantiseq_data):
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/data/quanTIseq_SimRNAseq_read_fractions_small.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|