rectanglepy 1.5.0__tar.gz → 1.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/.bumpversion.cfg +1 -1
  2. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/PKG-INFO +11 -3
  3. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/README.md +9 -1
  4. rectanglepy-1.6.0/docs/_static/rectangle_workflow.png +0 -0
  5. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/api.md +1 -0
  6. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/index.md +1 -0
  7. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/references.bib +9 -0
  8. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/pyproject.toml +1 -1
  9. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/src/rectanglepy/parameters.py +1 -1
  10. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/src/rectanglepy/pp/create_signature.py +27 -25
  11. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/src/rectanglepy/rectangle.py +2 -2
  12. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/src/rectanglepy/tl/deconvolution.py +217 -122
  13. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/tests/test_pp.py +2 -2
  14. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/tests/test_tl.py +1 -1
  15. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/.cruft.json +0 -0
  16. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/.editorconfig +0 -0
  17. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
  18. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/.github/ISSUE_TEMPLATE/config.yml +0 -0
  19. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
  20. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  21. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/.github/workflows/build.yaml +0 -0
  22. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/.github/workflows/release.yaml +0 -0
  23. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/.github/workflows/release_testpypi.yaml +0 -0
  24. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/.github/workflows/test.yaml +0 -0
  25. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/.gitignore +0 -0
  26. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/.pre-commit-config.yaml +0 -0
  27. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/.readthedocs.yaml +0 -0
  28. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/CHANGELOG.md +0 -0
  29. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/CLA.md +0 -0
  30. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/LICENSE +0 -0
  31. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/LICENSE-Commercial.md +0 -0
  32. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/Makefile +0 -0
  33. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/_static/.gitkeep +0 -0
  34. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/_static/rec_logo.001.png +0 -0
  35. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/_templates/.gitkeep +0 -0
  36. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/_templates/autosummary/class.rst +0 -0
  37. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/changelog.md +0 -0
  38. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/conf.py +0 -0
  39. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/contributing.md +0 -0
  40. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/extensions/typed_returns.py +0 -0
  41. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/installation.md +0 -0
  42. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/make.bat +0 -0
  43. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/notebooks/example.ipynb +0 -0
  44. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/references.md +0 -0
  45. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/docs/tutorials.md +0 -0
  46. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/src/rectanglepy/__init__.py +0 -0
  47. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/src/rectanglepy/data/hao1_annotations_small.zip +0 -0
  48. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/src/rectanglepy/data/hao1_counts_small.zip +0 -0
  49. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/src/rectanglepy/data/small_fino_bulks.zip +0 -0
  50. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/src/rectanglepy/pp/__init__.py +0 -0
  51. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/src/rectanglepy/pp/rectangle_signature.py +0 -0
  52. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/src/rectanglepy/tl/__init__.py +0 -0
  53. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/tests/data/TIL10_signature.txt +0 -0
  54. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/tests/data/bulk_small.csv +0 -0
  55. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/tests/data/cell_annotations_small.txt +0 -0
  56. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/tests/data/quanTIseq_SimRNAseq_mixture_smaller.csv +0 -0
  57. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/tests/data/quanTIseq_SimRNAseq_read_fractions_small.txt +0 -0
  58. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/tests/data/sc_object_small.csv +0 -0
  59. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/tests/data/signature_hao1.csv +0 -0
  60. {rectanglepy-1.5.0 → rectanglepy-1.6.0}/tests/test_rectangle.py +0 -0
@@ -1,5 +1,5 @@
1
1
  [bumpversion]
2
- current_version = 1.5.0
2
+ current_version = 1.6.0
3
3
  tag = True
4
4
  commit = True
5
5
 
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: rectanglepy
3
- Version: 1.5.0
3
+ Version: 1.6.0
4
4
  Summary: Hierarchical deconvolution of bulk transcriptomics
5
5
  Project-URL: Documentation, https://rectanglepy.readthedocs.io/
6
6
  Project-URL: Source, https://github.com/ComputationalBiomedicineGroup/Rectangle
@@ -72,6 +72,12 @@ Install the latest release of `Rectangle` from `PyPI` <https://pypi.org/project/
72
72
  pip install rectanglepy
73
73
  ```
74
74
 
75
+ ## How Rectangle works
76
+
77
+ ![Overview of the Rectangle signature-building and multiscale deconvolution workflow](docs/_static/rectangle_workflow.png)
78
+
79
+ Rectangle performs robust multiscale deconvolution informed by single-cell transcriptomics. Rectangle takes annotated scRNA-seq data and a bulk RNA-seq mixture as input. In the signature-building phase, it performs bootstrapped pseudobulking of the scRNA-seq data, followed by signature-gene selection based on log-fold-change (logFC) and p-value, and signature-matrix optimization (minimizing the condition number and correcting for cell-type-specific mRNA content bias). Based on this, it constructs two signature matrices: a “direct signature”, resolving individual cell types, and a “clustered signature” grouping transcriptionally similar cell types. In the deconvolution phase, Rectangle first uses the clustered signature to deconvolve the bulk data into coarse cell-type estimates. These are then used to constrain a second deconvolution step that leverages the direct signature to resolve individual cell-type fractions. Finally, the resulting estimates are scaled to account for unknown cellular content not represented in the reference.
80
+
75
81
  ## License
76
82
 
77
83
  Rectangle is dual-licensed: BSD-3-Clause OR Commercial.
@@ -100,7 +106,9 @@ For commercial licensing: **innovation-psb@uibk.ac.at**
100
106
 
101
107
  ## Citation
102
108
 
103
- > If you use Rectangle in your project, please cite: (TBA)
109
+ If you use Rectangle in your project, please cite:
110
+
111
+ > Eder B, Rigato I, Dietrich A, Merotto L, Sturm G, Treis T, List M, Theis FJ, Finotello F. Rectangle: robust and scalable multiscale deconvolution informed by single-cell RNA sequencing data. bioRxiv. 2026. doi:[10.64898/2026.07.07.736950](https://doi.org/10.64898/2026.07.07.736950)
104
112
 
105
113
  [scverse-discourse]: https://discourse.scverse.org/
106
114
  [issue-tracker]: https://github.com/ComputationalBiomedicineGroup/Rectangle/issues
@@ -33,6 +33,12 @@ Install the latest release of `Rectangle` from `PyPI` <https://pypi.org/project/
33
33
  pip install rectanglepy
34
34
  ```
35
35
 
36
+ ## How Rectangle works
37
+
38
+ ![Overview of the Rectangle signature-building and multiscale deconvolution workflow](docs/_static/rectangle_workflow.png)
39
+
40
+ Rectangle performs robust multiscale deconvolution informed by single-cell transcriptomics. Rectangle takes annotated scRNA-seq data and a bulk RNA-seq mixture as input. In the signature-building phase, it performs bootstrapped pseudobulking of the scRNA-seq data, followed by signature-gene selection based on log-fold-change (logFC) and p-value, and signature-matrix optimization (minimizing the condition number and correcting for cell-type-specific mRNA content bias). Based on this, it constructs two signature matrices: a “direct signature”, resolving individual cell types, and a “clustered signature” grouping transcriptionally similar cell types. In the deconvolution phase, Rectangle first uses the clustered signature to deconvolve the bulk data into coarse cell-type estimates. These are then used to constrain a second deconvolution step that leverages the direct signature to resolve individual cell-type fractions. Finally, the resulting estimates are scaled to account for unknown cellular content not represented in the reference.
41
+
36
42
  ## License
37
43
 
38
44
  Rectangle is dual-licensed: BSD-3-Clause OR Commercial.
@@ -61,7 +67,9 @@ For commercial licensing: **innovation-psb@uibk.ac.at**
61
67
 
62
68
  ## Citation
63
69
 
64
- > If you use Rectangle in your project, please cite: (TBA)
70
+ If you use Rectangle in your project, please cite:
71
+
72
+ > Eder B, Rigato I, Dietrich A, Merotto L, Sturm G, Treis T, List M, Theis FJ, Finotello F. Rectangle: robust and scalable multiscale deconvolution informed by single-cell RNA sequencing data. bioRxiv. 2026. doi:[10.64898/2026.07.07.736950](https://doi.org/10.64898/2026.07.07.736950)
65
73
 
66
74
  [scverse-discourse]: https://discourse.scverse.org/
67
75
  [issue-tracker]: https://github.com/ComputationalBiomedicineGroup/Rectangle/issues
@@ -12,4 +12,5 @@
12
12
  pp.build_rectangle_signatures
13
13
  tl.deconvolution
14
14
  pp.RectangleSignatureResult
15
+ RectangleAdvancedParameters
15
16
  ```
@@ -1,4 +1,5 @@
1
1
  ```{include} ../README.md
2
+ :relative-images:
2
3
 
3
4
  ```
4
5
 
@@ -16,6 +16,15 @@
16
16
  url = {https://doi.org/10.1186/s13059-017-1382-0}
17
17
  }
18
18
 
19
+ @article{Eder2026,
20
+ author = {Bernhard Eder and Irene Rigato and Alexander Dietrich and Lorenzo Merotto and Gregor Sturm and Tim Treis and Markus List and Fabian J. Theis and Francesca Finotello},
21
+ title = {Rectangle: robust and scalable multiscale deconvolution informed by single-cell RNA sequencing data},
22
+ journal = {bioRxiv},
23
+ year = {2026},
24
+ doi = {10.64898/2026.07.07.736950},
25
+ url = {https://doi.org/10.64898/2026.07.07.736950}
26
+ }
27
+
19
28
  @article{Finotello2019,
20
29
  author = {Francesca Finotello and Gregor Mayer and Christian Plattner and Cornelia Laschober and Dietmar Rieder and Peter Hackl and Lukas Krogsdam and Michael L. T. Helmberg and Zlatko Trajanoski},
21
30
  title = {Molecular and pharmacological modulators of the tumor immune contexture revealed by deconvolution of RNA-seq data},
@@ -4,7 +4,7 @@ requires = ["hatchling"]
4
4
 
5
5
  [project]
6
6
  name = "rectanglepy"
7
- version = "1.5.0"
7
+ version = "1.6.0"
8
8
  description = "Hierarchical deconvolution of bulk transcriptomics"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -7,5 +7,5 @@ from dataclasses import dataclass
7
7
  class RectangleAdvancedParameters:
8
8
  """Advanced parameters for Rectangle internals."""
9
9
 
10
- number_of_bootstraps: int = 7
10
+ number_of_bootstraps: int = 20
11
11
  grid_search_split_size: int = 50
@@ -134,25 +134,21 @@ def _run_deseq2(
134
134
  sc_data,
135
135
  annotations: pd.Series,
136
136
  n_cpus: int = None,
137
- gene_expression_threshold=0.5,
138
- number_of_bootstraps: int = 7,
137
+ gene_expression_threshold=0.4,
138
+ number_of_bootstraps: int = 20,
139
139
  ) -> dict[str | int, pd.DataFrame]:
140
140
  results = {}
141
141
  inference = DefaultInference(n_cpus=n_cpus)
142
142
  bootstrapped_signature = _create_bootstrap_signature(countsig, sc_data, annotations, number_of_bootstraps)
143
143
  np.random.seed(42)
144
144
  for _i, cell_type in enumerate(countsig.columns):
145
- bootstrapped_signature_copy = bootstrapped_signature.copy()
146
- countsig_copy = countsig.copy()
147
145
  sc_data_filtered = sc_data.T[annotations == cell_type]
148
146
  expressed_cells = (sc_data_filtered > 0).sum(axis=0)
149
147
  if expressed_cells.ndim > 1: # needed for sparse matrices
150
148
  expressed_cells = np.squeeze(np.asarray(expressed_cells))
151
- # make dense out of sparse
152
- sc_data_filtered = sc_data_filtered.toarray()
153
149
  threshold = gene_expression_threshold * sc_data_filtered.shape[0]
154
- genes = countsig_copy.index[expressed_cells > threshold].tolist()
155
- bootstrapped_signature_copy = bootstrapped_signature_copy.loc[genes].T
150
+ genes = countsig.index[expressed_cells > threshold].tolist()
151
+ bootstrapped_signature_copy = bootstrapped_signature.loc[genes].T
156
152
  logger.info(f"Running DE analysis for {cell_type}")
157
153
  condition = ["B" if (cell_type + "_") in x else "A" for x in bootstrapped_signature_copy.index]
158
154
  clinical_df = pd.DataFrame({"condition": condition}, index=bootstrapped_signature_copy.index)
@@ -173,24 +169,29 @@ def _run_deseq2(
173
169
  return results
174
170
 
175
171
 
176
- def _create_bootstrap_signature(countsig, sc_data, annotations, number_of_bootstraps: int = 7) -> pd.DataFrame:
177
- if scipy.sparse.issparse(sc_data):
178
- sc_data = sc_data.toarray()
172
+ def _create_bootstrap_signature(countsig, sc_data, annotations, number_of_bootstraps: int = 20) -> pd.DataFrame:
173
+ # Cells are only ever summed here, never read one by one, so sparse input is kept sparse:
174
+ # densifying the whole matrix costs genes * cells * 8 bytes, which is gigabytes on an
175
+ # atlas-sized reference. Dense input deliberately stays dense - gathering rows out of a
176
+ # dense array is faster than routing them through a sparse format.
177
+ cells_by_gene = sc_data.T.tocsr() if scipy.sparse.issparse(sc_data) else sc_data.T
179
178
  celltypes = countsig.columns
180
- bootstrapped_signature = pd.DataFrame()
181
- samples_per_bootstrap = 500
179
+ columns = {}
180
+ samples_per_bootstrap = 1000
182
181
  np.random.seed(42)
183
182
  for celltype in celltypes:
184
- sc_data_filtered = sc_data.T[annotations == celltype]
183
+ sc_data_filtered = cells_by_gene[(annotations == celltype).to_numpy()]
184
+ # sparse matrices do not support len(); shape[0] is the cell count for both formats
185
+ number_of_cells = sc_data_filtered.shape[0]
185
186
  for i in range(number_of_bootstraps):
186
- selected_rows = np.random.choice(len(sc_data_filtered), samples_per_bootstrap, replace=True)
187
+ selected_rows = np.random.choice(number_of_cells, samples_per_bootstrap, replace=True)
187
188
  summed_rows = sc_data_filtered[selected_rows].sum(axis=0)
188
- bootstrapped_signature[f"{celltype}_{i}"] = list(summed_rows)
189
- bootstrapped_signature.index = countsig.index
189
+ # a sparse sum is a (1, genes) matrix, a dense one a flat array
190
+ columns[f"{celltype}_{i}"] = np.asarray(summed_rows).ravel()
191
+ # built in one go: assigning column by column repeatedly reallocates the frame
192
+ bootstrapped_signature = pd.DataFrame(columns, index=countsig.index)
190
193
  # to int
191
- samples_per_bootstrap = samples_per_bootstrap / 2.5
192
- bootstrapped_signature = bootstrapped_signature.astype(int)
193
- return bootstrapped_signature
194
+ return bootstrapped_signature.astype(int)
194
195
 
195
196
 
196
197
  def _de_analysis(
@@ -202,7 +203,7 @@ def _de_analysis(
202
203
  optimize_cutoffs: bool,
203
204
  n_cpus: int = None,
204
205
  genes=None,
205
- gene_expression_threshold=0.5,
206
+ gene_expression_threshold=0.4,
206
207
  advanced_parameters: RectangleAdvancedParameters = None,
207
208
  ) -> tuple[Series, dict[str, [str]] :, DataFrame | None]:
208
209
  logger.info("Starting DE analysis")
@@ -307,7 +308,7 @@ def build_rectangle_signatures(
307
308
  p=0.015,
308
309
  lfc=1.5,
309
310
  n_cpus: int = None,
310
- gene_expression_threshold=0.5,
311
+ gene_expression_threshold=0.4,
311
312
  advanced_parameters: RectangleAdvancedParameters = None,
312
313
  ) -> RectangleSignatureResult:
313
314
  r"""Builds rectangle signatures based on single-cell count data and annotations.
@@ -333,7 +334,7 @@ def build_rectangle_signatures(
333
334
  n_cpus
334
335
  The number of cpus to use for the DE analysis. Defaults to the number of cpus available.
335
336
  gene_expression_threshold
336
- The gene expression threshold for the DE analysis. How many cells need to express a gene to be considered in DGE
337
+ The gene expression threshold for the DE analysis. The fraction of cells that must express a gene to be considered in DGE. Defaults to 0.4
337
338
  advanced_parameters
338
339
  Optional advanced Rectangle parameters. Defaults are used when not provided.
339
340
 
@@ -425,9 +426,10 @@ def build_rectangle_signatures(
425
426
  def _create_pseudo_count_sig(sc_counts: np.ndarray, annotations: pd.Series, var_names) -> pd.DataFrame:
426
427
  unique_labels, label_indices = np.unique(annotations, return_inverse=True)
427
428
  grouped_sum = np.zeros((len(unique_labels), sc_counts.shape[0]))
429
+ cells_by_gene = sc_counts.T
428
430
  for i, _label in enumerate(unique_labels):
429
431
  label_columns = label_indices == i
430
- grouped_sum[i, :] = np.sum(sc_counts.T[label_columns, :], axis=0)
432
+ grouped_sum[i, :] = np.sum(cells_by_gene[label_columns, :], axis=0)
431
433
  grouped_sum = grouped_sum.T
432
434
 
433
435
  grouped_sum = pd.DataFrame(grouped_sum, index=var_names, columns=unique_labels).astype(int)
@@ -443,7 +445,7 @@ def _optimize_parameters(
443
445
  grid_search_split_size: int = 50,
444
446
  ) -> pd.DataFrame:
445
447
  # search space for p and lfc
446
- lfcs = [x / 100 for x in range(160, 230, 10)]
448
+ lfcs = [1.5, 1.75, 2.0, 2.25, 2.5, 2.75, 3.0]
447
449
  ps = [x / 1000 for x in range(50, 51, 1)]
448
450
 
449
451
  results = []
@@ -24,7 +24,7 @@ def rectangle(
24
24
  p=0.015,
25
25
  lfc=1.5,
26
26
  n_cpus: int = None,
27
- gene_expression_threshold=0.5,
27
+ gene_expression_threshold=0.4,
28
28
  advanced_parameters: RectangleAdvancedParameters = None,
29
29
  ) -> tuple[DataFrame, RectangleSignatureResult]:
30
30
  r"""All in one deconvolution method. Creates signatures and deconvolutes the bulk data. Has options for subsampling and consensus runs.
@@ -52,7 +52,7 @@ def rectangle(
52
52
  correct_mrna_bias : bool
53
53
  A flag indicating whether to correct for mRNA bias. Defaults to True.
54
54
  gene_expression_threshold : float
55
- The threshold for gene expression. Genes with expression below this threshold are removed from the analysis.
55
+ The threshold for gene expression. Genes must be expressed in at least this fraction of cells. Defaults to 0.4.
56
56
  advanced_parameters
57
57
  Optional advanced Rectangle parameters. Defaults are used when not provided.
58
58
 
@@ -7,18 +7,166 @@ import numpy as np
7
7
  import osqp
8
8
  import pandas as pd
9
9
  import scipy.sparse as sp
10
- import statsmodels.api as sm
11
10
  from joblib import Parallel, delayed, parallel_backend
12
11
  from loguru import logger
13
12
 
14
13
  from rectanglepy.pp.rectangle_signature import RectangleSignatureResult
15
14
 
15
+ # OSQP can return noticeably different solutions from active-set QP when P is singular/ill-conditioned.
16
+ # A tiny ridge makes the problem strictly convex and closer to quadprog behavior.
17
+ QP_RIDGE = 1e-8
18
+
16
19
 
17
20
  def _scale_weights(weights: np.ndarray) -> np.ndarray:
18
21
  min_weight = np.nextafter(min(weights), np.float64(1.0)) # prevent division by zero
19
22
  return weights / min_weight
20
23
 
21
24
 
25
+ def _upper_triangular_csc_pattern(n_vars: int) -> tuple[np.ndarray, np.ndarray]:
26
+ """Returns the (indices, indptr) of a fully populated upper triangular CSC matrix.
27
+
28
+ OSQP only stores the upper triangle of P, so handing it the triangle directly saves the
29
+ conversion of the full symmetric matrix on every solve.
30
+ """
31
+ indices = np.concatenate([np.arange(column + 1) for column in range(n_vars)]).astype(np.int32)
32
+ indptr = np.concatenate(([0], np.cumsum(np.arange(1, n_vars + 1)))).astype(np.int32)
33
+ return indices, indptr
34
+
35
+
36
+ class _QuadraticProgram:
37
+ """A deconvolution QP for one (signature, bulk) pair, solvable at several dampening weights.
38
+
39
+ Everything that does not depend on the weights - the numpy views of the inputs and the constraint
40
+ matrix with its bounds - is built once and shared by all solves, which keeps it out of the
41
+ dampened least squares iteration.
42
+
43
+ Note that each solve still gets its own OSQP workspace: reusing one across solves makes OSQP
44
+ keep the problem scaling of the first objective, which perturbs the fixed point the iteration
45
+ converges to (and is not faster at these problem sizes).
46
+ """
47
+
48
+ def __init__(
49
+ self,
50
+ signature: pd.DataFrame,
51
+ bulk: pd.Series,
52
+ prev_assignments: list[int | str] = None,
53
+ prev_solution: pd.Series = None,
54
+ ):
55
+ if not signature.index.equals(bulk.index):
56
+ bulk = bulk.reindex(signature.index)
57
+ self.signature = signature.to_numpy(dtype=np.float64)
58
+ self.bulk = bulk.to_numpy(dtype=np.float64)
59
+ self.n_vars = self.signature.shape[1] # number of cell types / fractions
60
+
61
+ self._triu_indices, self._triu_indptr = _upper_triangular_csc_pattern(self.n_vars)
62
+ self._triu_selection = np.tril_indices(self.n_vars) # reads P.T column-major upper triangle
63
+ self._constraints, self._lower_bounds = self._build_constraints(prev_assignments, prev_solution)
64
+ self._upper_bounds = np.full_like(self._lower_bounds, np.inf, dtype=np.float64)
65
+
66
+ def _build_constraints(
67
+ self, prev_assignments: list[int | str], prev_solution: pd.Series
68
+ ) -> tuple[sp.csc_matrix, np.ndarray]:
69
+ # ----- Constraints in the original quadprog form: C.T x >= b
70
+ # We build C (n_vars, n_constraints) then map to OSQP: A = C.T, l=b, u=+inf
71
+ C_cols = []
72
+ b_list = []
73
+
74
+ # Constraint 1: sum(x) <= 1 -> -sum(x) >= -1
75
+ C_cols.append(-np.ones((self.n_vars, 1), dtype=np.float64))
76
+ b_list.append(np.array([-1.0], dtype=np.float64))
77
+
78
+ # Constraint 2: x >= 0 -> I x >= 0
79
+ C_cols.append(np.eye(self.n_vars, dtype=np.float64))
80
+ b_list.append(np.zeros(self.n_vars, dtype=np.float64))
81
+
82
+ # Constraint 3: keep close to prev_solution (this encoding already turns upper bounds into >= via negation)
83
+ if prev_solution is not None:
84
+ if prev_assignments is None:
85
+ raise ValueError("prev_assignments must be provided when prev_solution is provided.")
86
+
87
+ for cluster in prev_solution.index:
88
+ # x_cluster <= upper -> -x_cluster >= -upper
89
+ C_upper = np.array(
90
+ [-1.0 if str(x) == str(cluster) else 0.0 for x in prev_assignments],
91
+ dtype=np.float64,
92
+ ).reshape(-1, 1)
93
+
94
+ # x_cluster >= lower -> +x_cluster >= +lower
95
+ C_lower = np.array(
96
+ [1.0 if str(x) == str(cluster) else 0.0 for x in prev_assignments],
97
+ dtype=np.float64,
98
+ ).reshape(-1, 1)
99
+
100
+ prev_weight = float(prev_solution.loc[cluster])
101
+ upper = min(1.0, prev_weight + 0.03)
102
+ lower = max(0.0, prev_weight - 0.03)
103
+
104
+ C_cols.append(C_upper)
105
+ b_list.append(np.array([-upper], dtype=np.float64))
106
+
107
+ C_cols.append(C_lower)
108
+ b_list.append(np.array([lower], dtype=np.float64))
109
+
110
+ C = np.concatenate(C_cols, axis=1) # (n_vars, n_constraints)
111
+ b = np.concatenate(b_list, axis=0).astype(np.float64) # (n_constraints,)
112
+
113
+ # OSQP uses l <= A x <= u
114
+ return sp.csc_matrix(C.T), b
115
+
116
+ def _objective(self, gld: np.ndarray, multiplier: int) -> tuple[np.ndarray, np.ndarray]:
117
+ # Minimize 1/2 x^T G x - a^T x
118
+ if multiplier is None:
119
+ G = self.signature.T @ self.signature
120
+ a = self.signature.T @ self.bulk
121
+ else:
122
+ weights = np.square(1 / (self.signature @ gld))
123
+ weights_dampened = np.clip(_scale_weights(weights), None, multiplier)
124
+ # equivalent to signature.T @ diag(weights) @ signature, without the dense (genes, genes) matrix
125
+ G = self.signature.T @ (weights_dampened[:, None] * self.signature)
126
+ a = self.signature.T @ (weights_dampened * self.bulk)
127
+
128
+ # ----- Map to OSQP: minimize 1/2 x^T P x + q^T x -> q = -a
129
+ P = G
130
+ q = -a
131
+
132
+ # Optional scaling
133
+ scale = np.linalg.norm(P)
134
+ if scale > 0:
135
+ P = P / scale
136
+ q = q / scale
137
+
138
+ P = ((P + P.T) / 2.0) + QP_RIDGE * np.eye(self.n_vars, dtype=np.float64)
139
+ return P, q
140
+
141
+ def solve(self, gld: np.ndarray = None, multiplier: int = None) -> np.ndarray:
142
+ P, q = self._objective(gld, multiplier)
143
+ P_data = P.T[self._triu_selection] # upper triangle in CSC (column major) order
144
+ P_sp = sp.csc_matrix((P_data, self._triu_indices, self._triu_indptr), shape=(self.n_vars, self.n_vars))
145
+
146
+ solver = osqp.OSQP()
147
+ solver.setup(
148
+ P=P_sp,
149
+ q=q,
150
+ A=self._constraints,
151
+ l=self._lower_bounds,
152
+ u=self._upper_bounds,
153
+ verbose=False,
154
+ eps_abs=1e-7,
155
+ eps_rel=1e-7,
156
+ max_iter=50000,
157
+ polish=True,
158
+ warm_start=False,
159
+ scaled_termination=False,
160
+ )
161
+
162
+ res = solver.solve()
163
+
164
+ if res.info.status_val not in (1,): # 1 = solved
165
+ raise RuntimeError(f"OSQP did not solve the problem: {res.info.status}")
166
+
167
+ return res.x
168
+
169
+
22
170
  def solve_qp(
23
171
  signature: pd.DataFrame,
24
172
  bulk: pd.Series,
@@ -52,143 +200,89 @@ def solve_qp(
52
200
  Notes
53
201
  -----
54
202
  This function uses quadratic programming to solve the deconvolution problem. The objective is to minimize the difference between the observed bulk data and the data predicted by the signature matrix and the cell fractions. The function also includes constraints to ensure that the cell fractions are non-negative and sum to 1, and to make the solution similar to the previous assignments and weights if they are provided.
203
+
204
+ Callers that solve the same problem at several dampening weights (e.g. the dampened least squares
205
+ iteration) should use :class:`_QuadraticProgram` directly, which shares the weight independent
206
+ parts of the problem across solves.
55
207
  """
56
- # ------------------ QP-based deconvolution
57
- # Minimize 1/2 x^T G x - a^T x
58
- # Subject to C.T x >= b
59
- if multiplier is None:
60
- a = (signature.T @ bulk).to_numpy(dtype=np.float64)
61
- G = (signature.T @ signature).to_numpy(dtype=np.float64)
62
- else:
63
- weights = np.square(1 / (signature @ gld))
64
- weights_dampened = np.clip(_scale_weights(weights), None, multiplier)
65
- W = np.diag(np.asarray(weights_dampened, dtype=np.float64))
66
- G = (signature.T.to_numpy() @ (W @ signature.to_numpy())).astype(np.float64)
67
- a = (signature.T.to_numpy() @ (W @ bulk.to_numpy())).astype(np.float64)
68
-
69
- n_vars = G.shape[0] # number of cell types / fractions
70
-
71
- # ----- Constraints in your original quadprog form: C.T x >= b
72
- # We'll build C (n_vars, n_constraints) then map to OSQP: A = C.T, l=b, u=+inf
73
- C_cols = []
74
- b_list = []
75
-
76
- # Constraint 1: sum(x) <= 1 -> -sum(x) >= -1
77
- C_cols.append(-np.ones((n_vars, 1), dtype=np.float64))
78
- b_list.append(np.array([-1.0], dtype=np.float64))
79
-
80
- # Constraint 2: x >= 0 -> I x >= 0
81
- C_cols.append(np.eye(n_vars, dtype=np.float64))
82
- b_list.append(np.zeros(n_vars, dtype=np.float64))
83
-
84
- # Constraint 3: keep close to prev_solution (your encoding already turns upper bounds into >= via negation)
85
- if prev_solution is not None:
86
- if prev_assignments is None:
87
- raise ValueError("prev_assignments must be provided when prev_solution is provided.")
88
-
89
- for cluster in prev_solution.index:
90
- # x_cluster <= upper -> -x_cluster >= -upper
91
- C_upper = np.array(
92
- [-1.0 if str(x) == str(cluster) else 0.0 for x in prev_assignments],
93
- dtype=np.float64,
94
- ).reshape(-1, 1)
95
-
96
- # x_cluster >= lower -> +x_cluster >= +lower
97
- C_lower = np.array(
98
- [1.0 if str(x) == str(cluster) else 0.0 for x in prev_assignments],
99
- dtype=np.float64,
100
- ).reshape(-1, 1)
101
-
102
- prev_weight = float(prev_solution.loc[cluster])
103
- upper = min(1.0, prev_weight + 0.03)
104
- lower = max(0.0, prev_weight - 0.03)
105
-
106
- C_cols.append(C_upper)
107
- b_list.append(np.array([-upper], dtype=np.float64))
108
-
109
- C_cols.append(C_lower)
110
- b_list.append(np.array([lower], dtype=np.float64))
111
-
112
- C = np.concatenate(C_cols, axis=1) # (n_vars, n_constraints)
113
- b = np.concatenate(b_list, axis=0).astype(np.float64) # (n_constraints,)
114
-
115
- # ----- Map to OSQP: minimize 1/2 x^T P x + q^T x
116
- # Your objective: 1/2 x^T G x - a^T x -> q = -a
117
- P = G
118
- q = -a
119
-
120
- # Optional scaling
121
- scale = np.linalg.norm(P)
122
- if scale > 0:
123
- P = P / scale
124
- q = q / scale
125
-
126
- # OSQP can return noticeably different solutions from active-set QP when P is singular/ill-conditioned.
127
- # Add tiny ridge to make the problem strictly convex and closer to quadprog behavior.
128
- ridge = 1e-8
129
- P = ((P + P.T) / 2.0) + ridge * np.eye(n_vars, dtype=np.float64)
130
-
131
- # OSQP uses l <= A x <= u
132
- A = C.T # (n_constraints, n_vars)
133
- l = b
134
- u = np.full_like(l, np.inf, dtype=np.float64)
135
-
136
- # Sparse matrices (required/expected)
137
- P_sp = sp.csc_matrix(P)
138
- A_sp = sp.csc_matrix(A)
139
-
140
- solver = osqp.OSQP()
141
- solver.setup(
142
- P=P_sp,
143
- q=q,
144
- A=A_sp,
145
- l=l,
146
- u=u,
147
- verbose=False,
148
- eps_abs=1e-7,
149
- eps_rel=1e-7,
150
- max_iter=50000,
151
- polish=True,
152
- warm_start=False,
153
- scaled_termination=False,
154
- )
208
+ problem = _QuadraticProgram(signature, bulk, prev_assignments, prev_solution)
209
+ return problem.solve(gld, multiplier)
210
+
211
+
212
+ def _batched_wls(
213
+ design: np.ndarray,
214
+ response: np.ndarray,
215
+ weights: np.ndarray,
216
+ subsets: np.ndarray,
217
+ max_bytes: int = 64 * 1024**2,
218
+ ) -> np.ndarray:
219
+ """Fits one weighted least squares model per gene subset in batch.
220
+
221
+ Mirrors ``statsmodels.api.WLS(...).fit()``: the design and the response are whitened with the
222
+ square root of the weights and the parameters are read off the pseudo inverse of the whitened
223
+ design. Stacking the subsets lets numpy loop over them in C instead of Python.
155
224
 
156
- res = solver.solve()
225
+ Parameters
226
+ ----------
227
+ design : np.ndarray
228
+ The (genes, cell types) design matrix.
229
+ response : np.ndarray
230
+ The (genes,) response vector.
231
+ weights : np.ndarray
232
+ The (genes,) regression weights.
233
+ subsets : np.ndarray
234
+ A (n_subsets, subset size) array of gene indices, one row per model.
235
+ max_bytes : int
236
+ Upper bound on the size of a single whitened design batch, used to chunk the subsets.
237
+
238
+ Returns
239
+ -------
240
+ np.ndarray
241
+ A (n_subsets, cell types) array of fitted parameters.
242
+ """
243
+ n_subsets, subset_size = subsets.shape
244
+ n_params = design.shape[1]
245
+ params = np.empty((n_subsets, n_params), dtype=np.float64)
157
246
 
158
- if res.info.status_val not in (1,): # 1 = solved
159
- raise RuntimeError(f"OSQP did not solve the problem: {res.info.status}")
247
+ chunk_size = max(1, int(max_bytes // (subset_size * n_params * design.itemsize)))
248
+ for start in range(0, n_subsets, chunk_size):
249
+ chunk = subsets[start : start + chunk_size]
250
+ sqrt_weights = np.sqrt(weights[chunk]) # (chunk, subset size)
251
+ whitened_design = design[chunk] * sqrt_weights[:, :, None] # (chunk, subset size, cell types)
252
+ whitened_response = response[chunk] * sqrt_weights
253
+ params[start : start + chunk_size] = np.einsum("bpg,bg->bp", np.linalg.pinv(whitened_design), whitened_response)
160
254
 
161
- return res.x
255
+ return params
162
256
 
163
257
 
164
258
  def _calculate_dampening_constant(signature: pd.DataFrame, bulk: pd.Series, qp_gld: np.ndarray) -> int:
165
259
  solutions_std = []
166
260
  np.random.seed(1)
167
- weights = np.square(1 / (np.dot(signature, qp_gld)))
261
+ signature_values = np.asarray(signature, dtype=np.float64)
262
+ bulk_values = np.asarray(bulk, dtype=np.float64)
263
+ n_genes = signature_values.shape[0]
264
+ weights = np.square(1 / (signature_values @ qp_gld))
168
265
  weights_scaled = _scale_weights(weights)
169
266
  weights_scaled_no_inf = weights_scaled[weights_scaled != np.inf]
170
267
  qp_gld_sum = sum(qp_gld)
268
+ subset_size = n_genes // 2
269
+ n_subsets = 100
171
270
  # try multiple values of the dampening constant (multiplier)
172
271
  # for each, calculate the variance of the dampened weighted solution for a subset of genes
173
272
  max_range = 40
174
273
  multiplier_range = min(max_range, math.ceil(np.log2(max(weights_scaled_no_inf))))
175
274
  for i in range(multiplier_range):
176
- solutions = []
177
275
  multiplier = 2**i
178
- weights_dampened = np.array([multiplier if multiplier <= x else x for x in weights_scaled]).astype("double")
179
- for _ in range(100):
180
- subset = np.random.choice(len(signature), size=len(signature) // 2, replace=False)
181
- bulk_subset = bulk.iloc[list(subset)]
182
- signature_subset = signature.iloc[subset, :]
183
- fit = sm.WLS(bulk_subset, -1 + signature_subset, weights=weights_dampened[subset]).fit()
184
- solution = fit.params * qp_gld_sum / sum(fit.params)
185
- solutions.append(solution)
186
- solutions_df = pd.DataFrame(solutions)
187
-
188
- solutions_std.append(solutions_df.std(axis=0))
189
- solutions_std_df = pd.DataFrame(solutions_std)
190
- means = solutions_std_df.apply(lambda x: np.mean(x**2), axis=1)
191
- best_dampening_constant = means.idxmin()
276
+ weights_dampened = np.minimum(weights_scaled, multiplier)
277
+ subsets = np.array([np.random.choice(n_genes, size=subset_size, replace=False) for _ in range(n_subsets)])
278
+ # WLS uses the signature directly; no intercept column is added.
279
+ params = _batched_wls(signature_values, bulk_values, weights_dampened, subsets)
280
+ solutions = params * qp_gld_sum / params.sum(axis=1, keepdims=True)
281
+
282
+ solutions_std.append(np.nanstd(solutions, axis=0, ddof=1))
283
+ solutions_std_array = np.array(solutions_std)
284
+ means = np.nanmean(np.square(solutions_std_array), axis=1)
285
+ best_dampening_constant = int(np.nanargmin(means))
192
286
  return best_dampening_constant
193
287
 
194
288
 
@@ -202,7 +296,8 @@ def _calculate_ls(
202
296
  signature = signature.loc[genes].sort_index()
203
297
  bulk = bulk.loc[genes].sort_index().astype("double")
204
298
 
205
- approximate_solution = solve_qp(signature, bulk, prev_assignments, prev_weights)
299
+ problem = _QuadraticProgram(signature, bulk, prev_assignments, prev_weights)
300
+ approximate_solution = problem.solve()
206
301
  dampening_constant = _calculate_dampening_constant(signature, bulk, approximate_solution)
207
302
  multiplier = 2**dampening_constant
208
303
 
@@ -212,7 +307,7 @@ def _calculate_ls(
212
307
  iterations = 2
213
308
  solutions_sum = approximate_solution
214
309
  while (change > convergence_threshold) and (iterations < max_iterations):
215
- dampened_solution = solve_qp(signature, bulk, prev_assignments, prev_weights, approximate_solution, multiplier)
310
+ dampened_solution = problem.solve(approximate_solution, multiplier)
216
311
  solutions_sum += dampened_solution
217
312
  solution_averages = solutions_sum / iterations
218
313
  change = np.linalg.norm(solution_averages - approximate_solution, 1)
@@ -236,12 +236,12 @@ def test_de_analysis(small_data):
236
236
  # test with sparse matrix
237
237
  _ = _de_analysis(sc_pseudo, adata_sparse.X.T, annotations, 0.4, 0.1, False, None, adata.var_names)
238
238
 
239
- assert 5 < len(r1) < 50
239
+ assert 5 < len(r1) < 100
240
240
  assert len(r2) == 3
241
241
 
242
242
 
243
243
  def test_create_bootstrap_signature(small_data):
244
- bootstraps_per_cell = 7
244
+ bootstraps_per_cell = RectangleAdvancedParameters().number_of_bootstraps
245
245
  sc_counts, annotations, bulk = small_data
246
246
  sc_counts = sc_counts.astype("int")
247
247
  sc_pseudo = sc_counts.groupby(annotations.values, axis=1).sum()
@@ -62,7 +62,7 @@ def test_simple_weighted_dampened_deconvolution(quantiseq_data):
62
62
  corr = np.corrcoef(result, expected)[0, 1]
63
63
  rsme = np.sqrt(np.mean((result - expected) ** 2))
64
64
 
65
- assert corr > 0.815 and rsme < 0.011
65
+ assert corr > 0.81 and rsme < 0.01
66
66
 
67
67
 
68
68
  def test_correct_for_unknown_cell_content(small_data, quantiseq_data):
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes