rectanglepy 1.4.2__tar.gz → 1.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.bumpversion.cfg +1 -1
  2. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.readthedocs.yaml +1 -1
  3. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/PKG-INFO +11 -3
  4. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/README.md +9 -1
  5. rectanglepy-1.6.0/docs/_static/rectangle_workflow.png +0 -0
  6. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/api.md +1 -0
  7. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/index.md +1 -0
  8. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/references.bib +9 -0
  9. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/pyproject.toml +1 -1
  10. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/__init__.py +2 -1
  11. rectanglepy-1.6.0/src/rectanglepy/parameters.py +11 -0
  12. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/pp/create_signature.py +75 -34
  13. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/rectangle.py +8 -3
  14. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/tl/deconvolution.py +217 -122
  15. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/test_pp.py +51 -2
  16. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/test_tl.py +1 -1
  17. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.cruft.json +0 -0
  18. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.editorconfig +0 -0
  19. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
  20. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/ISSUE_TEMPLATE/config.yml +0 -0
  21. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
  22. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  23. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/workflows/build.yaml +0 -0
  24. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/workflows/release.yaml +0 -0
  25. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/workflows/release_testpypi.yaml +0 -0
  26. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.github/workflows/test.yaml +0 -0
  27. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.gitignore +0 -0
  28. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/.pre-commit-config.yaml +0 -0
  29. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/CHANGELOG.md +0 -0
  30. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/CLA.md +0 -0
  31. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/LICENSE +0 -0
  32. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/LICENSE-Commercial.md +0 -0
  33. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/Makefile +0 -0
  34. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/_static/.gitkeep +0 -0
  35. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/_static/rec_logo.001.png +0 -0
  36. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/_templates/.gitkeep +0 -0
  37. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/_templates/autosummary/class.rst +0 -0
  38. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/changelog.md +0 -0
  39. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/conf.py +0 -0
  40. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/contributing.md +0 -0
  41. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/extensions/typed_returns.py +0 -0
  42. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/installation.md +0 -0
  43. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/make.bat +0 -0
  44. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/notebooks/example.ipynb +0 -0
  45. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/references.md +0 -0
  46. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/docs/tutorials.md +0 -0
  47. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/data/hao1_annotations_small.zip +0 -0
  48. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/data/hao1_counts_small.zip +0 -0
  49. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/data/small_fino_bulks.zip +0 -0
  50. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/pp/__init__.py +0 -0
  51. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/pp/rectangle_signature.py +0 -0
  52. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/src/rectanglepy/tl/__init__.py +0 -0
  53. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/data/TIL10_signature.txt +0 -0
  54. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/data/bulk_small.csv +0 -0
  55. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/data/cell_annotations_small.txt +0 -0
  56. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/data/quanTIseq_SimRNAseq_mixture_smaller.csv +0 -0
  57. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/data/quanTIseq_SimRNAseq_read_fractions_small.txt +0 -0
  58. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/data/sc_object_small.csv +0 -0
  59. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/data/signature_hao1.csv +0 -0
  60. {rectanglepy-1.4.2 → rectanglepy-1.6.0}/tests/test_rectangle.py +0 -0
@@ -1,5 +1,5 @@
1
1
  [bumpversion]
2
- current_version = 1.4.2
2
+ current_version = 1.6.0
3
3
  tag = True
4
4
  commit = True
5
5
 
@@ -1,7 +1,7 @@
1
1
  # https://docs.readthedocs.io/en/stable/config-file/v2.html
2
2
  version: 2
3
3
  build:
4
- os: ubuntu-20.04
4
+ os: ubuntu-24.04
5
5
  tools:
6
6
  python: "3.10"
7
7
  sphinx:
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: rectanglepy
3
- Version: 1.4.2
3
+ Version: 1.6.0
4
4
  Summary: Hierarchical deconvolution of bulk transcriptomics
5
5
  Project-URL: Documentation, https://rectanglepy.readthedocs.io/
6
6
  Project-URL: Source, https://github.com/ComputationalBiomedicineGroup/Rectangle
@@ -72,6 +72,12 @@ Install the latest release of `Rectangle` from `PyPI` <https://pypi.org/project/
72
72
  pip install rectanglepy
73
73
  ```
74
74
 
75
+ ## How Rectangle works
76
+
77
+ ![Overview of the Rectangle signature-building and multiscale deconvolution workflow](docs/_static/rectangle_workflow.png)
78
+
79
+ Rectangle performs robust multiscale deconvolution informed by single-cell transcriptomics. Rectangle takes annotated scRNA-seq data and a bulk RNA-seq mixture as input. In the signature-building phase, it performs bootstrapped pseudobulking of the scRNA-seq data, followed by signature-gene selection based on log-fold-change (logFC) and p-value, and signature-matrix optimization (minimizing the condition number and correcting for cell-type-specific mRNA content bias). Based on this, it constructs two signature matrices: a “direct signature”, resolving individual cell types, and a “clustered signature” grouping transcriptionally similar cell types. In the deconvolution phase, Rectangle first uses the clustered signature to deconvolve the bulk data into coarse cell-type estimates. These are then used to constrain a second deconvolution step that leverages the direct signature to resolve individual cell-type fractions. Finally, the resulting estimates are scaled to account for unknown cellular content not represented in the reference.
80
+
75
81
  ## License
76
82
 
77
83
  Rectangle is dual-licensed: BSD-3-Clause OR Commercial.
@@ -100,7 +106,9 @@ For commercial licensing: **innovation-psb@uibk.ac.at**
100
106
 
101
107
  ## Citation
102
108
 
103
- > If you use Rectangle in your project, please cite: (TBA)
109
+ If you use Rectangle in your project, please cite:
110
+
111
+ > Eder B, Rigato I, Dietrich A, Merotto L, Sturm G, Treis T, List M, Theis FJ, Finotello F. Rectangle: robust and scalable multiscale deconvolution informed by single-cell RNA sequencing data. bioRxiv. 2026. doi:[10.64898/2026.07.07.736950](https://doi.org/10.64898/2026.07.07.736950)
104
112
 
105
113
  [scverse-discourse]: https://discourse.scverse.org/
106
114
  [issue-tracker]: https://github.com/ComputationalBiomedicineGroup/Rectangle/issues
@@ -33,6 +33,12 @@ Install the latest release of `Rectangle` from `PyPI` <https://pypi.org/project/
33
33
  pip install rectanglepy
34
34
  ```
35
35
 
36
+ ## How Rectangle works
37
+
38
+ ![Overview of the Rectangle signature-building and multiscale deconvolution workflow](docs/_static/rectangle_workflow.png)
39
+
40
+ Rectangle performs robust multiscale deconvolution informed by single-cell transcriptomics. Rectangle takes annotated scRNA-seq data and a bulk RNA-seq mixture as input. In the signature-building phase, it performs bootstrapped pseudobulking of the scRNA-seq data, followed by signature-gene selection based on log-fold-change (logFC) and p-value, and signature-matrix optimization (minimizing the condition number and correcting for cell-type-specific mRNA content bias). Based on this, it constructs two signature matrices: a “direct signature”, resolving individual cell types, and a “clustered signature” grouping transcriptionally similar cell types. In the deconvolution phase, Rectangle first uses the clustered signature to deconvolve the bulk data into coarse cell-type estimates. These are then used to constrain a second deconvolution step that leverages the direct signature to resolve individual cell-type fractions. Finally, the resulting estimates are scaled to account for unknown cellular content not represented in the reference.
41
+
36
42
  ## License
37
43
 
38
44
  Rectangle is dual-licensed: BSD-3-Clause OR Commercial.
@@ -61,7 +67,9 @@ For commercial licensing: **innovation-psb@uibk.ac.at**
61
67
 
62
68
  ## Citation
63
69
 
64
- > If you use Rectangle in your project, please cite: (TBA)
70
+ If you use Rectangle in your project, please cite:
71
+
72
+ > Eder B, Rigato I, Dietrich A, Merotto L, Sturm G, Treis T, List M, Theis FJ, Finotello F. Rectangle: robust and scalable multiscale deconvolution informed by single-cell RNA sequencing data. bioRxiv. 2026. doi:[10.64898/2026.07.07.736950](https://doi.org/10.64898/2026.07.07.736950)
65
73
 
66
74
  [scverse-discourse]: https://discourse.scverse.org/
67
75
  [issue-tracker]: https://github.com/ComputationalBiomedicineGroup/Rectangle/issues
@@ -12,4 +12,5 @@
12
12
  pp.build_rectangle_signatures
13
13
  tl.deconvolution
14
14
  pp.RectangleSignatureResult
15
+ RectangleAdvancedParameters
15
16
  ```
@@ -1,4 +1,5 @@
1
1
  ```{include} ../README.md
2
+ :relative-images:
2
3
 
3
4
  ```
4
5
 
@@ -16,6 +16,15 @@
16
16
  url = {https://doi.org/10.1186/s13059-017-1382-0}
17
17
  }
18
18
 
19
+ @article{Eder2026,
20
+ author = {Bernhard Eder and Irene Rigato and Alexander Dietrich and Lorenzo Merotto and Gregor Sturm and Tim Treis and Markus List and Fabian J. Theis and Francesca Finotello},
21
+ title = {Rectangle: robust and scalable multiscale deconvolution informed by single-cell RNA sequencing data},
22
+ journal = {bioRxiv},
23
+ year = {2026},
24
+ doi = {10.64898/2026.07.07.736950},
25
+ url = {https://doi.org/10.64898/2026.07.07.736950}
26
+ }
27
+
19
28
  @article{Finotello2019,
20
29
  author = {Francesca Finotello and Gregor Mayer and Christian Plattner and Cornelia Laschober and Dietmar Rieder and Peter Hackl and Lukas Krogsdam and Michael L. T. Helmberg and Zlatko Trajanoski},
21
30
  title = {Molecular and pharmacological modulators of the tumor immune contexture revealed by deconvolution of RNA-seq data},
@@ -4,7 +4,7 @@ requires = ["hatchling"]
4
4
 
5
5
  [project]
6
6
  name = "rectanglepy"
7
- version = "1.4.2"
7
+ version = "1.6.0"
8
8
  description = "Hierarchical deconvolution of bulk transcriptomics"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -1,8 +1,9 @@
1
1
  from importlib.metadata import version
2
2
 
3
3
  from . import pp, tl
4
+ from .parameters import RectangleAdvancedParameters
4
5
  from .rectangle import load_tutorial_data, rectangle
5
6
 
6
- __all__ = ["pp", "tl", "load_tutorial_data", "rectangle"]
7
+ __all__ = ["pp", "tl", "RectangleAdvancedParameters", "load_tutorial_data", "rectangle"]
7
8
 
8
9
  __version__ = version("rectanglepy")
@@ -0,0 +1,11 @@
1
+ # SPDX-License-Identifier: BSD-3-Clause OR LicenseRef-Rectangle-Commercial
2
+
3
+ from dataclasses import dataclass
4
+
5
+
6
+ @dataclass(frozen=True)
7
+ class RectangleAdvancedParameters:
8
+ """Advanced parameters for Rectangle internals."""
9
+
10
+ number_of_bootstraps: int = 20
11
+ grid_search_split_size: int = 50
@@ -13,6 +13,7 @@ from scipy.cluster.hierarchy import fcluster, linkage
13
13
  from scipy.stats import pearsonr
14
14
  from sklearn.metrics import silhouette_score
15
15
 
16
+ from rectanglepy.parameters import RectangleAdvancedParameters
16
17
  from rectanglepy.tl.deconvolution import solve_qp
17
18
 
18
19
  from .rectangle_signature import RectangleSignatureResult
@@ -129,24 +130,25 @@ def _filter_de_analysis_results(de_analysis_result, p, logfc):
129
130
 
130
131
 
131
132
  def _run_deseq2(
132
- countsig: pd.DataFrame, sc_data, annotations: pd.Series, n_cpus: int = None, gene_expression_threshold=0.5
133
+ countsig: pd.DataFrame,
134
+ sc_data,
135
+ annotations: pd.Series,
136
+ n_cpus: int = None,
137
+ gene_expression_threshold=0.4,
138
+ number_of_bootstraps: int = 20,
133
139
  ) -> dict[str | int, pd.DataFrame]:
134
140
  results = {}
135
141
  inference = DefaultInference(n_cpus=n_cpus)
136
- bootstrapped_signature = _create_bootstrap_signature(countsig, sc_data, annotations)
142
+ bootstrapped_signature = _create_bootstrap_signature(countsig, sc_data, annotations, number_of_bootstraps)
137
143
  np.random.seed(42)
138
144
  for _i, cell_type in enumerate(countsig.columns):
139
- bootstrapped_signature_copy = bootstrapped_signature.copy()
140
- countsig_copy = countsig.copy()
141
145
  sc_data_filtered = sc_data.T[annotations == cell_type]
142
146
  expressed_cells = (sc_data_filtered > 0).sum(axis=0)
143
147
  if expressed_cells.ndim > 1: # needed for sparse matrices
144
148
  expressed_cells = np.squeeze(np.asarray(expressed_cells))
145
- # make dense out of sparse
146
- sc_data_filtered = sc_data_filtered.toarray()
147
149
  threshold = gene_expression_threshold * sc_data_filtered.shape[0]
148
- genes = countsig_copy.index[expressed_cells > threshold].tolist()
149
- bootstrapped_signature_copy = bootstrapped_signature_copy.loc[genes].T
150
+ genes = countsig.index[expressed_cells > threshold].tolist()
151
+ bootstrapped_signature_copy = bootstrapped_signature.loc[genes].T
150
152
  logger.info(f"Running DE analysis for {cell_type}")
151
153
  condition = ["B" if (cell_type + "_") in x else "A" for x in bootstrapped_signature_copy.index]
152
154
  clinical_df = pd.DataFrame({"condition": condition}, index=bootstrapped_signature_copy.index)
@@ -167,25 +169,29 @@ def _run_deseq2(
167
169
  return results
168
170
 
169
171
 
170
- def _create_bootstrap_signature(countsig, sc_data, annotations) -> pd.DataFrame:
171
- if scipy.sparse.issparse(sc_data):
172
- sc_data = sc_data.toarray()
172
+ def _create_bootstrap_signature(countsig, sc_data, annotations, number_of_bootstraps: int = 20) -> pd.DataFrame:
173
+ # Cells are only ever summed here, never read one by one, so sparse input is kept sparse:
174
+ # densifying the whole matrix costs genes * cells * 8 bytes, which is gigabytes on an
175
+ # atlas-sized reference. Dense input deliberately stays dense - gathering rows out of a
176
+ # dense array is faster than routing them through a sparse format.
177
+ cells_by_gene = sc_data.T.tocsr() if scipy.sparse.issparse(sc_data) else sc_data.T
173
178
  celltypes = countsig.columns
174
- bootstrapped_signature = pd.DataFrame()
175
- number_of_bootstraps = 7
176
- samples_per_bootstrap = 500
179
+ columns = {}
180
+ samples_per_bootstrap = 1000
177
181
  np.random.seed(42)
178
182
  for celltype in celltypes:
179
- sc_data_filtered = sc_data.T[annotations == celltype]
183
+ sc_data_filtered = cells_by_gene[(annotations == celltype).to_numpy()]
184
+ # sparse matrices do not support len(); shape[0] is the cell count for both formats
185
+ number_of_cells = sc_data_filtered.shape[0]
180
186
  for i in range(number_of_bootstraps):
181
- selected_rows = np.random.choice(len(sc_data_filtered), samples_per_bootstrap, replace=True)
187
+ selected_rows = np.random.choice(number_of_cells, samples_per_bootstrap, replace=True)
182
188
  summed_rows = sc_data_filtered[selected_rows].sum(axis=0)
183
- bootstrapped_signature[f"{celltype}_{i}"] = list(summed_rows)
184
- bootstrapped_signature.index = countsig.index
189
+ # a sparse sum is a (1, genes) matrix, a dense one a flat array
190
+ columns[f"{celltype}_{i}"] = np.asarray(summed_rows).ravel()
191
+ # built in one go: assigning column by column repeatedly reallocates the frame
192
+ bootstrapped_signature = pd.DataFrame(columns, index=countsig.index)
185
193
  # to int
186
- samples_per_bootstrap = samples_per_bootstrap / 2.5
187
- bootstrapped_signature = bootstrapped_signature.astype(int)
188
- return bootstrapped_signature
194
+ return bootstrapped_signature.astype(int)
189
195
 
190
196
 
191
197
  def _de_analysis(
@@ -197,15 +203,31 @@ def _de_analysis(
197
203
  optimize_cutoffs: bool,
198
204
  n_cpus: int = None,
199
205
  genes=None,
200
- gene_expression_threshold=0.5,
206
+ gene_expression_threshold=0.4,
207
+ advanced_parameters: RectangleAdvancedParameters = None,
201
208
  ) -> tuple[Series, dict[str, [str]] :, DataFrame | None]:
202
209
  logger.info("Starting DE analysis")
203
- deseq_results = _run_deseq2(pseudo_count_sig, sc_data, annotations, n_cpus, gene_expression_threshold)
210
+ advanced_parameters = advanced_parameters or RectangleAdvancedParameters()
211
+ deseq_results = _run_deseq2(
212
+ pseudo_count_sig,
213
+ sc_data,
214
+ annotations,
215
+ n_cpus,
216
+ gene_expression_threshold,
217
+ advanced_parameters.number_of_bootstraps,
218
+ )
204
219
  optimization_results = None
205
220
 
206
221
  if optimize_cutoffs:
207
222
  logger.info("Optimizing cutoff parameters p and lfc")
208
- optimization_results = _optimize_parameters(sc_data, annotations, pseudo_count_sig, deseq_results, genes)
223
+ optimization_results = _optimize_parameters(
224
+ sc_data,
225
+ annotations,
226
+ pseudo_count_sig,
227
+ deseq_results,
228
+ genes,
229
+ advanced_parameters.grid_search_split_size,
230
+ )
209
231
  p, lfc = optimization_results.iloc[0, 0:2]
210
232
  logger.info(f"Optimization done\n Best cutoffs p: {p} and lfc: {lfc}")
211
233
 
@@ -286,7 +308,8 @@ def build_rectangle_signatures(
286
308
  p=0.015,
287
309
  lfc=1.5,
288
310
  n_cpus: int = None,
289
- gene_expression_threshold=0.5,
311
+ gene_expression_threshold=0.4,
312
+ advanced_parameters: RectangleAdvancedParameters = None,
290
313
  ) -> RectangleSignatureResult:
291
314
  r"""Builds rectangle signatures based on single-cell count data and annotations.
292
315
 
@@ -303,7 +326,7 @@ def build_rectangle_signatures(
303
326
  raw
304
327
  A flag indicating whether to use the raw Anndata data. Defaults to False.
305
328
  optimize_cutoffs
306
- Indicates whether to optimize the p-value and log fold change cutoffs using gridsearch. Defaults to True.
329
+ Indicates whether to optimize the log fold change cutoffs using gridsearch. Defaults to True.
307
330
  p
308
331
  The p-value threshold for the DE analysis (only used if optimize_cutoffs is False).
309
332
  lfc
@@ -311,12 +334,15 @@ def build_rectangle_signatures(
311
334
  n_cpus
312
335
  The number of cpus to use for the DE analysis. Defaults to the number of cpus available.
313
336
  gene_expression_threshold
314
- The gene expression threshold for the DE analysis. How many cells need to express a gene to be considered in DGE
337
+ The gene expression threshold for the DE analysis. The fraction of cells that must express a gene to be considered in DGE. Defaults to 0.4
338
+ advanced_parameters
339
+ Optional advanced Rectangle parameters. Defaults are used when not provided.
315
340
 
316
341
  Returns
317
342
  -------
318
343
  The result of the rectangle signature analysis which is of type RectangleSignatureResult.
319
344
  """
345
+ advanced_parameters = advanced_parameters or RectangleAdvancedParameters()
320
346
  annotations = adata.obs[cell_type_col]
321
347
  adata = adata[:, adata.X.sum(axis=0) > len(annotations.value_counts())]
322
348
  assert adata.var_names.is_unique, "Duplicate gene found in adata"
@@ -344,7 +370,16 @@ def build_rectangle_signatures(
344
370
  m_rna_biasfactors = _create_bias_factors(pseudo_sig_counts, sc_counts, annotations)
345
371
 
346
372
  marker_genes, marker_genes_per_cell_type, optimization_result = _de_analysis(
347
- pseudo_sig_counts, sc_counts, annotations, p, lfc, optimize_cutoffs, n_cpus, genes, gene_expression_threshold
373
+ pseudo_sig_counts,
374
+ sc_counts,
375
+ annotations,
376
+ p,
377
+ lfc,
378
+ optimize_cutoffs,
379
+ n_cpus,
380
+ genes,
381
+ gene_expression_threshold,
382
+ advanced_parameters,
348
383
  )
349
384
  pseudo_sig_cpm = _convert_to_cpm(pseudo_sig_counts)
350
385
  logger.info("Starting rectangle cluster analysis")
@@ -371,6 +406,7 @@ def build_rectangle_signatures(
371
406
  lfc,
372
407
  False,
373
408
  gene_expression_threshold=gene_expression_threshold,
409
+ advanced_parameters=advanced_parameters,
374
410
  )
375
411
  clustered_signature = _convert_to_cpm(clustered_signature)
376
412
  return RectangleSignatureResult(
@@ -390,9 +426,10 @@ def build_rectangle_signatures(
390
426
  def _create_pseudo_count_sig(sc_counts: np.ndarray, annotations: pd.Series, var_names) -> pd.DataFrame:
391
427
  unique_labels, label_indices = np.unique(annotations, return_inverse=True)
392
428
  grouped_sum = np.zeros((len(unique_labels), sc_counts.shape[0]))
429
+ cells_by_gene = sc_counts.T
393
430
  for i, _label in enumerate(unique_labels):
394
431
  label_columns = label_indices == i
395
- grouped_sum[i, :] = np.sum(sc_counts.T[label_columns, :], axis=0)
432
+ grouped_sum[i, :] = np.sum(cells_by_gene[label_columns, :], axis=0)
396
433
  grouped_sum = grouped_sum.T
397
434
 
398
435
  grouped_sum = pd.DataFrame(grouped_sum, index=var_names, columns=unique_labels).astype(int)
@@ -400,15 +437,20 @@ def _create_pseudo_count_sig(sc_counts: np.ndarray, annotations: pd.Series, var_
400
437
 
401
438
 
402
439
  def _optimize_parameters(
403
- sc_data: pd.DataFrame, annotations: pd.Series, pseudo_signature_counts: pd.DataFrame, de_results, genes=None
440
+ sc_data: pd.DataFrame,
441
+ annotations: pd.Series,
442
+ pseudo_signature_counts: pd.DataFrame,
443
+ de_results,
444
+ genes=None,
445
+ grid_search_split_size: int = 50,
404
446
  ) -> pd.DataFrame:
405
447
  # search space for p and lfc
406
- lfcs = [x / 100 for x in range(160, 230, 10)]
448
+ lfcs = [1.5, 1.75, 2.0, 2.25, 2.5, 2.75, 3.0]
407
449
  ps = [x / 1000 for x in range(50, 51, 1)]
408
450
 
409
451
  results = []
410
452
  logger.info("generating pseudo bulks")
411
- bulks, real_fractions = _generate_pseudo_bulks(sc_data, annotations, genes)
453
+ bulks, real_fractions = _generate_pseudo_bulks(sc_data, annotations, genes, grid_search_split_size)
412
454
  for p in ps:
413
455
  for lfc in lfcs:
414
456
  try:
@@ -441,9 +483,8 @@ def _assess_parameter_fit(
441
483
  return rsme, pearson_r
442
484
 
443
485
 
444
- def _generate_pseudo_bulks(sc_data, annotations, genes=None):
486
+ def _generate_pseudo_bulks(sc_data, annotations, genes=None, split_size: int = 50):
445
487
  number_of_bulks = 50
446
- split_size = 50
447
488
  bulks = []
448
489
  real_fractions = []
449
490
  np.random.seed(42)
@@ -7,6 +7,7 @@ from anndata import AnnData
7
7
  from loguru import logger
8
8
  from pandas import DataFrame
9
9
 
10
+ from .parameters import RectangleAdvancedParameters
10
11
  from .pp import RectangleSignatureResult, build_rectangle_signatures
11
12
  from .tl import deconvolution
12
13
 
@@ -23,7 +24,8 @@ def rectangle(
23
24
  p=0.015,
24
25
  lfc=1.5,
25
26
  n_cpus: int = None,
26
- gene_expression_threshold=0.5,
27
+ gene_expression_threshold=0.4,
28
+ advanced_parameters: RectangleAdvancedParameters = None,
27
29
  ) -> tuple[DataFrame, RectangleSignatureResult]:
28
30
  r"""All in one deconvolution method. Creates signatures and deconvolutes the bulk data. Has options for subsampling and consensus runs.
29
31
 
@@ -40,7 +42,7 @@ def rectangle(
40
42
  raw
41
43
  A flag indicating whether to use the raw Anndata data.
42
44
  optimize_cutoffs
43
- Indicates whether to optimize the p-value and log fold change cutoffs using gridsearch.
45
+ Indicates whether to optimize the log fold change cutoffs using gridsearch.
44
46
  p
45
47
  The p-value threshold for the DE analysis (only used if optimize_cutoffs is False).
46
48
  lfc
@@ -50,7 +52,9 @@ def rectangle(
50
52
  correct_mrna_bias : bool
51
53
  A flag indicating whether to correct for mRNA bias. Defaults to True.
52
54
  gene_expression_threshold : float
53
- The threshold for gene expression. Genes with expression below this threshold are removed from the analysis.
55
+ The threshold for gene expression. Genes must be expressed in at least this fraction of cells. Defaults to 0.4.
56
+ advanced_parameters
57
+ Optional advanced Rectangle parameters. Defaults are used when not provided.
54
58
 
55
59
  Returns
56
60
  -------
@@ -71,6 +75,7 @@ def rectangle(
71
75
  lfc=lfc,
72
76
  n_cpus=n_cpus,
73
77
  gene_expression_threshold=gene_expression_threshold,
78
+ advanced_parameters=advanced_parameters,
74
79
  )
75
80
 
76
81
  estimations, bulk_err = deconvolution(signatures, bulks, correct_mrna_bias, n_cpus)
@@ -7,18 +7,166 @@ import numpy as np
7
7
  import osqp
8
8
  import pandas as pd
9
9
  import scipy.sparse as sp
10
- import statsmodels.api as sm
11
10
  from joblib import Parallel, delayed, parallel_backend
12
11
  from loguru import logger
13
12
 
14
13
  from rectanglepy.pp.rectangle_signature import RectangleSignatureResult
15
14
 
15
+ # OSQP can return noticeably different solutions from active-set QP when P is singular/ill-conditioned.
16
+ # A tiny ridge makes the problem strictly convex and closer to quadprog behavior.
17
+ QP_RIDGE = 1e-8
18
+
16
19
 
17
20
  def _scale_weights(weights: np.ndarray) -> np.ndarray:
18
21
  min_weight = np.nextafter(min(weights), np.float64(1.0)) # prevent division by zero
19
22
  return weights / min_weight
20
23
 
21
24
 
25
+ def _upper_triangular_csc_pattern(n_vars: int) -> tuple[np.ndarray, np.ndarray]:
26
+ """Returns the (indices, indptr) of a fully populated upper triangular CSC matrix.
27
+
28
+ OSQP only stores the upper triangle of P, so handing it the triangle directly saves the
29
+ conversion of the full symmetric matrix on every solve.
30
+ """
31
+ indices = np.concatenate([np.arange(column + 1) for column in range(n_vars)]).astype(np.int32)
32
+ indptr = np.concatenate(([0], np.cumsum(np.arange(1, n_vars + 1)))).astype(np.int32)
33
+ return indices, indptr
34
+
35
+
36
+ class _QuadraticProgram:
37
+ """A deconvolution QP for one (signature, bulk) pair, solvable at several dampening weights.
38
+
39
+ Everything that does not depend on the weights - the numpy views of the inputs and the constraint
40
+ matrix with its bounds - is built once and shared by all solves, which keeps it out of the
41
+ dampened least squares iteration.
42
+
43
+ Note that each solve still gets its own OSQP workspace: reusing one across solves makes OSQP
44
+ keep the problem scaling of the first objective, which perturbs the fixed point the iteration
45
+ converges to (and is not faster at these problem sizes).
46
+ """
47
+
48
+ def __init__(
49
+ self,
50
+ signature: pd.DataFrame,
51
+ bulk: pd.Series,
52
+ prev_assignments: list[int | str] = None,
53
+ prev_solution: pd.Series = None,
54
+ ):
55
+ if not signature.index.equals(bulk.index):
56
+ bulk = bulk.reindex(signature.index)
57
+ self.signature = signature.to_numpy(dtype=np.float64)
58
+ self.bulk = bulk.to_numpy(dtype=np.float64)
59
+ self.n_vars = self.signature.shape[1] # number of cell types / fractions
60
+
61
+ self._triu_indices, self._triu_indptr = _upper_triangular_csc_pattern(self.n_vars)
62
+ self._triu_selection = np.tril_indices(self.n_vars) # reads P.T column-major upper triangle
63
+ self._constraints, self._lower_bounds = self._build_constraints(prev_assignments, prev_solution)
64
+ self._upper_bounds = np.full_like(self._lower_bounds, np.inf, dtype=np.float64)
65
+
66
+ def _build_constraints(
67
+ self, prev_assignments: list[int | str], prev_solution: pd.Series
68
+ ) -> tuple[sp.csc_matrix, np.ndarray]:
69
+ # ----- Constraints in the original quadprog form: C.T x >= b
70
+ # We build C (n_vars, n_constraints) then map to OSQP: A = C.T, l=b, u=+inf
71
+ C_cols = []
72
+ b_list = []
73
+
74
+ # Constraint 1: sum(x) <= 1 -> -sum(x) >= -1
75
+ C_cols.append(-np.ones((self.n_vars, 1), dtype=np.float64))
76
+ b_list.append(np.array([-1.0], dtype=np.float64))
77
+
78
+ # Constraint 2: x >= 0 -> I x >= 0
79
+ C_cols.append(np.eye(self.n_vars, dtype=np.float64))
80
+ b_list.append(np.zeros(self.n_vars, dtype=np.float64))
81
+
82
+ # Constraint 3: keep close to prev_solution (this encoding already turns upper bounds into >= via negation)
83
+ if prev_solution is not None:
84
+ if prev_assignments is None:
85
+ raise ValueError("prev_assignments must be provided when prev_solution is provided.")
86
+
87
+ for cluster in prev_solution.index:
88
+ # x_cluster <= upper -> -x_cluster >= -upper
89
+ C_upper = np.array(
90
+ [-1.0 if str(x) == str(cluster) else 0.0 for x in prev_assignments],
91
+ dtype=np.float64,
92
+ ).reshape(-1, 1)
93
+
94
+ # x_cluster >= lower -> +x_cluster >= +lower
95
+ C_lower = np.array(
96
+ [1.0 if str(x) == str(cluster) else 0.0 for x in prev_assignments],
97
+ dtype=np.float64,
98
+ ).reshape(-1, 1)
99
+
100
+ prev_weight = float(prev_solution.loc[cluster])
101
+ upper = min(1.0, prev_weight + 0.03)
102
+ lower = max(0.0, prev_weight - 0.03)
103
+
104
+ C_cols.append(C_upper)
105
+ b_list.append(np.array([-upper], dtype=np.float64))
106
+
107
+ C_cols.append(C_lower)
108
+ b_list.append(np.array([lower], dtype=np.float64))
109
+
110
+ C = np.concatenate(C_cols, axis=1) # (n_vars, n_constraints)
111
+ b = np.concatenate(b_list, axis=0).astype(np.float64) # (n_constraints,)
112
+
113
+ # OSQP uses l <= A x <= u
114
+ return sp.csc_matrix(C.T), b
115
+
116
+ def _objective(self, gld: np.ndarray, multiplier: int) -> tuple[np.ndarray, np.ndarray]:
117
+ # Minimize 1/2 x^T G x - a^T x
118
+ if multiplier is None:
119
+ G = self.signature.T @ self.signature
120
+ a = self.signature.T @ self.bulk
121
+ else:
122
+ weights = np.square(1 / (self.signature @ gld))
123
+ weights_dampened = np.clip(_scale_weights(weights), None, multiplier)
124
+ # equivalent to signature.T @ diag(weights) @ signature, without the dense (genes, genes) matrix
125
+ G = self.signature.T @ (weights_dampened[:, None] * self.signature)
126
+ a = self.signature.T @ (weights_dampened * self.bulk)
127
+
128
+ # ----- Map to OSQP: minimize 1/2 x^T P x + q^T x -> q = -a
129
+ P = G
130
+ q = -a
131
+
132
+ # Optional scaling
133
+ scale = np.linalg.norm(P)
134
+ if scale > 0:
135
+ P = P / scale
136
+ q = q / scale
137
+
138
+ P = ((P + P.T) / 2.0) + QP_RIDGE * np.eye(self.n_vars, dtype=np.float64)
139
+ return P, q
140
+
141
+ def solve(self, gld: np.ndarray = None, multiplier: int = None) -> np.ndarray:
142
+ P, q = self._objective(gld, multiplier)
143
+ P_data = P.T[self._triu_selection] # upper triangle in CSC (column major) order
144
+ P_sp = sp.csc_matrix((P_data, self._triu_indices, self._triu_indptr), shape=(self.n_vars, self.n_vars))
145
+
146
+ solver = osqp.OSQP()
147
+ solver.setup(
148
+ P=P_sp,
149
+ q=q,
150
+ A=self._constraints,
151
+ l=self._lower_bounds,
152
+ u=self._upper_bounds,
153
+ verbose=False,
154
+ eps_abs=1e-7,
155
+ eps_rel=1e-7,
156
+ max_iter=50000,
157
+ polish=True,
158
+ warm_start=False,
159
+ scaled_termination=False,
160
+ )
161
+
162
+ res = solver.solve()
163
+
164
+ if res.info.status_val not in (1,): # 1 = solved
165
+ raise RuntimeError(f"OSQP did not solve the problem: {res.info.status}")
166
+
167
+ return res.x
168
+
169
+
22
170
  def solve_qp(
23
171
  signature: pd.DataFrame,
24
172
  bulk: pd.Series,
@@ -52,143 +200,89 @@ def solve_qp(
52
200
  Notes
53
201
  -----
54
202
  This function uses quadratic programming to solve the deconvolution problem. The objective is to minimize the difference between the observed bulk data and the data predicted by the signature matrix and the cell fractions. The function also includes constraints to ensure that the cell fractions are non-negative and sum to 1, and to make the solution similar to the previous assignments and weights if they are provided.
203
+
204
+ Callers that solve the same problem at several dampening weights (e.g. the dampened least squares
205
+ iteration) should use :class:`_QuadraticProgram` directly, which shares the weight independent
206
+ parts of the problem across solves.
55
207
  """
56
- # ------------------ QP-based deconvolution
57
- # Minimize 1/2 x^T G x - a^T x
58
- # Subject to C.T x >= b
59
- if multiplier is None:
60
- a = (signature.T @ bulk).to_numpy(dtype=np.float64)
61
- G = (signature.T @ signature).to_numpy(dtype=np.float64)
62
- else:
63
- weights = np.square(1 / (signature @ gld))
64
- weights_dampened = np.clip(_scale_weights(weights), None, multiplier)
65
- W = np.diag(np.asarray(weights_dampened, dtype=np.float64))
66
- G = (signature.T.to_numpy() @ (W @ signature.to_numpy())).astype(np.float64)
67
- a = (signature.T.to_numpy() @ (W @ bulk.to_numpy())).astype(np.float64)
68
-
69
- n_vars = G.shape[0] # number of cell types / fractions
70
-
71
- # ----- Constraints in your original quadprog form: C.T x >= b
72
- # We'll build C (n_vars, n_constraints) then map to OSQP: A = C.T, l=b, u=+inf
73
- C_cols = []
74
- b_list = []
75
-
76
- # Constraint 1: sum(x) <= 1 -> -sum(x) >= -1
77
- C_cols.append(-np.ones((n_vars, 1), dtype=np.float64))
78
- b_list.append(np.array([-1.0], dtype=np.float64))
79
-
80
- # Constraint 2: x >= 0 -> I x >= 0
81
- C_cols.append(np.eye(n_vars, dtype=np.float64))
82
- b_list.append(np.zeros(n_vars, dtype=np.float64))
83
-
84
- # Constraint 3: keep close to prev_solution (your encoding already turns upper bounds into >= via negation)
85
- if prev_solution is not None:
86
- if prev_assignments is None:
87
- raise ValueError("prev_assignments must be provided when prev_solution is provided.")
88
-
89
- for cluster in prev_solution.index:
90
- # x_cluster <= upper -> -x_cluster >= -upper
91
- C_upper = np.array(
92
- [-1.0 if str(x) == str(cluster) else 0.0 for x in prev_assignments],
93
- dtype=np.float64,
94
- ).reshape(-1, 1)
95
-
96
- # x_cluster >= lower -> +x_cluster >= +lower
97
- C_lower = np.array(
98
- [1.0 if str(x) == str(cluster) else 0.0 for x in prev_assignments],
99
- dtype=np.float64,
100
- ).reshape(-1, 1)
101
-
102
- prev_weight = float(prev_solution.loc[cluster])
103
- upper = min(1.0, prev_weight + 0.03)
104
- lower = max(0.0, prev_weight - 0.03)
105
-
106
- C_cols.append(C_upper)
107
- b_list.append(np.array([-upper], dtype=np.float64))
108
-
109
- C_cols.append(C_lower)
110
- b_list.append(np.array([lower], dtype=np.float64))
111
-
112
- C = np.concatenate(C_cols, axis=1) # (n_vars, n_constraints)
113
- b = np.concatenate(b_list, axis=0).astype(np.float64) # (n_constraints,)
114
-
115
- # ----- Map to OSQP: minimize 1/2 x^T P x + q^T x
116
- # Your objective: 1/2 x^T G x - a^T x -> q = -a
117
- P = G
118
- q = -a
119
-
120
- # Optional scaling
121
- scale = np.linalg.norm(P)
122
- if scale > 0:
123
- P = P / scale
124
- q = q / scale
125
-
126
- # OSQP can return noticeably different solutions from active-set QP when P is singular/ill-conditioned.
127
- # Add tiny ridge to make the problem strictly convex and closer to quadprog behavior.
128
- ridge = 1e-8
129
- P = ((P + P.T) / 2.0) + ridge * np.eye(n_vars, dtype=np.float64)
130
-
131
- # OSQP uses l <= A x <= u
132
- A = C.T # (n_constraints, n_vars)
133
- l = b
134
- u = np.full_like(l, np.inf, dtype=np.float64)
135
-
136
- # Sparse matrices (required/expected)
137
- P_sp = sp.csc_matrix(P)
138
- A_sp = sp.csc_matrix(A)
139
-
140
- solver = osqp.OSQP()
141
- solver.setup(
142
- P=P_sp,
143
- q=q,
144
- A=A_sp,
145
- l=l,
146
- u=u,
147
- verbose=False,
148
- eps_abs=1e-7,
149
- eps_rel=1e-7,
150
- max_iter=50000,
151
- polish=True,
152
- warm_start=False,
153
- scaled_termination=False,
154
- )
208
+ problem = _QuadraticProgram(signature, bulk, prev_assignments, prev_solution)
209
+ return problem.solve(gld, multiplier)
210
+
211
+
212
+ def _batched_wls(
213
+ design: np.ndarray,
214
+ response: np.ndarray,
215
+ weights: np.ndarray,
216
+ subsets: np.ndarray,
217
+ max_bytes: int = 64 * 1024**2,
218
+ ) -> np.ndarray:
219
+ """Fits one weighted least squares model per gene subset in batch.
220
+
221
+ Mirrors ``statsmodels.api.WLS(...).fit()``: the design and the response are whitened with the
222
+ square root of the weights and the parameters are read off the pseudo inverse of the whitened
223
+ design. Stacking the subsets lets numpy loop over them in C instead of Python.
155
224
 
156
- res = solver.solve()
225
+ Parameters
226
+ ----------
227
+ design : np.ndarray
228
+ The (genes, cell types) design matrix.
229
+ response : np.ndarray
230
+ The (genes,) response vector.
231
+ weights : np.ndarray
232
+ The (genes,) regression weights.
233
+ subsets : np.ndarray
234
+ A (n_subsets, subset size) array of gene indices, one row per model.
235
+ max_bytes : int
236
+ Upper bound on the size of a single whitened design batch, used to chunk the subsets.
237
+
238
+ Returns
239
+ -------
240
+ np.ndarray
241
+ A (n_subsets, cell types) array of fitted parameters.
242
+ """
243
+ n_subsets, subset_size = subsets.shape
244
+ n_params = design.shape[1]
245
+ params = np.empty((n_subsets, n_params), dtype=np.float64)
157
246
 
158
- if res.info.status_val not in (1,): # 1 = solved
159
- raise RuntimeError(f"OSQP did not solve the problem: {res.info.status}")
247
+ chunk_size = max(1, int(max_bytes // (subset_size * n_params * design.itemsize)))
248
+ for start in range(0, n_subsets, chunk_size):
249
+ chunk = subsets[start : start + chunk_size]
250
+ sqrt_weights = np.sqrt(weights[chunk]) # (chunk, subset size)
251
+ whitened_design = design[chunk] * sqrt_weights[:, :, None] # (chunk, subset size, cell types)
252
+ whitened_response = response[chunk] * sqrt_weights
253
+ params[start : start + chunk_size] = np.einsum("bpg,bg->bp", np.linalg.pinv(whitened_design), whitened_response)
160
254
 
161
- return res.x
255
+ return params
162
256
 
163
257
 
164
258
  def _calculate_dampening_constant(signature: pd.DataFrame, bulk: pd.Series, qp_gld: np.ndarray) -> int:
165
259
  solutions_std = []
166
260
  np.random.seed(1)
167
- weights = np.square(1 / (np.dot(signature, qp_gld)))
261
+ signature_values = np.asarray(signature, dtype=np.float64)
262
+ bulk_values = np.asarray(bulk, dtype=np.float64)
263
+ n_genes = signature_values.shape[0]
264
+ weights = np.square(1 / (signature_values @ qp_gld))
168
265
  weights_scaled = _scale_weights(weights)
169
266
  weights_scaled_no_inf = weights_scaled[weights_scaled != np.inf]
170
267
  qp_gld_sum = sum(qp_gld)
268
+ subset_size = n_genes // 2
269
+ n_subsets = 100
171
270
  # try multiple values of the dampening constant (multiplier)
172
271
  # for each, calculate the variance of the dampened weighted solution for a subset of genes
173
272
  max_range = 40
174
273
  multiplier_range = min(max_range, math.ceil(np.log2(max(weights_scaled_no_inf))))
175
274
  for i in range(multiplier_range):
176
- solutions = []
177
275
  multiplier = 2**i
178
- weights_dampened = np.array([multiplier if multiplier <= x else x for x in weights_scaled]).astype("double")
179
- for _ in range(100):
180
- subset = np.random.choice(len(signature), size=len(signature) // 2, replace=False)
181
- bulk_subset = bulk.iloc[list(subset)]
182
- signature_subset = signature.iloc[subset, :]
183
- fit = sm.WLS(bulk_subset, -1 + signature_subset, weights=weights_dampened[subset]).fit()
184
- solution = fit.params * qp_gld_sum / sum(fit.params)
185
- solutions.append(solution)
186
- solutions_df = pd.DataFrame(solutions)
187
-
188
- solutions_std.append(solutions_df.std(axis=0))
189
- solutions_std_df = pd.DataFrame(solutions_std)
190
- means = solutions_std_df.apply(lambda x: np.mean(x**2), axis=1)
191
- best_dampening_constant = means.idxmin()
276
+ weights_dampened = np.minimum(weights_scaled, multiplier)
277
+ subsets = np.array([np.random.choice(n_genes, size=subset_size, replace=False) for _ in range(n_subsets)])
278
+ # WLS uses the signature directly; no intercept column is added.
279
+ params = _batched_wls(signature_values, bulk_values, weights_dampened, subsets)
280
+ solutions = params * qp_gld_sum / params.sum(axis=1, keepdims=True)
281
+
282
+ solutions_std.append(np.nanstd(solutions, axis=0, ddof=1))
283
+ solutions_std_array = np.array(solutions_std)
284
+ means = np.nanmean(np.square(solutions_std_array), axis=1)
285
+ best_dampening_constant = int(np.nanargmin(means))
192
286
  return best_dampening_constant
193
287
 
194
288
 
@@ -202,7 +296,8 @@ def _calculate_ls(
202
296
  signature = signature.loc[genes].sort_index()
203
297
  bulk = bulk.loc[genes].sort_index().astype("double")
204
298
 
205
- approximate_solution = solve_qp(signature, bulk, prev_assignments, prev_weights)
299
+ problem = _QuadraticProgram(signature, bulk, prev_assignments, prev_weights)
300
+ approximate_solution = problem.solve()
206
301
  dampening_constant = _calculate_dampening_constant(signature, bulk, approximate_solution)
207
302
  multiplier = 2**dampening_constant
208
303
 
@@ -212,7 +307,7 @@ def _calculate_ls(
212
307
  iterations = 2
213
308
  solutions_sum = approximate_solution
214
309
  while (change > convergence_threshold) and (iterations < max_iterations):
215
- dampened_solution = solve_qp(signature, bulk, prev_assignments, prev_weights, approximate_solution, multiplier)
310
+ dampened_solution = problem.solve(approximate_solution, multiplier)
216
311
  solutions_sum += dampened_solution
217
312
  solution_averages = solutions_sum / iterations
218
313
  change = np.linalg.norm(solution_averages - approximate_solution, 1)
@@ -6,6 +6,7 @@ import pytest
6
6
  from anndata import AnnData
7
7
 
8
8
  import rectanglepy as rectangle
9
+ from rectanglepy import RectangleAdvancedParameters
9
10
  from rectanglepy.pp.create_signature import (
10
11
  _assess_parameter_fit,
11
12
  _calculate_cluster_range,
@@ -18,6 +19,7 @@ from rectanglepy.pp.create_signature import (
18
19
  _de_analysis,
19
20
  _generate_pseudo_bulks,
20
21
  _get_fcluster_assignments,
22
+ _optimize_parameters,
21
23
  _run_deseq2,
22
24
  build_rectangle_signatures,
23
25
  )
@@ -154,6 +156,7 @@ def test_generate_pseudo_bulks(small_data):
154
156
  sc_counts = sc_counts.astype("int")
155
157
  adata = AnnData(sc_counts.T, obs=annotations.to_frame(name="cell_type"))
156
158
  result, _ = _generate_pseudo_bulks(adata.X.T, annotations, adata.var_names)
159
+ custom_result, _ = _generate_pseudo_bulks(adata.X.T, annotations, adata.var_names, split_size=10)
157
160
 
158
161
  sc_data = sc_counts.astype(pd.SparseDtype("int"))
159
162
  csr_sparse_matrix = sc_data.sparse.to_coo().tocsr()
@@ -162,11 +165,40 @@ def test_generate_pseudo_bulks(small_data):
162
165
  result_sparse, _ = _generate_pseudo_bulks(adata_sparse.X.T, annotations, adata_sparse.var_names)
163
166
 
164
167
  assert len(result) == 1000 and len(result.columns) == 50
168
+ assert custom_result.shape == result.shape
169
+ assert not np.allclose(result, custom_result)
165
170
  # first gene should have all 0s
166
171
  assert result.iloc[0, :].sum() == 0
167
172
  assert np.allclose(result, result_sparse)
168
173
 
169
174
 
175
+ def test_optimize_parameters_uses_grid_search_split_size(monkeypatch):
176
+ seen = {}
177
+
178
+ def fake_generate_pseudo_bulks(sc_data, annotations, genes=None, split_size=50):
179
+ seen["split_size"] = split_size
180
+ bulks = pd.DataFrame([[1.0]], index=["gene"], columns=["bulk"])
181
+ real_fractions = pd.DataFrame([[1.0]], index=["cell_type"], columns=["bulk"])
182
+ return bulks, real_fractions
183
+
184
+ def fake_assess_parameter_fit(lfc, p, bulks, real_fractions, pseudo_signature_counts, de_results):
185
+ return 0.0, 1.0
186
+
187
+ monkeypatch.setattr(rectangle.pp.create_signature, "_generate_pseudo_bulks", fake_generate_pseudo_bulks)
188
+ monkeypatch.setattr(rectangle.pp.create_signature, "_assess_parameter_fit", fake_assess_parameter_fit)
189
+
190
+ results = _optimize_parameters(
191
+ pd.DataFrame(),
192
+ pd.Series(dtype=str),
193
+ pd.DataFrame(),
194
+ {},
195
+ grid_search_split_size=13,
196
+ )
197
+
198
+ assert seen["split_size"] == 13
199
+ assert results.iloc[0]["pearson_r"] == 1.0
200
+
201
+
170
202
  def test_asses_fit(small_data):
171
203
  sc_counts, annotations, bulk = small_data
172
204
  sc_counts = sc_counts.astype("int")
@@ -204,12 +236,12 @@ def test_de_analysis(small_data):
204
236
  # test with sparse matrix
205
237
  _ = _de_analysis(sc_pseudo, adata_sparse.X.T, annotations, 0.4, 0.1, False, None, adata.var_names)
206
238
 
207
- assert 5 < len(r1) < 50
239
+ assert 5 < len(r1) < 100
208
240
  assert len(r2) == 3
209
241
 
210
242
 
211
243
  def test_create_bootstrap_signature(small_data):
212
- bootstraps_per_cell = 7
244
+ bootstraps_per_cell = RectangleAdvancedParameters().number_of_bootstraps
213
245
  sc_counts, annotations, bulk = small_data
214
246
  sc_counts = sc_counts.astype("int")
215
247
  sc_pseudo = sc_counts.groupby(annotations.values, axis=1).sum()
@@ -217,3 +249,20 @@ def test_create_bootstrap_signature(small_data):
217
249
  bootstrap = _create_bootstrap_signature(sc_pseudo, adata.X.T, annotations)
218
250
 
219
251
  assert len(bootstrap.columns) == len(sc_pseudo.columns) * bootstraps_per_cell
252
+
253
+
254
+ def test_create_bootstrap_signature_with_advanced_parameter(small_data):
255
+ bootstraps_per_cell = 3
256
+ advanced_parameters = RectangleAdvancedParameters(number_of_bootstraps=bootstraps_per_cell)
257
+ sc_counts, annotations, bulk = small_data
258
+ sc_counts = sc_counts.astype("int")
259
+ sc_pseudo = sc_counts.groupby(annotations.values, axis=1).sum()
260
+ adata = AnnData(sc_counts.T, obs=annotations.to_frame(name="cell_type"))
261
+ bootstrap = _create_bootstrap_signature(
262
+ sc_pseudo,
263
+ adata.X.T,
264
+ annotations,
265
+ advanced_parameters.number_of_bootstraps,
266
+ )
267
+
268
+ assert len(bootstrap.columns) == len(sc_pseudo.columns) * bootstraps_per_cell
@@ -62,7 +62,7 @@ def test_simple_weighted_dampened_deconvolution(quantiseq_data):
62
62
  corr = np.corrcoef(result, expected)[0, 1]
63
63
  rsme = np.sqrt(np.mean((result - expected) ** 2))
64
64
 
65
- assert corr > 0.815 and rsme < 0.011
65
+ assert corr > 0.81 and rsme < 0.01
66
66
 
67
67
 
68
68
  def test_correct_for_unknown_cell_content(small_data, quantiseq_data):
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes