crest-sc 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. crest_sc-0.3.0/.github/workflows/CI.yml +218 -0
  2. crest_sc-0.3.0/.gitignore +117 -0
  3. crest_sc-0.3.0/.readthedocs.yaml +20 -0
  4. crest_sc-0.3.0/CHANGELOG.md +95 -0
  5. crest_sc-0.3.0/CLAUDE.md +182 -0
  6. crest_sc-0.3.0/Cargo.lock +1716 -0
  7. crest_sc-0.3.0/Cargo.toml +22 -0
  8. crest_sc-0.3.0/HANDOVER.md +124 -0
  9. crest_sc-0.3.0/LICENSE +21 -0
  10. crest_sc-0.3.0/PKG-INFO +232 -0
  11. crest_sc-0.3.0/README.md +179 -0
  12. crest_sc-0.3.0/bench/deseq2/compare_r.py +164 -0
  13. crest_sc-0.3.0/bench/deseq2/kang_pseudobulk.py +184 -0
  14. crest_sc-0.3.0/bench/deseq2/results/comparison.csv +7 -0
  15. crest_sc-0.3.0/bench/deseq2/results/comparison.md +8 -0
  16. crest_sc-0.3.0/bench/deseq2/results/kang_pseudobulk.csv +9 -0
  17. crest_sc-0.3.0/bench/deseq2/results/kang_pseudobulk.md +14 -0
  18. crest_sc-0.3.0/bench/doublets/compare_scrublet.py +82 -0
  19. crest_sc-0.3.0/bench/doublets/results/kang_scrublet.csv +7 -0
  20. crest_sc-0.3.0/bench/doublets/results/kang_scrublet.md +6 -0
  21. crest_sc-0.3.0/bench/harmony/compare_harmonypy.py +137 -0
  22. crest_sc-0.3.0/bench/harmony/results/kang_harmony.csv +8 -0
  23. crest_sc-0.3.0/bench/harmony/results/kang_harmony.md +13 -0
  24. crest_sc-0.3.0/bench/ingest/compare_scanpy_ingest.py +91 -0
  25. crest_sc-0.3.0/bench/ingest/results/kang_ingest.csv +3 -0
  26. crest_sc-0.3.0/bench/ingest/results/kang_ingest.md +22 -0
  27. crest_sc-0.3.0/bench/paper/accuracy.py +122 -0
  28. crest_sc-0.3.0/bench/paper/datasets.py +310 -0
  29. crest_sc-0.3.0/bench/paper/make_dataset.py +100 -0
  30. crest_sc-0.3.0/bench/paper/monitor.py +159 -0
  31. crest_sc-0.3.0/bench/paper/run_all.py +143 -0
  32. crest_sc-0.3.0/bench/paper/run_one.py +276 -0
  33. crest_sc-0.3.0/bench/paper/summarize.py +436 -0
  34. crest_sc-0.3.0/bench/whitepaper/results/crest-ooc_pbmc68k.json +56 -0
  35. crest_sc-0.3.0/bench/whitepaper/results/crest-ooc_pbmc_100k.json +56 -0
  36. crest_sc-0.3.0/bench/whitepaper/results/crest-ooc_pbmc_200k.json +56 -0
  37. crest_sc-0.3.0/bench/whitepaper/results/crest_pbmc68k.json +56 -0
  38. crest_sc-0.3.0/bench/whitepaper/results/crest_pbmc_100k.json +56 -0
  39. crest_sc-0.3.0/bench/whitepaper/results/crest_pbmc_200k.json +56 -0
  40. crest_sc-0.3.0/bench/whitepaper/results/fig_scaling.png +0 -0
  41. crest_sc-0.3.0/bench/whitepaper/results/fig_scaling.svg +1875 -0
  42. crest_sc-0.3.0/bench/whitepaper/results/fig_step_times.png +0 -0
  43. crest_sc-0.3.0/bench/whitepaper/results/fig_step_times.svg +3767 -0
  44. crest_sc-0.3.0/bench/whitepaper/results/report.md +66 -0
  45. crest_sc-0.3.0/bench/whitepaper/results/scanpy_pbmc68k.json +56 -0
  46. crest_sc-0.3.0/bench/whitepaper/results/scanpy_pbmc_100k.json +56 -0
  47. crest_sc-0.3.0/crest/__init__.py +14 -0
  48. crest_sc-0.3.0/crest/_loess.py +86 -0
  49. crest_sc-0.3.0/crest/core.py +372 -0
  50. crest_sc-0.3.0/crest/deseq2.py +452 -0
  51. crest_sc-0.3.0/crest/doublets.py +192 -0
  52. crest_sc-0.3.0/crest/harmony.py +90 -0
  53. crest_sc-0.3.0/crest/ingest.py +116 -0
  54. crest_sc-0.3.0/crest/io.py +185 -0
  55. crest_sc-0.3.0/crest/pp.py +330 -0
  56. crest_sc-0.3.0/crest/tl.py +332 -0
  57. crest_sc-0.3.0/docs/api.md +139 -0
  58. crest_sc-0.3.0/docs/benchmarks.md +99 -0
  59. crest_sc-0.3.0/docs/changelog.md +2 -0
  60. crest_sc-0.3.0/docs/concepts.md +128 -0
  61. crest_sc-0.3.0/docs/conf.py +67 -0
  62. crest_sc-0.3.0/docs/deseq2.md +95 -0
  63. crest_sc-0.3.0/docs/development.md +129 -0
  64. crest_sc-0.3.0/docs/downstream.md +119 -0
  65. crest_sc-0.3.0/docs/index.md +69 -0
  66. crest_sc-0.3.0/docs/installation.md +57 -0
  67. crest_sc-0.3.0/docs/memory_model.md +53 -0
  68. crest_sc-0.3.0/docs/quickstart.md +112 -0
  69. crest_sc-0.3.0/docs/readthedocs.md +92 -0
  70. crest_sc-0.3.0/docs/requirements.txt +9 -0
  71. crest_sc-0.3.0/pyproject.toml +50 -0
  72. crest_sc-0.3.0/scripts/crest_git_housekeeping.sh +118 -0
  73. crest_sc-0.3.0/scripts/crest_paper_bench.sh +287 -0
  74. crest_sc-0.3.0/scripts/export_chat.py +126 -0
  75. crest_sc-0.3.0/src/deseq/linalg.rs +161 -0
  76. crest_sc-0.3.0/src/deseq/lowess.rs +170 -0
  77. crest_sc-0.3.0/src/deseq/mod.rs +1267 -0
  78. crest_sc-0.3.0/src/harmony.rs +837 -0
  79. crest_sc-0.3.0/src/kernels.rs +662 -0
  80. crest_sc-0.3.0/src/knn.rs +575 -0
  81. crest_sc-0.3.0/src/leiden.rs +484 -0
  82. crest_sc-0.3.0/src/lib.rs +10 -0
  83. crest_sc-0.3.0/src/py.rs +680 -0
  84. crest_sc-0.3.0/src/simd.rs +23 -0
  85. crest_sc-0.3.0/src/umap/core.rs +48 -0
  86. crest_sc-0.3.0/src/umap/graph.rs +161 -0
  87. crest_sc-0.3.0/src/umap/mod.rs +5 -0
  88. crest_sc-0.3.0/src/umap/sgd.rs +290 -0
  89. crest_sc-0.3.0/src/umap/spectral.rs +168 -0
  90. crest_sc-0.3.0/src/umap/utils.rs +83 -0
  91. crest_sc-0.3.0/tests/data/deseq2_outlier_replacement_LRT_R.csv +401 -0
  92. crest_sc-0.3.0/tests/data/deseq2_outlier_replacement_LRT_reduced.txt +1 -0
  93. crest_sc-0.3.0/tests/data/deseq2_outlier_replacement_R.csv +401 -0
  94. crest_sc-0.3.0/tests/data/deseq2_outlier_replacement_input.csv +17 -0
  95. crest_sc-0.3.0/tests/data/deseq2_outlier_replacement_meta.json +1 -0
  96. crest_sc-0.3.0/tests/data/deseq2_three_level_batch_LRT_R.csv +401 -0
  97. crest_sc-0.3.0/tests/data/deseq2_three_level_batch_LRT_reduced.txt +1 -0
  98. crest_sc-0.3.0/tests/data/deseq2_three_level_batch_R.csv +401 -0
  99. crest_sc-0.3.0/tests/data/deseq2_three_level_batch_input.csv +10 -0
  100. crest_sc-0.3.0/tests/data/deseq2_three_level_batch_meta.json +1 -0
  101. crest_sc-0.3.0/tests/data/deseq2_two_vs_two_R.csv +401 -0
  102. crest_sc-0.3.0/tests/data/deseq2_two_vs_two_input.csv +5 -0
  103. crest_sc-0.3.0/tests/data/deseq2_two_vs_two_meta.json +1 -0
  104. crest_sc-0.3.0/tests/data/make_lrt_fixtures.R +21 -0
  105. crest_sc-0.3.0/tests/test_crest.py +455 -0
  106. crest_sc-0.3.0/work.md +4844 -0
@@ -0,0 +1,218 @@
1
+ # This file is autogenerated by maturin v1.12.4
2
+ # To update, run
3
+ #
4
+ # maturin generate-ci github
5
+ #
6
+ name: CI
7
+
8
+ on:
9
+ push:
10
+ branches:
11
+ - main
12
+ - master
13
+ tags:
14
+ - '*'
15
+ pull_request:
16
+ workflow_dispatch:
17
+
18
+ permissions:
19
+ contents: read
20
+
21
+ jobs:
22
+ test:
23
+ runs-on: ubuntu-22.04
24
+ steps:
25
+ - uses: actions/checkout@v6
26
+ - uses: actions/setup-python@v6
27
+ with:
28
+ python-version: "3.11"
29
+ - uses: dtolnay/rust-toolchain@stable
30
+ - name: Rust unit tests
31
+ run: cargo test --release
32
+ - name: Build and install
33
+ run: |
34
+ python -m venv .venv
35
+ . .venv/bin/activate
36
+ pip install maturin
37
+ maturin develop --release --extras test
38
+ - name: Python tests (incl. scanpy parity)
39
+ run: |
40
+ . .venv/bin/activate
41
+ pytest -q
42
+
43
+ docs:
44
+ # same build as Read the Docs (.readthedocs.yaml): no Rust, native module mocked
45
+ runs-on: ubuntu-22.04
46
+ steps:
47
+ - uses: actions/checkout@v6
48
+ - uses: actions/setup-python@v6
49
+ with:
50
+ python-version: "3.12"
51
+ - name: Build the documentation (warnings are errors)
52
+ run: |
53
+ pip install -r docs/requirements.txt
54
+ sphinx-build -W --keep-going -b html docs docs/_build/html
55
+
56
+ linux:
57
+ runs-on: ${{ matrix.platform.runner }}
58
+ strategy:
59
+ matrix:
60
+ platform:
61
+ - runner: ubuntu-22.04
62
+ target: x86_64
63
+ - runner: ubuntu-22.04
64
+ target: x86
65
+ - runner: ubuntu-22.04
66
+ target: aarch64
67
+ - runner: ubuntu-22.04
68
+ target: armv7
69
+ # s390x and ppc64le removed: psm crate assembly errors on these targets,
70
+ # and no single-cell analysis workloads run on mainframes
71
+ steps:
72
+ - uses: actions/checkout@v6
73
+ - uses: actions/setup-python@v6
74
+ with:
75
+ python-version: 3.x
76
+ - name: Build wheels
77
+ uses: PyO3/maturin-action@v1
78
+ with:
79
+ target: ${{ matrix.platform.target }}
80
+ args: --release --out dist --find-interpreter
81
+ sccache: ${{ !startsWith(github.ref, 'refs/tags/') }}
82
+ manylinux: auto
83
+ - name: Upload wheels
84
+ uses: actions/upload-artifact@v5
85
+ with:
86
+ name: wheels-linux-${{ matrix.platform.target }}
87
+ path: dist
88
+
89
+ musllinux:
90
+ runs-on: ${{ matrix.platform.runner }}
91
+ strategy:
92
+ matrix:
93
+ platform:
94
+ - runner: ubuntu-22.04
95
+ target: x86_64
96
+ - runner: ubuntu-22.04
97
+ target: x86
98
+ - runner: ubuntu-22.04
99
+ target: aarch64
100
+ - runner: ubuntu-22.04
101
+ target: armv7
102
+ steps:
103
+ - uses: actions/checkout@v6
104
+ - uses: actions/setup-python@v6
105
+ with:
106
+ python-version: 3.x
107
+ - name: Build wheels
108
+ uses: PyO3/maturin-action@v1
109
+ with:
110
+ target: ${{ matrix.platform.target }}
111
+ args: --release --out dist --find-interpreter
112
+ sccache: ${{ !startsWith(github.ref, 'refs/tags/') }}
113
+ manylinux: musllinux_1_2
114
+ - name: Upload wheels
115
+ uses: actions/upload-artifact@v5
116
+ with:
117
+ name: wheels-musllinux-${{ matrix.platform.target }}
118
+ path: dist
119
+
120
+ windows:
121
+ runs-on: ${{ matrix.platform.runner }}
122
+ strategy:
123
+ matrix:
124
+ platform:
125
+ - runner: windows-latest
126
+ target: x64
127
+ python_arch: x64
128
+ - runner: windows-latest
129
+ target: x86
130
+ python_arch: x86
131
+ - runner: windows-11-arm
132
+ target: aarch64
133
+ python_arch: arm64
134
+ steps:
135
+ - uses: actions/checkout@v6
136
+ - uses: actions/setup-python@v6
137
+ with:
138
+ python-version: 3.13
139
+ architecture: ${{ matrix.platform.python_arch }}
140
+ - name: Build wheels
141
+ uses: PyO3/maturin-action@v1
142
+ with:
143
+ target: ${{ matrix.platform.target }}
144
+ args: --release --out dist --find-interpreter
145
+ sccache: ${{ !startsWith(github.ref, 'refs/tags/') }}
146
+ - name: Upload wheels
147
+ uses: actions/upload-artifact@v5
148
+ with:
149
+ name: wheels-windows-${{ matrix.platform.target }}
150
+ path: dist
151
+
152
+ macos:
153
+ runs-on: ${{ matrix.platform.runner }}
154
+ strategy:
155
+ matrix:
156
+ platform:
157
+ - runner: macos-15-intel
158
+ target: x86_64
159
+ - runner: macos-latest
160
+ target: aarch64
161
+ steps:
162
+ - uses: actions/checkout@v6
163
+ - uses: actions/setup-python@v6
164
+ with:
165
+ python-version: 3.x
166
+ - name: Build wheels
167
+ uses: PyO3/maturin-action@v1
168
+ with:
169
+ target: ${{ matrix.platform.target }}
170
+ args: --release --out dist --find-interpreter
171
+ sccache: ${{ !startsWith(github.ref, 'refs/tags/') }}
172
+ - name: Upload wheels
173
+ uses: actions/upload-artifact@v5
174
+ with:
175
+ name: wheels-macos-${{ matrix.platform.target }}
176
+ path: dist
177
+
178
+ sdist:
179
+ runs-on: ubuntu-latest
180
+ steps:
181
+ - uses: actions/checkout@v6
182
+ - name: Build sdist
183
+ uses: PyO3/maturin-action@v1
184
+ with:
185
+ command: sdist
186
+ args: --out dist
187
+ - name: Upload sdist
188
+ uses: actions/upload-artifact@v5
189
+ with:
190
+ name: wheels-sdist
191
+ path: dist
192
+
193
+ release:
194
+ name: Release
195
+ runs-on: ubuntu-latest
196
+ if: ${{ startsWith(github.ref, 'refs/tags/') }}
197
+ needs: [test, linux, musllinux, windows, macos, sdist]
198
+ permissions:
199
+ # Use to sign the release artifacts
200
+ id-token: write
201
+ # Used to upload release artifacts
202
+ contents: write
203
+ # Used to generate artifact attestation
204
+ attestations: write
205
+ steps:
206
+ - uses: actions/download-artifact@v6
207
+ - name: Generate artifact attestation
208
+ uses: actions/attest-build-provenance@v3
209
+ with:
210
+ subject-path: 'wheels-*/*'
211
+ - name: Install uv
212
+ if: ${{ startsWith(github.ref, 'refs/tags/') }}
213
+ uses: astral-sh/setup-uv@v7
214
+ - name: Publish to PyPI
215
+ if: ${{ startsWith(github.ref, 'refs/tags/') }}
216
+ run: uv publish 'wheels-*/*'
217
+ env:
218
+ UV_PUBLISH_TOKEN: ${{ secrets.PYPI_API_TOKEN }}
@@ -0,0 +1,117 @@
1
+ /target
2
+
3
+ # Byte-compiled / optimized / DLL files
4
+ __pycache__/
5
+ .pytest_cache/
6
+ *.py[cod]
7
+
8
+ # Compiled extensions
9
+ *.so
10
+ *.dylib
11
+
12
+ # Distribution / packaging
13
+ .Python
14
+ .venv/
15
+ env/
16
+ bin/
17
+ build/
18
+ develop-eggs/
19
+ dist/
20
+ eggs/
21
+ lib/
22
+ lib64/
23
+ parts/
24
+ sdist/
25
+ var/
26
+ include/
27
+ man/
28
+ venv/
29
+ *.egg-info/
30
+ .installed.cfg
31
+ *.egg
32
+
33
+ # Installer logs
34
+ pip-log.txt
35
+ pip-delete-this-directory.txt
36
+ pip-selfcheck.json
37
+
38
+ # Unit test / coverage reports
39
+ htmlcov/
40
+ .tox/
41
+ .coverage
42
+ .cache
43
+ nosetests.xml
44
+ coverage.xml
45
+
46
+ # Translations
47
+ *.mo
48
+
49
+ # Mr Developer
50
+ .mr.developer.cfg
51
+ .project
52
+ .pydevproject
53
+
54
+ # Rope
55
+ .ropeproject
56
+
57
+ # Django stuff:
58
+ *.log
59
+ *.pot
60
+
61
+ .DS_Store
62
+
63
+ # Sphinx documentation
64
+ docs/_build/
65
+
66
+ # PyCharm
67
+ .idea/
68
+
69
+ # VSCode
70
+ .vscode/
71
+
72
+ # Pyenv
73
+ .python-version
74
+
75
+ # ================================
76
+ # BioPolars Project-Specific
77
+ # ================================
78
+
79
+ # Large datasets and data files
80
+ data/
81
+ !tests/data/
82
+ *.h5
83
+ *.mtx
84
+ *.mtx.gz
85
+ *.tsv
86
+ *.tsv.gz
87
+ *.parquet
88
+ *.h5ad
89
+
90
+ # Benchmark outputs
91
+ bench/__pycache__/
92
+ bench/*.png
93
+ bench/*.svg
94
+ bench/*.pdf
95
+
96
+ # Debug and scratch files
97
+ bench/debug_*.py
98
+
99
+ # Docker build artifacts
100
+ _docker_build/
101
+
102
+ # Deprecated/removed directories
103
+ bio-polars/
104
+ polars_src/
105
+ polars/
106
+ rust_analysis/
107
+ bench_results/
108
+ research/
109
+
110
+ # Misc
111
+ .ruff_cache/
112
+ .claude/
113
+
114
+ # Local benchmark runs
115
+ .venv-bench/
116
+ bench_data/
117
+ bench_results/
@@ -0,0 +1,20 @@
1
+ # Read the Docs build configuration: https://docs.readthedocs.io/en/stable/config-file/v2.html
2
+ #
3
+ # The docs are built with Sphinx from docs/. The package is NOT installed: CREST's
4
+ # compiled Rust extension would need a Rust build on Read the Docs. Instead docs/conf.py
5
+ # puts the repository on sys.path and mocks the native module, so autodoc can read every
6
+ # docstring from the pure-Python layer. See docs/readthedocs.md for the step-by-step setup.
7
+ version: 2
8
+
9
+ build:
10
+ os: ubuntu-24.04
11
+ tools:
12
+ python: "3.12"
13
+
14
+ sphinx:
15
+ configuration: docs/conf.py
16
+ fail_on_warning: false
17
+
18
+ python:
19
+ install:
20
+ - requirements: docs/requirements.txt
@@ -0,0 +1,95 @@
1
+ # Changelog
2
+
3
+ ## Unreleased
4
+
5
+ ## 0.3.0 (2026-09-29)
6
+
7
+ ### Changed
8
+ - The acronym now reads **Chunked** Rust Engine for Single-cell Transcriptomics (was
9
+ "Columnar"): the engine streams chunks of cells through fused Rust kernels; Polars is only
10
+ the table and Parquet layer. Package and import names are unchanged (`crest-sc`, `crest`).
11
+
12
+ ### New
13
+ - **Pseudobulk DESeq2 in Rust.** `crest.tl.DESeq2` ports DESeq2 1.42 `DESeq()` +
14
+ `results()`: median-of-ratios size factors, Cox-Reid gene-wise dispersions,
15
+ parametric trend, MAP shrinkage, NB-GLM Wald tests, Cook's filtering and
16
+ outlier replacement, and independent filtering. It matches R to ~1e-9 on
17
+ simulated designs, is ~18× faster than R and ~28× faster than pydeseq2 on
18
+ Kang 2018 pseudobulk. See `docs/deseq2.md`.
19
+ - `crest.tl.pseudobulk` (streamed raw-count sums per sample × group) and
20
+ `crest.tl.pseudobulk_de` (per-cell-type DESeq2 in one call).
21
+ - `bench/deseq2/`: R comparison on simulated designs and on Kang et al. 2018.
22
+ - **Harmony** batch integration in Rust (`crest.tl.harmony`), the harmony2 algorithm of
23
+ R harmony >= 1.2 / harmonypy 2.x; 3.5x faster than harmonypy at equal integration quality.
24
+ - **Scrublet** doublet detection (`crest.pp.scrublet`), sparse and streamed; 12x faster
25
+ than `scanpy.pp.scrublet` with the same AUROC on demuxlet-labelled doublets.
26
+ - `highly_variable_genes(flavor="seurat_v3" | "seurat_v3_paper", batch_key=...)`, with a
27
+ port of netlib loess (`crest._loess`); identical to scanpy.
28
+ - `crest.tl.leiden_sweep`: many resolutions/seeds on one graph in parallel, with
29
+ ARI-based stability; `crest.tl.adjusted_rand_index`.
30
+ - `crest.tl.ingest`: project a query onto a reference PCA, transfer labels and UMAP.
31
+ `uns['pca']['projection']` now stores what the projection needs.
32
+ - DESeq2 likelihood-ratio test: `DESeq2(test="LRT", reduced="~ ...")`, matching R.
33
+ - Native `knn_query` (reference -> query kNN, exact or IVF).
34
+ - `scripts/crest_paper_bench.sh` + `bench/paper/`: one-command publication benchmark
35
+ (datasets, per-core CPU/clock/memory monitoring, statistics, figures).
36
+ - `scripts/crest_git_housekeeping.sh`: repository housekeeping (dev branch, stale branches).
37
+ - **Documentation site** (Sphinx + MyST, Read the Docs): installation, quickstart, concepts,
38
+ full API reference, benchmarks, developer guide; `.readthedocs.yaml`; CI `docs` job.
39
+ - `work.md`: the development history; `CLAUDE.md` / `HANDOVER.md` rewritten for new sessions.
40
+
41
+ ### Removed
42
+ - The `.bio` Polars expression namespace and its Rust plugin code (duplicated the
43
+ validated API, was not validated itself, and tied the wheel to Polars' plugin ABI).
44
+ The `polars`/`pyo3-polars` Rust dependencies go with it.
45
+ - SLAF support (`BioFrame.from_slaf`, `crest/slaf_io.py`, the `slaf` extra).
46
+ - Legacy scripts: `bench/*.py` from the biopolars era, `notebooks/`, the Docker setup,
47
+ `docs/dev/`, and `bench/whitepaper/`'s scripts (superseded by `bench/paper/`; its
48
+ 0.2.0 results stay).
49
+
50
+ ### Fixed
51
+ - `BioFrame.to_anndata()` / `from_anndata()` no longer need pyarrow.
52
+ - `crest_paper_bench.sh` stops with a clear message when the work directory is not writable,
53
+ and records the work directory's device and filesystem.
54
+
55
+ ## 0.2.0
56
+
57
+ A correctness and performance release. Every step of the standard scanpy
58
+ workflow has been re-implemented and validated against scanpy.
59
+
60
+ ### Fixed (results from 0.1.0 should not be used)
61
+ - **kNN / Leiden / UMAP**: the HNSW index shuffled points and its id map was
62
+ ignored, so neighbour indices pointed at random cells (recall@15 = 0.005).
63
+ - **Leiden** was a single-level local-move heuristic (thousands of singleton
64
+ clusters) with a modularity gain off by a factor of 2; now Traag et al. 2019
65
+ Leiden (local moving, refinement, aggregation). Modularity equals or exceeds
66
+ `leidenalg`.
67
+ - **normalize_cpm / qc / filter_cells / scale / score_genes** Polars plugins
68
+ were declared elementwise, so Polars computed per-cell sums on partial
69
+ batches; `normalize_cpm` ignored `target_sum`.
70
+ - **rank_genes_groups** divided by non-zero counts instead of group sizes.
71
+ - **Randomized PCA** lost trailing components (no re-orthonormalisation).
72
+ - **HVG** "seurat" flavour was raw variance/mean (883/2000 overlap with scanpy).
73
+ - **Wilcoxon** p-values underflowed to 0.
74
+
75
+ ### New
76
+ - `crest.pp` / `crest.tl` scanpy-style API on `BioFrame`, with raw counts held
77
+ in compact CSR, a Polars triplet frame, or a Parquet dataset streamed from disk.
78
+ Filters and `normalize_total` / `log1p` / `scale` are applied lazily inside
79
+ fused native kernels: no normalised, scaled or dense copy is ever made.
80
+ - Exact PCA from a streamed Gram matrix with `sc.pp.scale(max_value)` applied
81
+ implicitly (0.000 deg from scanpy's ARPACK result).
82
+ - kNN: tiled GEMM brute force (small n) and IVF + NN-descent (large n).
83
+ - UMAP: umap-learn semantics, parallelised by domain decomposition.
84
+ - Sparse Wilcoxon ranking only non-zeros; all-groups Welch t-test.
85
+ - `score_genes` reproducing scanpy's control-gene sampling.
86
+ - Readers: `read_10x_h5` (optionally streamed to Parquet), `read_h5ad`,
87
+ `read_10x_mtx`; `write_parquet` / `read_parquet`; AnnData conversion.
88
+ - `bench/whitepaper/` benchmark harness.
89
+
90
+ ### Changed
91
+ - `bio.deseq2` renamed `bio.nb_glm` (it fits an NB GLM with given dispersions;
92
+ not the full DESeq2 procedure) and returns null when IRLS does not converge.
93
+ - In `bio.leiden` / `bio.umap`, `n_neighbors` now counts the cell itself (scanpy convention).
94
+ - `crest.tl.sparse_masked_pca` / `incremental_pca` removed (superseded by `crest.tl.pca`).
95
+ - Distribution renamed `crest-sc` (the import name stays `crest`); Python >= 3.10.
@@ -0,0 +1,182 @@
1
+ # CREST: operating manual for Claude Code
2
+
3
+ Read this file completely at the start of every session. Then read `HANDOVER.md` (current
4
+ status, open work, decisions). If you wonder *why* something is the way it is, search
5
+ `work.md`: it is the full history of the sessions that built the project, with every request
6
+ and every reply. Use `grep -n "<keyword>" work.md` rather than reading it top to bottom. The
7
+ "Context summary" sections are dense recaps. `scripts/export_chat.py` regenerates it from a
8
+ Claude Code transcript (`~/.claude/projects/<project>/<session>.jsonl`); append new sessions
9
+ rather than replacing the history.
10
+
11
+ ## 1. What this project is
12
+
13
+ CREST is a Python package (`pip install crest-sc`, `import crest`) with a Rust core
14
+ (PyO3 + maturin). It runs the scanpy single-cell RNA-seq workflow and gives **the same
15
+ results as scanpy** (and as R DESeq2 / harmonypy for those modules), while being several
16
+ times faster and using far less memory. The owner is a computational biologist who works
17
+ mostly in R (Seurat); the goal is a publishable package with an arXiv/bioRxiv-grade
18
+ performance paper.
19
+
20
+ The steps covered:
21
+
22
+ * **Workflow:** QC → filter → `normalize_total` → `log1p` → HVG → `scale` → PCA →
23
+ neighbours → Leiden → UMAP → marker genes (t-test / Wilcoxon) → `score_genes`.
24
+ * **After clustering:** pseudobulk DESeq2 (Wald + LRT), Harmony, Scrublet, `seurat_v3`
25
+ HVGs, Leiden resolution sweep, ingest (label transfer).
26
+
27
+ History in one paragraph: the project began (Feb 2026) as "biopolars", Polars expression
28
+ plugins for single-cell data. That design was wrong (per-cell sums over partial batches,
29
+ broken kNN id mapping, a fake Leiden). In Sept 2026 it was rebuilt around a `BioFrame` with
30
+ lazy transforms and fused Rust kernels, validated step by step against scanpy, and released
31
+ as 0.2.0. Downstream modules followed in 0.3.0. The Polars plugin layer was deleted; Polars
32
+ remains only as the metadata / table / Parquet layer. Hence the acronym now reads **Chunked**
33
+ Rust Engine for Single-cell Transcriptomics (it was "Columnar" until 0.3.0).
34
+
35
+ ## 2. The mental model (read before touching code)
36
+
37
+ **Raw counts are stored once and never modified.** Everything else is recorded, and applied
38
+ on the fly inside Rust kernels as the data is streamed chunk by chunk:
39
+
40
+ ```text
41
+ BioFrame
42
+ store : CSRStore (in-memory CSR) | FrameStore (Polars triplets) | ParquetStore (on disk, out-of-core)
43
+ obs : polars.DataFrame, kept cells; obs["cell_id"] indexes the store
44
+ var : polars.DataFrame, kept genes; var["gene_id"] indexes the store
45
+ ops : [("normalize_total", 1e4), ("log1p",)] <- recorded transforms
46
+ uns : {"scale": {...}, "hvg": {...}, "pca": {...}, "neighbors": {...}, ...}
47
+ obsm / varm : embeddings (X_pca, X_umap, X_pca_harmony) / PCs
48
+
49
+ for ctx in bf.iter_ctx(): # one raw chunk + cell_map/gene_map (-1 = filtered) + transform
50
+ _native.<kernel>(*ctx, outputs...) # Rust: filter + normalise + log1p (+ scale) per cell, accumulate
51
+ ```
52
+
53
+ The consequences are rules, not suggestions:
54
+
55
+ * **Filters return a new BioFrame that shares the store:** `bf = crest.pp.filter_cells(bf, ...)`.
56
+ All other functions mutate `bf` in place and also return it.
57
+ * **Nothing is ever cells × genes and dense.** Results are per-gene, per-cell, per-group, or
58
+ genes × genes (the PCA Gram matrix, with HVGs only; capped at 20k genes).
59
+ * **PCA with `scale` is exact without densifying.** Zeros map to a per-gene constant
60
+ `b_j`, so `Z = 1·bᵀ + S` with `S` sparse. Centring removes `1·bᵀ`, so the Gram matrix
61
+ is accumulated from sparse rows (`docs/memory_model.md`).
62
+ * **What each step reads:**
63
+ - raw counts: QC, filters, `seurat_v3`, Scrublet, pseudobulk / DESeq2;
64
+ - log-normalised values: `seurat` / `cell_ranger` HVG, DE, `score_genes`;
65
+ - scaled values: PCA;
66
+ - `obsm` embeddings: neighbours, Leiden, UMAP, Harmony, ingest.
67
+ * `ParquetStore` = out-of-core: one Parquet part in RAM at a time, ~1 GB peak at any size,
68
+ ~2× slower.
69
+
70
+ Full user-level explanation: `docs/concepts.md`. The chunk protocol tuple and the table of
71
+ all native functions: `docs/development.md`.
72
+
73
+ ## 3. Build, test, docs
74
+
75
+ ```bash
76
+ # one-time: Rust stable >= 1.80, Python >= 3.10; the owner uses uv
77
+ uv venv .venv --python 3.11 && source .venv/bin/activate && uv pip install maturin
78
+ maturin develop --release -E test # rebuild after ANY change under src/ (always --release)
79
+ cargo test --release # 26 Rust unit tests
80
+ pytest -q # 27 Python tests: scanpy / R-DESeq2 / harmonypy / skmisc / skimage parity
81
+ pip install -r docs/requirements.txt && sphinx-build -W -b html docs docs/_build/html # docs
82
+ ```
83
+
84
+ * A Python-only change needs no rebuild.
85
+ * If `maturin` says "Both VIRTUAL_ENV and CONDA_PREFIX are set", run `conda deactivate`
86
+ (the owner's shell starts in conda `base`).
87
+ * The R-DESeq2 reference outputs are stored in `tests/data/`, so the tests don't need R.
88
+ They are regenerated with `Rscript tests/data/make_lrt_fixtures.R` and the scripts in
89
+ `bench/deseq2/`.
90
+
91
+ ## 4. Where things are
92
+
93
+ | need | file |
94
+ |---|---|
95
+ | BioFrame, stores, `iter_ctx`, AnnData / Parquet I/O | `crest/core.py` |
96
+ | readers | `crest/io.py` |
97
+ | QC, filters, lazy transforms, HVG (3 flavours), neighbours | `crest/pp.py` |
98
+ | PCA, Leiden, `leiden_sweep`, UMAP, DE, `score_genes` | `crest/tl.py` |
99
+ | pseudobulk + DESeq2 (formulas, Wald / LRT, results, independent filtering) | `crest/deseq2.py` + `src/deseq/` |
100
+ | Harmony / Scrublet / ingest / loess | `crest/harmony.py` + `src/harmony.rs`, `crest/doublets.py`, `crest/ingest.py`, `crest/_loess.py` |
101
+ | streaming kernels | `src/kernels.rs` (`ChunkView`) |
102
+ | kNN (exact GEMM ≤ 20k cells; IVF + NN-descent above; `knn_query`) | `src/knn.rs` |
103
+ | Leiden, UMAP | `src/leiden.rs`, `src/umap/` |
104
+ | Python ↔ Rust bindings (validation, GIL release) | `src/py.rs` |
105
+ | tests | `tests/test_crest.py`, `tests/data/` |
106
+ | paper benchmark | `scripts/crest_paper_bench.sh` → `bench/paper/{datasets,run_one,run_all,monitor,accuracy,summarize}.py` |
107
+ | module benchmarks | `bench/{deseq2,harmony,doublets,ingest}/` (results in `results/`) |
108
+ | user docs (Read the Docs) | `docs/*.md`, `docs/conf.py`, `.readthedocs.yaml`, `docs/readthedocs.md` |
109
+ | design / validation notes | `docs/concepts.md`, `docs/memory_model.md`, `docs/deseq2.md`, `docs/downstream.md`, `docs/benchmarks.md` |
110
+ | history | `work.md`, `CHANGELOG.md` |
111
+
112
+ ## 5. Rules
113
+
114
+ 1. **Parity first.** Results must match the reference tool: scanpy 1.11 for the workflow,
115
+ R DESeq2 1.42 for `DESeq2`, harmonypy 2.x for Harmony, `skmisc` for loess and `skimage`
116
+ for `threshold_minimum`.
117
+ - A new method gets a parity test before it is optimised.
118
+ - If a change moves a parity test, find out why. **Never loosen a tolerance** to make
119
+ it pass.
120
+ - Where exact parity is impossible (random streams: Scrublet pairs, Harmony init), compare
121
+ outcomes statistically and document it.
122
+ 2. **Bounded memory.** Stream through `iter_ctx()`. Never build dense cells × genes, and never
123
+ call `to_scipy()` inside library code.
124
+ 3. **Speed lives in Rust.** Loops over non-zeros or cells go in `src/`. Release the GIL
125
+ (`py.allow_threads`), parallelise with rayon, and validate inputs in `py.rs` so errors
126
+ raise instead of panicking.
127
+ 4. **scanpy-compatible names and defaults.** Deviations are documented in the docstring and
128
+ in `docs/`.
129
+ 5. **Honest benchmarks.** These rules come from the owner:
130
+ - The headline comparison is the `core` pipeline; optional modules go in a separate table.
131
+ - Speed-ups are quoted at each tool's *best* thread count and at matched threads; the
132
+ conservative one is the headline.
133
+ - Report the steps where CREST is not faster, its thread-scaling limits, and every failed
134
+ run.
135
+ - No cherry-picking.
136
+ 6. **Docs move with code.**
137
+ - A new public function gets a docstring and an entry in `docs/api.md`.
138
+ - A behaviour change updates `docs/` and `CHANGELOG.md` (Unreleased).
139
+ - `sphinx-build -W` must stay clean (CI enforces it).
140
+
141
+ ## 6. Git and GitHub
142
+
143
+ * Branches: `main` (releases only; **never push to it**), `dev` (integration; PRs land
144
+ here), and feature branches. A release merges `dev` → `main` and tags `vX.Y.Z`.
145
+ * The version lives in `Cargo.toml`. `pyproject.toml` and `docs/conf.py` read it from there.
146
+ `crest/__init__.py` has a copy: bump both.
147
+ * Commit prefixes: `feat:`, `fix:`, `bench:`, `docs:`, `chore:`. Before pushing, run
148
+ `cargo test --release && pytest -q`.
149
+ * CI (`.github/workflows/CI.yml`) runs on PRs and pushes to main:
150
+ - the `test` job (cargo + pytest);
151
+ - the `docs` job (Sphinx, warnings are errors);
152
+ - wheel builds for all platforms, published to PyPI on a tag.
153
+ * Deleting or tagging shared branches is done by the owner with
154
+ `scripts/crest_git_housekeeping.sh`; don't try to work around permission refusals.
155
+ * Old branches are archived as tags `archive/feat-leiden-hnsw-faer` and
156
+ `archive/fix-production-readiness`.
157
+
158
+ ## 7. The owner's machine (where benchmarks run)
159
+
160
+ * **Machine:** `rinamochana`, AMD Threadripper PRO 3975WX (32 cores / 64 threads),
161
+ 128 GB DDR4-2667 ECC, Ubuntu 26.04.
162
+ * **Disks:**
163
+ - `/home` is small (OS only): **never write data or caches there**;
164
+ - `/mnt/scratch` is a 1.8 TB NVMe (ext4, `/dev/nvme0n1`); run benchmarks here;
165
+ - `/storage` is a large ZFS pool on HDDs.
166
+ * **Benchmark:**
167
+ - Run from `/mnt/scratch` with
168
+ `bash crest_paper_bench.sh --tier quick|standard|full`.
169
+ - Everything, including uv / cargo / numba caches, goes under `./crest-bench`.
170
+ - Results go to `crest-bench/results/<host>-<date>/`, and a `.tar.gz` is written to send
171
+ back.
172
+ * The owner uses `uv` for Python environments.
173
+
174
+ ## 8. Working style the owner expects
175
+
176
+ * **Think thoroughly and weigh trade-offs.** Quality comes first, and every claim should be
177
+ verified: run the code, rebuild, re-test.
178
+ * Scripts they run locally must work from the current directory, log everything, and fail
179
+ early with a clear message.
180
+ * When you finish a session that changed anything, update `HANDOVER.md` (Status and Next
181
+ steps), and add a `CHANGELOG.md` entry for user-visible changes.
182
+ * Never put model identifiers in commits, code or docs.