crest-sc 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crest_sc-0.3.0/.github/workflows/CI.yml +218 -0
- crest_sc-0.3.0/.gitignore +117 -0
- crest_sc-0.3.0/.readthedocs.yaml +20 -0
- crest_sc-0.3.0/CHANGELOG.md +95 -0
- crest_sc-0.3.0/CLAUDE.md +182 -0
- crest_sc-0.3.0/Cargo.lock +1716 -0
- crest_sc-0.3.0/Cargo.toml +22 -0
- crest_sc-0.3.0/HANDOVER.md +124 -0
- crest_sc-0.3.0/LICENSE +21 -0
- crest_sc-0.3.0/PKG-INFO +232 -0
- crest_sc-0.3.0/README.md +179 -0
- crest_sc-0.3.0/bench/deseq2/compare_r.py +164 -0
- crest_sc-0.3.0/bench/deseq2/kang_pseudobulk.py +184 -0
- crest_sc-0.3.0/bench/deseq2/results/comparison.csv +7 -0
- crest_sc-0.3.0/bench/deseq2/results/comparison.md +8 -0
- crest_sc-0.3.0/bench/deseq2/results/kang_pseudobulk.csv +9 -0
- crest_sc-0.3.0/bench/deseq2/results/kang_pseudobulk.md +14 -0
- crest_sc-0.3.0/bench/doublets/compare_scrublet.py +82 -0
- crest_sc-0.3.0/bench/doublets/results/kang_scrublet.csv +7 -0
- crest_sc-0.3.0/bench/doublets/results/kang_scrublet.md +6 -0
- crest_sc-0.3.0/bench/harmony/compare_harmonypy.py +137 -0
- crest_sc-0.3.0/bench/harmony/results/kang_harmony.csv +8 -0
- crest_sc-0.3.0/bench/harmony/results/kang_harmony.md +13 -0
- crest_sc-0.3.0/bench/ingest/compare_scanpy_ingest.py +91 -0
- crest_sc-0.3.0/bench/ingest/results/kang_ingest.csv +3 -0
- crest_sc-0.3.0/bench/ingest/results/kang_ingest.md +22 -0
- crest_sc-0.3.0/bench/paper/accuracy.py +122 -0
- crest_sc-0.3.0/bench/paper/datasets.py +310 -0
- crest_sc-0.3.0/bench/paper/make_dataset.py +100 -0
- crest_sc-0.3.0/bench/paper/monitor.py +159 -0
- crest_sc-0.3.0/bench/paper/run_all.py +143 -0
- crest_sc-0.3.0/bench/paper/run_one.py +276 -0
- crest_sc-0.3.0/bench/paper/summarize.py +436 -0
- crest_sc-0.3.0/bench/whitepaper/results/crest-ooc_pbmc68k.json +56 -0
- crest_sc-0.3.0/bench/whitepaper/results/crest-ooc_pbmc_100k.json +56 -0
- crest_sc-0.3.0/bench/whitepaper/results/crest-ooc_pbmc_200k.json +56 -0
- crest_sc-0.3.0/bench/whitepaper/results/crest_pbmc68k.json +56 -0
- crest_sc-0.3.0/bench/whitepaper/results/crest_pbmc_100k.json +56 -0
- crest_sc-0.3.0/bench/whitepaper/results/crest_pbmc_200k.json +56 -0
- crest_sc-0.3.0/bench/whitepaper/results/fig_scaling.png +0 -0
- crest_sc-0.3.0/bench/whitepaper/results/fig_scaling.svg +1875 -0
- crest_sc-0.3.0/bench/whitepaper/results/fig_step_times.png +0 -0
- crest_sc-0.3.0/bench/whitepaper/results/fig_step_times.svg +3767 -0
- crest_sc-0.3.0/bench/whitepaper/results/report.md +66 -0
- crest_sc-0.3.0/bench/whitepaper/results/scanpy_pbmc68k.json +56 -0
- crest_sc-0.3.0/bench/whitepaper/results/scanpy_pbmc_100k.json +56 -0
- crest_sc-0.3.0/crest/__init__.py +14 -0
- crest_sc-0.3.0/crest/_loess.py +86 -0
- crest_sc-0.3.0/crest/core.py +372 -0
- crest_sc-0.3.0/crest/deseq2.py +452 -0
- crest_sc-0.3.0/crest/doublets.py +192 -0
- crest_sc-0.3.0/crest/harmony.py +90 -0
- crest_sc-0.3.0/crest/ingest.py +116 -0
- crest_sc-0.3.0/crest/io.py +185 -0
- crest_sc-0.3.0/crest/pp.py +330 -0
- crest_sc-0.3.0/crest/tl.py +332 -0
- crest_sc-0.3.0/docs/api.md +139 -0
- crest_sc-0.3.0/docs/benchmarks.md +99 -0
- crest_sc-0.3.0/docs/changelog.md +2 -0
- crest_sc-0.3.0/docs/concepts.md +128 -0
- crest_sc-0.3.0/docs/conf.py +67 -0
- crest_sc-0.3.0/docs/deseq2.md +95 -0
- crest_sc-0.3.0/docs/development.md +129 -0
- crest_sc-0.3.0/docs/downstream.md +119 -0
- crest_sc-0.3.0/docs/index.md +69 -0
- crest_sc-0.3.0/docs/installation.md +57 -0
- crest_sc-0.3.0/docs/memory_model.md +53 -0
- crest_sc-0.3.0/docs/quickstart.md +112 -0
- crest_sc-0.3.0/docs/readthedocs.md +92 -0
- crest_sc-0.3.0/docs/requirements.txt +9 -0
- crest_sc-0.3.0/pyproject.toml +50 -0
- crest_sc-0.3.0/scripts/crest_git_housekeeping.sh +118 -0
- crest_sc-0.3.0/scripts/crest_paper_bench.sh +287 -0
- crest_sc-0.3.0/scripts/export_chat.py +126 -0
- crest_sc-0.3.0/src/deseq/linalg.rs +161 -0
- crest_sc-0.3.0/src/deseq/lowess.rs +170 -0
- crest_sc-0.3.0/src/deseq/mod.rs +1267 -0
- crest_sc-0.3.0/src/harmony.rs +837 -0
- crest_sc-0.3.0/src/kernels.rs +662 -0
- crest_sc-0.3.0/src/knn.rs +575 -0
- crest_sc-0.3.0/src/leiden.rs +484 -0
- crest_sc-0.3.0/src/lib.rs +10 -0
- crest_sc-0.3.0/src/py.rs +680 -0
- crest_sc-0.3.0/src/simd.rs +23 -0
- crest_sc-0.3.0/src/umap/core.rs +48 -0
- crest_sc-0.3.0/src/umap/graph.rs +161 -0
- crest_sc-0.3.0/src/umap/mod.rs +5 -0
- crest_sc-0.3.0/src/umap/sgd.rs +290 -0
- crest_sc-0.3.0/src/umap/spectral.rs +168 -0
- crest_sc-0.3.0/src/umap/utils.rs +83 -0
- crest_sc-0.3.0/tests/data/deseq2_outlier_replacement_LRT_R.csv +401 -0
- crest_sc-0.3.0/tests/data/deseq2_outlier_replacement_LRT_reduced.txt +1 -0
- crest_sc-0.3.0/tests/data/deseq2_outlier_replacement_R.csv +401 -0
- crest_sc-0.3.0/tests/data/deseq2_outlier_replacement_input.csv +17 -0
- crest_sc-0.3.0/tests/data/deseq2_outlier_replacement_meta.json +1 -0
- crest_sc-0.3.0/tests/data/deseq2_three_level_batch_LRT_R.csv +401 -0
- crest_sc-0.3.0/tests/data/deseq2_three_level_batch_LRT_reduced.txt +1 -0
- crest_sc-0.3.0/tests/data/deseq2_three_level_batch_R.csv +401 -0
- crest_sc-0.3.0/tests/data/deseq2_three_level_batch_input.csv +10 -0
- crest_sc-0.3.0/tests/data/deseq2_three_level_batch_meta.json +1 -0
- crest_sc-0.3.0/tests/data/deseq2_two_vs_two_R.csv +401 -0
- crest_sc-0.3.0/tests/data/deseq2_two_vs_two_input.csv +5 -0
- crest_sc-0.3.0/tests/data/deseq2_two_vs_two_meta.json +1 -0
- crest_sc-0.3.0/tests/data/make_lrt_fixtures.R +21 -0
- crest_sc-0.3.0/tests/test_crest.py +455 -0
- crest_sc-0.3.0/work.md +4844 -0
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
# This file is autogenerated by maturin v1.12.4
|
|
2
|
+
# To update, run
|
|
3
|
+
#
|
|
4
|
+
# maturin generate-ci github
|
|
5
|
+
#
|
|
6
|
+
name: CI
|
|
7
|
+
|
|
8
|
+
on:
|
|
9
|
+
push:
|
|
10
|
+
branches:
|
|
11
|
+
- main
|
|
12
|
+
- master
|
|
13
|
+
tags:
|
|
14
|
+
- '*'
|
|
15
|
+
pull_request:
|
|
16
|
+
workflow_dispatch:
|
|
17
|
+
|
|
18
|
+
permissions:
|
|
19
|
+
contents: read
|
|
20
|
+
|
|
21
|
+
jobs:
|
|
22
|
+
test:
|
|
23
|
+
runs-on: ubuntu-22.04
|
|
24
|
+
steps:
|
|
25
|
+
- uses: actions/checkout@v6
|
|
26
|
+
- uses: actions/setup-python@v6
|
|
27
|
+
with:
|
|
28
|
+
python-version: "3.11"
|
|
29
|
+
- uses: dtolnay/rust-toolchain@stable
|
|
30
|
+
- name: Rust unit tests
|
|
31
|
+
run: cargo test --release
|
|
32
|
+
- name: Build and install
|
|
33
|
+
run: |
|
|
34
|
+
python -m venv .venv
|
|
35
|
+
. .venv/bin/activate
|
|
36
|
+
pip install maturin
|
|
37
|
+
maturin develop --release --extras test
|
|
38
|
+
- name: Python tests (incl. scanpy parity)
|
|
39
|
+
run: |
|
|
40
|
+
. .venv/bin/activate
|
|
41
|
+
pytest -q
|
|
42
|
+
|
|
43
|
+
docs:
|
|
44
|
+
# same build as Read the Docs (.readthedocs.yaml): no Rust, native module mocked
|
|
45
|
+
runs-on: ubuntu-22.04
|
|
46
|
+
steps:
|
|
47
|
+
- uses: actions/checkout@v6
|
|
48
|
+
- uses: actions/setup-python@v6
|
|
49
|
+
with:
|
|
50
|
+
python-version: "3.12"
|
|
51
|
+
- name: Build the documentation (warnings are errors)
|
|
52
|
+
run: |
|
|
53
|
+
pip install -r docs/requirements.txt
|
|
54
|
+
sphinx-build -W --keep-going -b html docs docs/_build/html
|
|
55
|
+
|
|
56
|
+
linux:
|
|
57
|
+
runs-on: ${{ matrix.platform.runner }}
|
|
58
|
+
strategy:
|
|
59
|
+
matrix:
|
|
60
|
+
platform:
|
|
61
|
+
- runner: ubuntu-22.04
|
|
62
|
+
target: x86_64
|
|
63
|
+
- runner: ubuntu-22.04
|
|
64
|
+
target: x86
|
|
65
|
+
- runner: ubuntu-22.04
|
|
66
|
+
target: aarch64
|
|
67
|
+
- runner: ubuntu-22.04
|
|
68
|
+
target: armv7
|
|
69
|
+
# s390x and ppc64le removed: psm crate assembly errors on these targets,
|
|
70
|
+
# and no single-cell analysis workloads run on mainframes
|
|
71
|
+
steps:
|
|
72
|
+
- uses: actions/checkout@v6
|
|
73
|
+
- uses: actions/setup-python@v6
|
|
74
|
+
with:
|
|
75
|
+
python-version: 3.x
|
|
76
|
+
- name: Build wheels
|
|
77
|
+
uses: PyO3/maturin-action@v1
|
|
78
|
+
with:
|
|
79
|
+
target: ${{ matrix.platform.target }}
|
|
80
|
+
args: --release --out dist --find-interpreter
|
|
81
|
+
sccache: ${{ !startsWith(github.ref, 'refs/tags/') }}
|
|
82
|
+
manylinux: auto
|
|
83
|
+
- name: Upload wheels
|
|
84
|
+
uses: actions/upload-artifact@v5
|
|
85
|
+
with:
|
|
86
|
+
name: wheels-linux-${{ matrix.platform.target }}
|
|
87
|
+
path: dist
|
|
88
|
+
|
|
89
|
+
musllinux:
|
|
90
|
+
runs-on: ${{ matrix.platform.runner }}
|
|
91
|
+
strategy:
|
|
92
|
+
matrix:
|
|
93
|
+
platform:
|
|
94
|
+
- runner: ubuntu-22.04
|
|
95
|
+
target: x86_64
|
|
96
|
+
- runner: ubuntu-22.04
|
|
97
|
+
target: x86
|
|
98
|
+
- runner: ubuntu-22.04
|
|
99
|
+
target: aarch64
|
|
100
|
+
- runner: ubuntu-22.04
|
|
101
|
+
target: armv7
|
|
102
|
+
steps:
|
|
103
|
+
- uses: actions/checkout@v6
|
|
104
|
+
- uses: actions/setup-python@v6
|
|
105
|
+
with:
|
|
106
|
+
python-version: 3.x
|
|
107
|
+
- name: Build wheels
|
|
108
|
+
uses: PyO3/maturin-action@v1
|
|
109
|
+
with:
|
|
110
|
+
target: ${{ matrix.platform.target }}
|
|
111
|
+
args: --release --out dist --find-interpreter
|
|
112
|
+
sccache: ${{ !startsWith(github.ref, 'refs/tags/') }}
|
|
113
|
+
manylinux: musllinux_1_2
|
|
114
|
+
- name: Upload wheels
|
|
115
|
+
uses: actions/upload-artifact@v5
|
|
116
|
+
with:
|
|
117
|
+
name: wheels-musllinux-${{ matrix.platform.target }}
|
|
118
|
+
path: dist
|
|
119
|
+
|
|
120
|
+
windows:
|
|
121
|
+
runs-on: ${{ matrix.platform.runner }}
|
|
122
|
+
strategy:
|
|
123
|
+
matrix:
|
|
124
|
+
platform:
|
|
125
|
+
- runner: windows-latest
|
|
126
|
+
target: x64
|
|
127
|
+
python_arch: x64
|
|
128
|
+
- runner: windows-latest
|
|
129
|
+
target: x86
|
|
130
|
+
python_arch: x86
|
|
131
|
+
- runner: windows-11-arm
|
|
132
|
+
target: aarch64
|
|
133
|
+
python_arch: arm64
|
|
134
|
+
steps:
|
|
135
|
+
- uses: actions/checkout@v6
|
|
136
|
+
- uses: actions/setup-python@v6
|
|
137
|
+
with:
|
|
138
|
+
python-version: 3.13
|
|
139
|
+
architecture: ${{ matrix.platform.python_arch }}
|
|
140
|
+
- name: Build wheels
|
|
141
|
+
uses: PyO3/maturin-action@v1
|
|
142
|
+
with:
|
|
143
|
+
target: ${{ matrix.platform.target }}
|
|
144
|
+
args: --release --out dist --find-interpreter
|
|
145
|
+
sccache: ${{ !startsWith(github.ref, 'refs/tags/') }}
|
|
146
|
+
- name: Upload wheels
|
|
147
|
+
uses: actions/upload-artifact@v5
|
|
148
|
+
with:
|
|
149
|
+
name: wheels-windows-${{ matrix.platform.target }}
|
|
150
|
+
path: dist
|
|
151
|
+
|
|
152
|
+
macos:
|
|
153
|
+
runs-on: ${{ matrix.platform.runner }}
|
|
154
|
+
strategy:
|
|
155
|
+
matrix:
|
|
156
|
+
platform:
|
|
157
|
+
- runner: macos-15-intel
|
|
158
|
+
target: x86_64
|
|
159
|
+
- runner: macos-latest
|
|
160
|
+
target: aarch64
|
|
161
|
+
steps:
|
|
162
|
+
- uses: actions/checkout@v6
|
|
163
|
+
- uses: actions/setup-python@v6
|
|
164
|
+
with:
|
|
165
|
+
python-version: 3.x
|
|
166
|
+
- name: Build wheels
|
|
167
|
+
uses: PyO3/maturin-action@v1
|
|
168
|
+
with:
|
|
169
|
+
target: ${{ matrix.platform.target }}
|
|
170
|
+
args: --release --out dist --find-interpreter
|
|
171
|
+
sccache: ${{ !startsWith(github.ref, 'refs/tags/') }}
|
|
172
|
+
- name: Upload wheels
|
|
173
|
+
uses: actions/upload-artifact@v5
|
|
174
|
+
with:
|
|
175
|
+
name: wheels-macos-${{ matrix.platform.target }}
|
|
176
|
+
path: dist
|
|
177
|
+
|
|
178
|
+
sdist:
|
|
179
|
+
runs-on: ubuntu-latest
|
|
180
|
+
steps:
|
|
181
|
+
- uses: actions/checkout@v6
|
|
182
|
+
- name: Build sdist
|
|
183
|
+
uses: PyO3/maturin-action@v1
|
|
184
|
+
with:
|
|
185
|
+
command: sdist
|
|
186
|
+
args: --out dist
|
|
187
|
+
- name: Upload sdist
|
|
188
|
+
uses: actions/upload-artifact@v5
|
|
189
|
+
with:
|
|
190
|
+
name: wheels-sdist
|
|
191
|
+
path: dist
|
|
192
|
+
|
|
193
|
+
release:
|
|
194
|
+
name: Release
|
|
195
|
+
runs-on: ubuntu-latest
|
|
196
|
+
if: ${{ startsWith(github.ref, 'refs/tags/') }}
|
|
197
|
+
needs: [test, linux, musllinux, windows, macos, sdist]
|
|
198
|
+
permissions:
|
|
199
|
+
# Use to sign the release artifacts
|
|
200
|
+
id-token: write
|
|
201
|
+
# Used to upload release artifacts
|
|
202
|
+
contents: write
|
|
203
|
+
# Used to generate artifact attestation
|
|
204
|
+
attestations: write
|
|
205
|
+
steps:
|
|
206
|
+
- uses: actions/download-artifact@v6
|
|
207
|
+
- name: Generate artifact attestation
|
|
208
|
+
uses: actions/attest-build-provenance@v3
|
|
209
|
+
with:
|
|
210
|
+
subject-path: 'wheels-*/*'
|
|
211
|
+
- name: Install uv
|
|
212
|
+
if: ${{ startsWith(github.ref, 'refs/tags/') }}
|
|
213
|
+
uses: astral-sh/setup-uv@v7
|
|
214
|
+
- name: Publish to PyPI
|
|
215
|
+
if: ${{ startsWith(github.ref, 'refs/tags/') }}
|
|
216
|
+
run: uv publish 'wheels-*/*'
|
|
217
|
+
env:
|
|
218
|
+
UV_PUBLISH_TOKEN: ${{ secrets.PYPI_API_TOKEN }}
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
/target
|
|
2
|
+
|
|
3
|
+
# Byte-compiled / optimized / DLL files
|
|
4
|
+
__pycache__/
|
|
5
|
+
.pytest_cache/
|
|
6
|
+
*.py[cod]
|
|
7
|
+
|
|
8
|
+
# Compiled extensions
|
|
9
|
+
*.so
|
|
10
|
+
*.dylib
|
|
11
|
+
|
|
12
|
+
# Distribution / packaging
|
|
13
|
+
.Python
|
|
14
|
+
.venv/
|
|
15
|
+
env/
|
|
16
|
+
bin/
|
|
17
|
+
build/
|
|
18
|
+
develop-eggs/
|
|
19
|
+
dist/
|
|
20
|
+
eggs/
|
|
21
|
+
lib/
|
|
22
|
+
lib64/
|
|
23
|
+
parts/
|
|
24
|
+
sdist/
|
|
25
|
+
var/
|
|
26
|
+
include/
|
|
27
|
+
man/
|
|
28
|
+
venv/
|
|
29
|
+
*.egg-info/
|
|
30
|
+
.installed.cfg
|
|
31
|
+
*.egg
|
|
32
|
+
|
|
33
|
+
# Installer logs
|
|
34
|
+
pip-log.txt
|
|
35
|
+
pip-delete-this-directory.txt
|
|
36
|
+
pip-selfcheck.json
|
|
37
|
+
|
|
38
|
+
# Unit test / coverage reports
|
|
39
|
+
htmlcov/
|
|
40
|
+
.tox/
|
|
41
|
+
.coverage
|
|
42
|
+
.cache
|
|
43
|
+
nosetests.xml
|
|
44
|
+
coverage.xml
|
|
45
|
+
|
|
46
|
+
# Translations
|
|
47
|
+
*.mo
|
|
48
|
+
|
|
49
|
+
# Mr Developer
|
|
50
|
+
.mr.developer.cfg
|
|
51
|
+
.project
|
|
52
|
+
.pydevproject
|
|
53
|
+
|
|
54
|
+
# Rope
|
|
55
|
+
.ropeproject
|
|
56
|
+
|
|
57
|
+
# Django stuff:
|
|
58
|
+
*.log
|
|
59
|
+
*.pot
|
|
60
|
+
|
|
61
|
+
.DS_Store
|
|
62
|
+
|
|
63
|
+
# Sphinx documentation
|
|
64
|
+
docs/_build/
|
|
65
|
+
|
|
66
|
+
# PyCharm
|
|
67
|
+
.idea/
|
|
68
|
+
|
|
69
|
+
# VSCode
|
|
70
|
+
.vscode/
|
|
71
|
+
|
|
72
|
+
# Pyenv
|
|
73
|
+
.python-version
|
|
74
|
+
|
|
75
|
+
# ================================
|
|
76
|
+
# BioPolars Project-Specific
|
|
77
|
+
# ================================
|
|
78
|
+
|
|
79
|
+
# Large datasets and data files
|
|
80
|
+
data/
|
|
81
|
+
!tests/data/
|
|
82
|
+
*.h5
|
|
83
|
+
*.mtx
|
|
84
|
+
*.mtx.gz
|
|
85
|
+
*.tsv
|
|
86
|
+
*.tsv.gz
|
|
87
|
+
*.parquet
|
|
88
|
+
*.h5ad
|
|
89
|
+
|
|
90
|
+
# Benchmark outputs
|
|
91
|
+
bench/__pycache__/
|
|
92
|
+
bench/*.png
|
|
93
|
+
bench/*.svg
|
|
94
|
+
bench/*.pdf
|
|
95
|
+
|
|
96
|
+
# Debug and scratch files
|
|
97
|
+
bench/debug_*.py
|
|
98
|
+
|
|
99
|
+
# Docker build artifacts
|
|
100
|
+
_docker_build/
|
|
101
|
+
|
|
102
|
+
# Deprecated/removed directories
|
|
103
|
+
bio-polars/
|
|
104
|
+
polars_src/
|
|
105
|
+
polars/
|
|
106
|
+
rust_analysis/
|
|
107
|
+
bench_results/
|
|
108
|
+
research/
|
|
109
|
+
|
|
110
|
+
# Misc
|
|
111
|
+
.ruff_cache/
|
|
112
|
+
.claude/
|
|
113
|
+
|
|
114
|
+
# Local benchmark runs
|
|
115
|
+
.venv-bench/
|
|
116
|
+
bench_data/
|
|
117
|
+
bench_results/
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
# Read the Docs build configuration: https://docs.readthedocs.io/en/stable/config-file/v2.html
|
|
2
|
+
#
|
|
3
|
+
# The docs are built with Sphinx from docs/. The package is NOT installed: CREST's
|
|
4
|
+
# compiled Rust extension would need a Rust build on Read the Docs. Instead docs/conf.py
|
|
5
|
+
# puts the repository on sys.path and mocks the native module, so autodoc can read every
|
|
6
|
+
# docstring from the pure-Python layer. See docs/readthedocs.md for the step-by-step setup.
|
|
7
|
+
version: 2
|
|
8
|
+
|
|
9
|
+
build:
|
|
10
|
+
os: ubuntu-24.04
|
|
11
|
+
tools:
|
|
12
|
+
python: "3.12"
|
|
13
|
+
|
|
14
|
+
sphinx:
|
|
15
|
+
configuration: docs/conf.py
|
|
16
|
+
fail_on_warning: false
|
|
17
|
+
|
|
18
|
+
python:
|
|
19
|
+
install:
|
|
20
|
+
- requirements: docs/requirements.txt
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## Unreleased
|
|
4
|
+
|
|
5
|
+
## 0.3.0 (2026-09-29)
|
|
6
|
+
|
|
7
|
+
### Changed
|
|
8
|
+
- The acronym now reads **Chunked** Rust Engine for Single-cell Transcriptomics (was
|
|
9
|
+
"Columnar"): the engine streams chunks of cells through fused Rust kernels; Polars is only
|
|
10
|
+
the table and Parquet layer. Package and import names are unchanged (`crest-sc`, `crest`).
|
|
11
|
+
|
|
12
|
+
### New
|
|
13
|
+
- **Pseudobulk DESeq2 in Rust.** `crest.tl.DESeq2` ports DESeq2 1.42 `DESeq()` +
|
|
14
|
+
`results()`: median-of-ratios size factors, Cox-Reid gene-wise dispersions,
|
|
15
|
+
parametric trend, MAP shrinkage, NB-GLM Wald tests, Cook's filtering and
|
|
16
|
+
outlier replacement, and independent filtering. It matches R to ~1e-9 on
|
|
17
|
+
simulated designs, is ~18× faster than R and ~28× faster than pydeseq2 on
|
|
18
|
+
Kang 2018 pseudobulk. See `docs/deseq2.md`.
|
|
19
|
+
- `crest.tl.pseudobulk` (streamed raw-count sums per sample × group) and
|
|
20
|
+
`crest.tl.pseudobulk_de` (per-cell-type DESeq2 in one call).
|
|
21
|
+
- `bench/deseq2/`: R comparison on simulated designs and on Kang et al. 2018.
|
|
22
|
+
- **Harmony** batch integration in Rust (`crest.tl.harmony`), the harmony2 algorithm of
|
|
23
|
+
R harmony >= 1.2 / harmonypy 2.x; 3.5x faster than harmonypy at equal integration quality.
|
|
24
|
+
- **Scrublet** doublet detection (`crest.pp.scrublet`), sparse and streamed; 12x faster
|
|
25
|
+
than `scanpy.pp.scrublet` with the same AUROC on demuxlet-labelled doublets.
|
|
26
|
+
- `highly_variable_genes(flavor="seurat_v3" | "seurat_v3_paper", batch_key=...)`, with a
|
|
27
|
+
port of netlib loess (`crest._loess`); identical to scanpy.
|
|
28
|
+
- `crest.tl.leiden_sweep`: many resolutions/seeds on one graph in parallel, with
|
|
29
|
+
ARI-based stability; `crest.tl.adjusted_rand_index`.
|
|
30
|
+
- `crest.tl.ingest`: project a query onto a reference PCA, transfer labels and UMAP.
|
|
31
|
+
`uns['pca']['projection']` now stores what the projection needs.
|
|
32
|
+
- DESeq2 likelihood-ratio test: `DESeq2(test="LRT", reduced="~ ...")`, matching R.
|
|
33
|
+
- Native `knn_query` (reference -> query kNN, exact or IVF).
|
|
34
|
+
- `scripts/crest_paper_bench.sh` + `bench/paper/`: one-command publication benchmark
|
|
35
|
+
(datasets, per-core CPU/clock/memory monitoring, statistics, figures).
|
|
36
|
+
- `scripts/crest_git_housekeeping.sh`: repository housekeeping (dev branch, stale branches).
|
|
37
|
+
- **Documentation site** (Sphinx + MyST, Read the Docs): installation, quickstart, concepts,
|
|
38
|
+
full API reference, benchmarks, developer guide; `.readthedocs.yaml`; CI `docs` job.
|
|
39
|
+
- `work.md`: the development history; `CLAUDE.md` / `HANDOVER.md` rewritten for new sessions.
|
|
40
|
+
|
|
41
|
+
### Removed
|
|
42
|
+
- The `.bio` Polars expression namespace and its Rust plugin code (duplicated the
|
|
43
|
+
validated API, was not validated itself, and tied the wheel to Polars' plugin ABI).
|
|
44
|
+
The `polars`/`pyo3-polars` Rust dependencies go with it.
|
|
45
|
+
- SLAF support (`BioFrame.from_slaf`, `crest/slaf_io.py`, the `slaf` extra).
|
|
46
|
+
- Legacy scripts: `bench/*.py` from the biopolars era, `notebooks/`, the Docker setup,
|
|
47
|
+
`docs/dev/`, and `bench/whitepaper/`'s scripts (superseded by `bench/paper/`; its
|
|
48
|
+
0.2.0 results stay).
|
|
49
|
+
|
|
50
|
+
### Fixed
|
|
51
|
+
- `BioFrame.to_anndata()` / `from_anndata()` no longer need pyarrow.
|
|
52
|
+
- `crest_paper_bench.sh` stops with a clear message when the work directory is not writable,
|
|
53
|
+
and records the work directory's device and filesystem.
|
|
54
|
+
|
|
55
|
+
## 0.2.0
|
|
56
|
+
|
|
57
|
+
A correctness and performance release. Every step of the standard scanpy
|
|
58
|
+
workflow has been re-implemented and validated against scanpy.
|
|
59
|
+
|
|
60
|
+
### Fixed (results from 0.1.0 should not be used)
|
|
61
|
+
- **kNN / Leiden / UMAP**: the HNSW index shuffled points and its id map was
|
|
62
|
+
ignored, so neighbour indices pointed at random cells (recall@15 = 0.005).
|
|
63
|
+
- **Leiden** was a single-level local-move heuristic (thousands of singleton
|
|
64
|
+
clusters) with a modularity gain off by a factor of 2; now Traag et al. 2019
|
|
65
|
+
Leiden (local moving, refinement, aggregation). Modularity equals or exceeds
|
|
66
|
+
`leidenalg`.
|
|
67
|
+
- **normalize_cpm / qc / filter_cells / scale / score_genes** Polars plugins
|
|
68
|
+
were declared elementwise, so Polars computed per-cell sums on partial
|
|
69
|
+
batches; `normalize_cpm` ignored `target_sum`.
|
|
70
|
+
- **rank_genes_groups** divided by non-zero counts instead of group sizes.
|
|
71
|
+
- **Randomized PCA** lost trailing components (no re-orthonormalisation).
|
|
72
|
+
- **HVG** "seurat" flavour was raw variance/mean (883/2000 overlap with scanpy).
|
|
73
|
+
- **Wilcoxon** p-values underflowed to 0.
|
|
74
|
+
|
|
75
|
+
### New
|
|
76
|
+
- `crest.pp` / `crest.tl` scanpy-style API on `BioFrame`, with raw counts held
|
|
77
|
+
in compact CSR, a Polars triplet frame, or a Parquet dataset streamed from disk.
|
|
78
|
+
Filters and `normalize_total` / `log1p` / `scale` are applied lazily inside
|
|
79
|
+
fused native kernels: no normalised, scaled or dense copy is ever made.
|
|
80
|
+
- Exact PCA from a streamed Gram matrix with `sc.pp.scale(max_value)` applied
|
|
81
|
+
implicitly (0.000 deg from scanpy's ARPACK result).
|
|
82
|
+
- kNN: tiled GEMM brute force (small n) and IVF + NN-descent (large n).
|
|
83
|
+
- UMAP: umap-learn semantics, parallelised by domain decomposition.
|
|
84
|
+
- Sparse Wilcoxon ranking only non-zeros; all-groups Welch t-test.
|
|
85
|
+
- `score_genes` reproducing scanpy's control-gene sampling.
|
|
86
|
+
- Readers: `read_10x_h5` (optionally streamed to Parquet), `read_h5ad`,
|
|
87
|
+
`read_10x_mtx`; `write_parquet` / `read_parquet`; AnnData conversion.
|
|
88
|
+
- `bench/whitepaper/` benchmark harness.
|
|
89
|
+
|
|
90
|
+
### Changed
|
|
91
|
+
- `bio.deseq2` renamed `bio.nb_glm` (it fits an NB GLM with given dispersions;
|
|
92
|
+
not the full DESeq2 procedure) and returns null when IRLS does not converge.
|
|
93
|
+
- In `bio.leiden` / `bio.umap`, `n_neighbors` now counts the cell itself (scanpy convention).
|
|
94
|
+
- `crest.tl.sparse_masked_pca` / `incremental_pca` removed (superseded by `crest.tl.pca`).
|
|
95
|
+
- Distribution renamed `crest-sc` (the import name stays `crest`); Python >= 3.10.
|
crest_sc-0.3.0/CLAUDE.md
ADDED
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
# CREST: operating manual for Claude Code
|
|
2
|
+
|
|
3
|
+
Read this file completely at the start of every session. Then read `HANDOVER.md` (current
|
|
4
|
+
status, open work, decisions). If you wonder *why* something is the way it is, search
|
|
5
|
+
`work.md`: it is the full history of the sessions that built the project, with every request
|
|
6
|
+
and every reply. Use `grep -n "<keyword>" work.md` rather than reading it top to bottom. The
|
|
7
|
+
"Context summary" sections are dense recaps. `scripts/export_chat.py` regenerates it from a
|
|
8
|
+
Claude Code transcript (`~/.claude/projects/<project>/<session>.jsonl`); append new sessions
|
|
9
|
+
rather than replacing the history.
|
|
10
|
+
|
|
11
|
+
## 1. What this project is
|
|
12
|
+
|
|
13
|
+
CREST is a Python package (`pip install crest-sc`, `import crest`) with a Rust core
|
|
14
|
+
(PyO3 + maturin). It runs the scanpy single-cell RNA-seq workflow and gives **the same
|
|
15
|
+
results as scanpy** (and as R DESeq2 / harmonypy for those modules), while being several
|
|
16
|
+
times faster and using far less memory. The owner is a computational biologist who works
|
|
17
|
+
mostly in R (Seurat); the goal is a publishable package with an arXiv/bioRxiv-grade
|
|
18
|
+
performance paper.
|
|
19
|
+
|
|
20
|
+
The steps covered:
|
|
21
|
+
|
|
22
|
+
* **Workflow:** QC → filter → `normalize_total` → `log1p` → HVG → `scale` → PCA →
|
|
23
|
+
neighbours → Leiden → UMAP → marker genes (t-test / Wilcoxon) → `score_genes`.
|
|
24
|
+
* **After clustering:** pseudobulk DESeq2 (Wald + LRT), Harmony, Scrublet, `seurat_v3`
|
|
25
|
+
HVGs, Leiden resolution sweep, ingest (label transfer).
|
|
26
|
+
|
|
27
|
+
History in one paragraph: the project began (Feb 2026) as "biopolars", Polars expression
|
|
28
|
+
plugins for single-cell data. That design was wrong (per-cell sums over partial batches,
|
|
29
|
+
broken kNN id mapping, a fake Leiden). In Sept 2026 it was rebuilt around a `BioFrame` with
|
|
30
|
+
lazy transforms and fused Rust kernels, validated step by step against scanpy, and released
|
|
31
|
+
as 0.2.0. Downstream modules followed in 0.3.0. The Polars plugin layer was deleted; Polars
|
|
32
|
+
remains only as the metadata / table / Parquet layer. Hence the acronym now reads **Chunked**
|
|
33
|
+
Rust Engine for Single-cell Transcriptomics (it was "Columnar" until 0.3.0).
|
|
34
|
+
|
|
35
|
+
## 2. The mental model (read before touching code)
|
|
36
|
+
|
|
37
|
+
**Raw counts are stored once and never modified.** Everything else is recorded, and applied
|
|
38
|
+
on the fly inside Rust kernels as the data is streamed chunk by chunk:
|
|
39
|
+
|
|
40
|
+
```text
|
|
41
|
+
BioFrame
|
|
42
|
+
store : CSRStore (in-memory CSR) | FrameStore (Polars triplets) | ParquetStore (on disk, out-of-core)
|
|
43
|
+
obs : polars.DataFrame, kept cells; obs["cell_id"] indexes the store
|
|
44
|
+
var : polars.DataFrame, kept genes; var["gene_id"] indexes the store
|
|
45
|
+
ops : [("normalize_total", 1e4), ("log1p",)] <- recorded transforms
|
|
46
|
+
uns : {"scale": {...}, "hvg": {...}, "pca": {...}, "neighbors": {...}, ...}
|
|
47
|
+
obsm / varm : embeddings (X_pca, X_umap, X_pca_harmony) / PCs
|
|
48
|
+
|
|
49
|
+
for ctx in bf.iter_ctx(): # one raw chunk + cell_map/gene_map (-1 = filtered) + transform
|
|
50
|
+
_native.<kernel>(*ctx, outputs...) # Rust: filter + normalise + log1p (+ scale) per cell, accumulate
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
The consequences are rules, not suggestions:
|
|
54
|
+
|
|
55
|
+
* **Filters return a new BioFrame that shares the store:** `bf = crest.pp.filter_cells(bf, ...)`.
|
|
56
|
+
All other functions mutate `bf` in place and also return it.
|
|
57
|
+
* **Nothing is ever cells × genes and dense.** Results are per-gene, per-cell, per-group, or
|
|
58
|
+
genes × genes (the PCA Gram matrix, with HVGs only; capped at 20k genes).
|
|
59
|
+
* **PCA with `scale` is exact without densifying.** Zeros map to a per-gene constant
|
|
60
|
+
`b_j`, so `Z = 1·bᵀ + S` with `S` sparse. Centring removes `1·bᵀ`, so the Gram matrix
|
|
61
|
+
is accumulated from sparse rows (`docs/memory_model.md`).
|
|
62
|
+
* **What each step reads:**
|
|
63
|
+
- raw counts: QC, filters, `seurat_v3`, Scrublet, pseudobulk / DESeq2;
|
|
64
|
+
- log-normalised values: `seurat` / `cell_ranger` HVG, DE, `score_genes`;
|
|
65
|
+
- scaled values: PCA;
|
|
66
|
+
- `obsm` embeddings: neighbours, Leiden, UMAP, Harmony, ingest.
|
|
67
|
+
* `ParquetStore` = out-of-core: one Parquet part in RAM at a time, ~1 GB peak at any size,
|
|
68
|
+
~2× slower.
|
|
69
|
+
|
|
70
|
+
Full user-level explanation: `docs/concepts.md`. The chunk protocol tuple and the table of
|
|
71
|
+
all native functions: `docs/development.md`.
|
|
72
|
+
|
|
73
|
+
## 3. Build, test, docs
|
|
74
|
+
|
|
75
|
+
```bash
|
|
76
|
+
# one-time: Rust stable >= 1.80, Python >= 3.10; the owner uses uv
|
|
77
|
+
uv venv .venv --python 3.11 && source .venv/bin/activate && uv pip install maturin
|
|
78
|
+
maturin develop --release -E test # rebuild after ANY change under src/ (always --release)
|
|
79
|
+
cargo test --release # 26 Rust unit tests
|
|
80
|
+
pytest -q # 27 Python tests: scanpy / R-DESeq2 / harmonypy / skmisc / skimage parity
|
|
81
|
+
pip install -r docs/requirements.txt && sphinx-build -W -b html docs docs/_build/html # docs
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
* A Python-only change needs no rebuild.
|
|
85
|
+
* If `maturin` says "Both VIRTUAL_ENV and CONDA_PREFIX are set", run `conda deactivate`
|
|
86
|
+
(the owner's shell starts in conda `base`).
|
|
87
|
+
* The R-DESeq2 reference outputs are stored in `tests/data/`, so the tests don't need R.
|
|
88
|
+
They are regenerated with `Rscript tests/data/make_lrt_fixtures.R` and the scripts in
|
|
89
|
+
`bench/deseq2/`.
|
|
90
|
+
|
|
91
|
+
## 4. Where things are
|
|
92
|
+
|
|
93
|
+
| need | file |
|
|
94
|
+
|---|---|
|
|
95
|
+
| BioFrame, stores, `iter_ctx`, AnnData / Parquet I/O | `crest/core.py` |
|
|
96
|
+
| readers | `crest/io.py` |
|
|
97
|
+
| QC, filters, lazy transforms, HVG (3 flavours), neighbours | `crest/pp.py` |
|
|
98
|
+
| PCA, Leiden, `leiden_sweep`, UMAP, DE, `score_genes` | `crest/tl.py` |
|
|
99
|
+
| pseudobulk + DESeq2 (formulas, Wald / LRT, results, independent filtering) | `crest/deseq2.py` + `src/deseq/` |
|
|
100
|
+
| Harmony / Scrublet / ingest / loess | `crest/harmony.py` + `src/harmony.rs`, `crest/doublets.py`, `crest/ingest.py`, `crest/_loess.py` |
|
|
101
|
+
| streaming kernels | `src/kernels.rs` (`ChunkView`) |
|
|
102
|
+
| kNN (exact GEMM ≤ 20k cells; IVF + NN-descent above; `knn_query`) | `src/knn.rs` |
|
|
103
|
+
| Leiden, UMAP | `src/leiden.rs`, `src/umap/` |
|
|
104
|
+
| Python ↔ Rust bindings (validation, GIL release) | `src/py.rs` |
|
|
105
|
+
| tests | `tests/test_crest.py`, `tests/data/` |
|
|
106
|
+
| paper benchmark | `scripts/crest_paper_bench.sh` → `bench/paper/{datasets,run_one,run_all,monitor,accuracy,summarize}.py` |
|
|
107
|
+
| module benchmarks | `bench/{deseq2,harmony,doublets,ingest}/` (results in `results/`) |
|
|
108
|
+
| user docs (Read the Docs) | `docs/*.md`, `docs/conf.py`, `.readthedocs.yaml`, `docs/readthedocs.md` |
|
|
109
|
+
| design / validation notes | `docs/concepts.md`, `docs/memory_model.md`, `docs/deseq2.md`, `docs/downstream.md`, `docs/benchmarks.md` |
|
|
110
|
+
| history | `work.md`, `CHANGELOG.md` |
|
|
111
|
+
|
|
112
|
+
## 5. Rules
|
|
113
|
+
|
|
114
|
+
1. **Parity first.** Results must match the reference tool: scanpy 1.11 for the workflow,
|
|
115
|
+
R DESeq2 1.42 for `DESeq2`, harmonypy 2.x for Harmony, `skmisc` for loess and `skimage`
|
|
116
|
+
for `threshold_minimum`.
|
|
117
|
+
- A new method gets a parity test before it is optimised.
|
|
118
|
+
- If a change moves a parity test, find out why. **Never loosen a tolerance** to make
|
|
119
|
+
it pass.
|
|
120
|
+
- Where exact parity is impossible (random streams: Scrublet pairs, Harmony init), compare
|
|
121
|
+
outcomes statistically and document it.
|
|
122
|
+
2. **Bounded memory.** Stream through `iter_ctx()`. Never build dense cells × genes, and never
|
|
123
|
+
call `to_scipy()` inside library code.
|
|
124
|
+
3. **Speed lives in Rust.** Loops over non-zeros or cells go in `src/`. Release the GIL
|
|
125
|
+
(`py.allow_threads`), parallelise with rayon, and validate inputs in `py.rs` so errors
|
|
126
|
+
raise instead of panicking.
|
|
127
|
+
4. **scanpy-compatible names and defaults.** Deviations are documented in the docstring and
|
|
128
|
+
in `docs/`.
|
|
129
|
+
5. **Honest benchmarks.** These rules come from the owner:
|
|
130
|
+
- The headline comparison is the `core` pipeline; optional modules go in a separate table.
|
|
131
|
+
- Speed-ups are quoted at each tool's *best* thread count and at matched threads; the
|
|
132
|
+
conservative one is the headline.
|
|
133
|
+
- Report the steps where CREST is not faster, its thread-scaling limits, and every failed
|
|
134
|
+
run.
|
|
135
|
+
- No cherry-picking.
|
|
136
|
+
6. **Docs move with code.**
|
|
137
|
+
- A new public function gets a docstring and an entry in `docs/api.md`.
|
|
138
|
+
- A behaviour change updates `docs/` and `CHANGELOG.md` (Unreleased).
|
|
139
|
+
- `sphinx-build -W` must stay clean (CI enforces it).
|
|
140
|
+
|
|
141
|
+
## 6. Git and GitHub
|
|
142
|
+
|
|
143
|
+
* Branches: `main` (releases only; **never push to it**), `dev` (integration; PRs land
|
|
144
|
+
here), and feature branches. A release merges `dev` → `main` and tags `vX.Y.Z`.
|
|
145
|
+
* The version lives in `Cargo.toml`. `pyproject.toml` and `docs/conf.py` read it from there.
|
|
146
|
+
`crest/__init__.py` has a copy: bump both.
|
|
147
|
+
* Commit prefixes: `feat:`, `fix:`, `bench:`, `docs:`, `chore:`. Before pushing, run
|
|
148
|
+
`cargo test --release && pytest -q`.
|
|
149
|
+
* CI (`.github/workflows/CI.yml`) runs on PRs and pushes to main:
|
|
150
|
+
- the `test` job (cargo + pytest);
|
|
151
|
+
- the `docs` job (Sphinx, warnings are errors);
|
|
152
|
+
- wheel builds for all platforms, published to PyPI on a tag.
|
|
153
|
+
* Deleting or tagging shared branches is done by the owner with
|
|
154
|
+
`scripts/crest_git_housekeeping.sh`; don't try to work around permission refusals.
|
|
155
|
+
* Old branches are archived as tags `archive/feat-leiden-hnsw-faer` and
|
|
156
|
+
`archive/fix-production-readiness`.
|
|
157
|
+
|
|
158
|
+
## 7. The owner's machine (where benchmarks run)
|
|
159
|
+
|
|
160
|
+
* **Machine:** `rinamochana`, AMD Threadripper PRO 3975WX (32 cores / 64 threads),
|
|
161
|
+
128 GB DDR4-2667 ECC, Ubuntu 26.04.
|
|
162
|
+
* **Disks:**
|
|
163
|
+
- `/home` is small (OS only): **never write data or caches there**;
|
|
164
|
+
- `/mnt/scratch` is a 1.8 TB NVMe (ext4, `/dev/nvme0n1`); run benchmarks here;
|
|
165
|
+
- `/storage` is a large ZFS pool on HDDs.
|
|
166
|
+
* **Benchmark:**
|
|
167
|
+
- Run from `/mnt/scratch` with
|
|
168
|
+
`bash crest_paper_bench.sh --tier quick|standard|full`.
|
|
169
|
+
- Everything, including uv / cargo / numba caches, goes under `./crest-bench`.
|
|
170
|
+
- Results go to `crest-bench/results/<host>-<date>/`, and a `.tar.gz` is written to send
|
|
171
|
+
back.
|
|
172
|
+
* The owner uses `uv` for Python environments.
|
|
173
|
+
|
|
174
|
+
## 8. Working style the owner expects
|
|
175
|
+
|
|
176
|
+
* **Think thoroughly and weigh trade-offs.** Quality comes first, and every claim should be
|
|
177
|
+
verified: run the code, rebuild, re-test.
|
|
178
|
+
* Scripts they run locally must work from the current directory, log everything, and fail
|
|
179
|
+
early with a clear message.
|
|
180
|
+
* When you finish a session that changed anything, update `HANDOVER.md` (Status and Next
|
|
181
|
+
steps), and add a `CHANGELOG.md` entry for user-visible changes.
|
|
182
|
+
* Never put model identifiers in commits, code or docs.
|