pros-sketch 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pros_sketch-0.1.0/.gitignore +16 -0
- pros_sketch-0.1.0/CITATION.cff +17 -0
- pros_sketch-0.1.0/LICENSE +22 -0
- pros_sketch-0.1.0/PKG-INFO +222 -0
- pros_sketch-0.1.0/README.md +154 -0
- pros_sketch-0.1.0/data/README.md +13 -0
- pros_sketch-0.1.0/data/synthetic_clusters.npz +0 -0
- pros_sketch-0.1.0/docs/api.md +16 -0
- pros_sketch-0.1.0/docs/configuration.md +48 -0
- pros_sketch-0.1.0/docs/examples.md +18 -0
- pros_sketch-0.1.0/docs/index.md +10 -0
- pros_sketch-0.1.0/docs/installation.md +19 -0
- pros_sketch-0.1.0/docs/publishing.md +61 -0
- pros_sketch-0.1.0/docs/quickstart.md +30 -0
- pros_sketch-0.1.0/docs/theory.md +39 -0
- pros_sketch-0.1.0/examples/01_basic_numpy.py +27 -0
- pros_sketch-0.1.0/examples/02_ann_data.py +29 -0
- pros_sketch-0.1.0/examples/03_certification.py +23 -0
- pros_sketch-0.1.0/examples/04_compare_configs.py +27 -0
- pros_sketch-0.1.0/mkdocs.yml +17 -0
- pros_sketch-0.1.0/pros/__init__.py +22 -0
- pros_sketch-0.1.0/pros/_validation.py +43 -0
- pros_sketch-0.1.0/pros/allocate.py +308 -0
- pros_sketch-0.1.0/pros/certify.py +273 -0
- pros_sketch-0.1.0/pros/core.py +541 -0
- pros_sketch-0.1.0/pros/geometry.py +370 -0
- pros_sketch-0.1.0/pros/partition.py +161 -0
- pros_sketch-0.1.0/pros/py.typed +0 -0
- pros_sketch-0.1.0/pros/select.py +293 -0
- pros_sketch-0.1.0/pyproject.toml +105 -0
- pros_sketch-0.1.0/tests/test_allocate.py +51 -0
- pros_sketch-0.1.0/tests/test_certify.py +64 -0
- pros_sketch-0.1.0/tests/test_core.py +87 -0
- pros_sketch-0.1.0/tests/test_geometry.py +36 -0
- pros_sketch-0.1.0/tests/test_partition.py +15 -0
- pros_sketch-0.1.0/tests/test_public_api.py +18 -0
- pros_sketch-0.1.0/tests/test_release_regressions.py +276 -0
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
__pycache__/
|
|
2
|
+
*.py[cod]
|
|
3
|
+
.pytest_cache/
|
|
4
|
+
.ruff_cache/
|
|
5
|
+
.coverage
|
|
6
|
+
htmlcov/
|
|
7
|
+
build/
|
|
8
|
+
dist/
|
|
9
|
+
*.egg-info/
|
|
10
|
+
.venv/
|
|
11
|
+
|
|
12
|
+
# Generated parameter-experiment artifacts are reproducible locally.
|
|
13
|
+
test_para/data/*.npz
|
|
14
|
+
test_para/results/*.csv
|
|
15
|
+
test_para/results/*.md
|
|
16
|
+
test_para/results/*.html
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
cff-version: 1.2.0
|
|
2
|
+
title: "PROS: Partitioned and Refined Oversampling Sketches"
|
|
3
|
+
message: "If you use PROS in your work, please cite this software."
|
|
4
|
+
type: software
|
|
5
|
+
abstract: "A Python package for partitioned, oversampled, and globally refined geometric data sketches."
|
|
6
|
+
authors:
|
|
7
|
+
- family-names: Li
|
|
8
|
+
given-names: Lei
|
|
9
|
+
repository-code: "https://github.com/LeiLi-Uchicago/PROS"
|
|
10
|
+
license: MIT
|
|
11
|
+
keywords:
|
|
12
|
+
- geometric sketching
|
|
13
|
+
- representative sampling
|
|
14
|
+
- k-center
|
|
15
|
+
- coreset
|
|
16
|
+
version: 0.1.0
|
|
17
|
+
date-released: 2026-08-15
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Lei Li
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
22
|
+
|
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: pros-sketch
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Partitioned and Refined Oversampling Sketches for certified geometric sketching.
|
|
5
|
+
Project-URL: Homepage, https://github.com/LeiLi-Uchicago/PROS
|
|
6
|
+
Project-URL: Issues, https://github.com/LeiLi-Uchicago/PROS/issues
|
|
7
|
+
Author: Lei Li
|
|
8
|
+
License: MIT License
|
|
9
|
+
|
|
10
|
+
Copyright (c) 2026 Lei Li
|
|
11
|
+
|
|
12
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
13
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
14
|
+
in the Software without restriction, including without limitation the rights
|
|
15
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
16
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
17
|
+
furnished to do so, subject to the following conditions:
|
|
18
|
+
|
|
19
|
+
The above copyright notice and this permission notice shall be included in all
|
|
20
|
+
copies or substantial portions of the Software.
|
|
21
|
+
|
|
22
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
23
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
24
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
25
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
26
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
27
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
28
|
+
SOFTWARE.
|
|
29
|
+
|
|
30
|
+
License-File: LICENSE
|
|
31
|
+
Keywords: coreset,k-center,single-cell,sketching,subsampling
|
|
32
|
+
Classifier: Development Status :: 4 - Beta
|
|
33
|
+
Classifier: Intended Audience :: Science/Research
|
|
34
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
35
|
+
Classifier: Programming Language :: Python :: 3
|
|
36
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
37
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
38
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
39
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
40
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
41
|
+
Classifier: Typing :: Typed
|
|
42
|
+
Requires-Python: >=3.10
|
|
43
|
+
Requires-Dist: numpy>=1.23
|
|
44
|
+
Requires-Dist: scikit-learn>=1.2
|
|
45
|
+
Requires-Dist: scipy>=1.9
|
|
46
|
+
Provides-Extra: adata
|
|
47
|
+
Requires-Dist: anndata>=0.9; extra == 'adata'
|
|
48
|
+
Provides-Extra: all
|
|
49
|
+
Requires-Dist: anndata>=0.9; extra == 'all'
|
|
50
|
+
Requires-Dist: mkdocs-material>=9.5; extra == 'all'
|
|
51
|
+
Requires-Dist: mkdocs>=1.6; extra == 'all'
|
|
52
|
+
Requires-Dist: pytest>=8; extra == 'all'
|
|
53
|
+
Requires-Dist: ruff>=0.6; extra == 'all'
|
|
54
|
+
Requires-Dist: scanpy>=1.9; extra == 'all'
|
|
55
|
+
Requires-Dist: scsampler; extra == 'all'
|
|
56
|
+
Provides-Extra: dev
|
|
57
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
58
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
59
|
+
Provides-Extra: docs
|
|
60
|
+
Requires-Dist: mkdocs-material>=9.5; extra == 'docs'
|
|
61
|
+
Requires-Dist: mkdocs>=1.6; extra == 'docs'
|
|
62
|
+
Provides-Extra: scanpy
|
|
63
|
+
Requires-Dist: anndata>=0.9; extra == 'scanpy'
|
|
64
|
+
Requires-Dist: scanpy>=1.9; extra == 'scanpy'
|
|
65
|
+
Provides-Extra: scsampler
|
|
66
|
+
Requires-Dist: scsampler; extra == 'scsampler'
|
|
67
|
+
Description-Content-Type: text/markdown
|
|
68
|
+
|
|
69
|
+
# PROS
|
|
70
|
+
|
|
71
|
+
PROS (Partitioned and Refined Oversampling Sketches) selects a small,
|
|
72
|
+
diversity-preserving subset of rows from a large real-valued coordinate matrix.
|
|
73
|
+
It builds an oversampled candidate pool within partitions and performs one
|
|
74
|
+
global refinement pass over that pool. The returned subset is represented by
|
|
75
|
+
indices into the original matrix.
|
|
76
|
+
|
|
77
|
+
The package is domain-agnostic. It can be used with reduced single-cell
|
|
78
|
+
embeddings, as well as other feature matrices where Euclidean distance is a
|
|
79
|
+
meaningful geometry.
|
|
80
|
+
|
|
81
|
+
## Installation
|
|
82
|
+
|
|
83
|
+
PROS requires Python 3.10 or newer. The distribution name is `pros-sketch`;
|
|
84
|
+
the Python import remains `pros`. The commands below install this checkout.
|
|
85
|
+
|
|
86
|
+
```bash
|
|
87
|
+
python -m pip install -e .
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Optional AnnData support:
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
python -m pip install -e ".[adata]"
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
For testing and linting:
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
python -m pip install -e ".[dev]"
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
## Quick start
|
|
103
|
+
|
|
104
|
+
```python
|
|
105
|
+
import numpy as np
|
|
106
|
+
from pros import sketch
|
|
107
|
+
|
|
108
|
+
X = np.random.default_rng(0).random((5_000, 50))
|
|
109
|
+
result = sketch(X, n=500, seed=0)
|
|
110
|
+
|
|
111
|
+
indices = result.indices
|
|
112
|
+
X_sketch = X[indices]
|
|
113
|
+
print(result.timings)
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
`result.indices` is a sorted, unique `int64` array of length `n`. By default,
|
|
117
|
+
PROS uses k-means partitioning, water-filling allocation, an oversampling ratio
|
|
118
|
+
of `r=10.0`, and global farthest-first refinement. Pass an explicit `seed` to
|
|
119
|
+
make a stochastic configuration reproducible.
|
|
120
|
+
|
|
121
|
+
## Certification
|
|
122
|
+
|
|
123
|
+
Set `certify=True` to compute the final covering radius, the candidate-pool
|
|
124
|
+
radius, and bounds derived from a full-data farthest-first traversal:
|
|
125
|
+
|
|
126
|
+
```python
|
|
127
|
+
result = sketch(X, n=500, seed=0, certify=True)
|
|
128
|
+
|
|
129
|
+
print(result.radius) # R(X, S): final covering radius
|
|
130
|
+
print(result.rho) # R(X, P): candidate-pool covering radius
|
|
131
|
+
print(result.opt_lower) # lower bound on the optimal n-centre radius
|
|
132
|
+
print(result.ratio_upper) # upper bound on R(X, S) / OPT_n(X)
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
Certification performs additional full-data work and is not included in
|
|
136
|
+
`result.timings["total"]`. Degenerate data may have `opt_lower == 0`; in that
|
|
137
|
+
case a finite approximation-ratio bound is not defined.
|
|
138
|
+
|
|
139
|
+
For an existing set of indices, use `pros.certificate`. `pros.opt_bounds`,
|
|
140
|
+
`pros.estimate_opt_scale`, and `pros.choose_r` are available for certificate
|
|
141
|
+
and oversampling-ratio workflows.
|
|
142
|
+
|
|
143
|
+
## AnnData
|
|
144
|
+
|
|
145
|
+
```python
|
|
146
|
+
from pros import sketch_adata
|
|
147
|
+
|
|
148
|
+
result = sketch_adata(adata, n=5_000, use_rep="X_pca", seed=0)
|
|
149
|
+
subset = adata[result.indices].copy()
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
`sketch_adata` also accepts an `.h5ad` path and reads it in backed mode before
|
|
153
|
+
extracting the requested representation.
|
|
154
|
+
|
|
155
|
+
## Configuration
|
|
156
|
+
|
|
157
|
+
```python
|
|
158
|
+
result = sketch(
|
|
159
|
+
X,
|
|
160
|
+
n=1_000,
|
|
161
|
+
partitioner="kmeans",
|
|
162
|
+
n_blocks="auto",
|
|
163
|
+
allocator="water_filling",
|
|
164
|
+
r=10.0,
|
|
165
|
+
selector="fft",
|
|
166
|
+
refiner="fft",
|
|
167
|
+
seed=0,
|
|
168
|
+
)
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
The `partitioner`, within-partition `selector`, `allocator`, and global
|
|
172
|
+
`refiner` are modular. See [`docs/configuration.md`](docs/configuration.md) for
|
|
173
|
+
the supported values and [`docs/theory.md`](docs/theory.md) for the covering
|
|
174
|
+
objective and certificate definitions.
|
|
175
|
+
|
|
176
|
+
## Repository layout
|
|
177
|
+
|
|
178
|
+
- `pros/`: installable library code.
|
|
179
|
+
- `tests/`: automated package tests.
|
|
180
|
+
- `examples/`: small runnable examples using synthetic data in `data/`.
|
|
181
|
+
- `test_para/`: standalone parameter experiments and their generated outputs;
|
|
182
|
+
this reproduction material is intentionally not packaged with the library.
|
|
183
|
+
- `docs/`: MkDocs source pages.
|
|
184
|
+
|
|
185
|
+
Run the examples from the repository root after installation:
|
|
186
|
+
|
|
187
|
+
```bash
|
|
188
|
+
python examples/01_basic_numpy.py
|
|
189
|
+
python examples/03_certification.py
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
## Citation
|
|
193
|
+
|
|
194
|
+
See [`CITATION.cff`](CITATION.cff) for software citation metadata.
|
|
195
|
+
|
|
196
|
+
## License
|
|
197
|
+
|
|
198
|
+
PROS is distributed under the [MIT License](LICENSE).
|
|
199
|
+
|
|
200
|
+
## Validated behavior and limits
|
|
201
|
+
|
|
202
|
+
- Water-filling is fused with FFT and requires `selector="fft"`. Choose another
|
|
203
|
+
allocator to use random or scSampler selection. Missing or broken scSampler
|
|
204
|
+
installations raise an error; no alternate algorithm is substituted.
|
|
205
|
+
- `mix>0` currently requires `refiner="fft"` so reserved points are retained.
|
|
206
|
+
`refiner="none"` uniformly downsamples the candidate pool to `n` rows.
|
|
207
|
+
- `certificate` computes the a posteriori ratio for any valid sketch. Its
|
|
208
|
+
`theory_bound` and `bound_slack` are NaN unless FFT refinement is explicitly
|
|
209
|
+
asserted; `sketch` sets this assertion only for unmixed FFT refinement.
|
|
210
|
+
Caches are tied to the exact data and sketch size. Floating-point results
|
|
211
|
+
are numerical bounds, not interval-arithmetic certificates.
|
|
212
|
+
- Internal stage timings exclude input conversion, validation, AnnData I/O,
|
|
213
|
+
and certification. Measure an external wall clock for runtime comparisons.
|
|
214
|
+
- `sketch_adata(..., return_adata=True, copy=False)` returns an in-memory view
|
|
215
|
+
without attaching metadata. Backed subset returns require `copy=True`.
|
|
216
|
+
Sparse `use_rep="X"` is rejected; supply a dense reduced embedding instead.
|
|
217
|
+
- Distances use SciPy directly and nearest-center evaluation tiles both axes.
|
|
218
|
+
Full input and block copies still require memory proportional to `N*d`;
|
|
219
|
+
optional swap refiners may allocate much larger matrices.
|
|
220
|
+
|
|
221
|
+
After installing development dependencies, run `python -m pytest`. Install
|
|
222
|
+
`.[dev,adata]` to include the optional AnnData integration regression test.
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
# PROS
|
|
2
|
+
|
|
3
|
+
PROS (Partitioned and Refined Oversampling Sketches) selects a small,
|
|
4
|
+
diversity-preserving subset of rows from a large real-valued coordinate matrix.
|
|
5
|
+
It builds an oversampled candidate pool within partitions and performs one
|
|
6
|
+
global refinement pass over that pool. The returned subset is represented by
|
|
7
|
+
indices into the original matrix.
|
|
8
|
+
|
|
9
|
+
The package is domain-agnostic. It can be used with reduced single-cell
|
|
10
|
+
embeddings, as well as other feature matrices where Euclidean distance is a
|
|
11
|
+
meaningful geometry.
|
|
12
|
+
|
|
13
|
+
## Installation
|
|
14
|
+
|
|
15
|
+
PROS requires Python 3.10 or newer. The distribution name is `pros-sketch`;
|
|
16
|
+
the Python import remains `pros`. The commands below install this checkout.
|
|
17
|
+
|
|
18
|
+
```bash
|
|
19
|
+
python -m pip install -e .
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
Optional AnnData support:
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
python -m pip install -e ".[adata]"
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
For testing and linting:
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
python -m pip install -e ".[dev]"
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## Quick start
|
|
35
|
+
|
|
36
|
+
```python
|
|
37
|
+
import numpy as np
|
|
38
|
+
from pros import sketch
|
|
39
|
+
|
|
40
|
+
X = np.random.default_rng(0).random((5_000, 50))
|
|
41
|
+
result = sketch(X, n=500, seed=0)
|
|
42
|
+
|
|
43
|
+
indices = result.indices
|
|
44
|
+
X_sketch = X[indices]
|
|
45
|
+
print(result.timings)
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
`result.indices` is a sorted, unique `int64` array of length `n`. By default,
|
|
49
|
+
PROS uses k-means partitioning, water-filling allocation, an oversampling ratio
|
|
50
|
+
of `r=10.0`, and global farthest-first refinement. Pass an explicit `seed` to
|
|
51
|
+
make a stochastic configuration reproducible.
|
|
52
|
+
|
|
53
|
+
## Certification
|
|
54
|
+
|
|
55
|
+
Set `certify=True` to compute the final covering radius, the candidate-pool
|
|
56
|
+
radius, and bounds derived from a full-data farthest-first traversal:
|
|
57
|
+
|
|
58
|
+
```python
|
|
59
|
+
result = sketch(X, n=500, seed=0, certify=True)
|
|
60
|
+
|
|
61
|
+
print(result.radius) # R(X, S): final covering radius
|
|
62
|
+
print(result.rho) # R(X, P): candidate-pool covering radius
|
|
63
|
+
print(result.opt_lower) # lower bound on the optimal n-centre radius
|
|
64
|
+
print(result.ratio_upper) # upper bound on R(X, S) / OPT_n(X)
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
Certification performs additional full-data work and is not included in
|
|
68
|
+
`result.timings["total"]`. Degenerate data may have `opt_lower == 0`; in that
|
|
69
|
+
case a finite approximation-ratio bound is not defined.
|
|
70
|
+
|
|
71
|
+
For an existing set of indices, use `pros.certificate`. `pros.opt_bounds`,
|
|
72
|
+
`pros.estimate_opt_scale`, and `pros.choose_r` are available for certificate
|
|
73
|
+
and oversampling-ratio workflows.
|
|
74
|
+
|
|
75
|
+
## AnnData
|
|
76
|
+
|
|
77
|
+
```python
|
|
78
|
+
from pros import sketch_adata
|
|
79
|
+
|
|
80
|
+
result = sketch_adata(adata, n=5_000, use_rep="X_pca", seed=0)
|
|
81
|
+
subset = adata[result.indices].copy()
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
`sketch_adata` also accepts an `.h5ad` path and reads it in backed mode before
|
|
85
|
+
extracting the requested representation.
|
|
86
|
+
|
|
87
|
+
## Configuration
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
result = sketch(
|
|
91
|
+
X,
|
|
92
|
+
n=1_000,
|
|
93
|
+
partitioner="kmeans",
|
|
94
|
+
n_blocks="auto",
|
|
95
|
+
allocator="water_filling",
|
|
96
|
+
r=10.0,
|
|
97
|
+
selector="fft",
|
|
98
|
+
refiner="fft",
|
|
99
|
+
seed=0,
|
|
100
|
+
)
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
The `partitioner`, within-partition `selector`, `allocator`, and global
|
|
104
|
+
`refiner` are modular. See [`docs/configuration.md`](docs/configuration.md) for
|
|
105
|
+
the supported values and [`docs/theory.md`](docs/theory.md) for the covering
|
|
106
|
+
objective and certificate definitions.
|
|
107
|
+
|
|
108
|
+
## Repository layout
|
|
109
|
+
|
|
110
|
+
- `pros/`: installable library code.
|
|
111
|
+
- `tests/`: automated package tests.
|
|
112
|
+
- `examples/`: small runnable examples using synthetic data in `data/`.
|
|
113
|
+
- `test_para/`: standalone parameter experiments and their generated outputs;
|
|
114
|
+
this reproduction material is intentionally not packaged with the library.
|
|
115
|
+
- `docs/`: MkDocs source pages.
|
|
116
|
+
|
|
117
|
+
Run the examples from the repository root after installation:
|
|
118
|
+
|
|
119
|
+
```bash
|
|
120
|
+
python examples/01_basic_numpy.py
|
|
121
|
+
python examples/03_certification.py
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
## Citation
|
|
125
|
+
|
|
126
|
+
See [`CITATION.cff`](CITATION.cff) for software citation metadata.
|
|
127
|
+
|
|
128
|
+
## License
|
|
129
|
+
|
|
130
|
+
PROS is distributed under the [MIT License](LICENSE).
|
|
131
|
+
|
|
132
|
+
## Validated behavior and limits
|
|
133
|
+
|
|
134
|
+
- Water-filling is fused with FFT and requires `selector="fft"`. Choose another
|
|
135
|
+
allocator to use random or scSampler selection. Missing or broken scSampler
|
|
136
|
+
installations raise an error; no alternate algorithm is substituted.
|
|
137
|
+
- `mix>0` currently requires `refiner="fft"` so reserved points are retained.
|
|
138
|
+
`refiner="none"` uniformly downsamples the candidate pool to `n` rows.
|
|
139
|
+
- `certificate` computes the a posteriori ratio for any valid sketch. Its
|
|
140
|
+
`theory_bound` and `bound_slack` are NaN unless FFT refinement is explicitly
|
|
141
|
+
asserted; `sketch` sets this assertion only for unmixed FFT refinement.
|
|
142
|
+
Caches are tied to the exact data and sketch size. Floating-point results
|
|
143
|
+
are numerical bounds, not interval-arithmetic certificates.
|
|
144
|
+
- Internal stage timings exclude input conversion, validation, AnnData I/O,
|
|
145
|
+
and certification. Measure an external wall clock for runtime comparisons.
|
|
146
|
+
- `sketch_adata(..., return_adata=True, copy=False)` returns an in-memory view
|
|
147
|
+
without attaching metadata. Backed subset returns require `copy=True`.
|
|
148
|
+
Sparse `use_rep="X"` is rejected; supply a dense reduced embedding instead.
|
|
149
|
+
- Distances use SciPy directly and nearest-center evaluation tiles both axes.
|
|
150
|
+
Full input and block copies still require memory proportional to `N*d`;
|
|
151
|
+
optional swap refiners may allocate much larger matrices.
|
|
152
|
+
|
|
153
|
+
After installing development dependencies, run `python -m pytest`. Install
|
|
154
|
+
`.[dev,adata]` to include the optional AnnData integration regression test.
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Example Data
|
|
2
|
+
|
|
3
|
+
`synthetic_clusters.npz` contains a small synthetic coordinate matrix for
|
|
4
|
+
examples and smoke tests.
|
|
5
|
+
|
|
6
|
+
Fields:
|
|
7
|
+
|
|
8
|
+
- `X`: `(900, 8)` float64 coordinates.
|
|
9
|
+
- `labels`: `(900,)` integer cluster labels.
|
|
10
|
+
|
|
11
|
+
The dataset is generated from Gaussian clusters and is not intended as a
|
|
12
|
+
scientific benchmark.
|
|
13
|
+
|
|
Binary file
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
# API
|
|
2
|
+
|
|
3
|
+
Public API:
|
|
4
|
+
|
|
5
|
+
- `pros.sketch(X, n, **kwargs)`: sketch a NumPy-compatible coordinate matrix.
|
|
6
|
+
- `pros.sketch_adata(adata, n, use_rep="X_pca", **kwargs)`: sketch an AnnData
|
|
7
|
+
embedding or `.h5ad` path.
|
|
8
|
+
- `pros.certificate(X, sketch_indices, pool_indices=None)`: compute a
|
|
9
|
+
certificate for an existing sketch.
|
|
10
|
+
- `pros.choose_r(X, n, tol=0.5)`: choose an oversampling ratio from a grid.
|
|
11
|
+
- `pros.estimate_opt_scale(X, n)`: estimate a target covering scale on a
|
|
12
|
+
subsample.
|
|
13
|
+
|
|
14
|
+
Lower-level implementation modules are intentionally not re-exported from
|
|
15
|
+
`pros`. They may be useful for local experiments, but are not part of the
|
|
16
|
+
stable public API.
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# Configuration
|
|
2
|
+
|
|
3
|
+
The default configuration is intended to be a strong starting point:
|
|
4
|
+
|
|
5
|
+
```python
|
|
6
|
+
res = sketch(
|
|
7
|
+
X,
|
|
8
|
+
n=1000,
|
|
9
|
+
partitioner="kmeans",
|
|
10
|
+
n_blocks="auto",
|
|
11
|
+
allocator="water_filling",
|
|
12
|
+
r=10.0,
|
|
13
|
+
selector="fft",
|
|
14
|
+
refiner="fft",
|
|
15
|
+
seed=0,
|
|
16
|
+
)
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
Useful options:
|
|
20
|
+
|
|
21
|
+
- `partitioner`: `"kmeans"`, `"pc_tree"`, `"random"`, or `"none"`.
|
|
22
|
+
- `allocator`: `"water_filling"`, `"proportional"`, `"power"`, `"volume"`, or
|
|
23
|
+
`"uniform"`.
|
|
24
|
+
- `selector`: `"fft"`, `"scsampler_maximin"`, or `"random"`.
|
|
25
|
+
- `refiner`: `"fft"`, `"maximin"`, `"local_swap"`, or `"none"`.
|
|
26
|
+
- `r`: oversampling ratio. Larger values make a denser pool and cost more.
|
|
27
|
+
- `certify`: compute certificate fields for evaluation.
|
|
28
|
+
|
|
29
|
+
For a single global FFT selection, use `partitioner="none", r=1,
|
|
30
|
+
refiner="none"`: stage 1 selects n points and stage 2 retains the whole pool.
|
|
31
|
+
With r>1, the single-block configuration still runs two selection stages.
|
|
32
|
+
|
|
33
|
+
Water-filling requires `selector="fft"`; incompatible combinations raise an
|
|
34
|
+
error. Other allocators support the other selectors. scSampler failures are
|
|
35
|
+
propagated, never replaced silently with FFT. Fixed seeds are reproducible
|
|
36
|
+
within a fixed numerical/software environment; record dependency versions and
|
|
37
|
+
thread settings for experiments.
|
|
38
|
+
|
|
39
|
+
`mix>0` requires FFT refinement to preserve the uniform reserve. With
|
|
40
|
+
`mix_source="data"`, reserve points may enlarge the pool beyond the stage-1
|
|
41
|
+
`budget`; `m_per_block` still describes stage 1. `block_radii` and
|
|
42
|
+
`rho_blockwise_max` are available only with water-filling. An unfunded nonempty
|
|
43
|
+
block has infinite local covering radius.
|
|
44
|
+
|
|
45
|
+
The pool budget is `min(ceil(r*n), N)`. Larger r need not monotonically improve
|
|
46
|
+
the greedy final radius. `choose_r` reports `tolerance_met` and certifies only
|
|
47
|
+
its returned sketch. Its pool-radius evaluations can still be expensive.
|
|
48
|
+
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Examples
|
|
2
|
+
|
|
3
|
+
The `examples/` directory contains small runnable scripts:
|
|
4
|
+
|
|
5
|
+
- `01_basic_numpy.py`: basic matrix sketching.
|
|
6
|
+
- `02_ann_data.py`: AnnData workflow.
|
|
7
|
+
- `03_certification.py`: certificate fields.
|
|
8
|
+
- `04_compare_configs.py`: simple configuration comparison.
|
|
9
|
+
|
|
10
|
+
Run an example from the repository root:
|
|
11
|
+
|
|
12
|
+
```bash
|
|
13
|
+
python examples/01_basic_numpy.py
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
The checkout examples use `partitioner="pc_tree"` to keep their behaviour
|
|
17
|
+
simple. The installed package includes `scikit-learn`, which is required by
|
|
18
|
+
the default `partitioner="kmeans"` workflow.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
# PROS
|
|
2
|
+
|
|
3
|
+
PROS builds diversity-preserving sketches of large coordinate matrices. It
|
|
4
|
+
first partitions the data, oversamples a candidate pool inside those blocks,
|
|
5
|
+
and then refines globally over the pool.
|
|
6
|
+
|
|
7
|
+
The returned sketch is an array of row indices into the original matrix. For
|
|
8
|
+
evaluation runs, PROS can also attach a certificate with the sketch covering
|
|
9
|
+
radius, pool covering radius, and a proven approximation-ratio upper bound.
|
|
10
|
+
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# Installation
|
|
2
|
+
|
|
3
|
+
Install from a local checkout:
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
python -m pip install -e .
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
Optional extras:
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
python -m pip install -e ".[adata]"
|
|
13
|
+
python -m pip install -e ".[dev]"
|
|
14
|
+
python -m pip install -e ".[docs]"
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
The core package depends on `numpy` and `scikit-learn`. AnnData and Scanpy are
|
|
18
|
+
optional so matrix-only users do not need the single-cell stack.
|
|
19
|
+
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
# Publishing pros-sketch
|
|
2
|
+
|
|
3
|
+
The distribution is `pros-sketch`, while the Python import is `pros`.
|
|
4
|
+
The workflow `.github/workflows/publish.yml` publishes to PyPI only when a
|
|
5
|
+
GitHub Release is published. Manual workflow runs perform validation and build
|
|
6
|
+
artifacts without uploading. No PyPI API token secret is required.
|
|
7
|
+
|
|
8
|
+
## One-time PyPI setup
|
|
9
|
+
|
|
10
|
+
For the first release, sign in to https://pypi.org/manage/account/publishing/
|
|
11
|
+
and add a pending publisher under GitHub:
|
|
12
|
+
|
|
13
|
+
| Field | Value |
|
|
14
|
+
| --- | --- |
|
|
15
|
+
| PyPI Project Name | `pros-sketch` |
|
|
16
|
+
| Owner | `LeiLi-Uchicago` |
|
|
17
|
+
| Repository name | `PROS` |
|
|
18
|
+
| Workflow name | `publish.yml` |
|
|
19
|
+
| Environment name | `pypi` |
|
|
20
|
+
|
|
21
|
+
Use the workflow filename, not its display title or full path. A pending
|
|
22
|
+
publisher does not reserve the project name; successful first upload creates
|
|
23
|
+
it. If this project already belongs to your account, add the publisher from
|
|
24
|
+
the project's Publishing settings instead.
|
|
25
|
+
|
|
26
|
+
In GitHub repository Settings → Environments, create the `pypi` environment.
|
|
27
|
+
Its name must match both the workflow and PyPI configuration. Optional review
|
|
28
|
+
and deployment restrictions can be configured there.
|
|
29
|
+
|
|
30
|
+
## Release procedure
|
|
31
|
+
|
|
32
|
+
1. Commit and push the reviewed code and publishing workflow to GitHub.
|
|
33
|
+
2. Run **Publish to PyPI** manually from Actions. This tests Python 3.10–3.12,
|
|
34
|
+
builds the sdist and wheel, checks metadata, and tests the installed wheel
|
|
35
|
+
outside the checkout. Confirm the run succeeds.
|
|
36
|
+
3. Ensure `pyproject.toml` and `pros/__init__.py` specify the same release
|
|
37
|
+
version. For the first release this is `0.1.0`.
|
|
38
|
+
4. Create a GitHub Release with tag `v0.1.0` pointing to the validated commit.
|
|
39
|
+
Publishing it triggers the workflow, repeats validation, and uploads the
|
|
40
|
+
built distributions through Trusted Publishing. A draft release does not.
|
|
41
|
+
5. Verify the project page and install `pros-sketch==0.1.0` in a new environment.
|
|
42
|
+
For AnnData support, install `pros-sketch[adata]==0.1.0`.
|
|
43
|
+
|
|
44
|
+
For later releases, update both version values and use the corresponding tag.
|
|
45
|
+
PyPI does not allow overwriting previously uploaded distribution filenames.
|
|
46
|
+
If an upload fails, inspect the logs and which artifacts reached PyPI before
|
|
47
|
+
retrying; this workflow intentionally does not hide existing-file errors.
|
|
48
|
+
|
|
49
|
+
TestPyPI is not configured by this workflow. It requires a separate account
|
|
50
|
+
and trusted publisher if a rehearsal upload is wanted.
|
|
51
|
+
|
|
52
|
+
## Scope of validation
|
|
53
|
+
|
|
54
|
+
The workflow tests core and AnnData functionality. The scSampler adapter is
|
|
55
|
+
covered by contract tests, not an end-to-end installation of scSampler. Large
|
|
56
|
+
benchmark datasets and manuscript runtime reproduction are not run in CI.
|
|
57
|
+
|
|
58
|
+
References:
|
|
59
|
+
- https://docs.pypi.org/trusted-publishers/creating-a-project-through-oidc/
|
|
60
|
+
- https://docs.pypi.org/trusted-publishers/adding-a-publisher/
|
|
61
|
+
- https://github.com/pypa/gh-action-pypi-publish
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Quickstart
|
|
2
|
+
|
|
3
|
+
```python
|
|
4
|
+
import numpy as np
|
|
5
|
+
from pros import sketch
|
|
6
|
+
|
|
7
|
+
rng = np.random.default_rng(0)
|
|
8
|
+
X = rng.random((5000, 50))
|
|
9
|
+
|
|
10
|
+
res = sketch(X, n=500, seed=0)
|
|
11
|
+
X_sketch = X[res.indices]
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
The result is dict-like and also supports attribute access:
|
|
15
|
+
|
|
16
|
+
```python
|
|
17
|
+
print(res["indices"])
|
|
18
|
+
print(res.indices)
|
|
19
|
+
print(res.timings)
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
For AnnData:
|
|
23
|
+
|
|
24
|
+
```python
|
|
25
|
+
from pros import sketch_adata
|
|
26
|
+
|
|
27
|
+
res = sketch_adata(adata, n=5000, use_rep="X_pca")
|
|
28
|
+
subset = adata[res.indices]
|
|
29
|
+
```
|
|
30
|
+
|