pros-sketch 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. pros_sketch-0.1.0/.gitignore +16 -0
  2. pros_sketch-0.1.0/CITATION.cff +17 -0
  3. pros_sketch-0.1.0/LICENSE +22 -0
  4. pros_sketch-0.1.0/PKG-INFO +222 -0
  5. pros_sketch-0.1.0/README.md +154 -0
  6. pros_sketch-0.1.0/data/README.md +13 -0
  7. pros_sketch-0.1.0/data/synthetic_clusters.npz +0 -0
  8. pros_sketch-0.1.0/docs/api.md +16 -0
  9. pros_sketch-0.1.0/docs/configuration.md +48 -0
  10. pros_sketch-0.1.0/docs/examples.md +18 -0
  11. pros_sketch-0.1.0/docs/index.md +10 -0
  12. pros_sketch-0.1.0/docs/installation.md +19 -0
  13. pros_sketch-0.1.0/docs/publishing.md +61 -0
  14. pros_sketch-0.1.0/docs/quickstart.md +30 -0
  15. pros_sketch-0.1.0/docs/theory.md +39 -0
  16. pros_sketch-0.1.0/examples/01_basic_numpy.py +27 -0
  17. pros_sketch-0.1.0/examples/02_ann_data.py +29 -0
  18. pros_sketch-0.1.0/examples/03_certification.py +23 -0
  19. pros_sketch-0.1.0/examples/04_compare_configs.py +27 -0
  20. pros_sketch-0.1.0/mkdocs.yml +17 -0
  21. pros_sketch-0.1.0/pros/__init__.py +22 -0
  22. pros_sketch-0.1.0/pros/_validation.py +43 -0
  23. pros_sketch-0.1.0/pros/allocate.py +308 -0
  24. pros_sketch-0.1.0/pros/certify.py +273 -0
  25. pros_sketch-0.1.0/pros/core.py +541 -0
  26. pros_sketch-0.1.0/pros/geometry.py +370 -0
  27. pros_sketch-0.1.0/pros/partition.py +161 -0
  28. pros_sketch-0.1.0/pros/py.typed +0 -0
  29. pros_sketch-0.1.0/pros/select.py +293 -0
  30. pros_sketch-0.1.0/pyproject.toml +105 -0
  31. pros_sketch-0.1.0/tests/test_allocate.py +51 -0
  32. pros_sketch-0.1.0/tests/test_certify.py +64 -0
  33. pros_sketch-0.1.0/tests/test_core.py +87 -0
  34. pros_sketch-0.1.0/tests/test_geometry.py +36 -0
  35. pros_sketch-0.1.0/tests/test_partition.py +15 -0
  36. pros_sketch-0.1.0/tests/test_public_api.py +18 -0
  37. pros_sketch-0.1.0/tests/test_release_regressions.py +276 -0
@@ -0,0 +1,16 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ .pytest_cache/
4
+ .ruff_cache/
5
+ .coverage
6
+ htmlcov/
7
+ build/
8
+ dist/
9
+ *.egg-info/
10
+ .venv/
11
+
12
+ # Generated parameter-experiment artifacts are reproducible locally.
13
+ test_para/data/*.npz
14
+ test_para/results/*.csv
15
+ test_para/results/*.md
16
+ test_para/results/*.html
@@ -0,0 +1,17 @@
1
+ cff-version: 1.2.0
2
+ title: "PROS: Partitioned and Refined Oversampling Sketches"
3
+ message: "If you use PROS in your work, please cite this software."
4
+ type: software
5
+ abstract: "A Python package for partitioned, oversampled, and globally refined geometric data sketches."
6
+ authors:
7
+ - family-names: Li
8
+ given-names: Lei
9
+ repository-code: "https://github.com/LeiLi-Uchicago/PROS"
10
+ license: MIT
11
+ keywords:
12
+ - geometric sketching
13
+ - representative sampling
14
+ - k-center
15
+ - coreset
16
+ version: 0.1.0
17
+ date-released: 2026-08-15
@@ -0,0 +1,22 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Lei Li
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
22
+
@@ -0,0 +1,222 @@
1
+ Metadata-Version: 2.5
2
+ Name: pros-sketch
3
+ Version: 0.1.0
4
+ Summary: Partitioned and Refined Oversampling Sketches for certified geometric sketching.
5
+ Project-URL: Homepage, https://github.com/LeiLi-Uchicago/PROS
6
+ Project-URL: Issues, https://github.com/LeiLi-Uchicago/PROS/issues
7
+ Author: Lei Li
8
+ License: MIT License
9
+
10
+ Copyright (c) 2026 Lei Li
11
+
12
+ Permission is hereby granted, free of charge, to any person obtaining a copy
13
+ of this software and associated documentation files (the "Software"), to deal
14
+ in the Software without restriction, including without limitation the rights
15
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
16
+ copies of the Software, and to permit persons to whom the Software is
17
+ furnished to do so, subject to the following conditions:
18
+
19
+ The above copyright notice and this permission notice shall be included in all
20
+ copies or substantial portions of the Software.
21
+
22
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
23
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
24
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
25
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
26
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
27
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
28
+ SOFTWARE.
29
+
30
+ License-File: LICENSE
31
+ Keywords: coreset,k-center,single-cell,sketching,subsampling
32
+ Classifier: Development Status :: 4 - Beta
33
+ Classifier: Intended Audience :: Science/Research
34
+ Classifier: License :: OSI Approved :: MIT License
35
+ Classifier: Programming Language :: Python :: 3
36
+ Classifier: Programming Language :: Python :: 3.10
37
+ Classifier: Programming Language :: Python :: 3.11
38
+ Classifier: Programming Language :: Python :: 3.12
39
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
40
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
41
+ Classifier: Typing :: Typed
42
+ Requires-Python: >=3.10
43
+ Requires-Dist: numpy>=1.23
44
+ Requires-Dist: scikit-learn>=1.2
45
+ Requires-Dist: scipy>=1.9
46
+ Provides-Extra: adata
47
+ Requires-Dist: anndata>=0.9; extra == 'adata'
48
+ Provides-Extra: all
49
+ Requires-Dist: anndata>=0.9; extra == 'all'
50
+ Requires-Dist: mkdocs-material>=9.5; extra == 'all'
51
+ Requires-Dist: mkdocs>=1.6; extra == 'all'
52
+ Requires-Dist: pytest>=8; extra == 'all'
53
+ Requires-Dist: ruff>=0.6; extra == 'all'
54
+ Requires-Dist: scanpy>=1.9; extra == 'all'
55
+ Requires-Dist: scsampler; extra == 'all'
56
+ Provides-Extra: dev
57
+ Requires-Dist: pytest>=8; extra == 'dev'
58
+ Requires-Dist: ruff>=0.6; extra == 'dev'
59
+ Provides-Extra: docs
60
+ Requires-Dist: mkdocs-material>=9.5; extra == 'docs'
61
+ Requires-Dist: mkdocs>=1.6; extra == 'docs'
62
+ Provides-Extra: scanpy
63
+ Requires-Dist: anndata>=0.9; extra == 'scanpy'
64
+ Requires-Dist: scanpy>=1.9; extra == 'scanpy'
65
+ Provides-Extra: scsampler
66
+ Requires-Dist: scsampler; extra == 'scsampler'
67
+ Description-Content-Type: text/markdown
68
+
69
+ # PROS
70
+
71
+ PROS (Partitioned and Refined Oversampling Sketches) selects a small,
72
+ diversity-preserving subset of rows from a large real-valued coordinate matrix.
73
+ It builds an oversampled candidate pool within partitions and performs one
74
+ global refinement pass over that pool. The returned subset is represented by
75
+ indices into the original matrix.
76
+
77
+ The package is domain-agnostic. It can be used with reduced single-cell
78
+ embeddings, as well as other feature matrices where Euclidean distance is a
79
+ meaningful geometry.
80
+
81
+ ## Installation
82
+
83
+ PROS requires Python 3.10 or newer. The distribution name is `pros-sketch`;
84
+ the Python import remains `pros`. The commands below install this checkout.
85
+
86
+ ```bash
87
+ python -m pip install -e .
88
+ ```
89
+
90
+ Optional AnnData support:
91
+
92
+ ```bash
93
+ python -m pip install -e ".[adata]"
94
+ ```
95
+
96
+ For testing and linting:
97
+
98
+ ```bash
99
+ python -m pip install -e ".[dev]"
100
+ ```
101
+
102
+ ## Quick start
103
+
104
+ ```python
105
+ import numpy as np
106
+ from pros import sketch
107
+
108
+ X = np.random.default_rng(0).random((5_000, 50))
109
+ result = sketch(X, n=500, seed=0)
110
+
111
+ indices = result.indices
112
+ X_sketch = X[indices]
113
+ print(result.timings)
114
+ ```
115
+
116
+ `result.indices` is a sorted, unique `int64` array of length `n`. By default,
117
+ PROS uses k-means partitioning, water-filling allocation, an oversampling ratio
118
+ of `r=10.0`, and global farthest-first refinement. Pass an explicit `seed` to
119
+ make a stochastic configuration reproducible.
120
+
121
+ ## Certification
122
+
123
+ Set `certify=True` to compute the final covering radius, the candidate-pool
124
+ radius, and bounds derived from a full-data farthest-first traversal:
125
+
126
+ ```python
127
+ result = sketch(X, n=500, seed=0, certify=True)
128
+
129
+ print(result.radius) # R(X, S): final covering radius
130
+ print(result.rho) # R(X, P): candidate-pool covering radius
131
+ print(result.opt_lower) # lower bound on the optimal n-centre radius
132
+ print(result.ratio_upper) # upper bound on R(X, S) / OPT_n(X)
133
+ ```
134
+
135
+ Certification performs additional full-data work and is not included in
136
+ `result.timings["total"]`. Degenerate data may have `opt_lower == 0`; in that
137
+ case a finite approximation-ratio bound is not defined.
138
+
139
+ For an existing set of indices, use `pros.certificate`. `pros.opt_bounds`,
140
+ `pros.estimate_opt_scale`, and `pros.choose_r` are available for certificate
141
+ and oversampling-ratio workflows.
142
+
143
+ ## AnnData
144
+
145
+ ```python
146
+ from pros import sketch_adata
147
+
148
+ result = sketch_adata(adata, n=5_000, use_rep="X_pca", seed=0)
149
+ subset = adata[result.indices].copy()
150
+ ```
151
+
152
+ `sketch_adata` also accepts an `.h5ad` path and reads it in backed mode before
153
+ extracting the requested representation.
154
+
155
+ ## Configuration
156
+
157
+ ```python
158
+ result = sketch(
159
+ X,
160
+ n=1_000,
161
+ partitioner="kmeans",
162
+ n_blocks="auto",
163
+ allocator="water_filling",
164
+ r=10.0,
165
+ selector="fft",
166
+ refiner="fft",
167
+ seed=0,
168
+ )
169
+ ```
170
+
171
+ The `partitioner`, within-partition `selector`, `allocator`, and global
172
+ `refiner` are modular. See [`docs/configuration.md`](docs/configuration.md) for
173
+ the supported values and [`docs/theory.md`](docs/theory.md) for the covering
174
+ objective and certificate definitions.
175
+
176
+ ## Repository layout
177
+
178
+ - `pros/`: installable library code.
179
+ - `tests/`: automated package tests.
180
+ - `examples/`: small runnable examples using synthetic data in `data/`.
181
+ - `test_para/`: standalone parameter experiments and their generated outputs;
182
+ this reproduction material is intentionally not packaged with the library.
183
+ - `docs/`: MkDocs source pages.
184
+
185
+ Run the examples from the repository root after installation:
186
+
187
+ ```bash
188
+ python examples/01_basic_numpy.py
189
+ python examples/03_certification.py
190
+ ```
191
+
192
+ ## Citation
193
+
194
+ See [`CITATION.cff`](CITATION.cff) for software citation metadata.
195
+
196
+ ## License
197
+
198
+ PROS is distributed under the [MIT License](LICENSE).
199
+
200
+ ## Validated behavior and limits
201
+
202
+ - Water-filling is fused with FFT and requires `selector="fft"`. Choose another
203
+ allocator to use random or scSampler selection. Missing or broken scSampler
204
+ installations raise an error; no alternate algorithm is substituted.
205
+ - `mix>0` currently requires `refiner="fft"` so reserved points are retained.
206
+ `refiner="none"` uniformly downsamples the candidate pool to `n` rows.
207
+ - `certificate` computes the a posteriori ratio for any valid sketch. Its
208
+ `theory_bound` and `bound_slack` are NaN unless FFT refinement is explicitly
209
+ asserted; `sketch` sets this assertion only for unmixed FFT refinement.
210
+ Caches are tied to the exact data and sketch size. Floating-point results
211
+ are numerical bounds, not interval-arithmetic certificates.
212
+ - Internal stage timings exclude input conversion, validation, AnnData I/O,
213
+ and certification. Measure an external wall clock for runtime comparisons.
214
+ - `sketch_adata(..., return_adata=True, copy=False)` returns an in-memory view
215
+ without attaching metadata. Backed subset returns require `copy=True`.
216
+ Sparse `use_rep="X"` is rejected; supply a dense reduced embedding instead.
217
+ - Distances use SciPy directly and nearest-center evaluation tiles both axes.
218
+ Full input and block copies still require memory proportional to `N*d`;
219
+ optional swap refiners may allocate much larger matrices.
220
+
221
+ After installing development dependencies, run `python -m pytest`. Install
222
+ `.[dev,adata]` to include the optional AnnData integration regression test.
@@ -0,0 +1,154 @@
1
+ # PROS
2
+
3
+ PROS (Partitioned and Refined Oversampling Sketches) selects a small,
4
+ diversity-preserving subset of rows from a large real-valued coordinate matrix.
5
+ It builds an oversampled candidate pool within partitions and performs one
6
+ global refinement pass over that pool. The returned subset is represented by
7
+ indices into the original matrix.
8
+
9
+ The package is domain-agnostic. It can be used with reduced single-cell
10
+ embeddings, as well as other feature matrices where Euclidean distance is a
11
+ meaningful geometry.
12
+
13
+ ## Installation
14
+
15
+ PROS requires Python 3.10 or newer. The distribution name is `pros-sketch`;
16
+ the Python import remains `pros`. The commands below install this checkout.
17
+
18
+ ```bash
19
+ python -m pip install -e .
20
+ ```
21
+
22
+ Optional AnnData support:
23
+
24
+ ```bash
25
+ python -m pip install -e ".[adata]"
26
+ ```
27
+
28
+ For testing and linting:
29
+
30
+ ```bash
31
+ python -m pip install -e ".[dev]"
32
+ ```
33
+
34
+ ## Quick start
35
+
36
+ ```python
37
+ import numpy as np
38
+ from pros import sketch
39
+
40
+ X = np.random.default_rng(0).random((5_000, 50))
41
+ result = sketch(X, n=500, seed=0)
42
+
43
+ indices = result.indices
44
+ X_sketch = X[indices]
45
+ print(result.timings)
46
+ ```
47
+
48
+ `result.indices` is a sorted, unique `int64` array of length `n`. By default,
49
+ PROS uses k-means partitioning, water-filling allocation, an oversampling ratio
50
+ of `r=10.0`, and global farthest-first refinement. Pass an explicit `seed` to
51
+ make a stochastic configuration reproducible.
52
+
53
+ ## Certification
54
+
55
+ Set `certify=True` to compute the final covering radius, the candidate-pool
56
+ radius, and bounds derived from a full-data farthest-first traversal:
57
+
58
+ ```python
59
+ result = sketch(X, n=500, seed=0, certify=True)
60
+
61
+ print(result.radius) # R(X, S): final covering radius
62
+ print(result.rho) # R(X, P): candidate-pool covering radius
63
+ print(result.opt_lower) # lower bound on the optimal n-centre radius
64
+ print(result.ratio_upper) # upper bound on R(X, S) / OPT_n(X)
65
+ ```
66
+
67
+ Certification performs additional full-data work and is not included in
68
+ `result.timings["total"]`. Degenerate data may have `opt_lower == 0`; in that
69
+ case a finite approximation-ratio bound is not defined.
70
+
71
+ For an existing set of indices, use `pros.certificate`. `pros.opt_bounds`,
72
+ `pros.estimate_opt_scale`, and `pros.choose_r` are available for certificate
73
+ and oversampling-ratio workflows.
74
+
75
+ ## AnnData
76
+
77
+ ```python
78
+ from pros import sketch_adata
79
+
80
+ result = sketch_adata(adata, n=5_000, use_rep="X_pca", seed=0)
81
+ subset = adata[result.indices].copy()
82
+ ```
83
+
84
+ `sketch_adata` also accepts an `.h5ad` path and reads it in backed mode before
85
+ extracting the requested representation.
86
+
87
+ ## Configuration
88
+
89
+ ```python
90
+ result = sketch(
91
+ X,
92
+ n=1_000,
93
+ partitioner="kmeans",
94
+ n_blocks="auto",
95
+ allocator="water_filling",
96
+ r=10.0,
97
+ selector="fft",
98
+ refiner="fft",
99
+ seed=0,
100
+ )
101
+ ```
102
+
103
+ The `partitioner`, within-partition `selector`, `allocator`, and global
104
+ `refiner` are modular. See [`docs/configuration.md`](docs/configuration.md) for
105
+ the supported values and [`docs/theory.md`](docs/theory.md) for the covering
106
+ objective and certificate definitions.
107
+
108
+ ## Repository layout
109
+
110
+ - `pros/`: installable library code.
111
+ - `tests/`: automated package tests.
112
+ - `examples/`: small runnable examples using synthetic data in `data/`.
113
+ - `test_para/`: standalone parameter experiments and their generated outputs;
114
+ this reproduction material is intentionally not packaged with the library.
115
+ - `docs/`: MkDocs source pages.
116
+
117
+ Run the examples from the repository root after installation:
118
+
119
+ ```bash
120
+ python examples/01_basic_numpy.py
121
+ python examples/03_certification.py
122
+ ```
123
+
124
+ ## Citation
125
+
126
+ See [`CITATION.cff`](CITATION.cff) for software citation metadata.
127
+
128
+ ## License
129
+
130
+ PROS is distributed under the [MIT License](LICENSE).
131
+
132
+ ## Validated behavior and limits
133
+
134
+ - Water-filling is fused with FFT and requires `selector="fft"`. Choose another
135
+ allocator to use random or scSampler selection. Missing or broken scSampler
136
+ installations raise an error; no alternate algorithm is substituted.
137
+ - `mix>0` currently requires `refiner="fft"` so reserved points are retained.
138
+ `refiner="none"` uniformly downsamples the candidate pool to `n` rows.
139
+ - `certificate` computes the a posteriori ratio for any valid sketch. Its
140
+ `theory_bound` and `bound_slack` are NaN unless FFT refinement is explicitly
141
+ asserted; `sketch` sets this assertion only for unmixed FFT refinement.
142
+ Caches are tied to the exact data and sketch size. Floating-point results
143
+ are numerical bounds, not interval-arithmetic certificates.
144
+ - Internal stage timings exclude input conversion, validation, AnnData I/O,
145
+ and certification. Measure an external wall clock for runtime comparisons.
146
+ - `sketch_adata(..., return_adata=True, copy=False)` returns an in-memory view
147
+ without attaching metadata. Backed subset returns require `copy=True`.
148
+ Sparse `use_rep="X"` is rejected; supply a dense reduced embedding instead.
149
+ - Distances use SciPy directly and nearest-center evaluation tiles both axes.
150
+ Full input and block copies still require memory proportional to `N*d`;
151
+ optional swap refiners may allocate much larger matrices.
152
+
153
+ After installing development dependencies, run `python -m pytest`. Install
154
+ `.[dev,adata]` to include the optional AnnData integration regression test.
@@ -0,0 +1,13 @@
1
+ # Example Data
2
+
3
+ `synthetic_clusters.npz` contains a small synthetic coordinate matrix for
4
+ examples and smoke tests.
5
+
6
+ Fields:
7
+
8
+ - `X`: `(900, 8)` float64 coordinates.
9
+ - `labels`: `(900,)` integer cluster labels.
10
+
11
+ The dataset is generated from Gaussian clusters and is not intended as a
12
+ scientific benchmark.
13
+
@@ -0,0 +1,16 @@
1
+ # API
2
+
3
+ Public API:
4
+
5
+ - `pros.sketch(X, n, **kwargs)`: sketch a NumPy-compatible coordinate matrix.
6
+ - `pros.sketch_adata(adata, n, use_rep="X_pca", **kwargs)`: sketch an AnnData
7
+ embedding or `.h5ad` path.
8
+ - `pros.certificate(X, sketch_indices, pool_indices=None)`: compute a
9
+ certificate for an existing sketch.
10
+ - `pros.choose_r(X, n, tol=0.5)`: choose an oversampling ratio from a grid.
11
+ - `pros.estimate_opt_scale(X, n)`: estimate a target covering scale on a
12
+ subsample.
13
+
14
+ Lower-level implementation modules are intentionally not re-exported from
15
+ `pros`. They may be useful for local experiments, but are not part of the
16
+ stable public API.
@@ -0,0 +1,48 @@
1
+ # Configuration
2
+
3
+ The default configuration is intended to be a strong starting point:
4
+
5
+ ```python
6
+ res = sketch(
7
+ X,
8
+ n=1000,
9
+ partitioner="kmeans",
10
+ n_blocks="auto",
11
+ allocator="water_filling",
12
+ r=10.0,
13
+ selector="fft",
14
+ refiner="fft",
15
+ seed=0,
16
+ )
17
+ ```
18
+
19
+ Useful options:
20
+
21
+ - `partitioner`: `"kmeans"`, `"pc_tree"`, `"random"`, or `"none"`.
22
+ - `allocator`: `"water_filling"`, `"proportional"`, `"power"`, `"volume"`, or
23
+ `"uniform"`.
24
+ - `selector`: `"fft"`, `"scsampler_maximin"`, or `"random"`.
25
+ - `refiner`: `"fft"`, `"maximin"`, `"local_swap"`, or `"none"`.
26
+ - `r`: oversampling ratio. Larger values make a denser pool and cost more.
27
+ - `certify`: compute certificate fields for evaluation.
28
+
29
+ For a single global FFT selection, use `partitioner="none", r=1,
30
+ refiner="none"`: stage 1 selects n points and stage 2 retains the whole pool.
31
+ With r>1, the single-block configuration still runs two selection stages.
32
+
33
+ Water-filling requires `selector="fft"`; incompatible combinations raise an
34
+ error. Other allocators support the other selectors. scSampler failures are
35
+ propagated, never replaced silently with FFT. Fixed seeds are reproducible
36
+ within a fixed numerical/software environment; record dependency versions and
37
+ thread settings for experiments.
38
+
39
+ `mix>0` requires FFT refinement to preserve the uniform reserve. With
40
+ `mix_source="data"`, reserve points may enlarge the pool beyond the stage-1
41
+ `budget`; `m_per_block` still describes stage 1. `block_radii` and
42
+ `rho_blockwise_max` are available only with water-filling. An unfunded nonempty
43
+ block has infinite local covering radius.
44
+
45
+ The pool budget is `min(ceil(r*n), N)`. Larger r need not monotonically improve
46
+ the greedy final radius. `choose_r` reports `tolerance_met` and certifies only
47
+ its returned sketch. Its pool-radius evaluations can still be expensive.
48
+
@@ -0,0 +1,18 @@
1
+ # Examples
2
+
3
+ The `examples/` directory contains small runnable scripts:
4
+
5
+ - `01_basic_numpy.py`: basic matrix sketching.
6
+ - `02_ann_data.py`: AnnData workflow.
7
+ - `03_certification.py`: certificate fields.
8
+ - `04_compare_configs.py`: simple configuration comparison.
9
+
10
+ Run an example from the repository root:
11
+
12
+ ```bash
13
+ python examples/01_basic_numpy.py
14
+ ```
15
+
16
+ The checkout examples use `partitioner="pc_tree"` to keep their behaviour
17
+ simple. The installed package includes `scikit-learn`, which is required by
18
+ the default `partitioner="kmeans"` workflow.
@@ -0,0 +1,10 @@
1
+ # PROS
2
+
3
+ PROS builds diversity-preserving sketches of large coordinate matrices. It
4
+ first partitions the data, oversamples a candidate pool inside those blocks,
5
+ and then refines globally over the pool.
6
+
7
+ The returned sketch is an array of row indices into the original matrix. For
8
+ evaluation runs, PROS can also attach a certificate with the sketch covering
9
+ radius, pool covering radius, and a proven approximation-ratio upper bound.
10
+
@@ -0,0 +1,19 @@
1
+ # Installation
2
+
3
+ Install from a local checkout:
4
+
5
+ ```bash
6
+ python -m pip install -e .
7
+ ```
8
+
9
+ Optional extras:
10
+
11
+ ```bash
12
+ python -m pip install -e ".[adata]"
13
+ python -m pip install -e ".[dev]"
14
+ python -m pip install -e ".[docs]"
15
+ ```
16
+
17
+ The core package depends on `numpy` and `scikit-learn`. AnnData and Scanpy are
18
+ optional so matrix-only users do not need the single-cell stack.
19
+
@@ -0,0 +1,61 @@
1
+ # Publishing pros-sketch
2
+
3
+ The distribution is `pros-sketch`, while the Python import is `pros`.
4
+ The workflow `.github/workflows/publish.yml` publishes to PyPI only when a
5
+ GitHub Release is published. Manual workflow runs perform validation and build
6
+ artifacts without uploading. No PyPI API token secret is required.
7
+
8
+ ## One-time PyPI setup
9
+
10
+ For the first release, sign in to https://pypi.org/manage/account/publishing/
11
+ and add a pending publisher under GitHub:
12
+
13
+ | Field | Value |
14
+ | --- | --- |
15
+ | PyPI Project Name | `pros-sketch` |
16
+ | Owner | `LeiLi-Uchicago` |
17
+ | Repository name | `PROS` |
18
+ | Workflow name | `publish.yml` |
19
+ | Environment name | `pypi` |
20
+
21
+ Use the workflow filename, not its display title or full path. A pending
22
+ publisher does not reserve the project name; successful first upload creates
23
+ it. If this project already belongs to your account, add the publisher from
24
+ the project's Publishing settings instead.
25
+
26
+ In GitHub repository Settings → Environments, create the `pypi` environment.
27
+ Its name must match both the workflow and PyPI configuration. Optional review
28
+ and deployment restrictions can be configured there.
29
+
30
+ ## Release procedure
31
+
32
+ 1. Commit and push the reviewed code and publishing workflow to GitHub.
33
+ 2. Run **Publish to PyPI** manually from Actions. This tests Python 3.10–3.12,
34
+ builds the sdist and wheel, checks metadata, and tests the installed wheel
35
+ outside the checkout. Confirm the run succeeds.
36
+ 3. Ensure `pyproject.toml` and `pros/__init__.py` specify the same release
37
+ version. For the first release this is `0.1.0`.
38
+ 4. Create a GitHub Release with tag `v0.1.0` pointing to the validated commit.
39
+ Publishing it triggers the workflow, repeats validation, and uploads the
40
+ built distributions through Trusted Publishing. A draft release does not.
41
+ 5. Verify the project page and install `pros-sketch==0.1.0` in a new environment.
42
+ For AnnData support, install `pros-sketch[adata]==0.1.0`.
43
+
44
+ For later releases, update both version values and use the corresponding tag.
45
+ PyPI does not allow overwriting previously uploaded distribution filenames.
46
+ If an upload fails, inspect the logs and which artifacts reached PyPI before
47
+ retrying; this workflow intentionally does not hide existing-file errors.
48
+
49
+ TestPyPI is not configured by this workflow. It requires a separate account
50
+ and trusted publisher if a rehearsal upload is wanted.
51
+
52
+ ## Scope of validation
53
+
54
+ The workflow tests core and AnnData functionality. The scSampler adapter is
55
+ covered by contract tests, not an end-to-end installation of scSampler. Large
56
+ benchmark datasets and manuscript runtime reproduction are not run in CI.
57
+
58
+ References:
59
+ - https://docs.pypi.org/trusted-publishers/creating-a-project-through-oidc/
60
+ - https://docs.pypi.org/trusted-publishers/adding-a-publisher/
61
+ - https://github.com/pypa/gh-action-pypi-publish
@@ -0,0 +1,30 @@
1
+ # Quickstart
2
+
3
+ ```python
4
+ import numpy as np
5
+ from pros import sketch
6
+
7
+ rng = np.random.default_rng(0)
8
+ X = rng.random((5000, 50))
9
+
10
+ res = sketch(X, n=500, seed=0)
11
+ X_sketch = X[res.indices]
12
+ ```
13
+
14
+ The result is dict-like and also supports attribute access:
15
+
16
+ ```python
17
+ print(res["indices"])
18
+ print(res.indices)
19
+ print(res.timings)
20
+ ```
21
+
22
+ For AnnData:
23
+
24
+ ```python
25
+ from pros import sketch_adata
26
+
27
+ res = sketch_adata(adata, n=5000, use_rep="X_pca")
28
+ subset = adata[res.indices]
29
+ ```
30
+