auroraomics 0.1.0.dev1__tar.gz → 0.1.0.dev3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. {auroraomics-0.1.0.dev1/src/auroraomics.egg-info → auroraomics-0.1.0.dev3}/PKG-INFO +76 -74
  2. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/README.md +73 -72
  3. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/pyproject.toml +25 -19
  4. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/__init__.py +20 -15
  5. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/MANIFEST.json +1 -1
  6. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/public-api.tokens.json +21 -3
  7. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_text.py +14 -1
  8. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/bulk.py +28 -34
  9. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/cli.py +329 -93
  10. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/__init__.py +8 -4
  11. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/_generated.py +28 -31
  12. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/_http.py +4 -4
  13. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/api.py +135 -61
  14. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/credentials.py +14 -3
  15. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/errors.py +27 -15
  16. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/jobs.py +13 -15
  17. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/uploads.py +1 -2
  18. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/contracts.py +32 -30
  19. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/embed.py +452 -192
  20. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/errors.py +7 -5
  21. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/genes.py +10 -21
  22. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/h5ad.py +8 -17
  23. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/mcp/server.py +5 -5
  24. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/mcp/tools.py +2 -3
  25. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/pack.py +141 -57
  26. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/postprocess.py +4 -7
  27. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/qc.py +16 -15
  28. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/runtimes/__init__.py +1 -9
  29. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/runtimes/deepspotm.py +250 -124
  30. auroraomics-0.1.0.dev3/src/auroraomics/slide.py +784 -0
  31. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/spatial.py +9 -2
  32. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/subsample.py +1 -2
  33. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3/src/auroraomics.egg-info}/PKG-INFO +76 -74
  34. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/requires.txt +1 -0
  35. auroraomics-0.1.0.dev1/src/auroraomics/slide.py +0 -325
  36. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/LICENSE +0 -0
  37. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/MANIFEST.in +0 -0
  38. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/setup.cfg +0 -0
  39. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_assets/ASSETS.json +0 -0
  40. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_assets/README.md +0 -0
  41. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_assets/calibration-tile.png +0 -0
  42. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/public-api-counters.tokens.json +0 -0
  43. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/public-api-input-kinds.tokens.json +0 -0
  44. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/mcp/__init__.py +0 -0
  45. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/py.typed +0 -0
  46. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/SOURCES.txt +0 -0
  47. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/dependency_links.txt +0 -0
  48. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/entry_points.txt +0 -0
  49. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/top_level.txt +0 -0
@@ -1,7 +1,7 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: auroraomics
3
- Version: 0.1.0.dev1
4
- Summary: Virtual spatial transcriptomics from H&E histology: patch QC, tile packing and the .h5ad result contract.
3
+ Version: 0.1.0.dev3
4
+ Summary: Prepare H&E slides on your own machine, submit them to Aurora, and read the spatial gene expression it predicts.
5
5
  Author: Kalin Nonchev
6
6
  License-Expression: PolyForm-Noncommercial-1.0.0
7
7
  Keywords: histopathology,spatial-transcriptomics,h5ad,anndata,whole-slide-image
@@ -33,6 +33,7 @@ Requires-Dist: mcp<3,>=2; extra == "mcp"
33
33
  Provides-Extra: dev
34
34
  Requires-Dist: auroraomics[mcp]; extra == "dev"
35
35
  Requires-Dist: pytest>=7; extra == "dev"
36
+ Requires-Dist: packaging>=22; extra == "dev"
36
37
  Requires-Dist: pytest-cov>=5; extra == "dev"
37
38
  Requires-Dist: setuptools>=77; extra == "dev"
38
39
  Requires-Dist: wheel; extra == "dev"
@@ -43,10 +44,14 @@ Dynamic: license-file
43
44
 
44
45
  # auroraomics
45
46
 
46
- Virtual spatial transcriptomics from H&E histology.
47
+ Virtual spatial transcriptomics from H&E images.
48
+
49
+ This package is the client of the **Aurora API**, the hosted service that runs
50
+ the prediction: from Python, from a shell with the `auroraomics` command, or
51
+ from an agent through its MCP server.
47
52
 
48
53
  This package holds the pieces of that pipeline that are pure Python, so the
49
- same code runs on your laptop, in a GPU container and on the service:
54
+ same code runs on your machine and on the service:
50
55
 
51
56
  - **`auroraomics.qc`** — the three patch-quality predicates (foreground, blur,
52
57
  stained tissue) applied to every candidate tile before a model sees it, and
@@ -58,13 +63,13 @@ same code runs on your laptop, in a GPU container and on the service:
58
63
  batch rather than the whole matrix.
59
64
  - **`auroraomics.postprocess`** — apply a model card's `postprocess` entries
60
65
  to a result file: a registry of entries, one runner, `h5py` alone, streaming
61
- one chunk at a time. Ships the unmeasured-gene mask, which states per gene
62
- how many training datasets measured it and blanks the ones none did.
66
+ one chunk at a time. Ships the unmeasured-gene mask, which blanks the genes
67
+ the model never measured.
63
68
  - **`auroraomics.subsample`** — pick the densest contiguous square of spots
64
- when a slide yields more tiles than a run is allowed to spend.
65
- - **`auroraomics.genes`** — the model family's gene symbols mapped to Ensembl
66
- stable gene ids, which is what every result's `var` index is keyed by. Ships
67
- as one committed table, resolved in the same order the service resolves it.
69
+ when a slide yields more tiles than one submission takes.
70
+ - **`auroraomics.genes`** — Ensembl stable gene ids, which is what every
71
+ result's `var` index is keyed by, and the shape a gene the service resolved
72
+ comes back in.
68
73
  - **`auroraomics.contracts`** — the shared contract values (container layout,
69
74
  input-kind caps, result layout) as data, so nothing here re-types a number
70
75
  the service also reads.
@@ -88,16 +93,18 @@ pip install auroraomics
88
93
 
89
94
  That is the whole documented path: open a slide, judge and pack its tiles,
90
95
  submit, and read the result. One extra adds local embedding extraction
91
- (`embed`), because a tensor runtime is gigabytes and specific to the machine it
92
- was built for. No extra brings the model: predicting gene expression runs on
93
- the service, not here.
96
+ (`embed`). No extra brings the model: predicting gene expression runs on the
97
+ service, not here.
94
98
 
95
99
  ## Pack tiles, then look at the report
96
100
 
97
101
  ```python
98
102
  import auroraomics as ao
99
103
 
100
- report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thumbnail=thumb)
104
+ with ao.Client() as client:
105
+ thresholds = client.qc_thresholds() # the service's quality floors, no key needed
106
+
107
+ report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thresholds=thresholds, thumbnail=thumb)
101
108
  print(report.written, "tiles kept,", report.rejected, "dropped")
102
109
  print(report.rejected_by_reason) # {'foreground_ratio': 12, ...}
103
110
  ```
@@ -126,7 +133,8 @@ members do not match the container contract.
126
133
  ao.write_result(
127
134
  "result.h5ad",
128
135
  obs=obs, # per-spot columns, as plain arrays
129
- var=var, # per-gene columns, indexed by gene id
136
+ var=var, # per-gene columns
137
+ var_index=gene_ids, # the Ensembl gene id of each column
130
138
  spatial=coords, # (n_spots, 2) array -> obsm["spatial"]
131
139
  x=batches, # an array, or an iterable of row batches
132
140
  uns={"model": {"id": "..."}},
@@ -149,70 +157,69 @@ pip install "auroraomics[embed]"
149
157
  import auroraomics as ao
150
158
  from auroraomics.embed import available_encoders, describe_encoders
151
159
 
152
- print(available_encoders()) # ('deepspot-h', 'dinov2-b14', 'midnight')
153
-
154
160
  with ao.Client() as client:
155
161
  # Both are the service's, so the file records the encoder and the
156
- # thresholds the model it is submitted to was served with.
157
- encoder = ao.resolve_encoder(client.encoders())
162
+ # thresholds the model it is submitted to was served with. There is no
163
+ # list of encoders inside this package: every name comes from the registry
164
+ # the service publishes, which is why these calls take one.
165
+ registry = client.encoders()
158
166
  thresholds = client.qc_thresholds()
167
+ # The tile size the model's card states, for the model the file is for.
168
+ patch_um = client.model_card(model_id)["input_spec"]["patch_um"]
169
+
170
+ print(available_encoders(registry)) # the names this package will run today
171
+ encoder = ao.resolve_encoder(registry)
159
172
 
160
173
  report = ao.embed_tiles(
161
- slide, "sample.npz", mpp=0.499, patch_um=55, encoder=encoder, thresholds=thresholds
174
+ slide, "sample.npz", mpp=0.499, patch_um=patch_um, encoder=encoder, thresholds=thresholds
162
175
  )
163
176
  print(report.rows, "rows of", report.dim, "numbers from", report.encoder.name)
164
177
  ```
165
178
 
166
179
  `slide` is anything with a `(height, width, 3)` RGB `uint8` shape that can be
167
180
  sliced — an array, or a memory-mapped or lazily-read one, so a slide larger than
168
- memory works: only one crop exists at a time. Crops are taken at `patch_um`
169
- micrometres using the `mpp` you give, quality-checked with the same three
170
- predicates as `pack_tiles`, and pooled over the encoder's patch tokens.
171
-
172
- Every encoder is pinned to a repository **and** a revision sha, with its licence
173
- and access gate recorded beside it. `describe_encoders()` returns the whole
174
- registry, including the encoders this package refuses to run — a refusal names
175
- the licence or the approval you would have to accept, and which encoders remain.
176
-
177
- The written file carries one extra vector, `calibration`: the embedding of a
178
- fixed image shipped inside this package, through the same encoder, revision and
179
- preprocessing as your rows. Comparing that one vector against a known reference
180
- tells a reader whether the file was produced by the model it claims, before they
181
- look at a single row.
182
-
183
- The weights are downloaded from their publisher on first use, into the ordinary
184
- model cache. Pass `allow_download=False` to guarantee no request is made: for
185
- the duration of that load, name resolution and internet sockets are refused in
186
- this process, so the guarantee does not rest on a model library reading an
187
- environment variable it may already have read.
188
-
189
- An encoder runs at one input size: the size the served model's embeddings were
190
- computed at, which the registry pins and which is not always what the encoder's
191
- own repository declares. The DINOv2 releases carry a 518-px position grid; the
192
- model was trained on 224-px crops fed to that grid through an interpolated
193
- position embedding, so that is what this package does too. `resize_px` is
194
- checked against the pin rather than obeyed, because a vector computed at another
195
- size is a different measurement that nobody else's numbers can be compared with.
181
+ memory works: only one crop exists at a time. Crops are taken at the tile size
182
+ the model's card states, using the `mpp` you give, and quality-checked with the
183
+ same three predicates as `pack_tiles`.
184
+
185
+ Every encoder is pinned to an exact revision.
186
+ `describe_encoders(registry)` returns the whole registry, including the
187
+ encoders this package refuses to run: a refusal says why, and which encoders
188
+ remain.
189
+
190
+ The written file carries one extra embedding, `calibration`: computed from a
191
+ fixed image shipped inside this package, through the same encoder and revision
192
+ as your rows. Comparing that one embedding against a known
193
+ reference tells a reader whether the file was produced by the model it claims,
194
+ before they look at a single row.
195
+
196
+ The weights are downloaded on first use. Pass `allow_download=False` to
197
+ guarantee no request is made: for the duration of that load, name resolution
198
+ and internet sockets are refused in this process.
199
+
200
+ An encoder runs at the input size the registry pins for it. `resize_px` is
201
+ checked against that pin rather than obeyed, because an embedding computed at
202
+ another size cannot be compared with the served model's.
203
+
196
204
  ## Where a prediction runs
197
205
 
198
- Predicting gene expression runs on the service, and `predict_local` refuses —
199
- by design, and the design is the product rather than a limitation.
206
+ Predicting gene expression runs on the service, and `predict_local` refuses.
200
207
 
201
- This is split inference at the encoder. An open-weight pathology foundation
202
- model turns your tiles into patch embeddings **on your machine**, so your slides
203
- never leave it; our gene decoder turns embeddings into expression **on ours**, so
204
- its weights never leave it. What crosses between us is a vector from a model
205
- neither side owns.
208
+ **DeepSpot-H**, the foundation model for H&E images, turns each tile of your
209
+ slide into one embedding. It can run on your machine or on ours. **DeepSpot-M**
210
+ turns those embeddings into spatial gene expression. It only ever runs on ours.
206
211
 
207
212
  ```python
208
- from auroraomics.runtimes.deepspotm import available_models, check_archive
213
+ from auroraomics.runtimes.deepspotm import check_archive, describe_models
209
214
 
210
- available_models() # what the service will run
211
- check_archive("sample.zip", model=...) # are my tiles the right physical size?
215
+ with ao.Client() as client:
216
+ cards = client.models() # the catalogue is the service's
217
+ describe_models(cards) # what it will run, and on what
218
+ check_archive("sample.zip", model=model_id, cards=cards) # are my tiles the right size?
212
219
  ```
213
220
 
214
221
  Then submit with the API client and read the result back — the same `.h5ad`
215
- contract whichever lane ran it.
222
+ contract whichever runtime ran it.
216
223
 
217
224
  ## Post-process a result
218
225
 
@@ -236,17 +243,19 @@ skipped: a card is the description of a file, and a reader promised a layer
236
243
  cannot tell a missing one from a model that predicted zeros.
237
244
 
238
245
  The entry this version ships states what the model measured. `var` gains
239
- `n_train_datasets` — the number of training datasets each gene was measured in,
240
- because a gene seen in one and a gene seen in twenty are not the same claim —
241
- and `measured_in_training`, derived from that count. A gene the model never
242
- measured becomes `nan` in `X` and in every layer, not zero: zero is a
243
- measurement, and an unmeasured gene left as one averages, correlates and
246
+ `measured_in_training`, which is `False` for a gene the model never saw
247
+ measured. Such a gene becomes `nan` in `X` and in every layer, not zero: zero
248
+ is a measurement, and an unmeasured gene left as one averages, correlates and
244
249
  colours a heat map exactly like a prediction.
245
250
 
246
251
  ## Prepare a bulk RNA profile
247
252
 
248
253
  ```python
249
- report = ao.bulk_rna("sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts")
254
+ with ao.Client() as client:
255
+ report = ao.bulk_rna(
256
+ "sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts",
257
+ card=client.model_card(model_id), resolve=client.resolve_keys,
258
+ )
250
259
  print(report.rows, "genes;", f"{report.coverage:.0%} of", report.coverage_of)
251
260
  ```
252
261
 
@@ -266,13 +275,6 @@ your project.
266
275
 
267
276
  ## Licence
268
277
 
269
- The code in this package is licensed under
270
- [PolyForm Noncommercial 1.0.0](https://polyformproject.org/licenses/noncommercial/1.0.0),
271
- which permits use for any purpose that is not commercial. It is the same licence the
272
- model package this client is built for carries, so installing both puts you under one
273
- rule rather than two.
274
-
275
- The model weights are licensed separately by whoever publishes them, and access to them
276
- may be gated. Read those terms before you use a model: they are not this licence, and a
277
- permission granted here is not a permission granted there.
278
+ The code in this package is licensed for non-commercial use; its terms are in the
279
+ `LICENSE` file it ships with. Commercial evaluation and use are governed by a written agreement with Aurora.
278
280
 
@@ -1,9 +1,13 @@
1
1
  # auroraomics
2
2
 
3
- Virtual spatial transcriptomics from H&E histology.
3
+ Virtual spatial transcriptomics from H&E images.
4
+
5
+ This package is the client of the **Aurora API**, the hosted service that runs
6
+ the prediction: from Python, from a shell with the `auroraomics` command, or
7
+ from an agent through its MCP server.
4
8
 
5
9
  This package holds the pieces of that pipeline that are pure Python, so the
6
- same code runs on your laptop, in a GPU container and on the service:
10
+ same code runs on your machine and on the service:
7
11
 
8
12
  - **`auroraomics.qc`** — the three patch-quality predicates (foreground, blur,
9
13
  stained tissue) applied to every candidate tile before a model sees it, and
@@ -15,13 +19,13 @@ same code runs on your laptop, in a GPU container and on the service:
15
19
  batch rather than the whole matrix.
16
20
  - **`auroraomics.postprocess`** — apply a model card's `postprocess` entries
17
21
  to a result file: a registry of entries, one runner, `h5py` alone, streaming
18
- one chunk at a time. Ships the unmeasured-gene mask, which states per gene
19
- how many training datasets measured it and blanks the ones none did.
22
+ one chunk at a time. Ships the unmeasured-gene mask, which blanks the genes
23
+ the model never measured.
20
24
  - **`auroraomics.subsample`** — pick the densest contiguous square of spots
21
- when a slide yields more tiles than a run is allowed to spend.
22
- - **`auroraomics.genes`** — the model family's gene symbols mapped to Ensembl
23
- stable gene ids, which is what every result's `var` index is keyed by. Ships
24
- as one committed table, resolved in the same order the service resolves it.
25
+ when a slide yields more tiles than one submission takes.
26
+ - **`auroraomics.genes`** — Ensembl stable gene ids, which is what every
27
+ result's `var` index is keyed by, and the shape a gene the service resolved
28
+ comes back in.
25
29
  - **`auroraomics.contracts`** — the shared contract values (container layout,
26
30
  input-kind caps, result layout) as data, so nothing here re-types a number
27
31
  the service also reads.
@@ -45,16 +49,18 @@ pip install auroraomics
45
49
 
46
50
  That is the whole documented path: open a slide, judge and pack its tiles,
47
51
  submit, and read the result. One extra adds local embedding extraction
48
- (`embed`), because a tensor runtime is gigabytes and specific to the machine it
49
- was built for. No extra brings the model: predicting gene expression runs on
50
- the service, not here.
52
+ (`embed`). No extra brings the model: predicting gene expression runs on the
53
+ service, not here.
51
54
 
52
55
  ## Pack tiles, then look at the report
53
56
 
54
57
  ```python
55
58
  import auroraomics as ao
56
59
 
57
- report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thumbnail=thumb)
60
+ with ao.Client() as client:
61
+ thresholds = client.qc_thresholds() # the service's quality floors, no key needed
62
+
63
+ report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thresholds=thresholds, thumbnail=thumb)
58
64
  print(report.written, "tiles kept,", report.rejected, "dropped")
59
65
  print(report.rejected_by_reason) # {'foreground_ratio': 12, ...}
60
66
  ```
@@ -83,7 +89,8 @@ members do not match the container contract.
83
89
  ao.write_result(
84
90
  "result.h5ad",
85
91
  obs=obs, # per-spot columns, as plain arrays
86
- var=var, # per-gene columns, indexed by gene id
92
+ var=var, # per-gene columns
93
+ var_index=gene_ids, # the Ensembl gene id of each column
87
94
  spatial=coords, # (n_spots, 2) array -> obsm["spatial"]
88
95
  x=batches, # an array, or an iterable of row batches
89
96
  uns={"model": {"id": "..."}},
@@ -106,70 +113,69 @@ pip install "auroraomics[embed]"
106
113
  import auroraomics as ao
107
114
  from auroraomics.embed import available_encoders, describe_encoders
108
115
 
109
- print(available_encoders()) # ('deepspot-h', 'dinov2-b14', 'midnight')
110
-
111
116
  with ao.Client() as client:
112
117
  # Both are the service's, so the file records the encoder and the
113
- # thresholds the model it is submitted to was served with.
114
- encoder = ao.resolve_encoder(client.encoders())
118
+ # thresholds the model it is submitted to was served with. There is no
119
+ # list of encoders inside this package: every name comes from the registry
120
+ # the service publishes, which is why these calls take one.
121
+ registry = client.encoders()
115
122
  thresholds = client.qc_thresholds()
123
+ # The tile size the model's card states, for the model the file is for.
124
+ patch_um = client.model_card(model_id)["input_spec"]["patch_um"]
125
+
126
+ print(available_encoders(registry)) # the names this package will run today
127
+ encoder = ao.resolve_encoder(registry)
116
128
 
117
129
  report = ao.embed_tiles(
118
- slide, "sample.npz", mpp=0.499, patch_um=55, encoder=encoder, thresholds=thresholds
130
+ slide, "sample.npz", mpp=0.499, patch_um=patch_um, encoder=encoder, thresholds=thresholds
119
131
  )
120
132
  print(report.rows, "rows of", report.dim, "numbers from", report.encoder.name)
121
133
  ```
122
134
 
123
135
  `slide` is anything with a `(height, width, 3)` RGB `uint8` shape that can be
124
136
  sliced — an array, or a memory-mapped or lazily-read one, so a slide larger than
125
- memory works: only one crop exists at a time. Crops are taken at `patch_um`
126
- micrometres using the `mpp` you give, quality-checked with the same three
127
- predicates as `pack_tiles`, and pooled over the encoder's patch tokens.
128
-
129
- Every encoder is pinned to a repository **and** a revision sha, with its licence
130
- and access gate recorded beside it. `describe_encoders()` returns the whole
131
- registry, including the encoders this package refuses to run — a refusal names
132
- the licence or the approval you would have to accept, and which encoders remain.
133
-
134
- The written file carries one extra vector, `calibration`: the embedding of a
135
- fixed image shipped inside this package, through the same encoder, revision and
136
- preprocessing as your rows. Comparing that one vector against a known reference
137
- tells a reader whether the file was produced by the model it claims, before they
138
- look at a single row.
139
-
140
- The weights are downloaded from their publisher on first use, into the ordinary
141
- model cache. Pass `allow_download=False` to guarantee no request is made: for
142
- the duration of that load, name resolution and internet sockets are refused in
143
- this process, so the guarantee does not rest on a model library reading an
144
- environment variable it may already have read.
145
-
146
- An encoder runs at one input size: the size the served model's embeddings were
147
- computed at, which the registry pins and which is not always what the encoder's
148
- own repository declares. The DINOv2 releases carry a 518-px position grid; the
149
- model was trained on 224-px crops fed to that grid through an interpolated
150
- position embedding, so that is what this package does too. `resize_px` is
151
- checked against the pin rather than obeyed, because a vector computed at another
152
- size is a different measurement that nobody else's numbers can be compared with.
137
+ memory works: only one crop exists at a time. Crops are taken at the tile size
138
+ the model's card states, using the `mpp` you give, and quality-checked with the
139
+ same three predicates as `pack_tiles`.
140
+
141
+ Every encoder is pinned to an exact revision.
142
+ `describe_encoders(registry)` returns the whole registry, including the
143
+ encoders this package refuses to run: a refusal says why, and which encoders
144
+ remain.
145
+
146
+ The written file carries one extra embedding, `calibration`: computed from a
147
+ fixed image shipped inside this package, through the same encoder and revision
148
+ as your rows. Comparing that one embedding against a known
149
+ reference tells a reader whether the file was produced by the model it claims,
150
+ before they look at a single row.
151
+
152
+ The weights are downloaded on first use. Pass `allow_download=False` to
153
+ guarantee no request is made: for the duration of that load, name resolution
154
+ and internet sockets are refused in this process.
155
+
156
+ An encoder runs at the input size the registry pins for it. `resize_px` is
157
+ checked against that pin rather than obeyed, because an embedding computed at
158
+ another size cannot be compared with the served model's.
159
+
153
160
  ## Where a prediction runs
154
161
 
155
- Predicting gene expression runs on the service, and `predict_local` refuses —
156
- by design, and the design is the product rather than a limitation.
162
+ Predicting gene expression runs on the service, and `predict_local` refuses.
157
163
 
158
- This is split inference at the encoder. An open-weight pathology foundation
159
- model turns your tiles into patch embeddings **on your machine**, so your slides
160
- never leave it; our gene decoder turns embeddings into expression **on ours**, so
161
- its weights never leave it. What crosses between us is a vector from a model
162
- neither side owns.
164
+ **DeepSpot-H**, the foundation model for H&E images, turns each tile of your
165
+ slide into one embedding. It can run on your machine or on ours. **DeepSpot-M**
166
+ turns those embeddings into spatial gene expression. It only ever runs on ours.
163
167
 
164
168
  ```python
165
- from auroraomics.runtimes.deepspotm import available_models, check_archive
169
+ from auroraomics.runtimes.deepspotm import check_archive, describe_models
166
170
 
167
- available_models() # what the service will run
168
- check_archive("sample.zip", model=...) # are my tiles the right physical size?
171
+ with ao.Client() as client:
172
+ cards = client.models() # the catalogue is the service's
173
+ describe_models(cards) # what it will run, and on what
174
+ check_archive("sample.zip", model=model_id, cards=cards) # are my tiles the right size?
169
175
  ```
170
176
 
171
177
  Then submit with the API client and read the result back — the same `.h5ad`
172
- contract whichever lane ran it.
178
+ contract whichever runtime ran it.
173
179
 
174
180
  ## Post-process a result
175
181
 
@@ -193,17 +199,19 @@ skipped: a card is the description of a file, and a reader promised a layer
193
199
  cannot tell a missing one from a model that predicted zeros.
194
200
 
195
201
  The entry this version ships states what the model measured. `var` gains
196
- `n_train_datasets` — the number of training datasets each gene was measured in,
197
- because a gene seen in one and a gene seen in twenty are not the same claim —
198
- and `measured_in_training`, derived from that count. A gene the model never
199
- measured becomes `nan` in `X` and in every layer, not zero: zero is a
200
- measurement, and an unmeasured gene left as one averages, correlates and
202
+ `measured_in_training`, which is `False` for a gene the model never saw
203
+ measured. Such a gene becomes `nan` in `X` and in every layer, not zero: zero
204
+ is a measurement, and an unmeasured gene left as one averages, correlates and
201
205
  colours a heat map exactly like a prediction.
202
206
 
203
207
  ## Prepare a bulk RNA profile
204
208
 
205
209
  ```python
206
- report = ao.bulk_rna("sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts")
210
+ with ao.Client() as client:
211
+ report = ao.bulk_rna(
212
+ "sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts",
213
+ card=client.model_card(model_id), resolve=client.resolve_keys,
214
+ )
207
215
  print(report.rows, "genes;", f"{report.coverage:.0%} of", report.coverage_of)
208
216
  ```
209
217
 
@@ -223,13 +231,6 @@ your project.
223
231
 
224
232
  ## Licence
225
233
 
226
- The code in this package is licensed under
227
- [PolyForm Noncommercial 1.0.0](https://polyformproject.org/licenses/noncommercial/1.0.0),
228
- which permits use for any purpose that is not commercial. It is the same licence the
229
- model package this client is built for carries, so installing both puts you under one
230
- rule rather than two.
231
-
232
- The model weights are licensed separately by whoever publishes them, and access to them
233
- may be gated. Read those terms before you use a model: they are not this licence, and a
234
- permission granted here is not a permission granted there.
234
+ The code in this package is licensed for non-commercial use; its terms are in the
235
+ `LICENSE` file it ships with. Commercial evaluation and use are governed by a written agreement with Aurora.
235
236
 
@@ -4,14 +4,12 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "auroraomics"
7
- version = "0.1.0.dev1"
8
- description = "Virtual spatial transcriptomics from H&E histology: patch QC, tile packing and the .h5ad result contract."
7
+ version = "0.1.0.dev3"
8
+ description = "Prepare H&E slides on your own machine, submit them to Aurora, and read the spatial gene expression it predicts."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
11
- # Non-commercial, and the same licence as the model package this client is built
12
- # for, so a user installing both faces one rule rather than two. PEP 639 forbids
13
- # a licence CLASSIFIER beside an expression, so this line is the whole
14
- # declaration; the model weights carry their own separate terms.
11
+ # Non-commercial; the terms are in LICENSE. PEP 639 forbids a licence
12
+ # CLASSIFIER beside an expression, so this line is the whole declaration.
15
13
  license = "PolyForm-Noncommercial-1.0.0"
16
14
  license-files = ["LICENSE"]
17
15
  authors = [{ name = "Kalin Nonchev" }]
@@ -31,7 +29,7 @@ classifiers = [
31
29
  # install line the quickstart prints is an install line that can do what the
32
30
  # quickstart says.
33
31
  #
34
- # It was four names until #691, with the API client and the slide reader behind
32
+ # It was four names once, with the API client and the slide reader behind
35
33
  # extras of their own. The comment on the extras below has always ended "the
36
34
  # line is at the ENCODER, not at the size of an install"; these lists did not
37
35
  # say it. They put a half-megabyte HTTP client behind the same ceremony as a
@@ -39,7 +37,7 @@ classifiers = [
39
37
  # auroraomics` could not then perform a single documented step. They say it now.
40
38
  #
41
39
  # WHAT IT COSTS, measured rather than estimated (importlib.metadata over the
42
- # transitive closure, on the interpreter this package targets, 2026-09-11): four
40
+ # transitive closure, on the interpreter this package targets): four
43
41
  # distributions and 179 MB became twenty-eight and 537 MB. Neither of the two
44
42
  # that dominate is named here — `imagecodecs` (142 MB) is what lets `tifffile`
45
43
  # decode a real slide, and `pandas` plus `scipy` (198 MB between them) arrive
@@ -47,9 +45,9 @@ classifiers = [
47
45
  # all of it. That is the trade, taken deliberately: one install line that works,
48
46
  # over a smaller one that cannot reach the end of the page describing it.
49
47
  #
50
- # What is still NOT here is the name that matters to the GPU pipeline images —
51
- # no torch. They install this package with `--no-deps`, so what they resolve
52
- # does not change either way; what protects them is the encoder line below.
48
+ # What is still NOT here is torch. An environment that installs this package
49
+ # with `--no-deps` resolves the same either way; what protects it is the
50
+ # encoder line below.
53
51
  dependencies = [
54
52
  "numpy>=1.23",
55
53
  "h5py>=3.9",
@@ -79,7 +77,7 @@ dependencies = [
79
77
  # resolves the model, or the adapter and checkpoint stack that would carry one,
80
78
  # and `predict_local` refuses with a message that says so. What runs on a user's
81
79
  # own machine is cutting, filtering and EMBEDDING tiles — their pixels stay put,
82
- # and a vector from an open-weight encoder is what crosses. So the line is at the
80
+ # and an embedding from the encoder is what crosses. So the line is at the
83
81
  # ENCODER, not at the size of an install: `embed` below does resolve a tensor
84
82
  # runtime, on purpose, and everything on the CLIENT side of the encoder — the
85
83
  # HTTP client, the slide reader — sits in the core above, where nobody has to
@@ -88,7 +86,7 @@ dependencies = [
88
86
  # encoder's four stay named, no shipped module imports either, and the release
89
87
  # guard refuses a wheel carrying a module that is not on its published list.
90
88
  #
91
- # `client` and `slide` were extras until #691 and are GONE rather than kept as
89
+ # `client` and `slide` were extras once and are GONE rather than kept as
92
90
  # empty aliases. The published wheel declares both, so `pip install
93
91
  # "auroraomics[client]"` exists in the wild and in our own older prose; pip
94
92
  # WARNS on an extra a distribution does not provide and installs anyway —
@@ -121,7 +119,7 @@ mcp = ["mcp>=2,<3"]
121
119
  #
122
120
  # anndata used to be listed here as a test-only dependency, with a note saying
123
121
  # that depending on it at run time would put pandas and its stack into every
124
- # install. That is now exactly what happens, knowingly (#691) — the note above
122
+ # install. That is now exactly what happens, knowingly — the note above
125
123
  # the core list records the price — so this extra no longer names anndata, nor
126
124
  # `tifffile` or `imagecodecs`, because the core already does. A second bound
127
125
  # here would be a test environment resolving what no user resolves. anndata is
@@ -138,10 +136,18 @@ dev = [
138
136
  # install. The client needs no reference any more — it is the core.
139
137
  "auroraomics[mcp]",
140
138
  "pytest>=7",
139
+ # The release guard parses version strings with `packaging.version`, at module
140
+ # scope. It has always resolved, because more than one dependency in this list
141
+ # requires packaging — but arriving in a closure is not the same as being
142
+ # declared: the day whichever one carries it stops doing so, the check that
143
+ # decides whether a build may be uploaded stops being a check and becomes a
144
+ # collection error, which reports as one missing optional dependency rather
145
+ # than as a suite that no longer runs.
146
+ "packaging>=22",
141
147
  # Coverage is measured on every run, with a floor, because this package is
142
148
  # the one that SHIPS: a user installs it and calls it, so a branch nothing
143
149
  # here exercises is a branch discovered in the field. The floor sat at
144
- # nothing until 2026-09-06, when the suite measured 94%.
150
+ # nothing until the suite first measured 94%.
145
151
  "pytest-cov>=5",
146
152
  # setuptools and wheel are TEST dependencies as well as build ones: the
147
153
  # release guard builds with --no-isolation so that it needs no network, and
@@ -223,8 +229,8 @@ source = ["src/auroraomics"]
223
229
  # when a real gap closes; never lower it to make a change pass.
224
230
  #
225
231
  # 92 rather than the 94.18 measured hours earlier, and the difference is not a
226
- # regression in anything tested here. #519 turned `predict_local` into a refusal
227
- # and moved the forward pass to the unpublished distribution — but left
232
+ # regression in anything tested here. Turning `predict_local` into a refusal
233
+ # moved the forward pass to the unpublished distribution — but left
228
234
  # `runtimes/deepspotm.py`'s `_panel_rows` and `_selection` behind, and their
229
235
  # ONLY callers are now in the unpublished inference distribution's runner. So
230
236
  # that file went from 271 statements at 99% to 169 at 73%: 46 statements this
@@ -233,8 +239,8 @@ source = ["src/auroraomics"]
233
239
  #
234
240
  # Moving those two helpers to the package that uses them takes the floor back
235
241
  # above 94 without writing a single test. Recorded as an issue rather than done
236
- # here, because it is a change to the boundary #519 drew and belongs with that
237
- # work, not with a coverage floor.
242
+ # here, because it is a change to the boundary that refusal drew and belongs
243
+ # with that work, not with a coverage floor.
238
244
  fail_under = 92
239
245
  # Two decimals, because `fail_under` is compared at this precision and rounding
240
246
  # 93.6 to 94 would pass a suite that had slipped.