auroraomics 0.1.0.dev2__tar.gz → 0.1.0.dev3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. {auroraomics-0.1.0.dev2/src/auroraomics.egg-info → auroraomics-0.1.0.dev3}/PKG-INFO +75 -74
  2. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/README.md +72 -72
  3. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/pyproject.toml +16 -10
  4. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/__init__.py +15 -10
  5. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/MANIFEST.json +1 -1
  6. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/public-api.tokens.json +13 -1
  7. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_text.py +13 -0
  8. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/bulk.py +26 -32
  9. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/cli.py +166 -98
  10. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/__init__.py +8 -4
  11. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/_generated.py +28 -31
  12. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/api.py +77 -57
  13. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/errors.py +27 -15
  14. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/jobs.py +13 -15
  15. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/uploads.py +1 -2
  16. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/contracts.py +10 -22
  17. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/embed.py +103 -194
  18. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/errors.py +6 -4
  19. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/genes.py +10 -21
  20. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/h5ad.py +8 -17
  21. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/mcp/server.py +5 -5
  22. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/mcp/tools.py +2 -3
  23. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/pack.py +50 -83
  24. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/postprocess.py +4 -7
  25. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/qc.py +16 -16
  26. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/runtimes/__init__.py +1 -1
  27. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/runtimes/deepspotm.py +53 -71
  28. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/slide.py +46 -55
  29. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/spatial.py +2 -2
  30. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/subsample.py +1 -2
  31. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3/src/auroraomics.egg-info}/PKG-INFO +75 -74
  32. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/requires.txt +1 -0
  33. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/LICENSE +0 -0
  34. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/MANIFEST.in +0 -0
  35. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/setup.cfg +0 -0
  36. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_assets/ASSETS.json +0 -0
  37. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_assets/README.md +0 -0
  38. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_assets/calibration-tile.png +0 -0
  39. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/public-api-counters.tokens.json +0 -0
  40. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/public-api-input-kinds.tokens.json +0 -0
  41. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/_http.py +0 -0
  42. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/credentials.py +0 -0
  43. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/mcp/__init__.py +0 -0
  44. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/py.typed +0 -0
  45. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/SOURCES.txt +0 -0
  46. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/dependency_links.txt +0 -0
  47. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/entry_points.txt +0 -0
  48. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/top_level.txt +0 -0
@@ -1,7 +1,7 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: auroraomics
3
- Version: 0.1.0.dev2
4
- Summary: Virtual spatial transcriptomics from H&E histology: patch QC, tile packing and the .h5ad result contract.
3
+ Version: 0.1.0.dev3
4
+ Summary: Prepare H&E slides on your own machine, submit them to Aurora, and read the spatial gene expression it predicts.
5
5
  Author: Kalin Nonchev
6
6
  License-Expression: PolyForm-Noncommercial-1.0.0
7
7
  Keywords: histopathology,spatial-transcriptomics,h5ad,anndata,whole-slide-image
@@ -33,6 +33,7 @@ Requires-Dist: mcp<3,>=2; extra == "mcp"
33
33
  Provides-Extra: dev
34
34
  Requires-Dist: auroraomics[mcp]; extra == "dev"
35
35
  Requires-Dist: pytest>=7; extra == "dev"
36
+ Requires-Dist: packaging>=22; extra == "dev"
36
37
  Requires-Dist: pytest-cov>=5; extra == "dev"
37
38
  Requires-Dist: setuptools>=77; extra == "dev"
38
39
  Requires-Dist: wheel; extra == "dev"
@@ -43,10 +44,14 @@ Dynamic: license-file
43
44
 
44
45
  # auroraomics
45
46
 
46
- Virtual spatial transcriptomics from H&E histology.
47
+ Virtual spatial transcriptomics from H&E images.
48
+
49
+ This package is the client of the **Aurora API**, the hosted service that runs
50
+ the prediction: from Python, from a shell with the `auroraomics` command, or
51
+ from an agent through its MCP server.
47
52
 
48
53
  This package holds the pieces of that pipeline that are pure Python, so the
49
- same code runs on your laptop, in a GPU container and on the service:
54
+ same code runs on your machine and on the service:
50
55
 
51
56
  - **`auroraomics.qc`** — the three patch-quality predicates (foreground, blur,
52
57
  stained tissue) applied to every candidate tile before a model sees it, and
@@ -58,13 +63,13 @@ same code runs on your laptop, in a GPU container and on the service:
58
63
  batch rather than the whole matrix.
59
64
  - **`auroraomics.postprocess`** — apply a model card's `postprocess` entries
60
65
  to a result file: a registry of entries, one runner, `h5py` alone, streaming
61
- one chunk at a time. Ships the unmeasured-gene mask, which states per gene
62
- how many training datasets measured it and blanks the ones none did.
66
+ one chunk at a time. Ships the unmeasured-gene mask, which blanks the genes
67
+ the model never measured.
63
68
  - **`auroraomics.subsample`** — pick the densest contiguous square of spots
64
- when a slide yields more tiles than a run is allowed to spend.
65
- - **`auroraomics.genes`** — the model family's gene symbols mapped to Ensembl
66
- stable gene ids, which is what every result's `var` index is keyed by. Ships
67
- as one committed table, resolved in the same order the service resolves it.
69
+ when a slide yields more tiles than one submission takes.
70
+ - **`auroraomics.genes`** — Ensembl stable gene ids, which is what every
71
+ result's `var` index is keyed by, and the shape a gene the service resolved
72
+ comes back in.
68
73
  - **`auroraomics.contracts`** — the shared contract values (container layout,
69
74
  input-kind caps, result layout) as data, so nothing here re-types a number
70
75
  the service also reads.
@@ -88,16 +93,18 @@ pip install auroraomics
88
93
 
89
94
  That is the whole documented path: open a slide, judge and pack its tiles,
90
95
  submit, and read the result. One extra adds local embedding extraction
91
- (`embed`), because a tensor runtime is gigabytes and specific to the machine it
92
- was built for. No extra brings the model: predicting gene expression runs on
93
- the service, not here.
96
+ (`embed`). No extra brings the model: predicting gene expression runs on the
97
+ service, not here.
94
98
 
95
99
  ## Pack tiles, then look at the report
96
100
 
97
101
  ```python
98
102
  import auroraomics as ao
99
103
 
100
- report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thumbnail=thumb)
104
+ with ao.Client() as client:
105
+ thresholds = client.qc_thresholds() # the service's quality floors, no key needed
106
+
107
+ report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thresholds=thresholds, thumbnail=thumb)
101
108
  print(report.written, "tiles kept,", report.rejected, "dropped")
102
109
  print(report.rejected_by_reason) # {'foreground_ratio': 12, ...}
103
110
  ```
@@ -126,7 +133,8 @@ members do not match the container contract.
126
133
  ao.write_result(
127
134
  "result.h5ad",
128
135
  obs=obs, # per-spot columns, as plain arrays
129
- var=var, # per-gene columns, indexed by gene id
136
+ var=var, # per-gene columns
137
+ var_index=gene_ids, # the Ensembl gene id of each column
130
138
  spatial=coords, # (n_spots, 2) array -> obsm["spatial"]
131
139
  x=batches, # an array, or an iterable of row batches
132
140
  uns={"model": {"id": "..."}},
@@ -149,71 +157,69 @@ pip install "auroraomics[embed]"
149
157
  import auroraomics as ao
150
158
  from auroraomics.embed import available_encoders, describe_encoders
151
159
 
152
- print(available_encoders()) # ('deepspot-h', 'dinov2-b14', 'midnight')
153
-
154
160
  with ao.Client() as client:
155
161
  # Both are the service's, so the file records the encoder and the
156
- # thresholds the model it is submitted to was served with.
157
- encoder = ao.resolve_encoder(client.encoders())
162
+ # thresholds the model it is submitted to was served with. There is no
163
+ # list of encoders inside this package: every name comes from the registry
164
+ # the service publishes, which is why these calls take one.
165
+ registry = client.encoders()
158
166
  thresholds = client.qc_thresholds()
167
+ # The tile size the model's card states, for the model the file is for.
168
+ patch_um = client.model_card(model_id)["input_spec"]["patch_um"]
169
+
170
+ print(available_encoders(registry)) # the names this package will run today
171
+ encoder = ao.resolve_encoder(registry)
159
172
 
160
173
  report = ao.embed_tiles(
161
- slide, "sample.npz", mpp=0.499, patch_um=55, encoder=encoder, thresholds=thresholds
174
+ slide, "sample.npz", mpp=0.499, patch_um=patch_um, encoder=encoder, thresholds=thresholds
162
175
  )
163
176
  print(report.rows, "rows of", report.dim, "numbers from", report.encoder.name)
164
177
  ```
165
178
 
166
179
  `slide` is anything with a `(height, width, 3)` RGB `uint8` shape that can be
167
180
  sliced — an array, or a memory-mapped or lazily-read one, so a slide larger than
168
- memory works: only one crop exists at a time. Crops are taken at `patch_um`
169
- micrometres using the `mpp` you give, quality-checked with the same three
170
- predicates as `pack_tiles`, and pooled over the encoder's patch tokens.
171
-
172
- Every encoder is pinned to a repository **and** a revision sha, with its licence
173
- and access gate recorded beside it. `describe_encoders()` returns the whole
174
- registry, including the encoders this package refuses to run — a refusal names
175
- the licence or the approval you would have to accept, and which encoders remain.
176
-
177
- The written file carries one extra vector, `calibration`: the embedding of a
178
- fixed image shipped inside this package, through the same encoder, revision and
179
- preprocessing as your rows. Comparing that one vector against a known reference
180
- tells a reader whether the file was produced by the model it claims, before they
181
- look at a single row.
182
-
183
- The weights are downloaded from their publisher on first use, into the ordinary
184
- model cache. Pass `allow_download=False` to guarantee no request is made: for
185
- the duration of that load, name resolution and internet sockets are refused in
186
- this process, so the guarantee does not rest on a model library reading an
187
- environment variable it may already have read.
188
-
189
- An encoder runs at one input size: the size the served model's embeddings were
190
- computed at, which the registry pins and which is not always what the encoder's
191
- own repository declares. The DINOv2 releases carry a 518-px position grid; the
192
- model was trained on 224-px crops fed to that grid through an interpolated
193
- position embedding, so that is what this package does too. `resize_px` is
194
- checked against the pin rather than obeyed, because a vector computed at another
195
- size is a different measurement that nobody else's numbers can be compared with.
181
+ memory works: only one crop exists at a time. Crops are taken at the tile size
182
+ the model's card states, using the `mpp` you give, and quality-checked with the
183
+ same three predicates as `pack_tiles`.
184
+
185
+ Every encoder is pinned to an exact revision.
186
+ `describe_encoders(registry)` returns the whole registry, including the
187
+ encoders this package refuses to run: a refusal says why, and which encoders
188
+ remain.
189
+
190
+ The written file carries one extra embedding, `calibration`: computed from a
191
+ fixed image shipped inside this package, through the same encoder and revision
192
+ as your rows. Comparing that one embedding against a known
193
+ reference tells a reader whether the file was produced by the model it claims,
194
+ before they look at a single row.
195
+
196
+ The weights are downloaded on first use. Pass `allow_download=False` to
197
+ guarantee no request is made: for the duration of that load, name resolution
198
+ and internet sockets are refused in this process.
199
+
200
+ An encoder runs at the input size the registry pins for it. `resize_px` is
201
+ checked against that pin rather than obeyed, because an embedding computed at
202
+ another size cannot be compared with the served model's.
203
+
196
204
  ## Where a prediction runs
197
205
 
198
- Predicting gene expression runs on the service, and `predict_local` refuses —
199
- by design, and the design is the product rather than a limitation.
206
+ Predicting gene expression runs on the service, and `predict_local` refuses.
200
207
 
201
- This is split inference at the encoder. An open-weight pathology foundation
202
- model turns your tiles into patch embeddings **on your machine**, so your slides
203
- never leave it; our gene decoder turns embeddings into expression **on ours**, so
204
- its weights never leave it. What crosses between us is a vector from a model
205
- neither side owns.
208
+ **DeepSpot-H**, the foundation model for H&E images, turns each tile of your
209
+ slide into one embedding. It can run on your machine or on ours. **DeepSpot-M**
210
+ turns those embeddings into spatial gene expression. It only ever runs on ours.
206
211
 
207
212
  ```python
208
213
  from auroraomics.runtimes.deepspotm import check_archive, describe_models
209
214
 
210
- cards = client.models() # the catalogue is the service's
211
- describe_models(cards) # what it will run, and on what
212
- check_archive("sample.zip", model=MODEL, cards=cards) # are my tiles the right size?
215
+ with ao.Client() as client:
216
+ cards = client.models() # the catalogue is the service's
217
+ describe_models(cards) # what it will run, and on what
218
+ check_archive("sample.zip", model=model_id, cards=cards) # are my tiles the right size?
213
219
  ```
214
220
 
215
221
  Then submit with the API client and read the result back — the same `.h5ad`
216
- contract whichever lane ran it.
222
+ contract whichever runtime ran it.
217
223
 
218
224
  ## Post-process a result
219
225
 
@@ -237,17 +243,19 @@ skipped: a card is the description of a file, and a reader promised a layer
237
243
  cannot tell a missing one from a model that predicted zeros.
238
244
 
239
245
  The entry this version ships states what the model measured. `var` gains
240
- `n_train_datasets` — the number of training datasets each gene was measured in,
241
- because a gene seen in one and a gene seen in twenty are not the same claim —
242
- and `measured_in_training`, derived from that count. A gene the model never
243
- measured becomes `nan` in `X` and in every layer, not zero: zero is a
244
- measurement, and an unmeasured gene left as one averages, correlates and
246
+ `measured_in_training`, which is `False` for a gene the model never saw
247
+ measured. Such a gene becomes `nan` in `X` and in every layer, not zero: zero
248
+ is a measurement, and an unmeasured gene left as one averages, correlates and
245
249
  colours a heat map exactly like a prediction.
246
250
 
247
251
  ## Prepare a bulk RNA profile
248
252
 
249
253
  ```python
250
- report = ao.bulk_rna("sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts")
254
+ with ao.Client() as client:
255
+ report = ao.bulk_rna(
256
+ "sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts",
257
+ card=client.model_card(model_id), resolve=client.resolve_keys,
258
+ )
251
259
  print(report.rows, "genes;", f"{report.coverage:.0%} of", report.coverage_of)
252
260
  ```
253
261
 
@@ -267,13 +275,6 @@ your project.
267
275
 
268
276
  ## Licence
269
277
 
270
- The code in this package is licensed under
271
- [PolyForm Noncommercial 1.0.0](https://polyformproject.org/licenses/noncommercial/1.0.0),
272
- which permits use for any purpose that is not commercial. It is the same licence the
273
- model package this client is built for carries, so installing both puts you under one
274
- rule rather than two.
275
-
276
- The model weights are licensed separately by whoever publishes them, and access to them
277
- may be gated. Read those terms before you use a model: they are not this licence, and a
278
- permission granted here is not a permission granted there.
278
+ The code in this package is licensed for non-commercial use; its terms are in the
279
+ `LICENSE` file it ships with. Commercial evaluation and use are governed by a written agreement with Aurora.
279
280
 
@@ -1,9 +1,13 @@
1
1
  # auroraomics
2
2
 
3
- Virtual spatial transcriptomics from H&E histology.
3
+ Virtual spatial transcriptomics from H&E images.
4
+
5
+ This package is the client of the **Aurora API**, the hosted service that runs
6
+ the prediction: from Python, from a shell with the `auroraomics` command, or
7
+ from an agent through its MCP server.
4
8
 
5
9
  This package holds the pieces of that pipeline that are pure Python, so the
6
- same code runs on your laptop, in a GPU container and on the service:
10
+ same code runs on your machine and on the service:
7
11
 
8
12
  - **`auroraomics.qc`** — the three patch-quality predicates (foreground, blur,
9
13
  stained tissue) applied to every candidate tile before a model sees it, and
@@ -15,13 +19,13 @@ same code runs on your laptop, in a GPU container and on the service:
15
19
  batch rather than the whole matrix.
16
20
  - **`auroraomics.postprocess`** — apply a model card's `postprocess` entries
17
21
  to a result file: a registry of entries, one runner, `h5py` alone, streaming
18
- one chunk at a time. Ships the unmeasured-gene mask, which states per gene
19
- how many training datasets measured it and blanks the ones none did.
22
+ one chunk at a time. Ships the unmeasured-gene mask, which blanks the genes
23
+ the model never measured.
20
24
  - **`auroraomics.subsample`** — pick the densest contiguous square of spots
21
- when a slide yields more tiles than a run is allowed to spend.
22
- - **`auroraomics.genes`** — the model family's gene symbols mapped to Ensembl
23
- stable gene ids, which is what every result's `var` index is keyed by. Ships
24
- as one committed table, resolved in the same order the service resolves it.
25
+ when a slide yields more tiles than one submission takes.
26
+ - **`auroraomics.genes`** — Ensembl stable gene ids, which is what every
27
+ result's `var` index is keyed by, and the shape a gene the service resolved
28
+ comes back in.
25
29
  - **`auroraomics.contracts`** — the shared contract values (container layout,
26
30
  input-kind caps, result layout) as data, so nothing here re-types a number
27
31
  the service also reads.
@@ -45,16 +49,18 @@ pip install auroraomics
45
49
 
46
50
  That is the whole documented path: open a slide, judge and pack its tiles,
47
51
  submit, and read the result. One extra adds local embedding extraction
48
- (`embed`), because a tensor runtime is gigabytes and specific to the machine it
49
- was built for. No extra brings the model: predicting gene expression runs on
50
- the service, not here.
52
+ (`embed`). No extra brings the model: predicting gene expression runs on the
53
+ service, not here.
51
54
 
52
55
  ## Pack tiles, then look at the report
53
56
 
54
57
  ```python
55
58
  import auroraomics as ao
56
59
 
57
- report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thumbnail=thumb)
60
+ with ao.Client() as client:
61
+ thresholds = client.qc_thresholds() # the service's quality floors, no key needed
62
+
63
+ report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thresholds=thresholds, thumbnail=thumb)
58
64
  print(report.written, "tiles kept,", report.rejected, "dropped")
59
65
  print(report.rejected_by_reason) # {'foreground_ratio': 12, ...}
60
66
  ```
@@ -83,7 +89,8 @@ members do not match the container contract.
83
89
  ao.write_result(
84
90
  "result.h5ad",
85
91
  obs=obs, # per-spot columns, as plain arrays
86
- var=var, # per-gene columns, indexed by gene id
92
+ var=var, # per-gene columns
93
+ var_index=gene_ids, # the Ensembl gene id of each column
87
94
  spatial=coords, # (n_spots, 2) array -> obsm["spatial"]
88
95
  x=batches, # an array, or an iterable of row batches
89
96
  uns={"model": {"id": "..."}},
@@ -106,71 +113,69 @@ pip install "auroraomics[embed]"
106
113
  import auroraomics as ao
107
114
  from auroraomics.embed import available_encoders, describe_encoders
108
115
 
109
- print(available_encoders()) # ('deepspot-h', 'dinov2-b14', 'midnight')
110
-
111
116
  with ao.Client() as client:
112
117
  # Both are the service's, so the file records the encoder and the
113
- # thresholds the model it is submitted to was served with.
114
- encoder = ao.resolve_encoder(client.encoders())
118
+ # thresholds the model it is submitted to was served with. There is no
119
+ # list of encoders inside this package: every name comes from the registry
120
+ # the service publishes, which is why these calls take one.
121
+ registry = client.encoders()
115
122
  thresholds = client.qc_thresholds()
123
+ # The tile size the model's card states, for the model the file is for.
124
+ patch_um = client.model_card(model_id)["input_spec"]["patch_um"]
125
+
126
+ print(available_encoders(registry)) # the names this package will run today
127
+ encoder = ao.resolve_encoder(registry)
116
128
 
117
129
  report = ao.embed_tiles(
118
- slide, "sample.npz", mpp=0.499, patch_um=55, encoder=encoder, thresholds=thresholds
130
+ slide, "sample.npz", mpp=0.499, patch_um=patch_um, encoder=encoder, thresholds=thresholds
119
131
  )
120
132
  print(report.rows, "rows of", report.dim, "numbers from", report.encoder.name)
121
133
  ```
122
134
 
123
135
  `slide` is anything with a `(height, width, 3)` RGB `uint8` shape that can be
124
136
  sliced — an array, or a memory-mapped or lazily-read one, so a slide larger than
125
- memory works: only one crop exists at a time. Crops are taken at `patch_um`
126
- micrometres using the `mpp` you give, quality-checked with the same three
127
- predicates as `pack_tiles`, and pooled over the encoder's patch tokens.
128
-
129
- Every encoder is pinned to a repository **and** a revision sha, with its licence
130
- and access gate recorded beside it. `describe_encoders()` returns the whole
131
- registry, including the encoders this package refuses to run — a refusal names
132
- the licence or the approval you would have to accept, and which encoders remain.
133
-
134
- The written file carries one extra vector, `calibration`: the embedding of a
135
- fixed image shipped inside this package, through the same encoder, revision and
136
- preprocessing as your rows. Comparing that one vector against a known reference
137
- tells a reader whether the file was produced by the model it claims, before they
138
- look at a single row.
139
-
140
- The weights are downloaded from their publisher on first use, into the ordinary
141
- model cache. Pass `allow_download=False` to guarantee no request is made: for
142
- the duration of that load, name resolution and internet sockets are refused in
143
- this process, so the guarantee does not rest on a model library reading an
144
- environment variable it may already have read.
145
-
146
- An encoder runs at one input size: the size the served model's embeddings were
147
- computed at, which the registry pins and which is not always what the encoder's
148
- own repository declares. The DINOv2 releases carry a 518-px position grid; the
149
- model was trained on 224-px crops fed to that grid through an interpolated
150
- position embedding, so that is what this package does too. `resize_px` is
151
- checked against the pin rather than obeyed, because a vector computed at another
152
- size is a different measurement that nobody else's numbers can be compared with.
137
+ memory works: only one crop exists at a time. Crops are taken at the tile size
138
+ the model's card states, using the `mpp` you give, and quality-checked with the
139
+ same three predicates as `pack_tiles`.
140
+
141
+ Every encoder is pinned to an exact revision.
142
+ `describe_encoders(registry)` returns the whole registry, including the
143
+ encoders this package refuses to run: a refusal says why, and which encoders
144
+ remain.
145
+
146
+ The written file carries one extra embedding, `calibration`: computed from a
147
+ fixed image shipped inside this package, through the same encoder and revision
148
+ as your rows. Comparing that one embedding against a known
149
+ reference tells a reader whether the file was produced by the model it claims,
150
+ before they look at a single row.
151
+
152
+ The weights are downloaded on first use. Pass `allow_download=False` to
153
+ guarantee no request is made: for the duration of that load, name resolution
154
+ and internet sockets are refused in this process.
155
+
156
+ An encoder runs at the input size the registry pins for it. `resize_px` is
157
+ checked against that pin rather than obeyed, because an embedding computed at
158
+ another size cannot be compared with the served model's.
159
+
153
160
  ## Where a prediction runs
154
161
 
155
- Predicting gene expression runs on the service, and `predict_local` refuses —
156
- by design, and the design is the product rather than a limitation.
162
+ Predicting gene expression runs on the service, and `predict_local` refuses.
157
163
 
158
- This is split inference at the encoder. An open-weight pathology foundation
159
- model turns your tiles into patch embeddings **on your machine**, so your slides
160
- never leave it; our gene decoder turns embeddings into expression **on ours**, so
161
- its weights never leave it. What crosses between us is a vector from a model
162
- neither side owns.
164
+ **DeepSpot-H**, the foundation model for H&E images, turns each tile of your
165
+ slide into one embedding. It can run on your machine or on ours. **DeepSpot-M**
166
+ turns those embeddings into spatial gene expression. It only ever runs on ours.
163
167
 
164
168
  ```python
165
169
  from auroraomics.runtimes.deepspotm import check_archive, describe_models
166
170
 
167
- cards = client.models() # the catalogue is the service's
168
- describe_models(cards) # what it will run, and on what
169
- check_archive("sample.zip", model=MODEL, cards=cards) # are my tiles the right size?
171
+ with ao.Client() as client:
172
+ cards = client.models() # the catalogue is the service's
173
+ describe_models(cards) # what it will run, and on what
174
+ check_archive("sample.zip", model=model_id, cards=cards) # are my tiles the right size?
170
175
  ```
171
176
 
172
177
  Then submit with the API client and read the result back — the same `.h5ad`
173
- contract whichever lane ran it.
178
+ contract whichever runtime ran it.
174
179
 
175
180
  ## Post-process a result
176
181
 
@@ -194,17 +199,19 @@ skipped: a card is the description of a file, and a reader promised a layer
194
199
  cannot tell a missing one from a model that predicted zeros.
195
200
 
196
201
  The entry this version ships states what the model measured. `var` gains
197
- `n_train_datasets` — the number of training datasets each gene was measured in,
198
- because a gene seen in one and a gene seen in twenty are not the same claim —
199
- and `measured_in_training`, derived from that count. A gene the model never
200
- measured becomes `nan` in `X` and in every layer, not zero: zero is a
201
- measurement, and an unmeasured gene left as one averages, correlates and
202
+ `measured_in_training`, which is `False` for a gene the model never saw
203
+ measured. Such a gene becomes `nan` in `X` and in every layer, not zero: zero
204
+ is a measurement, and an unmeasured gene left as one averages, correlates and
202
205
  colours a heat map exactly like a prediction.
203
206
 
204
207
  ## Prepare a bulk RNA profile
205
208
 
206
209
  ```python
207
- report = ao.bulk_rna("sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts")
210
+ with ao.Client() as client:
211
+ report = ao.bulk_rna(
212
+ "sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts",
213
+ card=client.model_card(model_id), resolve=client.resolve_keys,
214
+ )
208
215
  print(report.rows, "genes;", f"{report.coverage:.0%} of", report.coverage_of)
209
216
  ```
210
217
 
@@ -224,13 +231,6 @@ your project.
224
231
 
225
232
  ## Licence
226
233
 
227
- The code in this package is licensed under
228
- [PolyForm Noncommercial 1.0.0](https://polyformproject.org/licenses/noncommercial/1.0.0),
229
- which permits use for any purpose that is not commercial. It is the same licence the
230
- model package this client is built for carries, so installing both puts you under one
231
- rule rather than two.
232
-
233
- The model weights are licensed separately by whoever publishes them, and access to them
234
- may be gated. Read those terms before you use a model: they are not this licence, and a
235
- permission granted here is not a permission granted there.
234
+ The code in this package is licensed for non-commercial use; its terms are in the
235
+ `LICENSE` file it ships with. Commercial evaluation and use are governed by a written agreement with Aurora.
236
236
 
@@ -4,14 +4,12 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "auroraomics"
7
- version = "0.1.0.dev2"
8
- description = "Virtual spatial transcriptomics from H&E histology: patch QC, tile packing and the .h5ad result contract."
7
+ version = "0.1.0.dev3"
8
+ description = "Prepare H&E slides on your own machine, submit them to Aurora, and read the spatial gene expression it predicts."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
11
- # Non-commercial, and the same licence as the model package this client is built
12
- # for, so a user installing both faces one rule rather than two. PEP 639 forbids
13
- # a licence CLASSIFIER beside an expression, so this line is the whole
14
- # declaration; the model weights carry their own separate terms.
11
+ # Non-commercial; the terms are in LICENSE. PEP 639 forbids a licence
12
+ # CLASSIFIER beside an expression, so this line is the whole declaration.
15
13
  license = "PolyForm-Noncommercial-1.0.0"
16
14
  license-files = ["LICENSE"]
17
15
  authors = [{ name = "Kalin Nonchev" }]
@@ -47,9 +45,9 @@ classifiers = [
47
45
  # all of it. That is the trade, taken deliberately: one install line that works,
48
46
  # over a smaller one that cannot reach the end of the page describing it.
49
47
  #
50
- # What is still NOT here is the name that matters to the GPU pipeline images —
51
- # no torch. They install this package with `--no-deps`, so what they resolve
52
- # does not change either way; what protects them is the encoder line below.
48
+ # What is still NOT here is torch. An environment that installs this package
49
+ # with `--no-deps` resolves the same either way; what protects it is the
50
+ # encoder line below.
53
51
  dependencies = [
54
52
  "numpy>=1.23",
55
53
  "h5py>=3.9",
@@ -79,7 +77,7 @@ dependencies = [
79
77
  # resolves the model, or the adapter and checkpoint stack that would carry one,
80
78
  # and `predict_local` refuses with a message that says so. What runs on a user's
81
79
  # own machine is cutting, filtering and EMBEDDING tiles — their pixels stay put,
82
- # and a vector from the encoder is what crosses. So the line is at the
80
+ # and an embedding from the encoder is what crosses. So the line is at the
83
81
  # ENCODER, not at the size of an install: `embed` below does resolve a tensor
84
82
  # runtime, on purpose, and everything on the CLIENT side of the encoder — the
85
83
  # HTTP client, the slide reader — sits in the core above, where nobody has to
@@ -138,6 +136,14 @@ dev = [
138
136
  # install. The client needs no reference any more — it is the core.
139
137
  "auroraomics[mcp]",
140
138
  "pytest>=7",
139
+ # The release guard parses version strings with `packaging.version`, at module
140
+ # scope. It has always resolved, because more than one dependency in this list
141
+ # requires packaging — but arriving in a closure is not the same as being
142
+ # declared: the day whichever one carries it stops doing so, the check that
143
+ # decides whether a build may be uploaded stops being a check and becomes a
144
+ # collection error, which reports as one missing optional dependency rather
145
+ # than as a suite that no longer runs.
146
+ "packaging>=22",
141
147
  # Coverage is measured on every run, with a floor, because this package is
142
148
  # the one that SHIPS: a user installs it and calls it, so a branch nothing
143
149
  # here exercises is a branch discovered in the field. The floor sat at
@@ -1,17 +1,22 @@
1
- """Virtual spatial transcriptomics from H&E histology.
1
+ """Virtual spatial transcriptomics from H&E images.
2
2
 
3
3
  **One journey, and one word chooses which.** Send the tiles, or send only the
4
4
  embeddings computed from them; everything either side of that word is the same:
5
5
 
6
- import auroraomics as ao
6
+ ```python
7
+ import auroraomics as ao
7
8
 
8
- slide = ao.open_slide("slide.svs")
9
- report = ao.pack_tiles(slide, "sample.zip", mpp=slide.mpp, patch_um=55)
10
- job = ao.Client().predict(model=model_id, observations={"patches": report.path})
11
- job.wait().download("result.h5ad")
9
+ client = ao.Client()
10
+ slide = ao.open_slide("slide.svs")
11
+ report = ao.pack_tiles(slide, "sample.zip", mpp=slide.mpp, patch_um=55,
12
+ thresholds=client.qc_thresholds())
13
+ job = client.predict(model=model_id, observations={"patches": report.path})
14
+ job.wait().download("result.h5ad")
15
+ ```
12
16
 
13
- Write ``ao.embed_tiles`` in place of ``ao.pack_tiles`` and the slide's pixels
14
- never leave the machine: what crosses is a few hundred numbers per tile.
17
+ Write ``ao.embed_tiles`` in place of ``ao.pack_tiles``, with the encoder from
18
+ ``ao.resolve_encoder(client.encoders(), "deepspot-h")``, and the slide's pixels
19
+ never leave the machine: what crosses is a short row of numbers per tile.
15
20
 
16
21
  The names in that example, and the handful beside them a reader writes by hand,
17
22
  are what this module exports. Everything else in the package is still here and
@@ -19,7 +24,7 @@ still importable from the module that owns it — the front door is narrow on
19
24
  purpose, because a name a caller never types is a name they have to read past.
20
25
 
21
26
  The pieces of that pipeline that are pure Python live here, so the same code
22
- runs on a laptop, inside a GPU image and on the hosted service:
27
+ runs on your machine and on the hosted service:
23
28
 
24
29
  * [auroraomics.qc][] — the three patch-quality predicates and the cascade
25
30
  that decides which tiles are worth predicting on.
@@ -33,7 +38,7 @@ runs on a laptop, inside a GPU image and on the hosted service:
33
38
  * [auroraomics.bulk][] — prepare a bulk RNA profile as the ``bulk_rna``
34
39
  covariate a prediction accepts: one table, Ensembl ids, a declared unit.
35
40
  * [auroraomics.subsample][] — choose the densest contiguous square when a
36
- slide yields more tiles than a run may spend.
41
+ slide yields more tiles than one submission takes.
37
42
  * [auroraomics.genes][] — the model family's symbols mapped to Ensembl gene
38
43
  ids, which is what every result's ``var`` index is keyed by.
39
44
  * [auroraomics.errors][] — the one root every failure in this package
@@ -3,6 +3,6 @@
3
3
  "files": {
4
4
  "public-api-counters.tokens.json": "4f5dd14ce76f7001d36940dfc1d1f49feac9ec1be2e7646d47b29ab21c124a8f",
5
5
  "public-api-input-kinds.tokens.json": "1c80e2637d27f2491ea178e2fa87e2a435640e5445791b2fd20adc53a574e54d",
6
- "public-api.tokens.json": "a8d3253978aba375c1933aef3f3069bcb4b5e47c2edbf6062a08c95c0f255f4d"
6
+ "public-api.tokens.json": "75ac637df4a6b0258485e51dccbc05445b73bb42449e11b4be6e98ffb67c0fbb"
7
7
  }
8
8
  }
@@ -203,7 +203,9 @@
203
203
  "revoked"
204
204
  ],
205
205
  "requestPath": "/api-access",
206
- "termsVersion": "2026-09-05",
206
+ "termsVersion": "2026-09-27",
207
+ "termsPath": "/terms",
208
+ "termsDataUseSectionId": "model-development",
207
209
  "request": {
208
210
  "nameMaxChars": 120,
209
211
  "institutionMaxChars": 200,
@@ -265,6 +267,16 @@
265
267
  "commit",
266
268
  "qc"
267
269
  ],
270
+ "required": [
271
+ "emb",
272
+ "px_col",
273
+ "px_row",
274
+ "calibration",
275
+ "encoder",
276
+ "revision",
277
+ "patch_px",
278
+ "pooling"
279
+ ],
268
280
  "dtypes": [
269
281
  "float16",
270
282
  "float32"
@@ -55,6 +55,19 @@ into this wheel, so this is where the package can reach the name today. One
55
55
  site to move, not five.
56
56
  """
57
57
 
58
+ PREDICTION_COMPONENT = "DeepSpot-M"
59
+ """The step that turns embeddings into expression, named as a reader meets it.
60
+
61
+ The model's display name: the brand contract's `modelDisplayName`, which is
62
+ also this step's key in its `componentNames`, hyphenated like
63
+ `EMBEDDING_COMPONENT`. It is never an identifier: the package a runtime
64
+ installs and the repository it names keep their own spellings.
65
+
66
+ Typed here for the reason the constant above is, and held to the same fixture
67
+ by the same test, so the refusal that names both steps cannot spell one of
68
+ them the old way.
69
+ """
70
+
58
71
 
59
72
  def bounded(value: Any, *, limit: int = MAX_ECHOED_INPUT) -> str:
60
73
  """``value`` as text, cut to ``limit`` with a marker.