auroraomics 0.1.0.dev2__tar.gz → 0.1.0.dev4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. {auroraomics-0.1.0.dev2/src/auroraomics.egg-info → auroraomics-0.1.0.dev4}/PKG-INFO +104 -74
  2. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/README.md +96 -72
  3. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/pyproject.toml +44 -13
  4. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/__init__.py +16 -10
  5. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_contracts/MANIFEST.json +1 -1
  6. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_contracts/public-api.tokens.json +13 -1
  7. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_text.py +13 -0
  8. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/bulk.py +26 -32
  9. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/cli.py +238 -118
  10. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/__init__.py +8 -4
  11. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/_generated.py +36 -33
  12. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/api.py +214 -87
  13. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/errors.py +27 -15
  14. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/jobs.py +13 -15
  15. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/uploads.py +1 -2
  16. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/contracts.py +28 -22
  17. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/embed.py +115 -201
  18. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/errors.py +7 -4
  19. auroraomics-0.1.0.dev4/src/auroraomics/export.py +467 -0
  20. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/genes.py +10 -21
  21. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/h5ad.py +8 -17
  22. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/mcp/server.py +5 -5
  23. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/mcp/tools.py +12 -8
  24. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/pack.py +58 -84
  25. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/postprocess.py +4 -7
  26. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/qc.py +18 -18
  27. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/runtimes/__init__.py +1 -1
  28. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/runtimes/deepspotm.py +53 -71
  29. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/slide.py +47 -55
  30. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/spatial.py +2 -2
  31. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/subsample.py +1 -2
  32. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4/src/auroraomics.egg-info}/PKG-INFO +104 -74
  33. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics.egg-info/SOURCES.txt +1 -0
  34. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics.egg-info/requires.txt +9 -0
  35. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/LICENSE +0 -0
  36. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/MANIFEST.in +0 -0
  37. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/setup.cfg +0 -0
  38. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_assets/ASSETS.json +0 -0
  39. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_assets/README.md +0 -0
  40. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_assets/calibration-tile.png +0 -0
  41. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_contracts/public-api-counters.tokens.json +0 -0
  42. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_contracts/public-api-input-kinds.tokens.json +0 -0
  43. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/_http.py +0 -0
  44. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/credentials.py +0 -0
  45. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/mcp/__init__.py +0 -0
  46. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/py.typed +0 -0
  47. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics.egg-info/dependency_links.txt +0 -0
  48. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics.egg-info/entry_points.txt +0 -0
  49. {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics.egg-info/top_level.txt +0 -0
@@ -1,7 +1,7 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: auroraomics
3
- Version: 0.1.0.dev2
4
- Summary: Virtual spatial transcriptomics from H&E histology: patch QC, tile packing and the .h5ad result contract.
3
+ Version: 0.1.0.dev4
4
+ Summary: Prepare H&E slides on your own machine, submit them to Aurora, and read the spatial gene expression it predicts.
5
5
  Author: Kalin Nonchev
6
6
  License-Expression: PolyForm-Noncommercial-1.0.0
7
7
  Keywords: histopathology,spatial-transcriptomics,h5ad,anndata,whole-slide-image
@@ -23,16 +23,22 @@ Requires-Dist: pydantic>=2
23
23
  Requires-Dist: anndata>=0.10
24
24
  Requires-Dist: tifffile>=2023.7.10
25
25
  Requires-Dist: imagecodecs>=2023.3.16
26
+ Requires-Dist: scipy>1.8
26
27
  Provides-Extra: embed
27
28
  Requires-Dist: torch>=2.0; extra == "embed"
28
29
  Requires-Dist: timm>=1.0; extra == "embed"
29
30
  Requires-Dist: transformers>=4.40; extra == "embed"
30
31
  Requires-Dist: huggingface-hub>=0.23; extra == "embed"
32
+ Provides-Extra: spatialdata
33
+ Requires-Dist: spatialdata>=0.2; extra == "spatialdata"
34
+ Requires-Dist: setuptools<81; python_version < "3.12" and extra == "spatialdata"
31
35
  Provides-Extra: mcp
32
36
  Requires-Dist: mcp<3,>=2; extra == "mcp"
33
37
  Provides-Extra: dev
34
38
  Requires-Dist: auroraomics[mcp]; extra == "dev"
39
+ Requires-Dist: auroraomics[spatialdata]; extra == "dev"
35
40
  Requires-Dist: pytest>=7; extra == "dev"
41
+ Requires-Dist: packaging>=22; extra == "dev"
36
42
  Requires-Dist: pytest-cov>=5; extra == "dev"
37
43
  Requires-Dist: setuptools>=77; extra == "dev"
38
44
  Requires-Dist: wheel; extra == "dev"
@@ -43,10 +49,14 @@ Dynamic: license-file
43
49
 
44
50
  # auroraomics
45
51
 
46
- Virtual spatial transcriptomics from H&E histology.
52
+ Virtual spatial transcriptomics from H&E images.
53
+
54
+ This package is the client of the **Aurora API**, the hosted service that runs
55
+ the prediction: from Python, from a shell with the `auroraomics` command, or
56
+ from an agent through its MCP server.
47
57
 
48
58
  This package holds the pieces of that pipeline that are pure Python, so the
49
- same code runs on your laptop, in a GPU container and on the service:
59
+ same code runs on your machine and on the service:
50
60
 
51
61
  - **`auroraomics.qc`** — the three patch-quality predicates (foreground, blur,
52
62
  stained tissue) applied to every candidate tile before a model sees it, and
@@ -58,16 +68,19 @@ same code runs on your laptop, in a GPU container and on the service:
58
68
  batch rather than the whole matrix.
59
69
  - **`auroraomics.postprocess`** — apply a model card's `postprocess` entries
60
70
  to a result file: a registry of entries, one runner, `h5py` alone, streaming
61
- one chunk at a time. Ships the unmeasured-gene mask, which states per gene
62
- how many training datasets measured it and blanks the ones none did.
71
+ one chunk at a time. Ships the unmeasured-gene mask, which blanks the genes
72
+ the model never measured.
63
73
  - **`auroraomics.subsample`** — pick the densest contiguous square of spots
64
- when a slide yields more tiles than a run is allowed to spend.
65
- - **`auroraomics.genes`** — the model family's gene symbols mapped to Ensembl
66
- stable gene ids, which is what every result's `var` index is keyed by. Ships
67
- as one committed table, resolved in the same order the service resolves it.
74
+ when a slide yields more tiles than one submission takes.
75
+ - **`auroraomics.genes`** — Ensembl stable gene ids, which is what every
76
+ result's `var` index is keyed by, and the shape a gene the service resolved
77
+ comes back in.
68
78
  - **`auroraomics.contracts`** — the shared contract values (container layout,
69
79
  input-kind caps, result layout) as data, so nothing here re-types a number
70
80
  the service also reads.
81
+ - **`auroraomics.export`** — write a result as a SpatialData Zarr store
82
+ *(extra)* or as a folder Seurat loads, with the same matrix, coordinates and
83
+ gene names as the `.h5ad`.
71
84
  - **`auroraomics.embed`** *(extra)* — run a pinned image encoder over your
72
85
  tiles locally and write the embeddings container, so a prediction can be made
73
86
  from numbers instead of pixels.
@@ -88,16 +101,19 @@ pip install auroraomics
88
101
 
89
102
  That is the whole documented path: open a slide, judge and pack its tiles,
90
103
  submit, and read the result. One extra adds local embedding extraction
91
- (`embed`), because a tensor runtime is gigabytes and specific to the machine it
92
- was built for. No extra brings the model: predicting gene expression runs on
93
- the service, not here.
104
+ (`embed`), and one writes a result as a SpatialData store (`spatialdata`). No
105
+ extra brings the model: predicting gene expression runs on the service, not
106
+ here.
94
107
 
95
108
  ## Pack tiles, then look at the report
96
109
 
97
110
  ```python
98
111
  import auroraomics as ao
99
112
 
100
- report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thumbnail=thumb)
113
+ with ao.Client() as client:
114
+ thresholds = client.qc_thresholds() # the service's quality floors, no key needed
115
+
116
+ report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thresholds=thresholds, thumbnail=thumb)
101
117
  print(report.written, "tiles kept,", report.rejected, "dropped")
102
118
  print(report.rejected_by_reason) # {'foreground_ratio': 12, ...}
103
119
  ```
@@ -126,7 +142,8 @@ members do not match the container contract.
126
142
  ao.write_result(
127
143
  "result.h5ad",
128
144
  obs=obs, # per-spot columns, as plain arrays
129
- var=var, # per-gene columns, indexed by gene id
145
+ var=var, # per-gene columns
146
+ var_index=gene_ids, # the Ensembl gene id of each column
130
147
  spatial=coords, # (n_spots, 2) array -> obsm["spatial"]
131
148
  x=batches, # an array, or an iterable of row batches
132
149
  uns={"model": {"id": "..."}},
@@ -149,71 +166,69 @@ pip install "auroraomics[embed]"
149
166
  import auroraomics as ao
150
167
  from auroraomics.embed import available_encoders, describe_encoders
151
168
 
152
- print(available_encoders()) # ('deepspot-h', 'dinov2-b14', 'midnight')
153
-
154
169
  with ao.Client() as client:
155
170
  # Both are the service's, so the file records the encoder and the
156
- # thresholds the model it is submitted to was served with.
157
- encoder = ao.resolve_encoder(client.encoders())
171
+ # thresholds the model it is submitted to was served with. There is no
172
+ # list of encoders inside this package: every name comes from the registry
173
+ # the service publishes, which is why these calls take one.
174
+ registry = client.encoders()
158
175
  thresholds = client.qc_thresholds()
176
+ # The tile size the model's card states, for the model the file is for.
177
+ patch_um = client.model_card(model_id)["input_spec"]["patch_um"]
178
+
179
+ print(available_encoders(registry)) # the names this package will run today
180
+ encoder = ao.resolve_encoder(registry)
159
181
 
160
182
  report = ao.embed_tiles(
161
- slide, "sample.npz", mpp=0.499, patch_um=55, encoder=encoder, thresholds=thresholds
183
+ slide, "sample.npz", mpp=0.499, patch_um=patch_um, encoder=encoder, thresholds=thresholds
162
184
  )
163
185
  print(report.rows, "rows of", report.dim, "numbers from", report.encoder.name)
164
186
  ```
165
187
 
166
188
  `slide` is anything with a `(height, width, 3)` RGB `uint8` shape that can be
167
189
  sliced — an array, or a memory-mapped or lazily-read one, so a slide larger than
168
- memory works: only one crop exists at a time. Crops are taken at `patch_um`
169
- micrometres using the `mpp` you give, quality-checked with the same three
170
- predicates as `pack_tiles`, and pooled over the encoder's patch tokens.
171
-
172
- Every encoder is pinned to a repository **and** a revision sha, with its licence
173
- and access gate recorded beside it. `describe_encoders()` returns the whole
174
- registry, including the encoders this package refuses to run — a refusal names
175
- the licence or the approval you would have to accept, and which encoders remain.
176
-
177
- The written file carries one extra vector, `calibration`: the embedding of a
178
- fixed image shipped inside this package, through the same encoder, revision and
179
- preprocessing as your rows. Comparing that one vector against a known reference
180
- tells a reader whether the file was produced by the model it claims, before they
181
- look at a single row.
182
-
183
- The weights are downloaded from their publisher on first use, into the ordinary
184
- model cache. Pass `allow_download=False` to guarantee no request is made: for
185
- the duration of that load, name resolution and internet sockets are refused in
186
- this process, so the guarantee does not rest on a model library reading an
187
- environment variable it may already have read.
188
-
189
- An encoder runs at one input size: the size the served model's embeddings were
190
- computed at, which the registry pins and which is not always what the encoder's
191
- own repository declares. The DINOv2 releases carry a 518-px position grid; the
192
- model was trained on 224-px crops fed to that grid through an interpolated
193
- position embedding, so that is what this package does too. `resize_px` is
194
- checked against the pin rather than obeyed, because a vector computed at another
195
- size is a different measurement that nobody else's numbers can be compared with.
190
+ memory works: only one crop exists at a time. Crops are taken at the tile size
191
+ the model's card states, using the `mpp` you give, and quality-checked with the
192
+ same three predicates as `pack_tiles`.
193
+
194
+ Every encoder is pinned to an exact revision.
195
+ `describe_encoders(registry)` returns the whole registry, including the
196
+ encoders this package refuses to run: a refusal says why, and which encoders
197
+ remain.
198
+
199
+ The written file carries one extra embedding, `calibration`: computed from a
200
+ fixed image shipped inside this package, through the same encoder and revision
201
+ as your rows. Comparing that one embedding against a known
202
+ reference tells a reader whether the file was produced by the model it claims,
203
+ before they look at a single row.
204
+
205
+ The weights are downloaded on first use. Pass `allow_download=False` to
206
+ guarantee no request is made: for the duration of that load, name resolution
207
+ and internet sockets are refused in this process.
208
+
209
+ An encoder runs at the input size the registry pins for it. `resize_px` is
210
+ checked against that pin rather than obeyed, because an embedding computed at
211
+ another size cannot be compared with the served model's.
212
+
196
213
  ## Where a prediction runs
197
214
 
198
- Predicting gene expression runs on the service, and `predict_local` refuses —
199
- by design, and the design is the product rather than a limitation.
215
+ Predicting gene expression runs on the service, and `predict_local` refuses.
200
216
 
201
- This is split inference at the encoder. An open-weight pathology foundation
202
- model turns your tiles into patch embeddings **on your machine**, so your slides
203
- never leave it; our gene decoder turns embeddings into expression **on ours**, so
204
- its weights never leave it. What crosses between us is a vector from a model
205
- neither side owns.
217
+ **DeepSpot-H**, the foundation model for H&E images, turns each tile of your
218
+ slide into one embedding. It can run on your machine or on ours. **DeepSpot-M**
219
+ turns those embeddings into spatial gene expression. It only ever runs on ours.
206
220
 
207
221
  ```python
208
222
  from auroraomics.runtimes.deepspotm import check_archive, describe_models
209
223
 
210
- cards = client.models() # the catalogue is the service's
211
- describe_models(cards) # what it will run, and on what
212
- check_archive("sample.zip", model=MODEL, cards=cards) # are my tiles the right size?
224
+ with ao.Client() as client:
225
+ cards = client.models() # the catalogue is the service's
226
+ describe_models(cards) # what it will run, and on what
227
+ check_archive("sample.zip", model=model_id, cards=cards) # are my tiles the right size?
213
228
  ```
214
229
 
215
230
  Then submit with the API client and read the result back — the same `.h5ad`
216
- contract whichever lane ran it.
231
+ contract whichever runtime ran it.
217
232
 
218
233
  ## Post-process a result
219
234
 
@@ -237,17 +252,39 @@ skipped: a card is the description of a file, and a reader promised a layer
237
252
  cannot tell a missing one from a model that predicted zeros.
238
253
 
239
254
  The entry this version ships states what the model measured. `var` gains
240
- `n_train_datasets` — the number of training datasets each gene was measured in,
241
- because a gene seen in one and a gene seen in twenty are not the same claim —
242
- and `measured_in_training`, derived from that count. A gene the model never
243
- measured becomes `nan` in `X` and in every layer, not zero: zero is a
244
- measurement, and an unmeasured gene left as one averages, correlates and
255
+ `measured_in_training`, which is `False` for a gene the model never saw
256
+ measured. Such a gene becomes `nan` in `X` and in every layer, not zero: zero
257
+ is a measurement, and an unmeasured gene left as one averages, correlates and
245
258
  colours a heat map exactly like a prediction.
246
259
 
260
+ ## Read a result in SpatialData or Seurat
261
+
262
+ Every result is an `.h5ad`, which is an HDF5 file: `anndata` reads it, and so
263
+ does any HDF5 library. For tools that read other formats, write it again:
264
+
265
+ ```
266
+ auroraomics export result.h5ad --format seurat # result_seurat/, for Read10X
267
+ auroraomics export result.h5ad --format zarr # result.zarr, a SpatialData store
268
+ ```
269
+
270
+ ```python
271
+ from auroraomics.export import write_seurat_dir, write_zarr
272
+
273
+ write_seurat_dir("result.h5ad", "result_seurat")
274
+ write_zarr("result.h5ad", "result.zarr") # pip install "auroraomics[spatialdata]"
275
+ ```
276
+
277
+ Both keep the result's own values: a gene the model never measured stays blank
278
+ (`nan`) in either format, never zero.
279
+
247
280
  ## Prepare a bulk RNA profile
248
281
 
249
282
  ```python
250
- report = ao.bulk_rna("sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts")
283
+ with ao.Client() as client:
284
+ report = ao.bulk_rna(
285
+ "sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts",
286
+ card=client.model_card(model_id), resolve=client.resolve_keys,
287
+ )
251
288
  print(report.rows, "genes;", f"{report.coverage:.0%} of", report.coverage_of)
252
289
  ```
253
290
 
@@ -267,13 +304,6 @@ your project.
267
304
 
268
305
  ## Licence
269
306
 
270
- The code in this package is licensed under
271
- [PolyForm Noncommercial 1.0.0](https://polyformproject.org/licenses/noncommercial/1.0.0),
272
- which permits use for any purpose that is not commercial. It is the same licence the
273
- model package this client is built for carries, so installing both puts you under one
274
- rule rather than two.
275
-
276
- The model weights are licensed separately by whoever publishes them, and access to them
277
- may be gated. Read those terms before you use a model: they are not this licence, and a
278
- permission granted here is not a permission granted there.
307
+ The code in this package is licensed for non-commercial use; its terms are in the
308
+ `LICENSE` file it ships with. Commercial evaluation and use are governed by a written agreement with Aurora.
279
309
 
@@ -1,9 +1,13 @@
1
1
  # auroraomics
2
2
 
3
- Virtual spatial transcriptomics from H&E histology.
3
+ Virtual spatial transcriptomics from H&E images.
4
+
5
+ This package is the client of the **Aurora API**, the hosted service that runs
6
+ the prediction: from Python, from a shell with the `auroraomics` command, or
7
+ from an agent through its MCP server.
4
8
 
5
9
  This package holds the pieces of that pipeline that are pure Python, so the
6
- same code runs on your laptop, in a GPU container and on the service:
10
+ same code runs on your machine and on the service:
7
11
 
8
12
  - **`auroraomics.qc`** — the three patch-quality predicates (foreground, blur,
9
13
  stained tissue) applied to every candidate tile before a model sees it, and
@@ -15,16 +19,19 @@ same code runs on your laptop, in a GPU container and on the service:
15
19
  batch rather than the whole matrix.
16
20
  - **`auroraomics.postprocess`** — apply a model card's `postprocess` entries
17
21
  to a result file: a registry of entries, one runner, `h5py` alone, streaming
18
- one chunk at a time. Ships the unmeasured-gene mask, which states per gene
19
- how many training datasets measured it and blanks the ones none did.
22
+ one chunk at a time. Ships the unmeasured-gene mask, which blanks the genes
23
+ the model never measured.
20
24
  - **`auroraomics.subsample`** — pick the densest contiguous square of spots
21
- when a slide yields more tiles than a run is allowed to spend.
22
- - **`auroraomics.genes`** — the model family's gene symbols mapped to Ensembl
23
- stable gene ids, which is what every result's `var` index is keyed by. Ships
24
- as one committed table, resolved in the same order the service resolves it.
25
+ when a slide yields more tiles than one submission takes.
26
+ - **`auroraomics.genes`** — Ensembl stable gene ids, which is what every
27
+ result's `var` index is keyed by, and the shape a gene the service resolved
28
+ comes back in.
25
29
  - **`auroraomics.contracts`** — the shared contract values (container layout,
26
30
  input-kind caps, result layout) as data, so nothing here re-types a number
27
31
  the service also reads.
32
+ - **`auroraomics.export`** — write a result as a SpatialData Zarr store
33
+ *(extra)* or as a folder Seurat loads, with the same matrix, coordinates and
34
+ gene names as the `.h5ad`.
28
35
  - **`auroraomics.embed`** *(extra)* — run a pinned image encoder over your
29
36
  tiles locally and write the embeddings container, so a prediction can be made
30
37
  from numbers instead of pixels.
@@ -45,16 +52,19 @@ pip install auroraomics
45
52
 
46
53
  That is the whole documented path: open a slide, judge and pack its tiles,
47
54
  submit, and read the result. One extra adds local embedding extraction
48
- (`embed`), because a tensor runtime is gigabytes and specific to the machine it
49
- was built for. No extra brings the model: predicting gene expression runs on
50
- the service, not here.
55
+ (`embed`), and one writes a result as a SpatialData store (`spatialdata`). No
56
+ extra brings the model: predicting gene expression runs on the service, not
57
+ here.
51
58
 
52
59
  ## Pack tiles, then look at the report
53
60
 
54
61
  ```python
55
62
  import auroraomics as ao
56
63
 
57
- report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thumbnail=thumb)
64
+ with ao.Client() as client:
65
+ thresholds = client.qc_thresholds() # the service's quality floors, no key needed
66
+
67
+ report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thresholds=thresholds, thumbnail=thumb)
58
68
  print(report.written, "tiles kept,", report.rejected, "dropped")
59
69
  print(report.rejected_by_reason) # {'foreground_ratio': 12, ...}
60
70
  ```
@@ -83,7 +93,8 @@ members do not match the container contract.
83
93
  ao.write_result(
84
94
  "result.h5ad",
85
95
  obs=obs, # per-spot columns, as plain arrays
86
- var=var, # per-gene columns, indexed by gene id
96
+ var=var, # per-gene columns
97
+ var_index=gene_ids, # the Ensembl gene id of each column
87
98
  spatial=coords, # (n_spots, 2) array -> obsm["spatial"]
88
99
  x=batches, # an array, or an iterable of row batches
89
100
  uns={"model": {"id": "..."}},
@@ -106,71 +117,69 @@ pip install "auroraomics[embed]"
106
117
  import auroraomics as ao
107
118
  from auroraomics.embed import available_encoders, describe_encoders
108
119
 
109
- print(available_encoders()) # ('deepspot-h', 'dinov2-b14', 'midnight')
110
-
111
120
  with ao.Client() as client:
112
121
  # Both are the service's, so the file records the encoder and the
113
- # thresholds the model it is submitted to was served with.
114
- encoder = ao.resolve_encoder(client.encoders())
122
+ # thresholds the model it is submitted to was served with. There is no
123
+ # list of encoders inside this package: every name comes from the registry
124
+ # the service publishes, which is why these calls take one.
125
+ registry = client.encoders()
115
126
  thresholds = client.qc_thresholds()
127
+ # The tile size the model's card states, for the model the file is for.
128
+ patch_um = client.model_card(model_id)["input_spec"]["patch_um"]
129
+
130
+ print(available_encoders(registry)) # the names this package will run today
131
+ encoder = ao.resolve_encoder(registry)
116
132
 
117
133
  report = ao.embed_tiles(
118
- slide, "sample.npz", mpp=0.499, patch_um=55, encoder=encoder, thresholds=thresholds
134
+ slide, "sample.npz", mpp=0.499, patch_um=patch_um, encoder=encoder, thresholds=thresholds
119
135
  )
120
136
  print(report.rows, "rows of", report.dim, "numbers from", report.encoder.name)
121
137
  ```
122
138
 
123
139
  `slide` is anything with a `(height, width, 3)` RGB `uint8` shape that can be
124
140
  sliced — an array, or a memory-mapped or lazily-read one, so a slide larger than
125
- memory works: only one crop exists at a time. Crops are taken at `patch_um`
126
- micrometres using the `mpp` you give, quality-checked with the same three
127
- predicates as `pack_tiles`, and pooled over the encoder's patch tokens.
128
-
129
- Every encoder is pinned to a repository **and** a revision sha, with its licence
130
- and access gate recorded beside it. `describe_encoders()` returns the whole
131
- registry, including the encoders this package refuses to run — a refusal names
132
- the licence or the approval you would have to accept, and which encoders remain.
133
-
134
- The written file carries one extra vector, `calibration`: the embedding of a
135
- fixed image shipped inside this package, through the same encoder, revision and
136
- preprocessing as your rows. Comparing that one vector against a known reference
137
- tells a reader whether the file was produced by the model it claims, before they
138
- look at a single row.
139
-
140
- The weights are downloaded from their publisher on first use, into the ordinary
141
- model cache. Pass `allow_download=False` to guarantee no request is made: for
142
- the duration of that load, name resolution and internet sockets are refused in
143
- this process, so the guarantee does not rest on a model library reading an
144
- environment variable it may already have read.
145
-
146
- An encoder runs at one input size: the size the served model's embeddings were
147
- computed at, which the registry pins and which is not always what the encoder's
148
- own repository declares. The DINOv2 releases carry a 518-px position grid; the
149
- model was trained on 224-px crops fed to that grid through an interpolated
150
- position embedding, so that is what this package does too. `resize_px` is
151
- checked against the pin rather than obeyed, because a vector computed at another
152
- size is a different measurement that nobody else's numbers can be compared with.
141
+ memory works: only one crop exists at a time. Crops are taken at the tile size
142
+ the model's card states, using the `mpp` you give, and quality-checked with the
143
+ same three predicates as `pack_tiles`.
144
+
145
+ Every encoder is pinned to an exact revision.
146
+ `describe_encoders(registry)` returns the whole registry, including the
147
+ encoders this package refuses to run: a refusal says why, and which encoders
148
+ remain.
149
+
150
+ The written file carries one extra embedding, `calibration`: computed from a
151
+ fixed image shipped inside this package, through the same encoder and revision
152
+ as your rows. Comparing that one embedding against a known
153
+ reference tells a reader whether the file was produced by the model it claims,
154
+ before they look at a single row.
155
+
156
+ The weights are downloaded on first use. Pass `allow_download=False` to
157
+ guarantee no request is made: for the duration of that load, name resolution
158
+ and internet sockets are refused in this process.
159
+
160
+ An encoder runs at the input size the registry pins for it. `resize_px` is
161
+ checked against that pin rather than obeyed, because an embedding computed at
162
+ another size cannot be compared with the served model's.
163
+
153
164
  ## Where a prediction runs
154
165
 
155
- Predicting gene expression runs on the service, and `predict_local` refuses —
156
- by design, and the design is the product rather than a limitation.
166
+ Predicting gene expression runs on the service, and `predict_local` refuses.
157
167
 
158
- This is split inference at the encoder. An open-weight pathology foundation
159
- model turns your tiles into patch embeddings **on your machine**, so your slides
160
- never leave it; our gene decoder turns embeddings into expression **on ours**, so
161
- its weights never leave it. What crosses between us is a vector from a model
162
- neither side owns.
168
+ **DeepSpot-H**, the foundation model for H&E images, turns each tile of your
169
+ slide into one embedding. It can run on your machine or on ours. **DeepSpot-M**
170
+ turns those embeddings into spatial gene expression. It only ever runs on ours.
163
171
 
164
172
  ```python
165
173
  from auroraomics.runtimes.deepspotm import check_archive, describe_models
166
174
 
167
- cards = client.models() # the catalogue is the service's
168
- describe_models(cards) # what it will run, and on what
169
- check_archive("sample.zip", model=MODEL, cards=cards) # are my tiles the right size?
175
+ with ao.Client() as client:
176
+ cards = client.models() # the catalogue is the service's
177
+ describe_models(cards) # what it will run, and on what
178
+ check_archive("sample.zip", model=model_id, cards=cards) # are my tiles the right size?
170
179
  ```
171
180
 
172
181
  Then submit with the API client and read the result back — the same `.h5ad`
173
- contract whichever lane ran it.
182
+ contract whichever runtime ran it.
174
183
 
175
184
  ## Post-process a result
176
185
 
@@ -194,17 +203,39 @@ skipped: a card is the description of a file, and a reader promised a layer
194
203
  cannot tell a missing one from a model that predicted zeros.
195
204
 
196
205
  The entry this version ships states what the model measured. `var` gains
197
- `n_train_datasets` — the number of training datasets each gene was measured in,
198
- because a gene seen in one and a gene seen in twenty are not the same claim —
199
- and `measured_in_training`, derived from that count. A gene the model never
200
- measured becomes `nan` in `X` and in every layer, not zero: zero is a
201
- measurement, and an unmeasured gene left as one averages, correlates and
206
+ `measured_in_training`, which is `False` for a gene the model never saw
207
+ measured. Such a gene becomes `nan` in `X` and in every layer, not zero: zero
208
+ is a measurement, and an unmeasured gene left as one averages, correlates and
202
209
  colours a heat map exactly like a prediction.
203
210
 
211
+ ## Read a result in SpatialData or Seurat
212
+
213
+ Every result is an `.h5ad`, which is an HDF5 file: `anndata` reads it, and so
214
+ does any HDF5 library. For tools that read other formats, write it again:
215
+
216
+ ```
217
+ auroraomics export result.h5ad --format seurat # result_seurat/, for Read10X
218
+ auroraomics export result.h5ad --format zarr # result.zarr, a SpatialData store
219
+ ```
220
+
221
+ ```python
222
+ from auroraomics.export import write_seurat_dir, write_zarr
223
+
224
+ write_seurat_dir("result.h5ad", "result_seurat")
225
+ write_zarr("result.h5ad", "result.zarr") # pip install "auroraomics[spatialdata]"
226
+ ```
227
+
228
+ Both keep the result's own values: a gene the model never measured stays blank
229
+ (`nan`) in either format, never zero.
230
+
204
231
  ## Prepare a bulk RNA profile
205
232
 
206
233
  ```python
207
- report = ao.bulk_rna("sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts")
234
+ with ao.Client() as client:
235
+ report = ao.bulk_rna(
236
+ "sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts",
237
+ card=client.model_card(model_id), resolve=client.resolve_keys,
238
+ )
208
239
  print(report.rows, "genes;", f"{report.coverage:.0%} of", report.coverage_of)
209
240
  ```
210
241
 
@@ -224,13 +255,6 @@ your project.
224
255
 
225
256
  ## Licence
226
257
 
227
- The code in this package is licensed under
228
- [PolyForm Noncommercial 1.0.0](https://polyformproject.org/licenses/noncommercial/1.0.0),
229
- which permits use for any purpose that is not commercial. It is the same licence the
230
- model package this client is built for carries, so installing both puts you under one
231
- rule rather than two.
232
-
233
- The model weights are licensed separately by whoever publishes them, and access to them
234
- may be gated. Read those terms before you use a model: they are not this licence, and a
235
- permission granted here is not a permission granted there.
258
+ The code in this package is licensed for non-commercial use; its terms are in the
259
+ `LICENSE` file it ships with. Commercial evaluation and use are governed by a written agreement with Aurora.
236
260
 
@@ -4,14 +4,12 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "auroraomics"
7
- version = "0.1.0.dev2"
8
- description = "Virtual spatial transcriptomics from H&E histology: patch QC, tile packing and the .h5ad result contract."
7
+ version = "0.1.0.dev4"
8
+ description = "Prepare H&E slides on your own machine, submit them to Aurora, and read the spatial gene expression it predicts."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
11
- # Non-commercial, and the same licence as the model package this client is built
12
- # for, so a user installing both faces one rule rather than two. PEP 639 forbids
13
- # a licence CLASSIFIER beside an expression, so this line is the whole
14
- # declaration; the model weights carry their own separate terms.
11
+ # Non-commercial; the terms are in LICENSE. PEP 639 forbids a licence
12
+ # CLASSIFIER beside an expression, so this line is the whole declaration.
15
13
  license = "PolyForm-Noncommercial-1.0.0"
16
14
  license-files = ["LICENSE"]
17
15
  authors = [{ name = "Kalin Nonchev" }]
@@ -47,9 +45,9 @@ classifiers = [
47
45
  # all of it. That is the trade, taken deliberately: one install line that works,
48
46
  # over a smaller one that cannot reach the end of the page describing it.
49
47
  #
50
- # What is still NOT here is the name that matters to the GPU pipeline images —
51
- # no torch. They install this package with `--no-deps`, so what they resolve
52
- # does not change either way; what protects them is the encoder line below.
48
+ # What is still NOT here is torch. An environment that installs this package
49
+ # with `--no-deps` resolves the same either way; what protects it is the
50
+ # encoder line below.
53
51
  dependencies = [
54
52
  "numpy>=1.23",
55
53
  "h5py>=3.9",
@@ -66,12 +64,18 @@ dependencies = [
66
64
  # rather than being handed tiles they cut themselves.
67
65
  "tifffile>=2023.7.10",
68
66
  "imagecodecs>=2023.3.16",
67
+ # Writing a result as the folder Seurat loads (`auroraomics.export`): the
68
+ # matrix is written by scipy's Matrix Market writer. scipy already arrives
69
+ # behind anndata, so a plain install resolves nothing new for it; it is named
70
+ # because that module imports it directly, at anndata's own lower bound.
71
+ "scipy>1.8",
69
72
  ]
70
73
 
71
74
  [project.optional-dependencies]
72
- # TWO extras, and only one of them is a runtime: `embed` is the single opt-in
73
- # the documented path can need, and `mcp` is a protocol adapter for an agent.
74
- # Where that line is drawn is the whole of this comment.
75
+ # THREE extras, and only one of them is a model runtime: `embed` is the single
76
+ # opt-in the documented path can need, `mcp` is a protocol adapter for an agent,
77
+ # and `spatialdata` writes a result in one more format. Where the first line is
78
+ # drawn is the rest of this comment.
75
79
  #
76
80
  # There is no model-runtime extra, and that is the boundary rather than an
77
81
  # omission. Predicting gene expression runs on the service: the model definition
@@ -79,7 +83,7 @@ dependencies = [
79
83
  # resolves the model, or the adapter and checkpoint stack that would carry one,
80
84
  # and `predict_local` refuses with a message that says so. What runs on a user's
81
85
  # own machine is cutting, filtering and EMBEDDING tiles — their pixels stay put,
82
- # and a vector from the encoder is what crosses. So the line is at the
86
+ # and an embedding from the encoder is what crosses. So the line is at the
83
87
  # ENCODER, not at the size of an install: `embed` below does resolve a tensor
84
88
  # runtime, on purpose, and everything on the CLIENT side of the encoder — the
85
89
  # HTTP client, the slide reader — sits in the core above, where nobody has to
@@ -111,6 +115,21 @@ embed = [
111
115
  "transformers>=4.40",
112
116
  "huggingface-hub>=0.23",
113
117
  ]
118
+ # Writing a result as a SpatialData Zarr store (`auroraomics.export.write_zarr`,
119
+ # `auroraomics export --format zarr`). An extra because it is large — some
120
+ # seventy distributions and 700 MB on top of the core, measured on Python 3.10 —
121
+ # and only someone who wants SpatialData's own format needs it; the Seurat
122
+ # folder and the .h5ad need nothing from it.
123
+ #
124
+ # The setuptools bound is not a preference. Every SpatialData release before
125
+ # 0.8 imports a schema library that reads `pkg_resources`, which setuptools 81
126
+ # removed, so on a fresh environment the import fails with ModuleNotFoundError
127
+ # however the rest resolves. 0.8 dropped that library and requires Python 3.12,
128
+ # so the bound applies exactly where an older release is what pip can choose.
129
+ spatialdata = [
130
+ "spatialdata>=0.2",
131
+ "setuptools<81; python_version < '3.12'",
132
+ ]
114
133
  # The MCP server: `auroraomics mcp` over stdio, so an agent reaches the same
115
134
  # client the CLI does. Only the protocol adapter is here; the tool table and
116
135
  # every handler are in the client, which is in the core now, so this extra names
@@ -137,7 +156,19 @@ dev = [
137
156
  # that spelled its own bound would be testing a dependency set no user can
138
157
  # install. The client needs no reference any more — it is the core.
139
158
  "auroraomics[mcp]",
159
+ # The SpatialData extra by reference, for the same reason: the export tests
160
+ # read every store back with SpatialData's own reader, so the environment
161
+ # that runs them installs what a user who asks for the format installs.
162
+ "auroraomics[spatialdata]",
140
163
  "pytest>=7",
164
+ # The release guard parses version strings with `packaging.version`, at module
165
+ # scope. It has always resolved, because more than one dependency in this list
166
+ # requires packaging — but arriving in a closure is not the same as being
167
+ # declared: the day whichever one carries it stops doing so, the check that
168
+ # decides whether a build may be uploaded stops being a check and becomes a
169
+ # collection error, which reports as one missing optional dependency rather
170
+ # than as a suite that no longer runs.
171
+ "packaging>=22",
141
172
  # Coverage is measured on every run, with a floor, because this package is
142
173
  # the one that SHIPS: a user installs it and calls it, so a branch nothing
143
174
  # here exercises is a branch discovered in the field. The floor sat at