auroraomics 0.1.0.dev2__tar.gz → 0.1.0.dev4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {auroraomics-0.1.0.dev2/src/auroraomics.egg-info → auroraomics-0.1.0.dev4}/PKG-INFO +104 -74
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/README.md +96 -72
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/pyproject.toml +44 -13
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/__init__.py +16 -10
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_contracts/MANIFEST.json +1 -1
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_contracts/public-api.tokens.json +13 -1
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_text.py +13 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/bulk.py +26 -32
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/cli.py +238 -118
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/__init__.py +8 -4
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/_generated.py +36 -33
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/api.py +214 -87
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/errors.py +27 -15
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/jobs.py +13 -15
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/uploads.py +1 -2
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/contracts.py +28 -22
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/embed.py +115 -201
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/errors.py +7 -4
- auroraomics-0.1.0.dev4/src/auroraomics/export.py +467 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/genes.py +10 -21
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/h5ad.py +8 -17
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/mcp/server.py +5 -5
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/mcp/tools.py +12 -8
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/pack.py +58 -84
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/postprocess.py +4 -7
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/qc.py +18 -18
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/runtimes/__init__.py +1 -1
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/runtimes/deepspotm.py +53 -71
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/slide.py +47 -55
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/spatial.py +2 -2
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/subsample.py +1 -2
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4/src/auroraomics.egg-info}/PKG-INFO +104 -74
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics.egg-info/SOURCES.txt +1 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics.egg-info/requires.txt +9 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/LICENSE +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/MANIFEST.in +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/setup.cfg +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_assets/ASSETS.json +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_assets/README.md +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_assets/calibration-tile.png +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_contracts/public-api-counters.tokens.json +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/_contracts/public-api-input-kinds.tokens.json +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/_http.py +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/client/credentials.py +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/mcp/__init__.py +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics/py.typed +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics.egg-info/dependency_links.txt +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics.egg-info/entry_points.txt +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev4}/src/auroraomics.egg-info/top_level.txt +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: auroraomics
|
|
3
|
-
Version: 0.1.0.
|
|
4
|
-
Summary:
|
|
3
|
+
Version: 0.1.0.dev4
|
|
4
|
+
Summary: Prepare H&E slides on your own machine, submit them to Aurora, and read the spatial gene expression it predicts.
|
|
5
5
|
Author: Kalin Nonchev
|
|
6
6
|
License-Expression: PolyForm-Noncommercial-1.0.0
|
|
7
7
|
Keywords: histopathology,spatial-transcriptomics,h5ad,anndata,whole-slide-image
|
|
@@ -23,16 +23,22 @@ Requires-Dist: pydantic>=2
|
|
|
23
23
|
Requires-Dist: anndata>=0.10
|
|
24
24
|
Requires-Dist: tifffile>=2023.7.10
|
|
25
25
|
Requires-Dist: imagecodecs>=2023.3.16
|
|
26
|
+
Requires-Dist: scipy>1.8
|
|
26
27
|
Provides-Extra: embed
|
|
27
28
|
Requires-Dist: torch>=2.0; extra == "embed"
|
|
28
29
|
Requires-Dist: timm>=1.0; extra == "embed"
|
|
29
30
|
Requires-Dist: transformers>=4.40; extra == "embed"
|
|
30
31
|
Requires-Dist: huggingface-hub>=0.23; extra == "embed"
|
|
32
|
+
Provides-Extra: spatialdata
|
|
33
|
+
Requires-Dist: spatialdata>=0.2; extra == "spatialdata"
|
|
34
|
+
Requires-Dist: setuptools<81; python_version < "3.12" and extra == "spatialdata"
|
|
31
35
|
Provides-Extra: mcp
|
|
32
36
|
Requires-Dist: mcp<3,>=2; extra == "mcp"
|
|
33
37
|
Provides-Extra: dev
|
|
34
38
|
Requires-Dist: auroraomics[mcp]; extra == "dev"
|
|
39
|
+
Requires-Dist: auroraomics[spatialdata]; extra == "dev"
|
|
35
40
|
Requires-Dist: pytest>=7; extra == "dev"
|
|
41
|
+
Requires-Dist: packaging>=22; extra == "dev"
|
|
36
42
|
Requires-Dist: pytest-cov>=5; extra == "dev"
|
|
37
43
|
Requires-Dist: setuptools>=77; extra == "dev"
|
|
38
44
|
Requires-Dist: wheel; extra == "dev"
|
|
@@ -43,10 +49,14 @@ Dynamic: license-file
|
|
|
43
49
|
|
|
44
50
|
# auroraomics
|
|
45
51
|
|
|
46
|
-
Virtual spatial transcriptomics from H&E
|
|
52
|
+
Virtual spatial transcriptomics from H&E images.
|
|
53
|
+
|
|
54
|
+
This package is the client of the **Aurora API**, the hosted service that runs
|
|
55
|
+
the prediction: from Python, from a shell with the `auroraomics` command, or
|
|
56
|
+
from an agent through its MCP server.
|
|
47
57
|
|
|
48
58
|
This package holds the pieces of that pipeline that are pure Python, so the
|
|
49
|
-
same code runs on your
|
|
59
|
+
same code runs on your machine and on the service:
|
|
50
60
|
|
|
51
61
|
- **`auroraomics.qc`** — the three patch-quality predicates (foreground, blur,
|
|
52
62
|
stained tissue) applied to every candidate tile before a model sees it, and
|
|
@@ -58,16 +68,19 @@ same code runs on your laptop, in a GPU container and on the service:
|
|
|
58
68
|
batch rather than the whole matrix.
|
|
59
69
|
- **`auroraomics.postprocess`** — apply a model card's `postprocess` entries
|
|
60
70
|
to a result file: a registry of entries, one runner, `h5py` alone, streaming
|
|
61
|
-
one chunk at a time. Ships the unmeasured-gene mask, which
|
|
62
|
-
|
|
71
|
+
one chunk at a time. Ships the unmeasured-gene mask, which blanks the genes
|
|
72
|
+
the model never measured.
|
|
63
73
|
- **`auroraomics.subsample`** — pick the densest contiguous square of spots
|
|
64
|
-
when a slide yields more tiles than
|
|
65
|
-
- **`auroraomics.genes`** —
|
|
66
|
-
|
|
67
|
-
|
|
74
|
+
when a slide yields more tiles than one submission takes.
|
|
75
|
+
- **`auroraomics.genes`** — Ensembl stable gene ids, which is what every
|
|
76
|
+
result's `var` index is keyed by, and the shape a gene the service resolved
|
|
77
|
+
comes back in.
|
|
68
78
|
- **`auroraomics.contracts`** — the shared contract values (container layout,
|
|
69
79
|
input-kind caps, result layout) as data, so nothing here re-types a number
|
|
70
80
|
the service also reads.
|
|
81
|
+
- **`auroraomics.export`** — write a result as a SpatialData Zarr store
|
|
82
|
+
*(extra)* or as a folder Seurat loads, with the same matrix, coordinates and
|
|
83
|
+
gene names as the `.h5ad`.
|
|
71
84
|
- **`auroraomics.embed`** *(extra)* — run a pinned image encoder over your
|
|
72
85
|
tiles locally and write the embeddings container, so a prediction can be made
|
|
73
86
|
from numbers instead of pixels.
|
|
@@ -88,16 +101,19 @@ pip install auroraomics
|
|
|
88
101
|
|
|
89
102
|
That is the whole documented path: open a slide, judge and pack its tiles,
|
|
90
103
|
submit, and read the result. One extra adds local embedding extraction
|
|
91
|
-
(`embed`),
|
|
92
|
-
|
|
93
|
-
|
|
104
|
+
(`embed`), and one writes a result as a SpatialData store (`spatialdata`). No
|
|
105
|
+
extra brings the model: predicting gene expression runs on the service, not
|
|
106
|
+
here.
|
|
94
107
|
|
|
95
108
|
## Pack tiles, then look at the report
|
|
96
109
|
|
|
97
110
|
```python
|
|
98
111
|
import auroraomics as ao
|
|
99
112
|
|
|
100
|
-
|
|
113
|
+
with ao.Client() as client:
|
|
114
|
+
thresholds = client.qc_thresholds() # the service's quality floors, no key needed
|
|
115
|
+
|
|
116
|
+
report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thresholds=thresholds, thumbnail=thumb)
|
|
101
117
|
print(report.written, "tiles kept,", report.rejected, "dropped")
|
|
102
118
|
print(report.rejected_by_reason) # {'foreground_ratio': 12, ...}
|
|
103
119
|
```
|
|
@@ -126,7 +142,8 @@ members do not match the container contract.
|
|
|
126
142
|
ao.write_result(
|
|
127
143
|
"result.h5ad",
|
|
128
144
|
obs=obs, # per-spot columns, as plain arrays
|
|
129
|
-
var=var, # per-gene columns
|
|
145
|
+
var=var, # per-gene columns
|
|
146
|
+
var_index=gene_ids, # the Ensembl gene id of each column
|
|
130
147
|
spatial=coords, # (n_spots, 2) array -> obsm["spatial"]
|
|
131
148
|
x=batches, # an array, or an iterable of row batches
|
|
132
149
|
uns={"model": {"id": "..."}},
|
|
@@ -149,71 +166,69 @@ pip install "auroraomics[embed]"
|
|
|
149
166
|
import auroraomics as ao
|
|
150
167
|
from auroraomics.embed import available_encoders, describe_encoders
|
|
151
168
|
|
|
152
|
-
print(available_encoders()) # ('deepspot-h', 'dinov2-b14', 'midnight')
|
|
153
|
-
|
|
154
169
|
with ao.Client() as client:
|
|
155
170
|
# Both are the service's, so the file records the encoder and the
|
|
156
|
-
# thresholds the model it is submitted to was served with.
|
|
157
|
-
|
|
171
|
+
# thresholds the model it is submitted to was served with. There is no
|
|
172
|
+
# list of encoders inside this package: every name comes from the registry
|
|
173
|
+
# the service publishes, which is why these calls take one.
|
|
174
|
+
registry = client.encoders()
|
|
158
175
|
thresholds = client.qc_thresholds()
|
|
176
|
+
# The tile size the model's card states, for the model the file is for.
|
|
177
|
+
patch_um = client.model_card(model_id)["input_spec"]["patch_um"]
|
|
178
|
+
|
|
179
|
+
print(available_encoders(registry)) # the names this package will run today
|
|
180
|
+
encoder = ao.resolve_encoder(registry)
|
|
159
181
|
|
|
160
182
|
report = ao.embed_tiles(
|
|
161
|
-
slide, "sample.npz", mpp=0.499, patch_um=
|
|
183
|
+
slide, "sample.npz", mpp=0.499, patch_um=patch_um, encoder=encoder, thresholds=thresholds
|
|
162
184
|
)
|
|
163
185
|
print(report.rows, "rows of", report.dim, "numbers from", report.encoder.name)
|
|
164
186
|
```
|
|
165
187
|
|
|
166
188
|
`slide` is anything with a `(height, width, 3)` RGB `uint8` shape that can be
|
|
167
189
|
sliced — an array, or a memory-mapped or lazily-read one, so a slide larger than
|
|
168
|
-
memory works: only one crop exists at a time. Crops are taken at
|
|
169
|
-
|
|
170
|
-
predicates as `pack_tiles
|
|
171
|
-
|
|
172
|
-
Every encoder is pinned to
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
The written file carries one extra
|
|
178
|
-
fixed image shipped inside this package, through the same encoder
|
|
179
|
-
|
|
180
|
-
tells a reader whether the file was produced by the model it claims,
|
|
181
|
-
look at a single row.
|
|
182
|
-
|
|
183
|
-
The weights are downloaded
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
own repository declares. The DINOv2 releases carry a 518-px position grid; the
|
|
192
|
-
model was trained on 224-px crops fed to that grid through an interpolated
|
|
193
|
-
position embedding, so that is what this package does too. `resize_px` is
|
|
194
|
-
checked against the pin rather than obeyed, because a vector computed at another
|
|
195
|
-
size is a different measurement that nobody else's numbers can be compared with.
|
|
190
|
+
memory works: only one crop exists at a time. Crops are taken at the tile size
|
|
191
|
+
the model's card states, using the `mpp` you give, and quality-checked with the
|
|
192
|
+
same three predicates as `pack_tiles`.
|
|
193
|
+
|
|
194
|
+
Every encoder is pinned to an exact revision.
|
|
195
|
+
`describe_encoders(registry)` returns the whole registry, including the
|
|
196
|
+
encoders this package refuses to run: a refusal says why, and which encoders
|
|
197
|
+
remain.
|
|
198
|
+
|
|
199
|
+
The written file carries one extra embedding, `calibration`: computed from a
|
|
200
|
+
fixed image shipped inside this package, through the same encoder and revision
|
|
201
|
+
as your rows. Comparing that one embedding against a known
|
|
202
|
+
reference tells a reader whether the file was produced by the model it claims,
|
|
203
|
+
before they look at a single row.
|
|
204
|
+
|
|
205
|
+
The weights are downloaded on first use. Pass `allow_download=False` to
|
|
206
|
+
guarantee no request is made: for the duration of that load, name resolution
|
|
207
|
+
and internet sockets are refused in this process.
|
|
208
|
+
|
|
209
|
+
An encoder runs at the input size the registry pins for it. `resize_px` is
|
|
210
|
+
checked against that pin rather than obeyed, because an embedding computed at
|
|
211
|
+
another size cannot be compared with the served model's.
|
|
212
|
+
|
|
196
213
|
## Where a prediction runs
|
|
197
214
|
|
|
198
|
-
Predicting gene expression runs on the service, and `predict_local` refuses
|
|
199
|
-
by design, and the design is the product rather than a limitation.
|
|
215
|
+
Predicting gene expression runs on the service, and `predict_local` refuses.
|
|
200
216
|
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
its weights never leave it. What crosses between us is a vector from a model
|
|
205
|
-
neither side owns.
|
|
217
|
+
**DeepSpot-H**, the foundation model for H&E images, turns each tile of your
|
|
218
|
+
slide into one embedding. It can run on your machine or on ours. **DeepSpot-M**
|
|
219
|
+
turns those embeddings into spatial gene expression. It only ever runs on ours.
|
|
206
220
|
|
|
207
221
|
```python
|
|
208
222
|
from auroraomics.runtimes.deepspotm import check_archive, describe_models
|
|
209
223
|
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
224
|
+
with ao.Client() as client:
|
|
225
|
+
cards = client.models() # the catalogue is the service's
|
|
226
|
+
describe_models(cards) # what it will run, and on what
|
|
227
|
+
check_archive("sample.zip", model=model_id, cards=cards) # are my tiles the right size?
|
|
213
228
|
```
|
|
214
229
|
|
|
215
230
|
Then submit with the API client and read the result back — the same `.h5ad`
|
|
216
|
-
contract whichever
|
|
231
|
+
contract whichever runtime ran it.
|
|
217
232
|
|
|
218
233
|
## Post-process a result
|
|
219
234
|
|
|
@@ -237,17 +252,39 @@ skipped: a card is the description of a file, and a reader promised a layer
|
|
|
237
252
|
cannot tell a missing one from a model that predicted zeros.
|
|
238
253
|
|
|
239
254
|
The entry this version ships states what the model measured. `var` gains
|
|
240
|
-
`
|
|
241
|
-
|
|
242
|
-
and
|
|
243
|
-
measured becomes `nan` in `X` and in every layer, not zero: zero is a
|
|
244
|
-
measurement, and an unmeasured gene left as one averages, correlates and
|
|
255
|
+
`measured_in_training`, which is `False` for a gene the model never saw
|
|
256
|
+
measured. Such a gene becomes `nan` in `X` and in every layer, not zero: zero
|
|
257
|
+
is a measurement, and an unmeasured gene left as one averages, correlates and
|
|
245
258
|
colours a heat map exactly like a prediction.
|
|
246
259
|
|
|
260
|
+
## Read a result in SpatialData or Seurat
|
|
261
|
+
|
|
262
|
+
Every result is an `.h5ad`, which is an HDF5 file: `anndata` reads it, and so
|
|
263
|
+
does any HDF5 library. For tools that read other formats, write it again:
|
|
264
|
+
|
|
265
|
+
```
|
|
266
|
+
auroraomics export result.h5ad --format seurat # result_seurat/, for Read10X
|
|
267
|
+
auroraomics export result.h5ad --format zarr # result.zarr, a SpatialData store
|
|
268
|
+
```
|
|
269
|
+
|
|
270
|
+
```python
|
|
271
|
+
from auroraomics.export import write_seurat_dir, write_zarr
|
|
272
|
+
|
|
273
|
+
write_seurat_dir("result.h5ad", "result_seurat")
|
|
274
|
+
write_zarr("result.h5ad", "result.zarr") # pip install "auroraomics[spatialdata]"
|
|
275
|
+
```
|
|
276
|
+
|
|
277
|
+
Both keep the result's own values: a gene the model never measured stays blank
|
|
278
|
+
(`nan`) in either format, never zero.
|
|
279
|
+
|
|
247
280
|
## Prepare a bulk RNA profile
|
|
248
281
|
|
|
249
282
|
```python
|
|
250
|
-
|
|
283
|
+
with ao.Client() as client:
|
|
284
|
+
report = ao.bulk_rna(
|
|
285
|
+
"sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts",
|
|
286
|
+
card=client.model_card(model_id), resolve=client.resolve_keys,
|
|
287
|
+
)
|
|
251
288
|
print(report.rows, "genes;", f"{report.coverage:.0%} of", report.coverage_of)
|
|
252
289
|
```
|
|
253
290
|
|
|
@@ -267,13 +304,6 @@ your project.
|
|
|
267
304
|
|
|
268
305
|
## Licence
|
|
269
306
|
|
|
270
|
-
The code in this package is licensed
|
|
271
|
-
|
|
272
|
-
which permits use for any purpose that is not commercial. It is the same licence the
|
|
273
|
-
model package this client is built for carries, so installing both puts you under one
|
|
274
|
-
rule rather than two.
|
|
275
|
-
|
|
276
|
-
The model weights are licensed separately by whoever publishes them, and access to them
|
|
277
|
-
may be gated. Read those terms before you use a model: they are not this licence, and a
|
|
278
|
-
permission granted here is not a permission granted there.
|
|
307
|
+
The code in this package is licensed for non-commercial use; its terms are in the
|
|
308
|
+
`LICENSE` file it ships with. Commercial evaluation and use are governed by a written agreement with Aurora.
|
|
279
309
|
|
|
@@ -1,9 +1,13 @@
|
|
|
1
1
|
# auroraomics
|
|
2
2
|
|
|
3
|
-
Virtual spatial transcriptomics from H&E
|
|
3
|
+
Virtual spatial transcriptomics from H&E images.
|
|
4
|
+
|
|
5
|
+
This package is the client of the **Aurora API**, the hosted service that runs
|
|
6
|
+
the prediction: from Python, from a shell with the `auroraomics` command, or
|
|
7
|
+
from an agent through its MCP server.
|
|
4
8
|
|
|
5
9
|
This package holds the pieces of that pipeline that are pure Python, so the
|
|
6
|
-
same code runs on your
|
|
10
|
+
same code runs on your machine and on the service:
|
|
7
11
|
|
|
8
12
|
- **`auroraomics.qc`** — the three patch-quality predicates (foreground, blur,
|
|
9
13
|
stained tissue) applied to every candidate tile before a model sees it, and
|
|
@@ -15,16 +19,19 @@ same code runs on your laptop, in a GPU container and on the service:
|
|
|
15
19
|
batch rather than the whole matrix.
|
|
16
20
|
- **`auroraomics.postprocess`** — apply a model card's `postprocess` entries
|
|
17
21
|
to a result file: a registry of entries, one runner, `h5py` alone, streaming
|
|
18
|
-
one chunk at a time. Ships the unmeasured-gene mask, which
|
|
19
|
-
|
|
22
|
+
one chunk at a time. Ships the unmeasured-gene mask, which blanks the genes
|
|
23
|
+
the model never measured.
|
|
20
24
|
- **`auroraomics.subsample`** — pick the densest contiguous square of spots
|
|
21
|
-
when a slide yields more tiles than
|
|
22
|
-
- **`auroraomics.genes`** —
|
|
23
|
-
|
|
24
|
-
|
|
25
|
+
when a slide yields more tiles than one submission takes.
|
|
26
|
+
- **`auroraomics.genes`** — Ensembl stable gene ids, which is what every
|
|
27
|
+
result's `var` index is keyed by, and the shape a gene the service resolved
|
|
28
|
+
comes back in.
|
|
25
29
|
- **`auroraomics.contracts`** — the shared contract values (container layout,
|
|
26
30
|
input-kind caps, result layout) as data, so nothing here re-types a number
|
|
27
31
|
the service also reads.
|
|
32
|
+
- **`auroraomics.export`** — write a result as a SpatialData Zarr store
|
|
33
|
+
*(extra)* or as a folder Seurat loads, with the same matrix, coordinates and
|
|
34
|
+
gene names as the `.h5ad`.
|
|
28
35
|
- **`auroraomics.embed`** *(extra)* — run a pinned image encoder over your
|
|
29
36
|
tiles locally and write the embeddings container, so a prediction can be made
|
|
30
37
|
from numbers instead of pixels.
|
|
@@ -45,16 +52,19 @@ pip install auroraomics
|
|
|
45
52
|
|
|
46
53
|
That is the whole documented path: open a slide, judge and pack its tiles,
|
|
47
54
|
submit, and read the result. One extra adds local embedding extraction
|
|
48
|
-
(`embed`),
|
|
49
|
-
|
|
50
|
-
|
|
55
|
+
(`embed`), and one writes a result as a SpatialData store (`spatialdata`). No
|
|
56
|
+
extra brings the model: predicting gene expression runs on the service, not
|
|
57
|
+
here.
|
|
51
58
|
|
|
52
59
|
## Pack tiles, then look at the report
|
|
53
60
|
|
|
54
61
|
```python
|
|
55
62
|
import auroraomics as ao
|
|
56
63
|
|
|
57
|
-
|
|
64
|
+
with ao.Client() as client:
|
|
65
|
+
thresholds = client.qc_thresholds() # the service's quality floors, no key needed
|
|
66
|
+
|
|
67
|
+
report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thresholds=thresholds, thumbnail=thumb)
|
|
58
68
|
print(report.written, "tiles kept,", report.rejected, "dropped")
|
|
59
69
|
print(report.rejected_by_reason) # {'foreground_ratio': 12, ...}
|
|
60
70
|
```
|
|
@@ -83,7 +93,8 @@ members do not match the container contract.
|
|
|
83
93
|
ao.write_result(
|
|
84
94
|
"result.h5ad",
|
|
85
95
|
obs=obs, # per-spot columns, as plain arrays
|
|
86
|
-
var=var, # per-gene columns
|
|
96
|
+
var=var, # per-gene columns
|
|
97
|
+
var_index=gene_ids, # the Ensembl gene id of each column
|
|
87
98
|
spatial=coords, # (n_spots, 2) array -> obsm["spatial"]
|
|
88
99
|
x=batches, # an array, or an iterable of row batches
|
|
89
100
|
uns={"model": {"id": "..."}},
|
|
@@ -106,71 +117,69 @@ pip install "auroraomics[embed]"
|
|
|
106
117
|
import auroraomics as ao
|
|
107
118
|
from auroraomics.embed import available_encoders, describe_encoders
|
|
108
119
|
|
|
109
|
-
print(available_encoders()) # ('deepspot-h', 'dinov2-b14', 'midnight')
|
|
110
|
-
|
|
111
120
|
with ao.Client() as client:
|
|
112
121
|
# Both are the service's, so the file records the encoder and the
|
|
113
|
-
# thresholds the model it is submitted to was served with.
|
|
114
|
-
|
|
122
|
+
# thresholds the model it is submitted to was served with. There is no
|
|
123
|
+
# list of encoders inside this package: every name comes from the registry
|
|
124
|
+
# the service publishes, which is why these calls take one.
|
|
125
|
+
registry = client.encoders()
|
|
115
126
|
thresholds = client.qc_thresholds()
|
|
127
|
+
# The tile size the model's card states, for the model the file is for.
|
|
128
|
+
patch_um = client.model_card(model_id)["input_spec"]["patch_um"]
|
|
129
|
+
|
|
130
|
+
print(available_encoders(registry)) # the names this package will run today
|
|
131
|
+
encoder = ao.resolve_encoder(registry)
|
|
116
132
|
|
|
117
133
|
report = ao.embed_tiles(
|
|
118
|
-
slide, "sample.npz", mpp=0.499, patch_um=
|
|
134
|
+
slide, "sample.npz", mpp=0.499, patch_um=patch_um, encoder=encoder, thresholds=thresholds
|
|
119
135
|
)
|
|
120
136
|
print(report.rows, "rows of", report.dim, "numbers from", report.encoder.name)
|
|
121
137
|
```
|
|
122
138
|
|
|
123
139
|
`slide` is anything with a `(height, width, 3)` RGB `uint8` shape that can be
|
|
124
140
|
sliced — an array, or a memory-mapped or lazily-read one, so a slide larger than
|
|
125
|
-
memory works: only one crop exists at a time. Crops are taken at
|
|
126
|
-
|
|
127
|
-
predicates as `pack_tiles
|
|
128
|
-
|
|
129
|
-
Every encoder is pinned to
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
The written file carries one extra
|
|
135
|
-
fixed image shipped inside this package, through the same encoder
|
|
136
|
-
|
|
137
|
-
tells a reader whether the file was produced by the model it claims,
|
|
138
|
-
look at a single row.
|
|
139
|
-
|
|
140
|
-
The weights are downloaded
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
own repository declares. The DINOv2 releases carry a 518-px position grid; the
|
|
149
|
-
model was trained on 224-px crops fed to that grid through an interpolated
|
|
150
|
-
position embedding, so that is what this package does too. `resize_px` is
|
|
151
|
-
checked against the pin rather than obeyed, because a vector computed at another
|
|
152
|
-
size is a different measurement that nobody else's numbers can be compared with.
|
|
141
|
+
memory works: only one crop exists at a time. Crops are taken at the tile size
|
|
142
|
+
the model's card states, using the `mpp` you give, and quality-checked with the
|
|
143
|
+
same three predicates as `pack_tiles`.
|
|
144
|
+
|
|
145
|
+
Every encoder is pinned to an exact revision.
|
|
146
|
+
`describe_encoders(registry)` returns the whole registry, including the
|
|
147
|
+
encoders this package refuses to run: a refusal says why, and which encoders
|
|
148
|
+
remain.
|
|
149
|
+
|
|
150
|
+
The written file carries one extra embedding, `calibration`: computed from a
|
|
151
|
+
fixed image shipped inside this package, through the same encoder and revision
|
|
152
|
+
as your rows. Comparing that one embedding against a known
|
|
153
|
+
reference tells a reader whether the file was produced by the model it claims,
|
|
154
|
+
before they look at a single row.
|
|
155
|
+
|
|
156
|
+
The weights are downloaded on first use. Pass `allow_download=False` to
|
|
157
|
+
guarantee no request is made: for the duration of that load, name resolution
|
|
158
|
+
and internet sockets are refused in this process.
|
|
159
|
+
|
|
160
|
+
An encoder runs at the input size the registry pins for it. `resize_px` is
|
|
161
|
+
checked against that pin rather than obeyed, because an embedding computed at
|
|
162
|
+
another size cannot be compared with the served model's.
|
|
163
|
+
|
|
153
164
|
## Where a prediction runs
|
|
154
165
|
|
|
155
|
-
Predicting gene expression runs on the service, and `predict_local` refuses
|
|
156
|
-
by design, and the design is the product rather than a limitation.
|
|
166
|
+
Predicting gene expression runs on the service, and `predict_local` refuses.
|
|
157
167
|
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
its weights never leave it. What crosses between us is a vector from a model
|
|
162
|
-
neither side owns.
|
|
168
|
+
**DeepSpot-H**, the foundation model for H&E images, turns each tile of your
|
|
169
|
+
slide into one embedding. It can run on your machine or on ours. **DeepSpot-M**
|
|
170
|
+
turns those embeddings into spatial gene expression. It only ever runs on ours.
|
|
163
171
|
|
|
164
172
|
```python
|
|
165
173
|
from auroraomics.runtimes.deepspotm import check_archive, describe_models
|
|
166
174
|
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
175
|
+
with ao.Client() as client:
|
|
176
|
+
cards = client.models() # the catalogue is the service's
|
|
177
|
+
describe_models(cards) # what it will run, and on what
|
|
178
|
+
check_archive("sample.zip", model=model_id, cards=cards) # are my tiles the right size?
|
|
170
179
|
```
|
|
171
180
|
|
|
172
181
|
Then submit with the API client and read the result back — the same `.h5ad`
|
|
173
|
-
contract whichever
|
|
182
|
+
contract whichever runtime ran it.
|
|
174
183
|
|
|
175
184
|
## Post-process a result
|
|
176
185
|
|
|
@@ -194,17 +203,39 @@ skipped: a card is the description of a file, and a reader promised a layer
|
|
|
194
203
|
cannot tell a missing one from a model that predicted zeros.
|
|
195
204
|
|
|
196
205
|
The entry this version ships states what the model measured. `var` gains
|
|
197
|
-
`
|
|
198
|
-
|
|
199
|
-
and
|
|
200
|
-
measured becomes `nan` in `X` and in every layer, not zero: zero is a
|
|
201
|
-
measurement, and an unmeasured gene left as one averages, correlates and
|
|
206
|
+
`measured_in_training`, which is `False` for a gene the model never saw
|
|
207
|
+
measured. Such a gene becomes `nan` in `X` and in every layer, not zero: zero
|
|
208
|
+
is a measurement, and an unmeasured gene left as one averages, correlates and
|
|
202
209
|
colours a heat map exactly like a prediction.
|
|
203
210
|
|
|
211
|
+
## Read a result in SpatialData or Seurat
|
|
212
|
+
|
|
213
|
+
Every result is an `.h5ad`, which is an HDF5 file: `anndata` reads it, and so
|
|
214
|
+
does any HDF5 library. For tools that read other formats, write it again:
|
|
215
|
+
|
|
216
|
+
```
|
|
217
|
+
auroraomics export result.h5ad --format seurat # result_seurat/, for Read10X
|
|
218
|
+
auroraomics export result.h5ad --format zarr # result.zarr, a SpatialData store
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
```python
|
|
222
|
+
from auroraomics.export import write_seurat_dir, write_zarr
|
|
223
|
+
|
|
224
|
+
write_seurat_dir("result.h5ad", "result_seurat")
|
|
225
|
+
write_zarr("result.h5ad", "result.zarr") # pip install "auroraomics[spatialdata]"
|
|
226
|
+
```
|
|
227
|
+
|
|
228
|
+
Both keep the result's own values: a gene the model never measured stays blank
|
|
229
|
+
(`nan`) in either format, never zero.
|
|
230
|
+
|
|
204
231
|
## Prepare a bulk RNA profile
|
|
205
232
|
|
|
206
233
|
```python
|
|
207
|
-
|
|
234
|
+
with ao.Client() as client:
|
|
235
|
+
report = ao.bulk_rna(
|
|
236
|
+
"sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts",
|
|
237
|
+
card=client.model_card(model_id), resolve=client.resolve_keys,
|
|
238
|
+
)
|
|
208
239
|
print(report.rows, "genes;", f"{report.coverage:.0%} of", report.coverage_of)
|
|
209
240
|
```
|
|
210
241
|
|
|
@@ -224,13 +255,6 @@ your project.
|
|
|
224
255
|
|
|
225
256
|
## Licence
|
|
226
257
|
|
|
227
|
-
The code in this package is licensed
|
|
228
|
-
|
|
229
|
-
which permits use for any purpose that is not commercial. It is the same licence the
|
|
230
|
-
model package this client is built for carries, so installing both puts you under one
|
|
231
|
-
rule rather than two.
|
|
232
|
-
|
|
233
|
-
The model weights are licensed separately by whoever publishes them, and access to them
|
|
234
|
-
may be gated. Read those terms before you use a model: they are not this licence, and a
|
|
235
|
-
permission granted here is not a permission granted there.
|
|
258
|
+
The code in this package is licensed for non-commercial use; its terms are in the
|
|
259
|
+
`LICENSE` file it ships with. Commercial evaluation and use are governed by a written agreement with Aurora.
|
|
236
260
|
|
|
@@ -4,14 +4,12 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "auroraomics"
|
|
7
|
-
version = "0.1.0.
|
|
8
|
-
description = "
|
|
7
|
+
version = "0.1.0.dev4"
|
|
8
|
+
description = "Prepare H&E slides on your own machine, submit them to Aurora, and read the spatial gene expression it predicts."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
11
|
-
# Non-commercial
|
|
12
|
-
#
|
|
13
|
-
# a licence CLASSIFIER beside an expression, so this line is the whole
|
|
14
|
-
# declaration; the model weights carry their own separate terms.
|
|
11
|
+
# Non-commercial; the terms are in LICENSE. PEP 639 forbids a licence
|
|
12
|
+
# CLASSIFIER beside an expression, so this line is the whole declaration.
|
|
15
13
|
license = "PolyForm-Noncommercial-1.0.0"
|
|
16
14
|
license-files = ["LICENSE"]
|
|
17
15
|
authors = [{ name = "Kalin Nonchev" }]
|
|
@@ -47,9 +45,9 @@ classifiers = [
|
|
|
47
45
|
# all of it. That is the trade, taken deliberately: one install line that works,
|
|
48
46
|
# over a smaller one that cannot reach the end of the page describing it.
|
|
49
47
|
#
|
|
50
|
-
# What is still NOT here is
|
|
51
|
-
# no
|
|
52
|
-
#
|
|
48
|
+
# What is still NOT here is torch. An environment that installs this package
|
|
49
|
+
# with `--no-deps` resolves the same either way; what protects it is the
|
|
50
|
+
# encoder line below.
|
|
53
51
|
dependencies = [
|
|
54
52
|
"numpy>=1.23",
|
|
55
53
|
"h5py>=3.9",
|
|
@@ -66,12 +64,18 @@ dependencies = [
|
|
|
66
64
|
# rather than being handed tiles they cut themselves.
|
|
67
65
|
"tifffile>=2023.7.10",
|
|
68
66
|
"imagecodecs>=2023.3.16",
|
|
67
|
+
# Writing a result as the folder Seurat loads (`auroraomics.export`): the
|
|
68
|
+
# matrix is written by scipy's Matrix Market writer. scipy already arrives
|
|
69
|
+
# behind anndata, so a plain install resolves nothing new for it; it is named
|
|
70
|
+
# because that module imports it directly, at anndata's own lower bound.
|
|
71
|
+
"scipy>1.8",
|
|
69
72
|
]
|
|
70
73
|
|
|
71
74
|
[project.optional-dependencies]
|
|
72
|
-
#
|
|
73
|
-
# the documented path can need,
|
|
74
|
-
#
|
|
75
|
+
# THREE extras, and only one of them is a model runtime: `embed` is the single
|
|
76
|
+
# opt-in the documented path can need, `mcp` is a protocol adapter for an agent,
|
|
77
|
+
# and `spatialdata` writes a result in one more format. Where the first line is
|
|
78
|
+
# drawn is the rest of this comment.
|
|
75
79
|
#
|
|
76
80
|
# There is no model-runtime extra, and that is the boundary rather than an
|
|
77
81
|
# omission. Predicting gene expression runs on the service: the model definition
|
|
@@ -79,7 +83,7 @@ dependencies = [
|
|
|
79
83
|
# resolves the model, or the adapter and checkpoint stack that would carry one,
|
|
80
84
|
# and `predict_local` refuses with a message that says so. What runs on a user's
|
|
81
85
|
# own machine is cutting, filtering and EMBEDDING tiles — their pixels stay put,
|
|
82
|
-
# and
|
|
86
|
+
# and an embedding from the encoder is what crosses. So the line is at the
|
|
83
87
|
# ENCODER, not at the size of an install: `embed` below does resolve a tensor
|
|
84
88
|
# runtime, on purpose, and everything on the CLIENT side of the encoder — the
|
|
85
89
|
# HTTP client, the slide reader — sits in the core above, where nobody has to
|
|
@@ -111,6 +115,21 @@ embed = [
|
|
|
111
115
|
"transformers>=4.40",
|
|
112
116
|
"huggingface-hub>=0.23",
|
|
113
117
|
]
|
|
118
|
+
# Writing a result as a SpatialData Zarr store (`auroraomics.export.write_zarr`,
|
|
119
|
+
# `auroraomics export --format zarr`). An extra because it is large — some
|
|
120
|
+
# seventy distributions and 700 MB on top of the core, measured on Python 3.10 —
|
|
121
|
+
# and only someone who wants SpatialData's own format needs it; the Seurat
|
|
122
|
+
# folder and the .h5ad need nothing from it.
|
|
123
|
+
#
|
|
124
|
+
# The setuptools bound is not a preference. Every SpatialData release before
|
|
125
|
+
# 0.8 imports a schema library that reads `pkg_resources`, which setuptools 81
|
|
126
|
+
# removed, so on a fresh environment the import fails with ModuleNotFoundError
|
|
127
|
+
# however the rest resolves. 0.8 dropped that library and requires Python 3.12,
|
|
128
|
+
# so the bound applies exactly where an older release is what pip can choose.
|
|
129
|
+
spatialdata = [
|
|
130
|
+
"spatialdata>=0.2",
|
|
131
|
+
"setuptools<81; python_version < '3.12'",
|
|
132
|
+
]
|
|
114
133
|
# The MCP server: `auroraomics mcp` over stdio, so an agent reaches the same
|
|
115
134
|
# client the CLI does. Only the protocol adapter is here; the tool table and
|
|
116
135
|
# every handler are in the client, which is in the core now, so this extra names
|
|
@@ -137,7 +156,19 @@ dev = [
|
|
|
137
156
|
# that spelled its own bound would be testing a dependency set no user can
|
|
138
157
|
# install. The client needs no reference any more — it is the core.
|
|
139
158
|
"auroraomics[mcp]",
|
|
159
|
+
# The SpatialData extra by reference, for the same reason: the export tests
|
|
160
|
+
# read every store back with SpatialData's own reader, so the environment
|
|
161
|
+
# that runs them installs what a user who asks for the format installs.
|
|
162
|
+
"auroraomics[spatialdata]",
|
|
140
163
|
"pytest>=7",
|
|
164
|
+
# The release guard parses version strings with `packaging.version`, at module
|
|
165
|
+
# scope. It has always resolved, because more than one dependency in this list
|
|
166
|
+
# requires packaging — but arriving in a closure is not the same as being
|
|
167
|
+
# declared: the day whichever one carries it stops doing so, the check that
|
|
168
|
+
# decides whether a build may be uploaded stops being a check and becomes a
|
|
169
|
+
# collection error, which reports as one missing optional dependency rather
|
|
170
|
+
# than as a suite that no longer runs.
|
|
171
|
+
"packaging>=22",
|
|
141
172
|
# Coverage is measured on every run, with a floor, because this package is
|
|
142
173
|
# the one that SHIPS: a user installs it and calls it, so a branch nothing
|
|
143
174
|
# here exercises is a branch discovered in the field. The floor sat at
|