auroraomics 0.1.0.dev2__tar.gz → 0.1.0.dev3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {auroraomics-0.1.0.dev2/src/auroraomics.egg-info → auroraomics-0.1.0.dev3}/PKG-INFO +75 -74
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/README.md +72 -72
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/pyproject.toml +16 -10
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/__init__.py +15 -10
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/MANIFEST.json +1 -1
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/public-api.tokens.json +13 -1
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_text.py +13 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/bulk.py +26 -32
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/cli.py +166 -98
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/__init__.py +8 -4
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/_generated.py +28 -31
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/api.py +77 -57
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/errors.py +27 -15
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/jobs.py +13 -15
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/uploads.py +1 -2
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/contracts.py +10 -22
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/embed.py +103 -194
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/errors.py +6 -4
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/genes.py +10 -21
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/h5ad.py +8 -17
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/mcp/server.py +5 -5
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/mcp/tools.py +2 -3
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/pack.py +50 -83
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/postprocess.py +4 -7
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/qc.py +16 -16
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/runtimes/__init__.py +1 -1
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/runtimes/deepspotm.py +53 -71
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/slide.py +46 -55
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/spatial.py +2 -2
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/subsample.py +1 -2
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3/src/auroraomics.egg-info}/PKG-INFO +75 -74
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/requires.txt +1 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/LICENSE +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/MANIFEST.in +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/setup.cfg +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_assets/ASSETS.json +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_assets/README.md +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_assets/calibration-tile.png +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/public-api-counters.tokens.json +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/public-api-input-kinds.tokens.json +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/_http.py +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/client/credentials.py +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/mcp/__init__.py +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/py.typed +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/SOURCES.txt +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/dependency_links.txt +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/entry_points.txt +0 -0
- {auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/top_level.txt +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: auroraomics
|
|
3
|
-
Version: 0.1.0.
|
|
4
|
-
Summary:
|
|
3
|
+
Version: 0.1.0.dev3
|
|
4
|
+
Summary: Prepare H&E slides on your own machine, submit them to Aurora, and read the spatial gene expression it predicts.
|
|
5
5
|
Author: Kalin Nonchev
|
|
6
6
|
License-Expression: PolyForm-Noncommercial-1.0.0
|
|
7
7
|
Keywords: histopathology,spatial-transcriptomics,h5ad,anndata,whole-slide-image
|
|
@@ -33,6 +33,7 @@ Requires-Dist: mcp<3,>=2; extra == "mcp"
|
|
|
33
33
|
Provides-Extra: dev
|
|
34
34
|
Requires-Dist: auroraomics[mcp]; extra == "dev"
|
|
35
35
|
Requires-Dist: pytest>=7; extra == "dev"
|
|
36
|
+
Requires-Dist: packaging>=22; extra == "dev"
|
|
36
37
|
Requires-Dist: pytest-cov>=5; extra == "dev"
|
|
37
38
|
Requires-Dist: setuptools>=77; extra == "dev"
|
|
38
39
|
Requires-Dist: wheel; extra == "dev"
|
|
@@ -43,10 +44,14 @@ Dynamic: license-file
|
|
|
43
44
|
|
|
44
45
|
# auroraomics
|
|
45
46
|
|
|
46
|
-
Virtual spatial transcriptomics from H&E
|
|
47
|
+
Virtual spatial transcriptomics from H&E images.
|
|
48
|
+
|
|
49
|
+
This package is the client of the **Aurora API**, the hosted service that runs
|
|
50
|
+
the prediction: from Python, from a shell with the `auroraomics` command, or
|
|
51
|
+
from an agent through its MCP server.
|
|
47
52
|
|
|
48
53
|
This package holds the pieces of that pipeline that are pure Python, so the
|
|
49
|
-
same code runs on your
|
|
54
|
+
same code runs on your machine and on the service:
|
|
50
55
|
|
|
51
56
|
- **`auroraomics.qc`** — the three patch-quality predicates (foreground, blur,
|
|
52
57
|
stained tissue) applied to every candidate tile before a model sees it, and
|
|
@@ -58,13 +63,13 @@ same code runs on your laptop, in a GPU container and on the service:
|
|
|
58
63
|
batch rather than the whole matrix.
|
|
59
64
|
- **`auroraomics.postprocess`** — apply a model card's `postprocess` entries
|
|
60
65
|
to a result file: a registry of entries, one runner, `h5py` alone, streaming
|
|
61
|
-
one chunk at a time. Ships the unmeasured-gene mask, which
|
|
62
|
-
|
|
66
|
+
one chunk at a time. Ships the unmeasured-gene mask, which blanks the genes
|
|
67
|
+
the model never measured.
|
|
63
68
|
- **`auroraomics.subsample`** — pick the densest contiguous square of spots
|
|
64
|
-
when a slide yields more tiles than
|
|
65
|
-
- **`auroraomics.genes`** —
|
|
66
|
-
|
|
67
|
-
|
|
69
|
+
when a slide yields more tiles than one submission takes.
|
|
70
|
+
- **`auroraomics.genes`** — Ensembl stable gene ids, which is what every
|
|
71
|
+
result's `var` index is keyed by, and the shape a gene the service resolved
|
|
72
|
+
comes back in.
|
|
68
73
|
- **`auroraomics.contracts`** — the shared contract values (container layout,
|
|
69
74
|
input-kind caps, result layout) as data, so nothing here re-types a number
|
|
70
75
|
the service also reads.
|
|
@@ -88,16 +93,18 @@ pip install auroraomics
|
|
|
88
93
|
|
|
89
94
|
That is the whole documented path: open a slide, judge and pack its tiles,
|
|
90
95
|
submit, and read the result. One extra adds local embedding extraction
|
|
91
|
-
(`embed`)
|
|
92
|
-
|
|
93
|
-
the service, not here.
|
|
96
|
+
(`embed`). No extra brings the model: predicting gene expression runs on the
|
|
97
|
+
service, not here.
|
|
94
98
|
|
|
95
99
|
## Pack tiles, then look at the report
|
|
96
100
|
|
|
97
101
|
```python
|
|
98
102
|
import auroraomics as ao
|
|
99
103
|
|
|
100
|
-
|
|
104
|
+
with ao.Client() as client:
|
|
105
|
+
thresholds = client.qc_thresholds() # the service's quality floors, no key needed
|
|
106
|
+
|
|
107
|
+
report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thresholds=thresholds, thumbnail=thumb)
|
|
101
108
|
print(report.written, "tiles kept,", report.rejected, "dropped")
|
|
102
109
|
print(report.rejected_by_reason) # {'foreground_ratio': 12, ...}
|
|
103
110
|
```
|
|
@@ -126,7 +133,8 @@ members do not match the container contract.
|
|
|
126
133
|
ao.write_result(
|
|
127
134
|
"result.h5ad",
|
|
128
135
|
obs=obs, # per-spot columns, as plain arrays
|
|
129
|
-
var=var, # per-gene columns
|
|
136
|
+
var=var, # per-gene columns
|
|
137
|
+
var_index=gene_ids, # the Ensembl gene id of each column
|
|
130
138
|
spatial=coords, # (n_spots, 2) array -> obsm["spatial"]
|
|
131
139
|
x=batches, # an array, or an iterable of row batches
|
|
132
140
|
uns={"model": {"id": "..."}},
|
|
@@ -149,71 +157,69 @@ pip install "auroraomics[embed]"
|
|
|
149
157
|
import auroraomics as ao
|
|
150
158
|
from auroraomics.embed import available_encoders, describe_encoders
|
|
151
159
|
|
|
152
|
-
print(available_encoders()) # ('deepspot-h', 'dinov2-b14', 'midnight')
|
|
153
|
-
|
|
154
160
|
with ao.Client() as client:
|
|
155
161
|
# Both are the service's, so the file records the encoder and the
|
|
156
|
-
# thresholds the model it is submitted to was served with.
|
|
157
|
-
|
|
162
|
+
# thresholds the model it is submitted to was served with. There is no
|
|
163
|
+
# list of encoders inside this package: every name comes from the registry
|
|
164
|
+
# the service publishes, which is why these calls take one.
|
|
165
|
+
registry = client.encoders()
|
|
158
166
|
thresholds = client.qc_thresholds()
|
|
167
|
+
# The tile size the model's card states, for the model the file is for.
|
|
168
|
+
patch_um = client.model_card(model_id)["input_spec"]["patch_um"]
|
|
169
|
+
|
|
170
|
+
print(available_encoders(registry)) # the names this package will run today
|
|
171
|
+
encoder = ao.resolve_encoder(registry)
|
|
159
172
|
|
|
160
173
|
report = ao.embed_tiles(
|
|
161
|
-
slide, "sample.npz", mpp=0.499, patch_um=
|
|
174
|
+
slide, "sample.npz", mpp=0.499, patch_um=patch_um, encoder=encoder, thresholds=thresholds
|
|
162
175
|
)
|
|
163
176
|
print(report.rows, "rows of", report.dim, "numbers from", report.encoder.name)
|
|
164
177
|
```
|
|
165
178
|
|
|
166
179
|
`slide` is anything with a `(height, width, 3)` RGB `uint8` shape that can be
|
|
167
180
|
sliced — an array, or a memory-mapped or lazily-read one, so a slide larger than
|
|
168
|
-
memory works: only one crop exists at a time. Crops are taken at
|
|
169
|
-
|
|
170
|
-
predicates as `pack_tiles
|
|
171
|
-
|
|
172
|
-
Every encoder is pinned to
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
The written file carries one extra
|
|
178
|
-
fixed image shipped inside this package, through the same encoder
|
|
179
|
-
|
|
180
|
-
tells a reader whether the file was produced by the model it claims,
|
|
181
|
-
look at a single row.
|
|
182
|
-
|
|
183
|
-
The weights are downloaded
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
own repository declares. The DINOv2 releases carry a 518-px position grid; the
|
|
192
|
-
model was trained on 224-px crops fed to that grid through an interpolated
|
|
193
|
-
position embedding, so that is what this package does too. `resize_px` is
|
|
194
|
-
checked against the pin rather than obeyed, because a vector computed at another
|
|
195
|
-
size is a different measurement that nobody else's numbers can be compared with.
|
|
181
|
+
memory works: only one crop exists at a time. Crops are taken at the tile size
|
|
182
|
+
the model's card states, using the `mpp` you give, and quality-checked with the
|
|
183
|
+
same three predicates as `pack_tiles`.
|
|
184
|
+
|
|
185
|
+
Every encoder is pinned to an exact revision.
|
|
186
|
+
`describe_encoders(registry)` returns the whole registry, including the
|
|
187
|
+
encoders this package refuses to run: a refusal says why, and which encoders
|
|
188
|
+
remain.
|
|
189
|
+
|
|
190
|
+
The written file carries one extra embedding, `calibration`: computed from a
|
|
191
|
+
fixed image shipped inside this package, through the same encoder and revision
|
|
192
|
+
as your rows. Comparing that one embedding against a known
|
|
193
|
+
reference tells a reader whether the file was produced by the model it claims,
|
|
194
|
+
before they look at a single row.
|
|
195
|
+
|
|
196
|
+
The weights are downloaded on first use. Pass `allow_download=False` to
|
|
197
|
+
guarantee no request is made: for the duration of that load, name resolution
|
|
198
|
+
and internet sockets are refused in this process.
|
|
199
|
+
|
|
200
|
+
An encoder runs at the input size the registry pins for it. `resize_px` is
|
|
201
|
+
checked against that pin rather than obeyed, because an embedding computed at
|
|
202
|
+
another size cannot be compared with the served model's.
|
|
203
|
+
|
|
196
204
|
## Where a prediction runs
|
|
197
205
|
|
|
198
|
-
Predicting gene expression runs on the service, and `predict_local` refuses
|
|
199
|
-
by design, and the design is the product rather than a limitation.
|
|
206
|
+
Predicting gene expression runs on the service, and `predict_local` refuses.
|
|
200
207
|
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
its weights never leave it. What crosses between us is a vector from a model
|
|
205
|
-
neither side owns.
|
|
208
|
+
**DeepSpot-H**, the foundation model for H&E images, turns each tile of your
|
|
209
|
+
slide into one embedding. It can run on your machine or on ours. **DeepSpot-M**
|
|
210
|
+
turns those embeddings into spatial gene expression. It only ever runs on ours.
|
|
206
211
|
|
|
207
212
|
```python
|
|
208
213
|
from auroraomics.runtimes.deepspotm import check_archive, describe_models
|
|
209
214
|
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
215
|
+
with ao.Client() as client:
|
|
216
|
+
cards = client.models() # the catalogue is the service's
|
|
217
|
+
describe_models(cards) # what it will run, and on what
|
|
218
|
+
check_archive("sample.zip", model=model_id, cards=cards) # are my tiles the right size?
|
|
213
219
|
```
|
|
214
220
|
|
|
215
221
|
Then submit with the API client and read the result back — the same `.h5ad`
|
|
216
|
-
contract whichever
|
|
222
|
+
contract whichever runtime ran it.
|
|
217
223
|
|
|
218
224
|
## Post-process a result
|
|
219
225
|
|
|
@@ -237,17 +243,19 @@ skipped: a card is the description of a file, and a reader promised a layer
|
|
|
237
243
|
cannot tell a missing one from a model that predicted zeros.
|
|
238
244
|
|
|
239
245
|
The entry this version ships states what the model measured. `var` gains
|
|
240
|
-
`
|
|
241
|
-
|
|
242
|
-
and
|
|
243
|
-
measured becomes `nan` in `X` and in every layer, not zero: zero is a
|
|
244
|
-
measurement, and an unmeasured gene left as one averages, correlates and
|
|
246
|
+
`measured_in_training`, which is `False` for a gene the model never saw
|
|
247
|
+
measured. Such a gene becomes `nan` in `X` and in every layer, not zero: zero
|
|
248
|
+
is a measurement, and an unmeasured gene left as one averages, correlates and
|
|
245
249
|
colours a heat map exactly like a prediction.
|
|
246
250
|
|
|
247
251
|
## Prepare a bulk RNA profile
|
|
248
252
|
|
|
249
253
|
```python
|
|
250
|
-
|
|
254
|
+
with ao.Client() as client:
|
|
255
|
+
report = ao.bulk_rna(
|
|
256
|
+
"sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts",
|
|
257
|
+
card=client.model_card(model_id), resolve=client.resolve_keys,
|
|
258
|
+
)
|
|
251
259
|
print(report.rows, "genes;", f"{report.coverage:.0%} of", report.coverage_of)
|
|
252
260
|
```
|
|
253
261
|
|
|
@@ -267,13 +275,6 @@ your project.
|
|
|
267
275
|
|
|
268
276
|
## Licence
|
|
269
277
|
|
|
270
|
-
The code in this package is licensed
|
|
271
|
-
|
|
272
|
-
which permits use for any purpose that is not commercial. It is the same licence the
|
|
273
|
-
model package this client is built for carries, so installing both puts you under one
|
|
274
|
-
rule rather than two.
|
|
275
|
-
|
|
276
|
-
The model weights are licensed separately by whoever publishes them, and access to them
|
|
277
|
-
may be gated. Read those terms before you use a model: they are not this licence, and a
|
|
278
|
-
permission granted here is not a permission granted there.
|
|
278
|
+
The code in this package is licensed for non-commercial use; its terms are in the
|
|
279
|
+
`LICENSE` file it ships with. Commercial evaluation and use are governed by a written agreement with Aurora.
|
|
279
280
|
|
|
@@ -1,9 +1,13 @@
|
|
|
1
1
|
# auroraomics
|
|
2
2
|
|
|
3
|
-
Virtual spatial transcriptomics from H&E
|
|
3
|
+
Virtual spatial transcriptomics from H&E images.
|
|
4
|
+
|
|
5
|
+
This package is the client of the **Aurora API**, the hosted service that runs
|
|
6
|
+
the prediction: from Python, from a shell with the `auroraomics` command, or
|
|
7
|
+
from an agent through its MCP server.
|
|
4
8
|
|
|
5
9
|
This package holds the pieces of that pipeline that are pure Python, so the
|
|
6
|
-
same code runs on your
|
|
10
|
+
same code runs on your machine and on the service:
|
|
7
11
|
|
|
8
12
|
- **`auroraomics.qc`** — the three patch-quality predicates (foreground, blur,
|
|
9
13
|
stained tissue) applied to every candidate tile before a model sees it, and
|
|
@@ -15,13 +19,13 @@ same code runs on your laptop, in a GPU container and on the service:
|
|
|
15
19
|
batch rather than the whole matrix.
|
|
16
20
|
- **`auroraomics.postprocess`** — apply a model card's `postprocess` entries
|
|
17
21
|
to a result file: a registry of entries, one runner, `h5py` alone, streaming
|
|
18
|
-
one chunk at a time. Ships the unmeasured-gene mask, which
|
|
19
|
-
|
|
22
|
+
one chunk at a time. Ships the unmeasured-gene mask, which blanks the genes
|
|
23
|
+
the model never measured.
|
|
20
24
|
- **`auroraomics.subsample`** — pick the densest contiguous square of spots
|
|
21
|
-
when a slide yields more tiles than
|
|
22
|
-
- **`auroraomics.genes`** —
|
|
23
|
-
|
|
24
|
-
|
|
25
|
+
when a slide yields more tiles than one submission takes.
|
|
26
|
+
- **`auroraomics.genes`** — Ensembl stable gene ids, which is what every
|
|
27
|
+
result's `var` index is keyed by, and the shape a gene the service resolved
|
|
28
|
+
comes back in.
|
|
25
29
|
- **`auroraomics.contracts`** — the shared contract values (container layout,
|
|
26
30
|
input-kind caps, result layout) as data, so nothing here re-types a number
|
|
27
31
|
the service also reads.
|
|
@@ -45,16 +49,18 @@ pip install auroraomics
|
|
|
45
49
|
|
|
46
50
|
That is the whole documented path: open a slide, judge and pack its tiles,
|
|
47
51
|
submit, and read the result. One extra adds local embedding extraction
|
|
48
|
-
(`embed`)
|
|
49
|
-
|
|
50
|
-
the service, not here.
|
|
52
|
+
(`embed`). No extra brings the model: predicting gene expression runs on the
|
|
53
|
+
service, not here.
|
|
51
54
|
|
|
52
55
|
## Pack tiles, then look at the report
|
|
53
56
|
|
|
54
57
|
```python
|
|
55
58
|
import auroraomics as ao
|
|
56
59
|
|
|
57
|
-
|
|
60
|
+
with ao.Client() as client:
|
|
61
|
+
thresholds = client.qc_thresholds() # the service's quality floors, no key needed
|
|
62
|
+
|
|
63
|
+
report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thresholds=thresholds, thumbnail=thumb)
|
|
58
64
|
print(report.written, "tiles kept,", report.rejected, "dropped")
|
|
59
65
|
print(report.rejected_by_reason) # {'foreground_ratio': 12, ...}
|
|
60
66
|
```
|
|
@@ -83,7 +89,8 @@ members do not match the container contract.
|
|
|
83
89
|
ao.write_result(
|
|
84
90
|
"result.h5ad",
|
|
85
91
|
obs=obs, # per-spot columns, as plain arrays
|
|
86
|
-
var=var, # per-gene columns
|
|
92
|
+
var=var, # per-gene columns
|
|
93
|
+
var_index=gene_ids, # the Ensembl gene id of each column
|
|
87
94
|
spatial=coords, # (n_spots, 2) array -> obsm["spatial"]
|
|
88
95
|
x=batches, # an array, or an iterable of row batches
|
|
89
96
|
uns={"model": {"id": "..."}},
|
|
@@ -106,71 +113,69 @@ pip install "auroraomics[embed]"
|
|
|
106
113
|
import auroraomics as ao
|
|
107
114
|
from auroraomics.embed import available_encoders, describe_encoders
|
|
108
115
|
|
|
109
|
-
print(available_encoders()) # ('deepspot-h', 'dinov2-b14', 'midnight')
|
|
110
|
-
|
|
111
116
|
with ao.Client() as client:
|
|
112
117
|
# Both are the service's, so the file records the encoder and the
|
|
113
|
-
# thresholds the model it is submitted to was served with.
|
|
114
|
-
|
|
118
|
+
# thresholds the model it is submitted to was served with. There is no
|
|
119
|
+
# list of encoders inside this package: every name comes from the registry
|
|
120
|
+
# the service publishes, which is why these calls take one.
|
|
121
|
+
registry = client.encoders()
|
|
115
122
|
thresholds = client.qc_thresholds()
|
|
123
|
+
# The tile size the model's card states, for the model the file is for.
|
|
124
|
+
patch_um = client.model_card(model_id)["input_spec"]["patch_um"]
|
|
125
|
+
|
|
126
|
+
print(available_encoders(registry)) # the names this package will run today
|
|
127
|
+
encoder = ao.resolve_encoder(registry)
|
|
116
128
|
|
|
117
129
|
report = ao.embed_tiles(
|
|
118
|
-
slide, "sample.npz", mpp=0.499, patch_um=
|
|
130
|
+
slide, "sample.npz", mpp=0.499, patch_um=patch_um, encoder=encoder, thresholds=thresholds
|
|
119
131
|
)
|
|
120
132
|
print(report.rows, "rows of", report.dim, "numbers from", report.encoder.name)
|
|
121
133
|
```
|
|
122
134
|
|
|
123
135
|
`slide` is anything with a `(height, width, 3)` RGB `uint8` shape that can be
|
|
124
136
|
sliced — an array, or a memory-mapped or lazily-read one, so a slide larger than
|
|
125
|
-
memory works: only one crop exists at a time. Crops are taken at
|
|
126
|
-
|
|
127
|
-
predicates as `pack_tiles
|
|
128
|
-
|
|
129
|
-
Every encoder is pinned to
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
The written file carries one extra
|
|
135
|
-
fixed image shipped inside this package, through the same encoder
|
|
136
|
-
|
|
137
|
-
tells a reader whether the file was produced by the model it claims,
|
|
138
|
-
look at a single row.
|
|
139
|
-
|
|
140
|
-
The weights are downloaded
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
own repository declares. The DINOv2 releases carry a 518-px position grid; the
|
|
149
|
-
model was trained on 224-px crops fed to that grid through an interpolated
|
|
150
|
-
position embedding, so that is what this package does too. `resize_px` is
|
|
151
|
-
checked against the pin rather than obeyed, because a vector computed at another
|
|
152
|
-
size is a different measurement that nobody else's numbers can be compared with.
|
|
137
|
+
memory works: only one crop exists at a time. Crops are taken at the tile size
|
|
138
|
+
the model's card states, using the `mpp` you give, and quality-checked with the
|
|
139
|
+
same three predicates as `pack_tiles`.
|
|
140
|
+
|
|
141
|
+
Every encoder is pinned to an exact revision.
|
|
142
|
+
`describe_encoders(registry)` returns the whole registry, including the
|
|
143
|
+
encoders this package refuses to run: a refusal says why, and which encoders
|
|
144
|
+
remain.
|
|
145
|
+
|
|
146
|
+
The written file carries one extra embedding, `calibration`: computed from a
|
|
147
|
+
fixed image shipped inside this package, through the same encoder and revision
|
|
148
|
+
as your rows. Comparing that one embedding against a known
|
|
149
|
+
reference tells a reader whether the file was produced by the model it claims,
|
|
150
|
+
before they look at a single row.
|
|
151
|
+
|
|
152
|
+
The weights are downloaded on first use. Pass `allow_download=False` to
|
|
153
|
+
guarantee no request is made: for the duration of that load, name resolution
|
|
154
|
+
and internet sockets are refused in this process.
|
|
155
|
+
|
|
156
|
+
An encoder runs at the input size the registry pins for it. `resize_px` is
|
|
157
|
+
checked against that pin rather than obeyed, because an embedding computed at
|
|
158
|
+
another size cannot be compared with the served model's.
|
|
159
|
+
|
|
153
160
|
## Where a prediction runs
|
|
154
161
|
|
|
155
|
-
Predicting gene expression runs on the service, and `predict_local` refuses
|
|
156
|
-
by design, and the design is the product rather than a limitation.
|
|
162
|
+
Predicting gene expression runs on the service, and `predict_local` refuses.
|
|
157
163
|
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
its weights never leave it. What crosses between us is a vector from a model
|
|
162
|
-
neither side owns.
|
|
164
|
+
**DeepSpot-H**, the foundation model for H&E images, turns each tile of your
|
|
165
|
+
slide into one embedding. It can run on your machine or on ours. **DeepSpot-M**
|
|
166
|
+
turns those embeddings into spatial gene expression. It only ever runs on ours.
|
|
163
167
|
|
|
164
168
|
```python
|
|
165
169
|
from auroraomics.runtimes.deepspotm import check_archive, describe_models
|
|
166
170
|
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
171
|
+
with ao.Client() as client:
|
|
172
|
+
cards = client.models() # the catalogue is the service's
|
|
173
|
+
describe_models(cards) # what it will run, and on what
|
|
174
|
+
check_archive("sample.zip", model=model_id, cards=cards) # are my tiles the right size?
|
|
170
175
|
```
|
|
171
176
|
|
|
172
177
|
Then submit with the API client and read the result back — the same `.h5ad`
|
|
173
|
-
contract whichever
|
|
178
|
+
contract whichever runtime ran it.
|
|
174
179
|
|
|
175
180
|
## Post-process a result
|
|
176
181
|
|
|
@@ -194,17 +199,19 @@ skipped: a card is the description of a file, and a reader promised a layer
|
|
|
194
199
|
cannot tell a missing one from a model that predicted zeros.
|
|
195
200
|
|
|
196
201
|
The entry this version ships states what the model measured. `var` gains
|
|
197
|
-
`
|
|
198
|
-
|
|
199
|
-
and
|
|
200
|
-
measured becomes `nan` in `X` and in every layer, not zero: zero is a
|
|
201
|
-
measurement, and an unmeasured gene left as one averages, correlates and
|
|
202
|
+
`measured_in_training`, which is `False` for a gene the model never saw
|
|
203
|
+
measured. Such a gene becomes `nan` in `X` and in every layer, not zero: zero
|
|
204
|
+
is a measurement, and an unmeasured gene left as one averages, correlates and
|
|
202
205
|
colours a heat map exactly like a prediction.
|
|
203
206
|
|
|
204
207
|
## Prepare a bulk RNA profile
|
|
205
208
|
|
|
206
209
|
```python
|
|
207
|
-
|
|
210
|
+
with ao.Client() as client:
|
|
211
|
+
report = ao.bulk_rna(
|
|
212
|
+
"sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts",
|
|
213
|
+
card=client.model_card(model_id), resolve=client.resolve_keys,
|
|
214
|
+
)
|
|
208
215
|
print(report.rows, "genes;", f"{report.coverage:.0%} of", report.coverage_of)
|
|
209
216
|
```
|
|
210
217
|
|
|
@@ -224,13 +231,6 @@ your project.
|
|
|
224
231
|
|
|
225
232
|
## Licence
|
|
226
233
|
|
|
227
|
-
The code in this package is licensed
|
|
228
|
-
|
|
229
|
-
which permits use for any purpose that is not commercial. It is the same licence the
|
|
230
|
-
model package this client is built for carries, so installing both puts you under one
|
|
231
|
-
rule rather than two.
|
|
232
|
-
|
|
233
|
-
The model weights are licensed separately by whoever publishes them, and access to them
|
|
234
|
-
may be gated. Read those terms before you use a model: they are not this licence, and a
|
|
235
|
-
permission granted here is not a permission granted there.
|
|
234
|
+
The code in this package is licensed for non-commercial use; its terms are in the
|
|
235
|
+
`LICENSE` file it ships with. Commercial evaluation and use are governed by a written agreement with Aurora.
|
|
236
236
|
|
|
@@ -4,14 +4,12 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "auroraomics"
|
|
7
|
-
version = "0.1.0.
|
|
8
|
-
description = "
|
|
7
|
+
version = "0.1.0.dev3"
|
|
8
|
+
description = "Prepare H&E slides on your own machine, submit them to Aurora, and read the spatial gene expression it predicts."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
11
|
-
# Non-commercial
|
|
12
|
-
#
|
|
13
|
-
# a licence CLASSIFIER beside an expression, so this line is the whole
|
|
14
|
-
# declaration; the model weights carry their own separate terms.
|
|
11
|
+
# Non-commercial; the terms are in LICENSE. PEP 639 forbids a licence
|
|
12
|
+
# CLASSIFIER beside an expression, so this line is the whole declaration.
|
|
15
13
|
license = "PolyForm-Noncommercial-1.0.0"
|
|
16
14
|
license-files = ["LICENSE"]
|
|
17
15
|
authors = [{ name = "Kalin Nonchev" }]
|
|
@@ -47,9 +45,9 @@ classifiers = [
|
|
|
47
45
|
# all of it. That is the trade, taken deliberately: one install line that works,
|
|
48
46
|
# over a smaller one that cannot reach the end of the page describing it.
|
|
49
47
|
#
|
|
50
|
-
# What is still NOT here is
|
|
51
|
-
# no
|
|
52
|
-
#
|
|
48
|
+
# What is still NOT here is torch. An environment that installs this package
|
|
49
|
+
# with `--no-deps` resolves the same either way; what protects it is the
|
|
50
|
+
# encoder line below.
|
|
53
51
|
dependencies = [
|
|
54
52
|
"numpy>=1.23",
|
|
55
53
|
"h5py>=3.9",
|
|
@@ -79,7 +77,7 @@ dependencies = [
|
|
|
79
77
|
# resolves the model, or the adapter and checkpoint stack that would carry one,
|
|
80
78
|
# and `predict_local` refuses with a message that says so. What runs on a user's
|
|
81
79
|
# own machine is cutting, filtering and EMBEDDING tiles — their pixels stay put,
|
|
82
|
-
# and
|
|
80
|
+
# and an embedding from the encoder is what crosses. So the line is at the
|
|
83
81
|
# ENCODER, not at the size of an install: `embed` below does resolve a tensor
|
|
84
82
|
# runtime, on purpose, and everything on the CLIENT side of the encoder — the
|
|
85
83
|
# HTTP client, the slide reader — sits in the core above, where nobody has to
|
|
@@ -138,6 +136,14 @@ dev = [
|
|
|
138
136
|
# install. The client needs no reference any more — it is the core.
|
|
139
137
|
"auroraomics[mcp]",
|
|
140
138
|
"pytest>=7",
|
|
139
|
+
# The release guard parses version strings with `packaging.version`, at module
|
|
140
|
+
# scope. It has always resolved, because more than one dependency in this list
|
|
141
|
+
# requires packaging — but arriving in a closure is not the same as being
|
|
142
|
+
# declared: the day whichever one carries it stops doing so, the check that
|
|
143
|
+
# decides whether a build may be uploaded stops being a check and becomes a
|
|
144
|
+
# collection error, which reports as one missing optional dependency rather
|
|
145
|
+
# than as a suite that no longer runs.
|
|
146
|
+
"packaging>=22",
|
|
141
147
|
# Coverage is measured on every run, with a floor, because this package is
|
|
142
148
|
# the one that SHIPS: a user installs it and calls it, so a branch nothing
|
|
143
149
|
# here exercises is a branch discovered in the field. The floor sat at
|
|
@@ -1,17 +1,22 @@
|
|
|
1
|
-
"""Virtual spatial transcriptomics from H&E
|
|
1
|
+
"""Virtual spatial transcriptomics from H&E images.
|
|
2
2
|
|
|
3
3
|
**One journey, and one word chooses which.** Send the tiles, or send only the
|
|
4
4
|
embeddings computed from them; everything either side of that word is the same:
|
|
5
5
|
|
|
6
|
-
|
|
6
|
+
```python
|
|
7
|
+
import auroraomics as ao
|
|
7
8
|
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
9
|
+
client = ao.Client()
|
|
10
|
+
slide = ao.open_slide("slide.svs")
|
|
11
|
+
report = ao.pack_tiles(slide, "sample.zip", mpp=slide.mpp, patch_um=55,
|
|
12
|
+
thresholds=client.qc_thresholds())
|
|
13
|
+
job = client.predict(model=model_id, observations={"patches": report.path})
|
|
14
|
+
job.wait().download("result.h5ad")
|
|
15
|
+
```
|
|
12
16
|
|
|
13
|
-
Write ``ao.embed_tiles`` in place of ``ao.pack_tiles
|
|
14
|
-
|
|
17
|
+
Write ``ao.embed_tiles`` in place of ``ao.pack_tiles``, with the encoder from
|
|
18
|
+
``ao.resolve_encoder(client.encoders(), "deepspot-h")``, and the slide's pixels
|
|
19
|
+
never leave the machine: what crosses is a short row of numbers per tile.
|
|
15
20
|
|
|
16
21
|
The names in that example, and the handful beside them a reader writes by hand,
|
|
17
22
|
are what this module exports. Everything else in the package is still here and
|
|
@@ -19,7 +24,7 @@ still importable from the module that owns it — the front door is narrow on
|
|
|
19
24
|
purpose, because a name a caller never types is a name they have to read past.
|
|
20
25
|
|
|
21
26
|
The pieces of that pipeline that are pure Python live here, so the same code
|
|
22
|
-
runs on
|
|
27
|
+
runs on your machine and on the hosted service:
|
|
23
28
|
|
|
24
29
|
* [auroraomics.qc][] — the three patch-quality predicates and the cascade
|
|
25
30
|
that decides which tiles are worth predicting on.
|
|
@@ -33,7 +38,7 @@ runs on a laptop, inside a GPU image and on the hosted service:
|
|
|
33
38
|
* [auroraomics.bulk][] — prepare a bulk RNA profile as the ``bulk_rna``
|
|
34
39
|
covariate a prediction accepts: one table, Ensembl ids, a declared unit.
|
|
35
40
|
* [auroraomics.subsample][] — choose the densest contiguous square when a
|
|
36
|
-
slide yields more tiles than
|
|
41
|
+
slide yields more tiles than one submission takes.
|
|
37
42
|
* [auroraomics.genes][] — the model family's symbols mapped to Ensembl gene
|
|
38
43
|
ids, which is what every result's ``var`` index is keyed by.
|
|
39
44
|
* [auroraomics.errors][] — the one root every failure in this package
|
|
@@ -3,6 +3,6 @@
|
|
|
3
3
|
"files": {
|
|
4
4
|
"public-api-counters.tokens.json": "4f5dd14ce76f7001d36940dfc1d1f49feac9ec1be2e7646d47b29ab21c124a8f",
|
|
5
5
|
"public-api-input-kinds.tokens.json": "1c80e2637d27f2491ea178e2fa87e2a435640e5445791b2fd20adc53a574e54d",
|
|
6
|
-
"public-api.tokens.json": "
|
|
6
|
+
"public-api.tokens.json": "75ac637df4a6b0258485e51dccbc05445b73bb42449e11b4be6e98ffb67c0fbb"
|
|
7
7
|
}
|
|
8
8
|
}
|
{auroraomics-0.1.0.dev2 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/public-api.tokens.json
RENAMED
|
@@ -203,7 +203,9 @@
|
|
|
203
203
|
"revoked"
|
|
204
204
|
],
|
|
205
205
|
"requestPath": "/api-access",
|
|
206
|
-
"termsVersion": "2026-09-
|
|
206
|
+
"termsVersion": "2026-09-27",
|
|
207
|
+
"termsPath": "/terms",
|
|
208
|
+
"termsDataUseSectionId": "model-development",
|
|
207
209
|
"request": {
|
|
208
210
|
"nameMaxChars": 120,
|
|
209
211
|
"institutionMaxChars": 200,
|
|
@@ -265,6 +267,16 @@
|
|
|
265
267
|
"commit",
|
|
266
268
|
"qc"
|
|
267
269
|
],
|
|
270
|
+
"required": [
|
|
271
|
+
"emb",
|
|
272
|
+
"px_col",
|
|
273
|
+
"px_row",
|
|
274
|
+
"calibration",
|
|
275
|
+
"encoder",
|
|
276
|
+
"revision",
|
|
277
|
+
"patch_px",
|
|
278
|
+
"pooling"
|
|
279
|
+
],
|
|
268
280
|
"dtypes": [
|
|
269
281
|
"float16",
|
|
270
282
|
"float32"
|
|
@@ -55,6 +55,19 @@ into this wheel, so this is where the package can reach the name today. One
|
|
|
55
55
|
site to move, not five.
|
|
56
56
|
"""
|
|
57
57
|
|
|
58
|
+
PREDICTION_COMPONENT = "DeepSpot-M"
|
|
59
|
+
"""The step that turns embeddings into expression, named as a reader meets it.
|
|
60
|
+
|
|
61
|
+
The model's display name: the brand contract's `modelDisplayName`, which is
|
|
62
|
+
also this step's key in its `componentNames`, hyphenated like
|
|
63
|
+
`EMBEDDING_COMPONENT`. It is never an identifier: the package a runtime
|
|
64
|
+
installs and the repository it names keep their own spellings.
|
|
65
|
+
|
|
66
|
+
Typed here for the reason the constant above is, and held to the same fixture
|
|
67
|
+
by the same test, so the refusal that names both steps cannot spell one of
|
|
68
|
+
them the old way.
|
|
69
|
+
"""
|
|
70
|
+
|
|
58
71
|
|
|
59
72
|
def bounded(value: Any, *, limit: int = MAX_ECHOED_INPUT) -> str:
|
|
60
73
|
"""``value`` as text, cut to ``limit`` with a marker.
|