auroraomics 0.1.0.dev1__tar.gz → 0.1.0.dev3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {auroraomics-0.1.0.dev1/src/auroraomics.egg-info → auroraomics-0.1.0.dev3}/PKG-INFO +76 -74
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/README.md +73 -72
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/pyproject.toml +25 -19
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/__init__.py +20 -15
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/MANIFEST.json +1 -1
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/public-api.tokens.json +21 -3
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_text.py +14 -1
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/bulk.py +28 -34
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/cli.py +329 -93
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/__init__.py +8 -4
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/_generated.py +28 -31
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/_http.py +4 -4
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/api.py +135 -61
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/credentials.py +14 -3
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/errors.py +27 -15
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/jobs.py +13 -15
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/client/uploads.py +1 -2
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/contracts.py +32 -30
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/embed.py +452 -192
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/errors.py +7 -5
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/genes.py +10 -21
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/h5ad.py +8 -17
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/mcp/server.py +5 -5
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/mcp/tools.py +2 -3
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/pack.py +141 -57
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/postprocess.py +4 -7
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/qc.py +16 -15
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/runtimes/__init__.py +1 -9
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/runtimes/deepspotm.py +250 -124
- auroraomics-0.1.0.dev3/src/auroraomics/slide.py +784 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/spatial.py +9 -2
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/subsample.py +1 -2
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3/src/auroraomics.egg-info}/PKG-INFO +76 -74
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/requires.txt +1 -0
- auroraomics-0.1.0.dev1/src/auroraomics/slide.py +0 -325
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/LICENSE +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/MANIFEST.in +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/setup.cfg +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_assets/ASSETS.json +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_assets/README.md +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_assets/calibration-tile.png +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/public-api-counters.tokens.json +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/_contracts/public-api-input-kinds.tokens.json +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/mcp/__init__.py +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics/py.typed +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/SOURCES.txt +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/dependency_links.txt +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/entry_points.txt +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev3}/src/auroraomics.egg-info/top_level.txt +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: auroraomics
|
|
3
|
-
Version: 0.1.0.
|
|
4
|
-
Summary:
|
|
3
|
+
Version: 0.1.0.dev3
|
|
4
|
+
Summary: Prepare H&E slides on your own machine, submit them to Aurora, and read the spatial gene expression it predicts.
|
|
5
5
|
Author: Kalin Nonchev
|
|
6
6
|
License-Expression: PolyForm-Noncommercial-1.0.0
|
|
7
7
|
Keywords: histopathology,spatial-transcriptomics,h5ad,anndata,whole-slide-image
|
|
@@ -33,6 +33,7 @@ Requires-Dist: mcp<3,>=2; extra == "mcp"
|
|
|
33
33
|
Provides-Extra: dev
|
|
34
34
|
Requires-Dist: auroraomics[mcp]; extra == "dev"
|
|
35
35
|
Requires-Dist: pytest>=7; extra == "dev"
|
|
36
|
+
Requires-Dist: packaging>=22; extra == "dev"
|
|
36
37
|
Requires-Dist: pytest-cov>=5; extra == "dev"
|
|
37
38
|
Requires-Dist: setuptools>=77; extra == "dev"
|
|
38
39
|
Requires-Dist: wheel; extra == "dev"
|
|
@@ -43,10 +44,14 @@ Dynamic: license-file
|
|
|
43
44
|
|
|
44
45
|
# auroraomics
|
|
45
46
|
|
|
46
|
-
Virtual spatial transcriptomics from H&E
|
|
47
|
+
Virtual spatial transcriptomics from H&E images.
|
|
48
|
+
|
|
49
|
+
This package is the client of the **Aurora API**, the hosted service that runs
|
|
50
|
+
the prediction: from Python, from a shell with the `auroraomics` command, or
|
|
51
|
+
from an agent through its MCP server.
|
|
47
52
|
|
|
48
53
|
This package holds the pieces of that pipeline that are pure Python, so the
|
|
49
|
-
same code runs on your
|
|
54
|
+
same code runs on your machine and on the service:
|
|
50
55
|
|
|
51
56
|
- **`auroraomics.qc`** — the three patch-quality predicates (foreground, blur,
|
|
52
57
|
stained tissue) applied to every candidate tile before a model sees it, and
|
|
@@ -58,13 +63,13 @@ same code runs on your laptop, in a GPU container and on the service:
|
|
|
58
63
|
batch rather than the whole matrix.
|
|
59
64
|
- **`auroraomics.postprocess`** — apply a model card's `postprocess` entries
|
|
60
65
|
to a result file: a registry of entries, one runner, `h5py` alone, streaming
|
|
61
|
-
one chunk at a time. Ships the unmeasured-gene mask, which
|
|
62
|
-
|
|
66
|
+
one chunk at a time. Ships the unmeasured-gene mask, which blanks the genes
|
|
67
|
+
the model never measured.
|
|
63
68
|
- **`auroraomics.subsample`** — pick the densest contiguous square of spots
|
|
64
|
-
when a slide yields more tiles than
|
|
65
|
-
- **`auroraomics.genes`** —
|
|
66
|
-
|
|
67
|
-
|
|
69
|
+
when a slide yields more tiles than one submission takes.
|
|
70
|
+
- **`auroraomics.genes`** — Ensembl stable gene ids, which is what every
|
|
71
|
+
result's `var` index is keyed by, and the shape a gene the service resolved
|
|
72
|
+
comes back in.
|
|
68
73
|
- **`auroraomics.contracts`** — the shared contract values (container layout,
|
|
69
74
|
input-kind caps, result layout) as data, so nothing here re-types a number
|
|
70
75
|
the service also reads.
|
|
@@ -88,16 +93,18 @@ pip install auroraomics
|
|
|
88
93
|
|
|
89
94
|
That is the whole documented path: open a slide, judge and pack its tiles,
|
|
90
95
|
submit, and read the result. One extra adds local embedding extraction
|
|
91
|
-
(`embed`)
|
|
92
|
-
|
|
93
|
-
the service, not here.
|
|
96
|
+
(`embed`). No extra brings the model: predicting gene expression runs on the
|
|
97
|
+
service, not here.
|
|
94
98
|
|
|
95
99
|
## Pack tiles, then look at the report
|
|
96
100
|
|
|
97
101
|
```python
|
|
98
102
|
import auroraomics as ao
|
|
99
103
|
|
|
100
|
-
|
|
104
|
+
with ao.Client() as client:
|
|
105
|
+
thresholds = client.qc_thresholds() # the service's quality floors, no key needed
|
|
106
|
+
|
|
107
|
+
report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thresholds=thresholds, thumbnail=thumb)
|
|
101
108
|
print(report.written, "tiles kept,", report.rejected, "dropped")
|
|
102
109
|
print(report.rejected_by_reason) # {'foreground_ratio': 12, ...}
|
|
103
110
|
```
|
|
@@ -126,7 +133,8 @@ members do not match the container contract.
|
|
|
126
133
|
ao.write_result(
|
|
127
134
|
"result.h5ad",
|
|
128
135
|
obs=obs, # per-spot columns, as plain arrays
|
|
129
|
-
var=var, # per-gene columns
|
|
136
|
+
var=var, # per-gene columns
|
|
137
|
+
var_index=gene_ids, # the Ensembl gene id of each column
|
|
130
138
|
spatial=coords, # (n_spots, 2) array -> obsm["spatial"]
|
|
131
139
|
x=batches, # an array, or an iterable of row batches
|
|
132
140
|
uns={"model": {"id": "..."}},
|
|
@@ -149,70 +157,69 @@ pip install "auroraomics[embed]"
|
|
|
149
157
|
import auroraomics as ao
|
|
150
158
|
from auroraomics.embed import available_encoders, describe_encoders
|
|
151
159
|
|
|
152
|
-
print(available_encoders()) # ('deepspot-h', 'dinov2-b14', 'midnight')
|
|
153
|
-
|
|
154
160
|
with ao.Client() as client:
|
|
155
161
|
# Both are the service's, so the file records the encoder and the
|
|
156
|
-
# thresholds the model it is submitted to was served with.
|
|
157
|
-
|
|
162
|
+
# thresholds the model it is submitted to was served with. There is no
|
|
163
|
+
# list of encoders inside this package: every name comes from the registry
|
|
164
|
+
# the service publishes, which is why these calls take one.
|
|
165
|
+
registry = client.encoders()
|
|
158
166
|
thresholds = client.qc_thresholds()
|
|
167
|
+
# The tile size the model's card states, for the model the file is for.
|
|
168
|
+
patch_um = client.model_card(model_id)["input_spec"]["patch_um"]
|
|
169
|
+
|
|
170
|
+
print(available_encoders(registry)) # the names this package will run today
|
|
171
|
+
encoder = ao.resolve_encoder(registry)
|
|
159
172
|
|
|
160
173
|
report = ao.embed_tiles(
|
|
161
|
-
slide, "sample.npz", mpp=0.499, patch_um=
|
|
174
|
+
slide, "sample.npz", mpp=0.499, patch_um=patch_um, encoder=encoder, thresholds=thresholds
|
|
162
175
|
)
|
|
163
176
|
print(report.rows, "rows of", report.dim, "numbers from", report.encoder.name)
|
|
164
177
|
```
|
|
165
178
|
|
|
166
179
|
`slide` is anything with a `(height, width, 3)` RGB `uint8` shape that can be
|
|
167
180
|
sliced — an array, or a memory-mapped or lazily-read one, so a slide larger than
|
|
168
|
-
memory works: only one crop exists at a time. Crops are taken at
|
|
169
|
-
|
|
170
|
-
predicates as `pack_tiles
|
|
171
|
-
|
|
172
|
-
Every encoder is pinned to
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
The written file carries one extra
|
|
178
|
-
fixed image shipped inside this package, through the same encoder
|
|
179
|
-
|
|
180
|
-
tells a reader whether the file was produced by the model it claims,
|
|
181
|
-
look at a single row.
|
|
182
|
-
|
|
183
|
-
The weights are downloaded
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
own repository declares. The DINOv2 releases carry a 518-px position grid; the
|
|
192
|
-
model was trained on 224-px crops fed to that grid through an interpolated
|
|
193
|
-
position embedding, so that is what this package does too. `resize_px` is
|
|
194
|
-
checked against the pin rather than obeyed, because a vector computed at another
|
|
195
|
-
size is a different measurement that nobody else's numbers can be compared with.
|
|
181
|
+
memory works: only one crop exists at a time. Crops are taken at the tile size
|
|
182
|
+
the model's card states, using the `mpp` you give, and quality-checked with the
|
|
183
|
+
same three predicates as `pack_tiles`.
|
|
184
|
+
|
|
185
|
+
Every encoder is pinned to an exact revision.
|
|
186
|
+
`describe_encoders(registry)` returns the whole registry, including the
|
|
187
|
+
encoders this package refuses to run: a refusal says why, and which encoders
|
|
188
|
+
remain.
|
|
189
|
+
|
|
190
|
+
The written file carries one extra embedding, `calibration`: computed from a
|
|
191
|
+
fixed image shipped inside this package, through the same encoder and revision
|
|
192
|
+
as your rows. Comparing that one embedding against a known
|
|
193
|
+
reference tells a reader whether the file was produced by the model it claims,
|
|
194
|
+
before they look at a single row.
|
|
195
|
+
|
|
196
|
+
The weights are downloaded on first use. Pass `allow_download=False` to
|
|
197
|
+
guarantee no request is made: for the duration of that load, name resolution
|
|
198
|
+
and internet sockets are refused in this process.
|
|
199
|
+
|
|
200
|
+
An encoder runs at the input size the registry pins for it. `resize_px` is
|
|
201
|
+
checked against that pin rather than obeyed, because an embedding computed at
|
|
202
|
+
another size cannot be compared with the served model's.
|
|
203
|
+
|
|
196
204
|
## Where a prediction runs
|
|
197
205
|
|
|
198
|
-
Predicting gene expression runs on the service, and `predict_local` refuses
|
|
199
|
-
by design, and the design is the product rather than a limitation.
|
|
206
|
+
Predicting gene expression runs on the service, and `predict_local` refuses.
|
|
200
207
|
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
its weights never leave it. What crosses between us is a vector from a model
|
|
205
|
-
neither side owns.
|
|
208
|
+
**DeepSpot-H**, the foundation model for H&E images, turns each tile of your
|
|
209
|
+
slide into one embedding. It can run on your machine or on ours. **DeepSpot-M**
|
|
210
|
+
turns those embeddings into spatial gene expression. It only ever runs on ours.
|
|
206
211
|
|
|
207
212
|
```python
|
|
208
|
-
from auroraomics.runtimes.deepspotm import
|
|
213
|
+
from auroraomics.runtimes.deepspotm import check_archive, describe_models
|
|
209
214
|
|
|
210
|
-
|
|
211
|
-
|
|
215
|
+
with ao.Client() as client:
|
|
216
|
+
cards = client.models() # the catalogue is the service's
|
|
217
|
+
describe_models(cards) # what it will run, and on what
|
|
218
|
+
check_archive("sample.zip", model=model_id, cards=cards) # are my tiles the right size?
|
|
212
219
|
```
|
|
213
220
|
|
|
214
221
|
Then submit with the API client and read the result back — the same `.h5ad`
|
|
215
|
-
contract whichever
|
|
222
|
+
contract whichever runtime ran it.
|
|
216
223
|
|
|
217
224
|
## Post-process a result
|
|
218
225
|
|
|
@@ -236,17 +243,19 @@ skipped: a card is the description of a file, and a reader promised a layer
|
|
|
236
243
|
cannot tell a missing one from a model that predicted zeros.
|
|
237
244
|
|
|
238
245
|
The entry this version ships states what the model measured. `var` gains
|
|
239
|
-
`
|
|
240
|
-
|
|
241
|
-
and
|
|
242
|
-
measured becomes `nan` in `X` and in every layer, not zero: zero is a
|
|
243
|
-
measurement, and an unmeasured gene left as one averages, correlates and
|
|
246
|
+
`measured_in_training`, which is `False` for a gene the model never saw
|
|
247
|
+
measured. Such a gene becomes `nan` in `X` and in every layer, not zero: zero
|
|
248
|
+
is a measurement, and an unmeasured gene left as one averages, correlates and
|
|
244
249
|
colours a heat map exactly like a prediction.
|
|
245
250
|
|
|
246
251
|
## Prepare a bulk RNA profile
|
|
247
252
|
|
|
248
253
|
```python
|
|
249
|
-
|
|
254
|
+
with ao.Client() as client:
|
|
255
|
+
report = ao.bulk_rna(
|
|
256
|
+
"sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts",
|
|
257
|
+
card=client.model_card(model_id), resolve=client.resolve_keys,
|
|
258
|
+
)
|
|
250
259
|
print(report.rows, "genes;", f"{report.coverage:.0%} of", report.coverage_of)
|
|
251
260
|
```
|
|
252
261
|
|
|
@@ -266,13 +275,6 @@ your project.
|
|
|
266
275
|
|
|
267
276
|
## Licence
|
|
268
277
|
|
|
269
|
-
The code in this package is licensed
|
|
270
|
-
|
|
271
|
-
which permits use for any purpose that is not commercial. It is the same licence the
|
|
272
|
-
model package this client is built for carries, so installing both puts you under one
|
|
273
|
-
rule rather than two.
|
|
274
|
-
|
|
275
|
-
The model weights are licensed separately by whoever publishes them, and access to them
|
|
276
|
-
may be gated. Read those terms before you use a model: they are not this licence, and a
|
|
277
|
-
permission granted here is not a permission granted there.
|
|
278
|
+
The code in this package is licensed for non-commercial use; its terms are in the
|
|
279
|
+
`LICENSE` file it ships with. Commercial evaluation and use are governed by a written agreement with Aurora.
|
|
278
280
|
|
|
@@ -1,9 +1,13 @@
|
|
|
1
1
|
# auroraomics
|
|
2
2
|
|
|
3
|
-
Virtual spatial transcriptomics from H&E
|
|
3
|
+
Virtual spatial transcriptomics from H&E images.
|
|
4
|
+
|
|
5
|
+
This package is the client of the **Aurora API**, the hosted service that runs
|
|
6
|
+
the prediction: from Python, from a shell with the `auroraomics` command, or
|
|
7
|
+
from an agent through its MCP server.
|
|
4
8
|
|
|
5
9
|
This package holds the pieces of that pipeline that are pure Python, so the
|
|
6
|
-
same code runs on your
|
|
10
|
+
same code runs on your machine and on the service:
|
|
7
11
|
|
|
8
12
|
- **`auroraomics.qc`** — the three patch-quality predicates (foreground, blur,
|
|
9
13
|
stained tissue) applied to every candidate tile before a model sees it, and
|
|
@@ -15,13 +19,13 @@ same code runs on your laptop, in a GPU container and on the service:
|
|
|
15
19
|
batch rather than the whole matrix.
|
|
16
20
|
- **`auroraomics.postprocess`** — apply a model card's `postprocess` entries
|
|
17
21
|
to a result file: a registry of entries, one runner, `h5py` alone, streaming
|
|
18
|
-
one chunk at a time. Ships the unmeasured-gene mask, which
|
|
19
|
-
|
|
22
|
+
one chunk at a time. Ships the unmeasured-gene mask, which blanks the genes
|
|
23
|
+
the model never measured.
|
|
20
24
|
- **`auroraomics.subsample`** — pick the densest contiguous square of spots
|
|
21
|
-
when a slide yields more tiles than
|
|
22
|
-
- **`auroraomics.genes`** —
|
|
23
|
-
|
|
24
|
-
|
|
25
|
+
when a slide yields more tiles than one submission takes.
|
|
26
|
+
- **`auroraomics.genes`** — Ensembl stable gene ids, which is what every
|
|
27
|
+
result's `var` index is keyed by, and the shape a gene the service resolved
|
|
28
|
+
comes back in.
|
|
25
29
|
- **`auroraomics.contracts`** — the shared contract values (container layout,
|
|
26
30
|
input-kind caps, result layout) as data, so nothing here re-types a number
|
|
27
31
|
the service also reads.
|
|
@@ -45,16 +49,18 @@ pip install auroraomics
|
|
|
45
49
|
|
|
46
50
|
That is the whole documented path: open a slide, judge and pack its tiles,
|
|
47
51
|
submit, and read the result. One extra adds local embedding extraction
|
|
48
|
-
(`embed`)
|
|
49
|
-
|
|
50
|
-
the service, not here.
|
|
52
|
+
(`embed`). No extra brings the model: predicting gene expression runs on the
|
|
53
|
+
service, not here.
|
|
51
54
|
|
|
52
55
|
## Pack tiles, then look at the report
|
|
53
56
|
|
|
54
57
|
```python
|
|
55
58
|
import auroraomics as ao
|
|
56
59
|
|
|
57
|
-
|
|
60
|
+
with ao.Client() as client:
|
|
61
|
+
thresholds = client.qc_thresholds() # the service's quality floors, no key needed
|
|
62
|
+
|
|
63
|
+
report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thresholds=thresholds, thumbnail=thumb)
|
|
58
64
|
print(report.written, "tiles kept,", report.rejected, "dropped")
|
|
59
65
|
print(report.rejected_by_reason) # {'foreground_ratio': 12, ...}
|
|
60
66
|
```
|
|
@@ -83,7 +89,8 @@ members do not match the container contract.
|
|
|
83
89
|
ao.write_result(
|
|
84
90
|
"result.h5ad",
|
|
85
91
|
obs=obs, # per-spot columns, as plain arrays
|
|
86
|
-
var=var, # per-gene columns
|
|
92
|
+
var=var, # per-gene columns
|
|
93
|
+
var_index=gene_ids, # the Ensembl gene id of each column
|
|
87
94
|
spatial=coords, # (n_spots, 2) array -> obsm["spatial"]
|
|
88
95
|
x=batches, # an array, or an iterable of row batches
|
|
89
96
|
uns={"model": {"id": "..."}},
|
|
@@ -106,70 +113,69 @@ pip install "auroraomics[embed]"
|
|
|
106
113
|
import auroraomics as ao
|
|
107
114
|
from auroraomics.embed import available_encoders, describe_encoders
|
|
108
115
|
|
|
109
|
-
print(available_encoders()) # ('deepspot-h', 'dinov2-b14', 'midnight')
|
|
110
|
-
|
|
111
116
|
with ao.Client() as client:
|
|
112
117
|
# Both are the service's, so the file records the encoder and the
|
|
113
|
-
# thresholds the model it is submitted to was served with.
|
|
114
|
-
|
|
118
|
+
# thresholds the model it is submitted to was served with. There is no
|
|
119
|
+
# list of encoders inside this package: every name comes from the registry
|
|
120
|
+
# the service publishes, which is why these calls take one.
|
|
121
|
+
registry = client.encoders()
|
|
115
122
|
thresholds = client.qc_thresholds()
|
|
123
|
+
# The tile size the model's card states, for the model the file is for.
|
|
124
|
+
patch_um = client.model_card(model_id)["input_spec"]["patch_um"]
|
|
125
|
+
|
|
126
|
+
print(available_encoders(registry)) # the names this package will run today
|
|
127
|
+
encoder = ao.resolve_encoder(registry)
|
|
116
128
|
|
|
117
129
|
report = ao.embed_tiles(
|
|
118
|
-
slide, "sample.npz", mpp=0.499, patch_um=
|
|
130
|
+
slide, "sample.npz", mpp=0.499, patch_um=patch_um, encoder=encoder, thresholds=thresholds
|
|
119
131
|
)
|
|
120
132
|
print(report.rows, "rows of", report.dim, "numbers from", report.encoder.name)
|
|
121
133
|
```
|
|
122
134
|
|
|
123
135
|
`slide` is anything with a `(height, width, 3)` RGB `uint8` shape that can be
|
|
124
136
|
sliced — an array, or a memory-mapped or lazily-read one, so a slide larger than
|
|
125
|
-
memory works: only one crop exists at a time. Crops are taken at
|
|
126
|
-
|
|
127
|
-
predicates as `pack_tiles
|
|
128
|
-
|
|
129
|
-
Every encoder is pinned to
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
The written file carries one extra
|
|
135
|
-
fixed image shipped inside this package, through the same encoder
|
|
136
|
-
|
|
137
|
-
tells a reader whether the file was produced by the model it claims,
|
|
138
|
-
look at a single row.
|
|
139
|
-
|
|
140
|
-
The weights are downloaded
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
own repository declares. The DINOv2 releases carry a 518-px position grid; the
|
|
149
|
-
model was trained on 224-px crops fed to that grid through an interpolated
|
|
150
|
-
position embedding, so that is what this package does too. `resize_px` is
|
|
151
|
-
checked against the pin rather than obeyed, because a vector computed at another
|
|
152
|
-
size is a different measurement that nobody else's numbers can be compared with.
|
|
137
|
+
memory works: only one crop exists at a time. Crops are taken at the tile size
|
|
138
|
+
the model's card states, using the `mpp` you give, and quality-checked with the
|
|
139
|
+
same three predicates as `pack_tiles`.
|
|
140
|
+
|
|
141
|
+
Every encoder is pinned to an exact revision.
|
|
142
|
+
`describe_encoders(registry)` returns the whole registry, including the
|
|
143
|
+
encoders this package refuses to run: a refusal says why, and which encoders
|
|
144
|
+
remain.
|
|
145
|
+
|
|
146
|
+
The written file carries one extra embedding, `calibration`: computed from a
|
|
147
|
+
fixed image shipped inside this package, through the same encoder and revision
|
|
148
|
+
as your rows. Comparing that one embedding against a known
|
|
149
|
+
reference tells a reader whether the file was produced by the model it claims,
|
|
150
|
+
before they look at a single row.
|
|
151
|
+
|
|
152
|
+
The weights are downloaded on first use. Pass `allow_download=False` to
|
|
153
|
+
guarantee no request is made: for the duration of that load, name resolution
|
|
154
|
+
and internet sockets are refused in this process.
|
|
155
|
+
|
|
156
|
+
An encoder runs at the input size the registry pins for it. `resize_px` is
|
|
157
|
+
checked against that pin rather than obeyed, because an embedding computed at
|
|
158
|
+
another size cannot be compared with the served model's.
|
|
159
|
+
|
|
153
160
|
## Where a prediction runs
|
|
154
161
|
|
|
155
|
-
Predicting gene expression runs on the service, and `predict_local` refuses
|
|
156
|
-
by design, and the design is the product rather than a limitation.
|
|
162
|
+
Predicting gene expression runs on the service, and `predict_local` refuses.
|
|
157
163
|
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
its weights never leave it. What crosses between us is a vector from a model
|
|
162
|
-
neither side owns.
|
|
164
|
+
**DeepSpot-H**, the foundation model for H&E images, turns each tile of your
|
|
165
|
+
slide into one embedding. It can run on your machine or on ours. **DeepSpot-M**
|
|
166
|
+
turns those embeddings into spatial gene expression. It only ever runs on ours.
|
|
163
167
|
|
|
164
168
|
```python
|
|
165
|
-
from auroraomics.runtimes.deepspotm import
|
|
169
|
+
from auroraomics.runtimes.deepspotm import check_archive, describe_models
|
|
166
170
|
|
|
167
|
-
|
|
168
|
-
|
|
171
|
+
with ao.Client() as client:
|
|
172
|
+
cards = client.models() # the catalogue is the service's
|
|
173
|
+
describe_models(cards) # what it will run, and on what
|
|
174
|
+
check_archive("sample.zip", model=model_id, cards=cards) # are my tiles the right size?
|
|
169
175
|
```
|
|
170
176
|
|
|
171
177
|
Then submit with the API client and read the result back — the same `.h5ad`
|
|
172
|
-
contract whichever
|
|
178
|
+
contract whichever runtime ran it.
|
|
173
179
|
|
|
174
180
|
## Post-process a result
|
|
175
181
|
|
|
@@ -193,17 +199,19 @@ skipped: a card is the description of a file, and a reader promised a layer
|
|
|
193
199
|
cannot tell a missing one from a model that predicted zeros.
|
|
194
200
|
|
|
195
201
|
The entry this version ships states what the model measured. `var` gains
|
|
196
|
-
`
|
|
197
|
-
|
|
198
|
-
and
|
|
199
|
-
measured becomes `nan` in `X` and in every layer, not zero: zero is a
|
|
200
|
-
measurement, and an unmeasured gene left as one averages, correlates and
|
|
202
|
+
`measured_in_training`, which is `False` for a gene the model never saw
|
|
203
|
+
measured. Such a gene becomes `nan` in `X` and in every layer, not zero: zero
|
|
204
|
+
is a measurement, and an unmeasured gene left as one averages, correlates and
|
|
201
205
|
colours a heat map exactly like a prediction.
|
|
202
206
|
|
|
203
207
|
## Prepare a bulk RNA profile
|
|
204
208
|
|
|
205
209
|
```python
|
|
206
|
-
|
|
210
|
+
with ao.Client() as client:
|
|
211
|
+
report = ao.bulk_rna(
|
|
212
|
+
"sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts",
|
|
213
|
+
card=client.model_card(model_id), resolve=client.resolve_keys,
|
|
214
|
+
)
|
|
207
215
|
print(report.rows, "genes;", f"{report.coverage:.0%} of", report.coverage_of)
|
|
208
216
|
```
|
|
209
217
|
|
|
@@ -223,13 +231,6 @@ your project.
|
|
|
223
231
|
|
|
224
232
|
## Licence
|
|
225
233
|
|
|
226
|
-
The code in this package is licensed
|
|
227
|
-
|
|
228
|
-
which permits use for any purpose that is not commercial. It is the same licence the
|
|
229
|
-
model package this client is built for carries, so installing both puts you under one
|
|
230
|
-
rule rather than two.
|
|
231
|
-
|
|
232
|
-
The model weights are licensed separately by whoever publishes them, and access to them
|
|
233
|
-
may be gated. Read those terms before you use a model: they are not this licence, and a
|
|
234
|
-
permission granted here is not a permission granted there.
|
|
234
|
+
The code in this package is licensed for non-commercial use; its terms are in the
|
|
235
|
+
`LICENSE` file it ships with. Commercial evaluation and use are governed by a written agreement with Aurora.
|
|
235
236
|
|
|
@@ -4,14 +4,12 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "auroraomics"
|
|
7
|
-
version = "0.1.0.
|
|
8
|
-
description = "
|
|
7
|
+
version = "0.1.0.dev3"
|
|
8
|
+
description = "Prepare H&E slides on your own machine, submit them to Aurora, and read the spatial gene expression it predicts."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
11
|
-
# Non-commercial
|
|
12
|
-
#
|
|
13
|
-
# a licence CLASSIFIER beside an expression, so this line is the whole
|
|
14
|
-
# declaration; the model weights carry their own separate terms.
|
|
11
|
+
# Non-commercial; the terms are in LICENSE. PEP 639 forbids a licence
|
|
12
|
+
# CLASSIFIER beside an expression, so this line is the whole declaration.
|
|
15
13
|
license = "PolyForm-Noncommercial-1.0.0"
|
|
16
14
|
license-files = ["LICENSE"]
|
|
17
15
|
authors = [{ name = "Kalin Nonchev" }]
|
|
@@ -31,7 +29,7 @@ classifiers = [
|
|
|
31
29
|
# install line the quickstart prints is an install line that can do what the
|
|
32
30
|
# quickstart says.
|
|
33
31
|
#
|
|
34
|
-
# It was four names
|
|
32
|
+
# It was four names once, with the API client and the slide reader behind
|
|
35
33
|
# extras of their own. The comment on the extras below has always ended "the
|
|
36
34
|
# line is at the ENCODER, not at the size of an install"; these lists did not
|
|
37
35
|
# say it. They put a half-megabyte HTTP client behind the same ceremony as a
|
|
@@ -39,7 +37,7 @@ classifiers = [
|
|
|
39
37
|
# auroraomics` could not then perform a single documented step. They say it now.
|
|
40
38
|
#
|
|
41
39
|
# WHAT IT COSTS, measured rather than estimated (importlib.metadata over the
|
|
42
|
-
# transitive closure, on the interpreter this package targets
|
|
40
|
+
# transitive closure, on the interpreter this package targets): four
|
|
43
41
|
# distributions and 179 MB became twenty-eight and 537 MB. Neither of the two
|
|
44
42
|
# that dominate is named here — `imagecodecs` (142 MB) is what lets `tifffile`
|
|
45
43
|
# decode a real slide, and `pandas` plus `scipy` (198 MB between them) arrive
|
|
@@ -47,9 +45,9 @@ classifiers = [
|
|
|
47
45
|
# all of it. That is the trade, taken deliberately: one install line that works,
|
|
48
46
|
# over a smaller one that cannot reach the end of the page describing it.
|
|
49
47
|
#
|
|
50
|
-
# What is still NOT here is
|
|
51
|
-
# no
|
|
52
|
-
#
|
|
48
|
+
# What is still NOT here is torch. An environment that installs this package
|
|
49
|
+
# with `--no-deps` resolves the same either way; what protects it is the
|
|
50
|
+
# encoder line below.
|
|
53
51
|
dependencies = [
|
|
54
52
|
"numpy>=1.23",
|
|
55
53
|
"h5py>=3.9",
|
|
@@ -79,7 +77,7 @@ dependencies = [
|
|
|
79
77
|
# resolves the model, or the adapter and checkpoint stack that would carry one,
|
|
80
78
|
# and `predict_local` refuses with a message that says so. What runs on a user's
|
|
81
79
|
# own machine is cutting, filtering and EMBEDDING tiles — their pixels stay put,
|
|
82
|
-
# and
|
|
80
|
+
# and an embedding from the encoder is what crosses. So the line is at the
|
|
83
81
|
# ENCODER, not at the size of an install: `embed` below does resolve a tensor
|
|
84
82
|
# runtime, on purpose, and everything on the CLIENT side of the encoder — the
|
|
85
83
|
# HTTP client, the slide reader — sits in the core above, where nobody has to
|
|
@@ -88,7 +86,7 @@ dependencies = [
|
|
|
88
86
|
# encoder's four stay named, no shipped module imports either, and the release
|
|
89
87
|
# guard refuses a wheel carrying a module that is not on its published list.
|
|
90
88
|
#
|
|
91
|
-
# `client` and `slide` were extras
|
|
89
|
+
# `client` and `slide` were extras once and are GONE rather than kept as
|
|
92
90
|
# empty aliases. The published wheel declares both, so `pip install
|
|
93
91
|
# "auroraomics[client]"` exists in the wild and in our own older prose; pip
|
|
94
92
|
# WARNS on an extra a distribution does not provide and installs anyway —
|
|
@@ -121,7 +119,7 @@ mcp = ["mcp>=2,<3"]
|
|
|
121
119
|
#
|
|
122
120
|
# anndata used to be listed here as a test-only dependency, with a note saying
|
|
123
121
|
# that depending on it at run time would put pandas and its stack into every
|
|
124
|
-
# install. That is now exactly what happens, knowingly
|
|
122
|
+
# install. That is now exactly what happens, knowingly — the note above
|
|
125
123
|
# the core list records the price — so this extra no longer names anndata, nor
|
|
126
124
|
# `tifffile` or `imagecodecs`, because the core already does. A second bound
|
|
127
125
|
# here would be a test environment resolving what no user resolves. anndata is
|
|
@@ -138,10 +136,18 @@ dev = [
|
|
|
138
136
|
# install. The client needs no reference any more — it is the core.
|
|
139
137
|
"auroraomics[mcp]",
|
|
140
138
|
"pytest>=7",
|
|
139
|
+
# The release guard parses version strings with `packaging.version`, at module
|
|
140
|
+
# scope. It has always resolved, because more than one dependency in this list
|
|
141
|
+
# requires packaging — but arriving in a closure is not the same as being
|
|
142
|
+
# declared: the day whichever one carries it stops doing so, the check that
|
|
143
|
+
# decides whether a build may be uploaded stops being a check and becomes a
|
|
144
|
+
# collection error, which reports as one missing optional dependency rather
|
|
145
|
+
# than as a suite that no longer runs.
|
|
146
|
+
"packaging>=22",
|
|
141
147
|
# Coverage is measured on every run, with a floor, because this package is
|
|
142
148
|
# the one that SHIPS: a user installs it and calls it, so a branch nothing
|
|
143
149
|
# here exercises is a branch discovered in the field. The floor sat at
|
|
144
|
-
# nothing until
|
|
150
|
+
# nothing until the suite first measured 94%.
|
|
145
151
|
"pytest-cov>=5",
|
|
146
152
|
# setuptools and wheel are TEST dependencies as well as build ones: the
|
|
147
153
|
# release guard builds with --no-isolation so that it needs no network, and
|
|
@@ -223,8 +229,8 @@ source = ["src/auroraomics"]
|
|
|
223
229
|
# when a real gap closes; never lower it to make a change pass.
|
|
224
230
|
#
|
|
225
231
|
# 92 rather than the 94.18 measured hours earlier, and the difference is not a
|
|
226
|
-
# regression in anything tested here.
|
|
227
|
-
#
|
|
232
|
+
# regression in anything tested here. Turning `predict_local` into a refusal
|
|
233
|
+
# moved the forward pass to the unpublished distribution — but left
|
|
228
234
|
# `runtimes/deepspotm.py`'s `_panel_rows` and `_selection` behind, and their
|
|
229
235
|
# ONLY callers are now in the unpublished inference distribution's runner. So
|
|
230
236
|
# that file went from 271 statements at 99% to 169 at 73%: 46 statements this
|
|
@@ -233,8 +239,8 @@ source = ["src/auroraomics"]
|
|
|
233
239
|
#
|
|
234
240
|
# Moving those two helpers to the package that uses them takes the floor back
|
|
235
241
|
# above 94 without writing a single test. Recorded as an issue rather than done
|
|
236
|
-
# here, because it is a change to the boundary
|
|
237
|
-
# work, not with a coverage floor.
|
|
242
|
+
# here, because it is a change to the boundary that refusal drew and belongs
|
|
243
|
+
# with that work, not with a coverage floor.
|
|
238
244
|
fail_under = 92
|
|
239
245
|
# Two decimals, because `fail_under` is compared at this precision and rounding
|
|
240
246
|
# 93.6 to 94 would pass a suite that had slipped.
|