auroraomics 0.1.0.dev0__tar.gz → 0.1.0.dev2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- auroraomics-0.1.0.dev2/PKG-INFO +279 -0
- auroraomics-0.1.0.dev2/README.md +236 -0
- auroraomics-0.1.0.dev2/pyproject.toml +243 -0
- auroraomics-0.1.0.dev2/src/auroraomics/__init__.py +169 -0
- auroraomics-0.1.0.dev2/src/auroraomics/_assets/ASSETS.json +16 -0
- auroraomics-0.1.0.dev2/src/auroraomics/_assets/README.md +26 -0
- auroraomics-0.1.0.dev2/src/auroraomics/_assets/calibration-tile.png +0 -0
- auroraomics-0.1.0.dev2/src/auroraomics/_contracts/MANIFEST.json +8 -0
- {auroraomics-0.1.0.dev0 → auroraomics-0.1.0.dev2}/src/auroraomics/_contracts/public-api-counters.tokens.json +2 -3
- {auroraomics-0.1.0.dev0 → auroraomics-0.1.0.dev2}/src/auroraomics/_contracts/public-api-input-kinds.tokens.json +1 -1
- {auroraomics-0.1.0.dev0 → auroraomics-0.1.0.dev2}/src/auroraomics/_contracts/public-api.tokens.json +86 -13
- auroraomics-0.1.0.dev2/src/auroraomics/_text.py +97 -0
- auroraomics-0.1.0.dev2/src/auroraomics/bulk.py +776 -0
- auroraomics-0.1.0.dev2/src/auroraomics/cli.py +1529 -0
- auroraomics-0.1.0.dev2/src/auroraomics/client/__init__.py +71 -0
- auroraomics-0.1.0.dev2/src/auroraomics/client/_generated.py +382 -0
- auroraomics-0.1.0.dev2/src/auroraomics/client/_http.py +412 -0
- auroraomics-0.1.0.dev2/src/auroraomics/client/api.py +858 -0
- auroraomics-0.1.0.dev2/src/auroraomics/client/credentials.py +286 -0
- auroraomics-0.1.0.dev2/src/auroraomics/client/errors.py +396 -0
- auroraomics-0.1.0.dev2/src/auroraomics/client/jobs.py +380 -0
- auroraomics-0.1.0.dev2/src/auroraomics/client/uploads.py +638 -0
- auroraomics-0.1.0.dev2/src/auroraomics/contracts.py +428 -0
- auroraomics-0.1.0.dev2/src/auroraomics/embed.py +1648 -0
- auroraomics-0.1.0.dev2/src/auroraomics/errors.py +81 -0
- auroraomics-0.1.0.dev2/src/auroraomics/genes.py +150 -0
- {auroraomics-0.1.0.dev0 → auroraomics-0.1.0.dev2}/src/auroraomics/h5ad.py +27 -19
- auroraomics-0.1.0.dev2/src/auroraomics/mcp/__init__.py +14 -0
- auroraomics-0.1.0.dev2/src/auroraomics/mcp/server.py +90 -0
- auroraomics-0.1.0.dev2/src/auroraomics/mcp/tools.py +115 -0
- auroraomics-0.1.0.dev2/src/auroraomics/pack.py +1263 -0
- auroraomics-0.1.0.dev2/src/auroraomics/postprocess.py +1126 -0
- auroraomics-0.1.0.dev2/src/auroraomics/qc.py +377 -0
- auroraomics-0.1.0.dev2/src/auroraomics/runtimes/__init__.py +19 -0
- auroraomics-0.1.0.dev2/src/auroraomics/runtimes/deepspotm.py +792 -0
- auroraomics-0.1.0.dev2/src/auroraomics/slide.py +793 -0
- auroraomics-0.1.0.dev2/src/auroraomics/spatial.py +239 -0
- {auroraomics-0.1.0.dev0 → auroraomics-0.1.0.dev2}/src/auroraomics/subsample.py +43 -12
- auroraomics-0.1.0.dev2/src/auroraomics.egg-info/PKG-INFO +279 -0
- auroraomics-0.1.0.dev2/src/auroraomics.egg-info/SOURCES.txt +46 -0
- auroraomics-0.1.0.dev2/src/auroraomics.egg-info/entry_points.txt +2 -0
- {auroraomics-0.1.0.dev0 → auroraomics-0.1.0.dev2}/src/auroraomics.egg-info/requires.txt +13 -10
- auroraomics-0.1.0.dev0/PKG-INFO +0 -134
- auroraomics-0.1.0.dev0/README.md +0 -95
- auroraomics-0.1.0.dev0/pyproject.toml +0 -91
- auroraomics-0.1.0.dev0/src/auroraomics/__init__.py +0 -74
- auroraomics-0.1.0.dev0/src/auroraomics/_contracts/MANIFEST.json +0 -11
- auroraomics-0.1.0.dev0/src/auroraomics/_contracts/genes/gene-table.2026-09-05.json +0 -19368
- auroraomics-0.1.0.dev0/src/auroraomics/_contracts/public-api/golden-16-tiles.manifest.json +0 -308
- auroraomics-0.1.0.dev0/src/auroraomics/_contracts/qc-thresholds.tokens.json +0 -10
- auroraomics-0.1.0.dev0/src/auroraomics/contracts.py +0 -155
- auroraomics-0.1.0.dev0/src/auroraomics/pack.py +0 -667
- auroraomics-0.1.0.dev0/src/auroraomics/qc.py +0 -288
- auroraomics-0.1.0.dev0/src/auroraomics.egg-info/PKG-INFO +0 -134
- auroraomics-0.1.0.dev0/src/auroraomics.egg-info/SOURCES.txt +0 -23
- {auroraomics-0.1.0.dev0 → auroraomics-0.1.0.dev2}/LICENSE +0 -0
- {auroraomics-0.1.0.dev0 → auroraomics-0.1.0.dev2}/MANIFEST.in +0 -0
- {auroraomics-0.1.0.dev0 → auroraomics-0.1.0.dev2}/setup.cfg +0 -0
- {auroraomics-0.1.0.dev0 → auroraomics-0.1.0.dev2}/src/auroraomics/py.typed +0 -0
- {auroraomics-0.1.0.dev0 → auroraomics-0.1.0.dev2}/src/auroraomics.egg-info/dependency_links.txt +0 -0
- {auroraomics-0.1.0.dev0 → auroraomics-0.1.0.dev2}/src/auroraomics.egg-info/top_level.txt +0 -0
|
@@ -0,0 +1,279 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: auroraomics
|
|
3
|
+
Version: 0.1.0.dev2
|
|
4
|
+
Summary: Virtual spatial transcriptomics from H&E histology: patch QC, tile packing and the .h5ad result contract.
|
|
5
|
+
Author: Kalin Nonchev
|
|
6
|
+
License-Expression: PolyForm-Noncommercial-1.0.0
|
|
7
|
+
Keywords: histopathology,spatial-transcriptomics,h5ad,anndata,whole-slide-image
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Intended Audience :: Science/Research
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
12
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
13
|
+
Classifier: Typing :: Typed
|
|
14
|
+
Requires-Python: >=3.10
|
|
15
|
+
Description-Content-Type: text/markdown
|
|
16
|
+
License-File: LICENSE
|
|
17
|
+
Requires-Dist: numpy>=1.23
|
|
18
|
+
Requires-Dist: h5py>=3.9
|
|
19
|
+
Requires-Dist: opencv-python-headless>=4.5
|
|
20
|
+
Requires-Dist: pillow>=9
|
|
21
|
+
Requires-Dist: httpx>=0.27
|
|
22
|
+
Requires-Dist: pydantic>=2
|
|
23
|
+
Requires-Dist: anndata>=0.10
|
|
24
|
+
Requires-Dist: tifffile>=2023.7.10
|
|
25
|
+
Requires-Dist: imagecodecs>=2023.3.16
|
|
26
|
+
Provides-Extra: embed
|
|
27
|
+
Requires-Dist: torch>=2.0; extra == "embed"
|
|
28
|
+
Requires-Dist: timm>=1.0; extra == "embed"
|
|
29
|
+
Requires-Dist: transformers>=4.40; extra == "embed"
|
|
30
|
+
Requires-Dist: huggingface-hub>=0.23; extra == "embed"
|
|
31
|
+
Provides-Extra: mcp
|
|
32
|
+
Requires-Dist: mcp<3,>=2; extra == "mcp"
|
|
33
|
+
Provides-Extra: dev
|
|
34
|
+
Requires-Dist: auroraomics[mcp]; extra == "dev"
|
|
35
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
36
|
+
Requires-Dist: pytest-cov>=5; extra == "dev"
|
|
37
|
+
Requires-Dist: setuptools>=77; extra == "dev"
|
|
38
|
+
Requires-Dist: wheel; extra == "dev"
|
|
39
|
+
Requires-Dist: build>=1.0; extra == "dev"
|
|
40
|
+
Requires-Dist: numpy<2; extra == "dev"
|
|
41
|
+
Requires-Dist: huggingface-hub>=0.23; extra == "dev"
|
|
42
|
+
Dynamic: license-file
|
|
43
|
+
|
|
44
|
+
# auroraomics
|
|
45
|
+
|
|
46
|
+
Virtual spatial transcriptomics from H&E histology.
|
|
47
|
+
|
|
48
|
+
This package holds the pieces of that pipeline that are pure Python, so the
|
|
49
|
+
same code runs on your laptop, in a GPU container and on the service:
|
|
50
|
+
|
|
51
|
+
- **`auroraomics.qc`** — the three patch-quality predicates (foreground, blur,
|
|
52
|
+
stained tissue) applied to every candidate tile before a model sees it, and
|
|
53
|
+
the short-circuiting cascade that combines them.
|
|
54
|
+
- **`auroraomics.pack`** — build the `patches` archive a prediction takes as
|
|
55
|
+
input (`pack_tiles`), and validate one you have been handed (`read_archive`).
|
|
56
|
+
- **`auroraomics.h5ad`** — write the standard `.h5ad` result with `h5py`
|
|
57
|
+
alone, streaming the expression matrix batch by batch so peak memory is one
|
|
58
|
+
batch rather than the whole matrix.
|
|
59
|
+
- **`auroraomics.postprocess`** — apply a model card's `postprocess` entries
|
|
60
|
+
to a result file: a registry of entries, one runner, `h5py` alone, streaming
|
|
61
|
+
one chunk at a time. Ships the unmeasured-gene mask, which states per gene
|
|
62
|
+
how many training datasets measured it and blanks the ones none did.
|
|
63
|
+
- **`auroraomics.subsample`** — pick the densest contiguous square of spots
|
|
64
|
+
when a slide yields more tiles than a run is allowed to spend.
|
|
65
|
+
- **`auroraomics.genes`** — the model family's gene symbols mapped to Ensembl
|
|
66
|
+
stable gene ids, which is what every result's `var` index is keyed by. Ships
|
|
67
|
+
as one committed table, resolved in the same order the service resolves it.
|
|
68
|
+
- **`auroraomics.contracts`** — the shared contract values (container layout,
|
|
69
|
+
input-kind caps, result layout) as data, so nothing here re-types a number
|
|
70
|
+
the service also reads.
|
|
71
|
+
- **`auroraomics.embed`** *(extra)* — run a pinned image encoder over your
|
|
72
|
+
tiles locally and write the embeddings container, so a prediction can be made
|
|
73
|
+
from numbers instead of pixels.
|
|
74
|
+
|
|
75
|
+
One module describes work that does not happen here:
|
|
76
|
+
|
|
77
|
+
- **`auroraomics.runtimes.deepspotm`** — the models the service will run, as
|
|
78
|
+
data: what each one accepts, what it returns, and whether an archive matches
|
|
79
|
+
it. There is no install that makes one of them run locally; the definition is
|
|
80
|
+
not distributed, and `predict_local` refuses and says where prediction
|
|
81
|
+
happens.
|
|
82
|
+
|
|
83
|
+
## Install
|
|
84
|
+
|
|
85
|
+
```
|
|
86
|
+
pip install auroraomics
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
That is the whole documented path: open a slide, judge and pack its tiles,
|
|
90
|
+
submit, and read the result. One extra adds local embedding extraction
|
|
91
|
+
(`embed`), because a tensor runtime is gigabytes and specific to the machine it
|
|
92
|
+
was built for. No extra brings the model: predicting gene expression runs on
|
|
93
|
+
the service, not here.
|
|
94
|
+
|
|
95
|
+
## Pack tiles, then look at the report
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
import auroraomics as ao
|
|
99
|
+
|
|
100
|
+
report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thumbnail=thumb)
|
|
101
|
+
print(report.written, "tiles kept,", report.rejected, "dropped")
|
|
102
|
+
print(report.rejected_by_reason) # {'foreground_ratio': 12, ...}
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
`tiles` is any iterable of `ao.Tile(image, x, y)`, where `image` is an RGB
|
|
106
|
+
`uint8` array and `x`/`y` are the tile's top-left position in full-resolution
|
|
107
|
+
pixels. Tiles are quality-checked as they stream past, and only the ones that
|
|
108
|
+
pass are written, so an iterable that reads a slide lazily never has to hold
|
|
109
|
+
more than one tile in memory.
|
|
110
|
+
|
|
111
|
+
## Validate an archive before trusting it
|
|
112
|
+
|
|
113
|
+
```python
|
|
114
|
+
archive = ao.read_archive("sample.zip") # raises ArchiveError on anything odd
|
|
115
|
+
for tile in archive.tiles(): # decoded one at a time
|
|
116
|
+
...
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
`read_archive` checks the member names, the manifest, the tile geometry and the
|
|
120
|
+
declared sizes *before* decoding a single pixel, and refuses an archive whose
|
|
121
|
+
members do not match the container contract.
|
|
122
|
+
|
|
123
|
+
## Write a result
|
|
124
|
+
|
|
125
|
+
```python
|
|
126
|
+
ao.write_result(
|
|
127
|
+
"result.h5ad",
|
|
128
|
+
obs=obs, # per-spot columns, as plain arrays
|
|
129
|
+
var=var, # per-gene columns, indexed by gene id
|
|
130
|
+
spatial=coords, # (n_spots, 2) array -> obsm["spatial"]
|
|
131
|
+
x=batches, # an array, or an iterable of row batches
|
|
132
|
+
uns={"model": {"id": "..."}},
|
|
133
|
+
layers={"image_only": other_batches},
|
|
134
|
+
)
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
The file reads back as an ordinary `AnnData` in anndata 0.10 and 0.11. Passing
|
|
138
|
+
an iterable for `x` streams it: each batch is compressed into the file as it
|
|
139
|
+
arrives and then dropped, which is what makes a matrix larger than memory
|
|
140
|
+
writable.
|
|
141
|
+
|
|
142
|
+
## Embed locally, and send numbers instead of pixels
|
|
143
|
+
|
|
144
|
+
```
|
|
145
|
+
pip install "auroraomics[embed]"
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
```python
|
|
149
|
+
import auroraomics as ao
|
|
150
|
+
from auroraomics.embed import available_encoders, describe_encoders
|
|
151
|
+
|
|
152
|
+
print(available_encoders()) # ('deepspot-h', 'dinov2-b14', 'midnight')
|
|
153
|
+
|
|
154
|
+
with ao.Client() as client:
|
|
155
|
+
# Both are the service's, so the file records the encoder and the
|
|
156
|
+
# thresholds the model it is submitted to was served with.
|
|
157
|
+
encoder = ao.resolve_encoder(client.encoders())
|
|
158
|
+
thresholds = client.qc_thresholds()
|
|
159
|
+
|
|
160
|
+
report = ao.embed_tiles(
|
|
161
|
+
slide, "sample.npz", mpp=0.499, patch_um=55, encoder=encoder, thresholds=thresholds
|
|
162
|
+
)
|
|
163
|
+
print(report.rows, "rows of", report.dim, "numbers from", report.encoder.name)
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
`slide` is anything with a `(height, width, 3)` RGB `uint8` shape that can be
|
|
167
|
+
sliced — an array, or a memory-mapped or lazily-read one, so a slide larger than
|
|
168
|
+
memory works: only one crop exists at a time. Crops are taken at `patch_um`
|
|
169
|
+
micrometres using the `mpp` you give, quality-checked with the same three
|
|
170
|
+
predicates as `pack_tiles`, and pooled over the encoder's patch tokens.
|
|
171
|
+
|
|
172
|
+
Every encoder is pinned to a repository **and** a revision sha, with its licence
|
|
173
|
+
and access gate recorded beside it. `describe_encoders()` returns the whole
|
|
174
|
+
registry, including the encoders this package refuses to run — a refusal names
|
|
175
|
+
the licence or the approval you would have to accept, and which encoders remain.
|
|
176
|
+
|
|
177
|
+
The written file carries one extra vector, `calibration`: the embedding of a
|
|
178
|
+
fixed image shipped inside this package, through the same encoder, revision and
|
|
179
|
+
preprocessing as your rows. Comparing that one vector against a known reference
|
|
180
|
+
tells a reader whether the file was produced by the model it claims, before they
|
|
181
|
+
look at a single row.
|
|
182
|
+
|
|
183
|
+
The weights are downloaded from their publisher on first use, into the ordinary
|
|
184
|
+
model cache. Pass `allow_download=False` to guarantee no request is made: for
|
|
185
|
+
the duration of that load, name resolution and internet sockets are refused in
|
|
186
|
+
this process, so the guarantee does not rest on a model library reading an
|
|
187
|
+
environment variable it may already have read.
|
|
188
|
+
|
|
189
|
+
An encoder runs at one input size: the size the served model's embeddings were
|
|
190
|
+
computed at, which the registry pins and which is not always what the encoder's
|
|
191
|
+
own repository declares. The DINOv2 releases carry a 518-px position grid; the
|
|
192
|
+
model was trained on 224-px crops fed to that grid through an interpolated
|
|
193
|
+
position embedding, so that is what this package does too. `resize_px` is
|
|
194
|
+
checked against the pin rather than obeyed, because a vector computed at another
|
|
195
|
+
size is a different measurement that nobody else's numbers can be compared with.
|
|
196
|
+
## Where a prediction runs
|
|
197
|
+
|
|
198
|
+
Predicting gene expression runs on the service, and `predict_local` refuses —
|
|
199
|
+
by design, and the design is the product rather than a limitation.
|
|
200
|
+
|
|
201
|
+
This is split inference at the encoder. An open-weight pathology foundation
|
|
202
|
+
model turns your tiles into patch embeddings **on your machine**, so your slides
|
|
203
|
+
never leave it; our gene decoder turns embeddings into expression **on ours**, so
|
|
204
|
+
its weights never leave it. What crosses between us is a vector from a model
|
|
205
|
+
neither side owns.
|
|
206
|
+
|
|
207
|
+
```python
|
|
208
|
+
from auroraomics.runtimes.deepspotm import check_archive, describe_models
|
|
209
|
+
|
|
210
|
+
cards = client.models() # the catalogue is the service's
|
|
211
|
+
describe_models(cards) # what it will run, and on what
|
|
212
|
+
check_archive("sample.zip", model=MODEL, cards=cards) # are my tiles the right size?
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
Then submit with the API client and read the result back — the same `.h5ad`
|
|
216
|
+
contract whichever lane ran it.
|
|
217
|
+
|
|
218
|
+
## Post-process a result
|
|
219
|
+
|
|
220
|
+
```python
|
|
221
|
+
from auroraomics.postprocess import apply_card
|
|
222
|
+
|
|
223
|
+
apply_card("result.h5ad", card=card) # -> ('unmeasured_gene_mask',)
|
|
224
|
+
```
|
|
225
|
+
|
|
226
|
+
Or, from a pipeline rule, without writing a script:
|
|
227
|
+
|
|
228
|
+
```
|
|
229
|
+
python -m auroraomics.postprocess result.h5ad --card card.json
|
|
230
|
+
```
|
|
231
|
+
|
|
232
|
+
The card's `postprocess` list decides what runs. Every entry writes only where
|
|
233
|
+
the result contract says its product lives — a matrix under `layers/`, a
|
|
234
|
+
per-gene statement as a `var` column — and an entry the installed version does
|
|
235
|
+
not implement is refused **by name**, before the file is opened, rather than
|
|
236
|
+
skipped: a card is the description of a file, and a reader promised a layer
|
|
237
|
+
cannot tell a missing one from a model that predicted zeros.
|
|
238
|
+
|
|
239
|
+
The entry this version ships states what the model measured. `var` gains
|
|
240
|
+
`n_train_datasets` — the number of training datasets each gene was measured in,
|
|
241
|
+
because a gene seen in one and a gene seen in twenty are not the same claim —
|
|
242
|
+
and `measured_in_training`, derived from that count. A gene the model never
|
|
243
|
+
measured becomes `nan` in `X` and in every layer, not zero: zero is a
|
|
244
|
+
measurement, and an unmeasured gene left as one averages, correlates and
|
|
245
|
+
colours a heat map exactly like a prediction.
|
|
246
|
+
|
|
247
|
+
## Prepare a bulk RNA profile
|
|
248
|
+
|
|
249
|
+
```python
|
|
250
|
+
report = ao.bulk_rna("sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts")
|
|
251
|
+
print(report.rows, "genes;", f"{report.coverage:.0%} of", report.coverage_of)
|
|
252
|
+
```
|
|
253
|
+
|
|
254
|
+
A long or wide table, CSV or TSV, or a GDC STAR gene-counts file exactly as
|
|
255
|
+
downloaded. Gene names are resolved to unversioned Ensembl ids through the same
|
|
256
|
+
table every result is keyed by; the value column of the written table is named
|
|
257
|
+
by the unit you declared, so the file states its own units; and a table that is
|
|
258
|
+
already log-transformed is refused rather than logged a second time. Given the
|
|
259
|
+
model's card, a profile that covers less of the model's measured genes than the
|
|
260
|
+
service requires is refused here, with the service's own code, before anything
|
|
261
|
+
is uploaded.
|
|
262
|
+
|
|
263
|
+
## Typing
|
|
264
|
+
|
|
265
|
+
The package ships `py.typed`, so annotations are visible to type checkers in
|
|
266
|
+
your project.
|
|
267
|
+
|
|
268
|
+
## Licence
|
|
269
|
+
|
|
270
|
+
The code in this package is licensed under
|
|
271
|
+
[PolyForm Noncommercial 1.0.0](https://polyformproject.org/licenses/noncommercial/1.0.0),
|
|
272
|
+
which permits use for any purpose that is not commercial. It is the same licence the
|
|
273
|
+
model package this client is built for carries, so installing both puts you under one
|
|
274
|
+
rule rather than two.
|
|
275
|
+
|
|
276
|
+
The model weights are licensed separately by whoever publishes them, and access to them
|
|
277
|
+
may be gated. Read those terms before you use a model: they are not this licence, and a
|
|
278
|
+
permission granted here is not a permission granted there.
|
|
279
|
+
|
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
# auroraomics
|
|
2
|
+
|
|
3
|
+
Virtual spatial transcriptomics from H&E histology.
|
|
4
|
+
|
|
5
|
+
This package holds the pieces of that pipeline that are pure Python, so the
|
|
6
|
+
same code runs on your laptop, in a GPU container and on the service:
|
|
7
|
+
|
|
8
|
+
- **`auroraomics.qc`** — the three patch-quality predicates (foreground, blur,
|
|
9
|
+
stained tissue) applied to every candidate tile before a model sees it, and
|
|
10
|
+
the short-circuiting cascade that combines them.
|
|
11
|
+
- **`auroraomics.pack`** — build the `patches` archive a prediction takes as
|
|
12
|
+
input (`pack_tiles`), and validate one you have been handed (`read_archive`).
|
|
13
|
+
- **`auroraomics.h5ad`** — write the standard `.h5ad` result with `h5py`
|
|
14
|
+
alone, streaming the expression matrix batch by batch so peak memory is one
|
|
15
|
+
batch rather than the whole matrix.
|
|
16
|
+
- **`auroraomics.postprocess`** — apply a model card's `postprocess` entries
|
|
17
|
+
to a result file: a registry of entries, one runner, `h5py` alone, streaming
|
|
18
|
+
one chunk at a time. Ships the unmeasured-gene mask, which states per gene
|
|
19
|
+
how many training datasets measured it and blanks the ones none did.
|
|
20
|
+
- **`auroraomics.subsample`** — pick the densest contiguous square of spots
|
|
21
|
+
when a slide yields more tiles than a run is allowed to spend.
|
|
22
|
+
- **`auroraomics.genes`** — the model family's gene symbols mapped to Ensembl
|
|
23
|
+
stable gene ids, which is what every result's `var` index is keyed by. Ships
|
|
24
|
+
as one committed table, resolved in the same order the service resolves it.
|
|
25
|
+
- **`auroraomics.contracts`** — the shared contract values (container layout,
|
|
26
|
+
input-kind caps, result layout) as data, so nothing here re-types a number
|
|
27
|
+
the service also reads.
|
|
28
|
+
- **`auroraomics.embed`** *(extra)* — run a pinned image encoder over your
|
|
29
|
+
tiles locally and write the embeddings container, so a prediction can be made
|
|
30
|
+
from numbers instead of pixels.
|
|
31
|
+
|
|
32
|
+
One module describes work that does not happen here:
|
|
33
|
+
|
|
34
|
+
- **`auroraomics.runtimes.deepspotm`** — the models the service will run, as
|
|
35
|
+
data: what each one accepts, what it returns, and whether an archive matches
|
|
36
|
+
it. There is no install that makes one of them run locally; the definition is
|
|
37
|
+
not distributed, and `predict_local` refuses and says where prediction
|
|
38
|
+
happens.
|
|
39
|
+
|
|
40
|
+
## Install
|
|
41
|
+
|
|
42
|
+
```
|
|
43
|
+
pip install auroraomics
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
That is the whole documented path: open a slide, judge and pack its tiles,
|
|
47
|
+
submit, and read the result. One extra adds local embedding extraction
|
|
48
|
+
(`embed`), because a tensor runtime is gigabytes and specific to the machine it
|
|
49
|
+
was built for. No extra brings the model: predicting gene expression runs on
|
|
50
|
+
the service, not here.
|
|
51
|
+
|
|
52
|
+
## Pack tiles, then look at the report
|
|
53
|
+
|
|
54
|
+
```python
|
|
55
|
+
import auroraomics as ao
|
|
56
|
+
|
|
57
|
+
report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thumbnail=thumb)
|
|
58
|
+
print(report.written, "tiles kept,", report.rejected, "dropped")
|
|
59
|
+
print(report.rejected_by_reason) # {'foreground_ratio': 12, ...}
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
`tiles` is any iterable of `ao.Tile(image, x, y)`, where `image` is an RGB
|
|
63
|
+
`uint8` array and `x`/`y` are the tile's top-left position in full-resolution
|
|
64
|
+
pixels. Tiles are quality-checked as they stream past, and only the ones that
|
|
65
|
+
pass are written, so an iterable that reads a slide lazily never has to hold
|
|
66
|
+
more than one tile in memory.
|
|
67
|
+
|
|
68
|
+
## Validate an archive before trusting it
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
archive = ao.read_archive("sample.zip") # raises ArchiveError on anything odd
|
|
72
|
+
for tile in archive.tiles(): # decoded one at a time
|
|
73
|
+
...
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
`read_archive` checks the member names, the manifest, the tile geometry and the
|
|
77
|
+
declared sizes *before* decoding a single pixel, and refuses an archive whose
|
|
78
|
+
members do not match the container contract.
|
|
79
|
+
|
|
80
|
+
## Write a result
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
ao.write_result(
|
|
84
|
+
"result.h5ad",
|
|
85
|
+
obs=obs, # per-spot columns, as plain arrays
|
|
86
|
+
var=var, # per-gene columns, indexed by gene id
|
|
87
|
+
spatial=coords, # (n_spots, 2) array -> obsm["spatial"]
|
|
88
|
+
x=batches, # an array, or an iterable of row batches
|
|
89
|
+
uns={"model": {"id": "..."}},
|
|
90
|
+
layers={"image_only": other_batches},
|
|
91
|
+
)
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
The file reads back as an ordinary `AnnData` in anndata 0.10 and 0.11. Passing
|
|
95
|
+
an iterable for `x` streams it: each batch is compressed into the file as it
|
|
96
|
+
arrives and then dropped, which is what makes a matrix larger than memory
|
|
97
|
+
writable.
|
|
98
|
+
|
|
99
|
+
## Embed locally, and send numbers instead of pixels
|
|
100
|
+
|
|
101
|
+
```
|
|
102
|
+
pip install "auroraomics[embed]"
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
```python
|
|
106
|
+
import auroraomics as ao
|
|
107
|
+
from auroraomics.embed import available_encoders, describe_encoders
|
|
108
|
+
|
|
109
|
+
print(available_encoders()) # ('deepspot-h', 'dinov2-b14', 'midnight')
|
|
110
|
+
|
|
111
|
+
with ao.Client() as client:
|
|
112
|
+
# Both are the service's, so the file records the encoder and the
|
|
113
|
+
# thresholds the model it is submitted to was served with.
|
|
114
|
+
encoder = ao.resolve_encoder(client.encoders())
|
|
115
|
+
thresholds = client.qc_thresholds()
|
|
116
|
+
|
|
117
|
+
report = ao.embed_tiles(
|
|
118
|
+
slide, "sample.npz", mpp=0.499, patch_um=55, encoder=encoder, thresholds=thresholds
|
|
119
|
+
)
|
|
120
|
+
print(report.rows, "rows of", report.dim, "numbers from", report.encoder.name)
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
`slide` is anything with a `(height, width, 3)` RGB `uint8` shape that can be
|
|
124
|
+
sliced — an array, or a memory-mapped or lazily-read one, so a slide larger than
|
|
125
|
+
memory works: only one crop exists at a time. Crops are taken at `patch_um`
|
|
126
|
+
micrometres using the `mpp` you give, quality-checked with the same three
|
|
127
|
+
predicates as `pack_tiles`, and pooled over the encoder's patch tokens.
|
|
128
|
+
|
|
129
|
+
Every encoder is pinned to a repository **and** a revision sha, with its licence
|
|
130
|
+
and access gate recorded beside it. `describe_encoders()` returns the whole
|
|
131
|
+
registry, including the encoders this package refuses to run — a refusal names
|
|
132
|
+
the licence or the approval you would have to accept, and which encoders remain.
|
|
133
|
+
|
|
134
|
+
The written file carries one extra vector, `calibration`: the embedding of a
|
|
135
|
+
fixed image shipped inside this package, through the same encoder, revision and
|
|
136
|
+
preprocessing as your rows. Comparing that one vector against a known reference
|
|
137
|
+
tells a reader whether the file was produced by the model it claims, before they
|
|
138
|
+
look at a single row.
|
|
139
|
+
|
|
140
|
+
The weights are downloaded from their publisher on first use, into the ordinary
|
|
141
|
+
model cache. Pass `allow_download=False` to guarantee no request is made: for
|
|
142
|
+
the duration of that load, name resolution and internet sockets are refused in
|
|
143
|
+
this process, so the guarantee does not rest on a model library reading an
|
|
144
|
+
environment variable it may already have read.
|
|
145
|
+
|
|
146
|
+
An encoder runs at one input size: the size the served model's embeddings were
|
|
147
|
+
computed at, which the registry pins and which is not always what the encoder's
|
|
148
|
+
own repository declares. The DINOv2 releases carry a 518-px position grid; the
|
|
149
|
+
model was trained on 224-px crops fed to that grid through an interpolated
|
|
150
|
+
position embedding, so that is what this package does too. `resize_px` is
|
|
151
|
+
checked against the pin rather than obeyed, because a vector computed at another
|
|
152
|
+
size is a different measurement that nobody else's numbers can be compared with.
|
|
153
|
+
## Where a prediction runs
|
|
154
|
+
|
|
155
|
+
Predicting gene expression runs on the service, and `predict_local` refuses —
|
|
156
|
+
by design, and the design is the product rather than a limitation.
|
|
157
|
+
|
|
158
|
+
This is split inference at the encoder. An open-weight pathology foundation
|
|
159
|
+
model turns your tiles into patch embeddings **on your machine**, so your slides
|
|
160
|
+
never leave it; our gene decoder turns embeddings into expression **on ours**, so
|
|
161
|
+
its weights never leave it. What crosses between us is a vector from a model
|
|
162
|
+
neither side owns.
|
|
163
|
+
|
|
164
|
+
```python
|
|
165
|
+
from auroraomics.runtimes.deepspotm import check_archive, describe_models
|
|
166
|
+
|
|
167
|
+
cards = client.models() # the catalogue is the service's
|
|
168
|
+
describe_models(cards) # what it will run, and on what
|
|
169
|
+
check_archive("sample.zip", model=MODEL, cards=cards) # are my tiles the right size?
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
Then submit with the API client and read the result back — the same `.h5ad`
|
|
173
|
+
contract whichever lane ran it.
|
|
174
|
+
|
|
175
|
+
## Post-process a result
|
|
176
|
+
|
|
177
|
+
```python
|
|
178
|
+
from auroraomics.postprocess import apply_card
|
|
179
|
+
|
|
180
|
+
apply_card("result.h5ad", card=card) # -> ('unmeasured_gene_mask',)
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
Or, from a pipeline rule, without writing a script:
|
|
184
|
+
|
|
185
|
+
```
|
|
186
|
+
python -m auroraomics.postprocess result.h5ad --card card.json
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
The card's `postprocess` list decides what runs. Every entry writes only where
|
|
190
|
+
the result contract says its product lives — a matrix under `layers/`, a
|
|
191
|
+
per-gene statement as a `var` column — and an entry the installed version does
|
|
192
|
+
not implement is refused **by name**, before the file is opened, rather than
|
|
193
|
+
skipped: a card is the description of a file, and a reader promised a layer
|
|
194
|
+
cannot tell a missing one from a model that predicted zeros.
|
|
195
|
+
|
|
196
|
+
The entry this version ships states what the model measured. `var` gains
|
|
197
|
+
`n_train_datasets` — the number of training datasets each gene was measured in,
|
|
198
|
+
because a gene seen in one and a gene seen in twenty are not the same claim —
|
|
199
|
+
and `measured_in_training`, derived from that count. A gene the model never
|
|
200
|
+
measured becomes `nan` in `X` and in every layer, not zero: zero is a
|
|
201
|
+
measurement, and an unmeasured gene left as one averages, correlates and
|
|
202
|
+
colours a heat map exactly like a prediction.
|
|
203
|
+
|
|
204
|
+
## Prepare a bulk RNA profile
|
|
205
|
+
|
|
206
|
+
```python
|
|
207
|
+
report = ao.bulk_rna("sample.star_gene_counts.tsv", "sample.bulk.tsv", units="counts")
|
|
208
|
+
print(report.rows, "genes;", f"{report.coverage:.0%} of", report.coverage_of)
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
A long or wide table, CSV or TSV, or a GDC STAR gene-counts file exactly as
|
|
212
|
+
downloaded. Gene names are resolved to unversioned Ensembl ids through the same
|
|
213
|
+
table every result is keyed by; the value column of the written table is named
|
|
214
|
+
by the unit you declared, so the file states its own units; and a table that is
|
|
215
|
+
already log-transformed is refused rather than logged a second time. Given the
|
|
216
|
+
model's card, a profile that covers less of the model's measured genes than the
|
|
217
|
+
service requires is refused here, with the service's own code, before anything
|
|
218
|
+
is uploaded.
|
|
219
|
+
|
|
220
|
+
## Typing
|
|
221
|
+
|
|
222
|
+
The package ships `py.typed`, so annotations are visible to type checkers in
|
|
223
|
+
your project.
|
|
224
|
+
|
|
225
|
+
## Licence
|
|
226
|
+
|
|
227
|
+
The code in this package is licensed under
|
|
228
|
+
[PolyForm Noncommercial 1.0.0](https://polyformproject.org/licenses/noncommercial/1.0.0),
|
|
229
|
+
which permits use for any purpose that is not commercial. It is the same licence the
|
|
230
|
+
model package this client is built for carries, so installing both puts you under one
|
|
231
|
+
rule rather than two.
|
|
232
|
+
|
|
233
|
+
The model weights are licensed separately by whoever publishes them, and access to them
|
|
234
|
+
may be gated. Read those terms before you use a model: they are not this licence, and a
|
|
235
|
+
permission granted here is not a permission granted there.
|
|
236
|
+
|