auroraomics 0.1.0.dev1__tar.gz → 0.1.0.dev2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {auroraomics-0.1.0.dev1/src/auroraomics.egg-info → auroraomics-0.1.0.dev2}/PKG-INFO +5 -4
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/README.md +4 -3
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/pyproject.toml +11 -11
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/__init__.py +5 -5
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_contracts/MANIFEST.json +1 -1
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_contracts/public-api.tokens.json +8 -2
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_text.py +1 -1
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/bulk.py +7 -7
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/cli.py +211 -43
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/_http.py +4 -4
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/api.py +64 -10
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/credentials.py +14 -3
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/contracts.py +26 -12
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/embed.py +449 -98
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/errors.py +1 -1
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/genes.py +3 -3
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/pack.py +125 -8
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/postprocess.py +1 -1
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/qc.py +4 -3
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/runtimes/__init__.py +0 -8
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/runtimes/deepspotm.py +210 -66
- auroraomics-0.1.0.dev2/src/auroraomics/slide.py +793 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/spatial.py +7 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2/src/auroraomics.egg-info}/PKG-INFO +5 -4
- auroraomics-0.1.0.dev1/src/auroraomics/slide.py +0 -325
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/LICENSE +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/MANIFEST.in +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/setup.cfg +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_assets/ASSETS.json +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_assets/README.md +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_assets/calibration-tile.png +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_contracts/public-api-counters.tokens.json +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_contracts/public-api-input-kinds.tokens.json +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/__init__.py +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/_generated.py +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/errors.py +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/jobs.py +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/uploads.py +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/h5ad.py +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/mcp/__init__.py +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/mcp/server.py +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/mcp/tools.py +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/py.typed +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/subsample.py +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics.egg-info/SOURCES.txt +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics.egg-info/dependency_links.txt +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics.egg-info/entry_points.txt +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics.egg-info/requires.txt +0 -0
- {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: auroraomics
|
|
3
|
-
Version: 0.1.0.
|
|
3
|
+
Version: 0.1.0.dev2
|
|
4
4
|
Summary: Virtual spatial transcriptomics from H&E histology: patch QC, tile packing and the .h5ad result contract.
|
|
5
5
|
Author: Kalin Nonchev
|
|
6
6
|
License-Expression: PolyForm-Noncommercial-1.0.0
|
|
@@ -205,10 +205,11 @@ its weights never leave it. What crosses between us is a vector from a model
|
|
|
205
205
|
neither side owns.
|
|
206
206
|
|
|
207
207
|
```python
|
|
208
|
-
from auroraomics.runtimes.deepspotm import
|
|
208
|
+
from auroraomics.runtimes.deepspotm import check_archive, describe_models
|
|
209
209
|
|
|
210
|
-
|
|
211
|
-
|
|
210
|
+
cards = client.models() # the catalogue is the service's
|
|
211
|
+
describe_models(cards) # what it will run, and on what
|
|
212
|
+
check_archive("sample.zip", model=MODEL, cards=cards) # are my tiles the right size?
|
|
212
213
|
```
|
|
213
214
|
|
|
214
215
|
Then submit with the API client and read the result back — the same `.h5ad`
|
|
@@ -162,10 +162,11 @@ its weights never leave it. What crosses between us is a vector from a model
|
|
|
162
162
|
neither side owns.
|
|
163
163
|
|
|
164
164
|
```python
|
|
165
|
-
from auroraomics.runtimes.deepspotm import
|
|
165
|
+
from auroraomics.runtimes.deepspotm import check_archive, describe_models
|
|
166
166
|
|
|
167
|
-
|
|
168
|
-
|
|
167
|
+
cards = client.models() # the catalogue is the service's
|
|
168
|
+
describe_models(cards) # what it will run, and on what
|
|
169
|
+
check_archive("sample.zip", model=MODEL, cards=cards) # are my tiles the right size?
|
|
169
170
|
```
|
|
170
171
|
|
|
171
172
|
Then submit with the API client and read the result back — the same `.h5ad`
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "auroraomics"
|
|
7
|
-
version = "0.1.0.
|
|
7
|
+
version = "0.1.0.dev2"
|
|
8
8
|
description = "Virtual spatial transcriptomics from H&E histology: patch QC, tile packing and the .h5ad result contract."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -31,7 +31,7 @@ classifiers = [
|
|
|
31
31
|
# install line the quickstart prints is an install line that can do what the
|
|
32
32
|
# quickstart says.
|
|
33
33
|
#
|
|
34
|
-
# It was four names
|
|
34
|
+
# It was four names once, with the API client and the slide reader behind
|
|
35
35
|
# extras of their own. The comment on the extras below has always ended "the
|
|
36
36
|
# line is at the ENCODER, not at the size of an install"; these lists did not
|
|
37
37
|
# say it. They put a half-megabyte HTTP client behind the same ceremony as a
|
|
@@ -39,7 +39,7 @@ classifiers = [
|
|
|
39
39
|
# auroraomics` could not then perform a single documented step. They say it now.
|
|
40
40
|
#
|
|
41
41
|
# WHAT IT COSTS, measured rather than estimated (importlib.metadata over the
|
|
42
|
-
# transitive closure, on the interpreter this package targets
|
|
42
|
+
# transitive closure, on the interpreter this package targets): four
|
|
43
43
|
# distributions and 179 MB became twenty-eight and 537 MB. Neither of the two
|
|
44
44
|
# that dominate is named here — `imagecodecs` (142 MB) is what lets `tifffile`
|
|
45
45
|
# decode a real slide, and `pandas` plus `scipy` (198 MB between them) arrive
|
|
@@ -79,7 +79,7 @@ dependencies = [
|
|
|
79
79
|
# resolves the model, or the adapter and checkpoint stack that would carry one,
|
|
80
80
|
# and `predict_local` refuses with a message that says so. What runs on a user's
|
|
81
81
|
# own machine is cutting, filtering and EMBEDDING tiles — their pixels stay put,
|
|
82
|
-
# and a vector from
|
|
82
|
+
# and a vector from the encoder is what crosses. So the line is at the
|
|
83
83
|
# ENCODER, not at the size of an install: `embed` below does resolve a tensor
|
|
84
84
|
# runtime, on purpose, and everything on the CLIENT side of the encoder — the
|
|
85
85
|
# HTTP client, the slide reader — sits in the core above, where nobody has to
|
|
@@ -88,7 +88,7 @@ dependencies = [
|
|
|
88
88
|
# encoder's four stay named, no shipped module imports either, and the release
|
|
89
89
|
# guard refuses a wheel carrying a module that is not on its published list.
|
|
90
90
|
#
|
|
91
|
-
# `client` and `slide` were extras
|
|
91
|
+
# `client` and `slide` were extras once and are GONE rather than kept as
|
|
92
92
|
# empty aliases. The published wheel declares both, so `pip install
|
|
93
93
|
# "auroraomics[client]"` exists in the wild and in our own older prose; pip
|
|
94
94
|
# WARNS on an extra a distribution does not provide and installs anyway —
|
|
@@ -121,7 +121,7 @@ mcp = ["mcp>=2,<3"]
|
|
|
121
121
|
#
|
|
122
122
|
# anndata used to be listed here as a test-only dependency, with a note saying
|
|
123
123
|
# that depending on it at run time would put pandas and its stack into every
|
|
124
|
-
# install. That is now exactly what happens, knowingly
|
|
124
|
+
# install. That is now exactly what happens, knowingly — the note above
|
|
125
125
|
# the core list records the price — so this extra no longer names anndata, nor
|
|
126
126
|
# `tifffile` or `imagecodecs`, because the core already does. A second bound
|
|
127
127
|
# here would be a test environment resolving what no user resolves. anndata is
|
|
@@ -141,7 +141,7 @@ dev = [
|
|
|
141
141
|
# Coverage is measured on every run, with a floor, because this package is
|
|
142
142
|
# the one that SHIPS: a user installs it and calls it, so a branch nothing
|
|
143
143
|
# here exercises is a branch discovered in the field. The floor sat at
|
|
144
|
-
# nothing until
|
|
144
|
+
# nothing until the suite first measured 94%.
|
|
145
145
|
"pytest-cov>=5",
|
|
146
146
|
# setuptools and wheel are TEST dependencies as well as build ones: the
|
|
147
147
|
# release guard builds with --no-isolation so that it needs no network, and
|
|
@@ -223,8 +223,8 @@ source = ["src/auroraomics"]
|
|
|
223
223
|
# when a real gap closes; never lower it to make a change pass.
|
|
224
224
|
#
|
|
225
225
|
# 92 rather than the 94.18 measured hours earlier, and the difference is not a
|
|
226
|
-
# regression in anything tested here.
|
|
227
|
-
#
|
|
226
|
+
# regression in anything tested here. Turning `predict_local` into a refusal
|
|
227
|
+
# moved the forward pass to the unpublished distribution — but left
|
|
228
228
|
# `runtimes/deepspotm.py`'s `_panel_rows` and `_selection` behind, and their
|
|
229
229
|
# ONLY callers are now in the unpublished inference distribution's runner. So
|
|
230
230
|
# that file went from 271 statements at 99% to 169 at 73%: 46 statements this
|
|
@@ -233,8 +233,8 @@ source = ["src/auroraomics"]
|
|
|
233
233
|
#
|
|
234
234
|
# Moving those two helpers to the package that uses them takes the floor back
|
|
235
235
|
# above 94 without writing a single test. Recorded as an issue rather than done
|
|
236
|
-
# here, because it is a change to the boundary
|
|
237
|
-
# work, not with a coverage floor.
|
|
236
|
+
# here, because it is a change to the boundary that refusal drew and belongs
|
|
237
|
+
# with that work, not with a coverage floor.
|
|
238
238
|
fail_under = 92
|
|
239
239
|
# Two decimals, because `fail_under` is compared at this precision and rounding
|
|
240
240
|
# 93.6 to 94 would pass a suite that had slipped.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
"""Virtual spatial transcriptomics from H&E histology.
|
|
2
2
|
|
|
3
3
|
**One journey, and one word chooses which.** Send the tiles, or send only the
|
|
4
|
-
|
|
4
|
+
embeddings computed from them; everything either side of that word is the same:
|
|
5
5
|
|
|
6
6
|
import auroraomics as ao
|
|
7
7
|
|
|
@@ -105,10 +105,10 @@ _EXPORTS: dict[str, str] = {
|
|
|
105
105
|
|
|
106
106
|
Two properties, and the table buys both.
|
|
107
107
|
|
|
108
|
-
``resolve_encoder`` is here because
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
108
|
+
``resolve_encoder`` is here because it is on the journey rather than beside it:
|
|
109
|
+
``embed_tiles`` takes a resolved encoder and the registry is the service's, so
|
|
110
|
+
``resolve_encoder(client.encoders(), name)`` is a line every caller who embeds
|
|
111
|
+
tiles writes by hand.
|
|
112
112
|
|
|
113
113
|
**It is short.** Thirty-three names stood here while the two journeys through
|
|
114
114
|
this package used two of them between them, because a module's whole public
|
|
@@ -3,6 +3,6 @@
|
|
|
3
3
|
"files": {
|
|
4
4
|
"public-api-counters.tokens.json": "4f5dd14ce76f7001d36940dfc1d1f49feac9ec1be2e7646d47b29ab21c124a8f",
|
|
5
5
|
"public-api-input-kinds.tokens.json": "1c80e2637d27f2491ea178e2fa87e2a435640e5445791b2fd20adc53a574e54d",
|
|
6
|
-
"public-api.tokens.json": "
|
|
6
|
+
"public-api.tokens.json": "a8d3253978aba375c1933aef3f3069bcb4b5e47c2edbf6062a08c95c0f255f4d"
|
|
7
7
|
}
|
|
8
8
|
}
|
{auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_contracts/public-api.tokens.json
RENAMED
|
@@ -354,8 +354,14 @@
|
|
|
354
354
|
"uns_keys": [
|
|
355
355
|
"model",
|
|
356
356
|
"input_spec",
|
|
357
|
-
"job"
|
|
358
|
-
|
|
357
|
+
"job",
|
|
358
|
+
"resolution"
|
|
359
|
+
],
|
|
360
|
+
"resolution_checks": {
|
|
361
|
+
"slide": "slide_metadata",
|
|
362
|
+
"none": "nothing",
|
|
363
|
+
"unstated": "unstated"
|
|
364
|
+
}
|
|
359
365
|
},
|
|
360
366
|
"rateBuckets": {
|
|
361
367
|
"read": {
|
|
@@ -38,7 +38,7 @@ def not_one_file(path, *, instead: str) -> str:
|
|
|
38
38
|
return f"{path} is not a regular file; {instead}"
|
|
39
39
|
|
|
40
40
|
EMBEDDING_COMPONENT = "DeepSpot-H"
|
|
41
|
-
"""The step that turns tiles into
|
|
41
|
+
"""The step that turns tiles into embeddings, named as a reader meets it.
|
|
42
42
|
|
|
43
43
|
Typed once, and imported by every message that names it. It was written out
|
|
44
44
|
five times across `cli.py` and `runtimes/deepspotm.py` — the defect the
|
|
@@ -157,10 +157,10 @@ Resolver = Callable[[Sequence[str]], Mapping[str, "str | None"]]
|
|
|
157
157
|
[Client.resolve_keys][auroraomics.client.api.Client.resolve_keys] is one, and so is
|
|
158
158
|
any callable a caller writes over their own mapping.
|
|
159
159
|
|
|
160
|
-
An ARGUMENT
|
|
161
|
-
and the service holds it — this module used to read a
|
|
162
|
-
the wheel, which was frozen at build time and was a
|
|
163
|
-
service's resolution order besides.
|
|
160
|
+
An ARGUMENT rather than a lookup this module performs, because the gene table is
|
|
161
|
+
the model's own output axis and the service holds it — this module used to read a
|
|
162
|
+
3.7 MB copy vendored into the wheel, which was frozen at build time and was a
|
|
163
|
+
second implementation of the service's resolution order besides.
|
|
164
164
|
"""
|
|
165
165
|
|
|
166
166
|
|
|
@@ -446,7 +446,7 @@ def _asked_once(resolve: Resolver) -> Resolver:
|
|
|
446
446
|
decides the layout by resolving ``header[1:]``, and then returns exactly
|
|
447
447
|
those names as the pairs — which ``_resolve_all`` resolves again. Two
|
|
448
448
|
identical passes over a transcriptome's worth of names, and since resolution
|
|
449
|
-
became the service's
|
|
449
|
+
became the service's, two passes over the NETWORK: measured on a
|
|
450
450
|
2,000-column table, the resolver was called twice with 2.0x the header.
|
|
451
451
|
|
|
452
452
|
Per CALL, not per process. A resolver is a live connection to one
|
|
@@ -513,7 +513,7 @@ def _coverage_reference(card: Mapping[str, Any]) -> tuple[frozenset[str], str]:
|
|
|
513
513
|
"""The gene set coverage is judged against: the model's measured set.
|
|
514
514
|
|
|
515
515
|
The card's, always. It used to fall back to the whole vendored gene table for
|
|
516
|
-
a caller who named no model, and that fallback went with the table
|
|
516
|
+
a caller who named no model, and that fallback went with the table —
|
|
517
517
|
which is no loss: coverage of "every gene the family can predict" was never
|
|
518
518
|
the number the service refuses on.
|
|
519
519
|
"""
|
|
@@ -571,7 +571,7 @@ def bulk_rna(
|
|
|
571
571
|
``output.gene_mask`` is the measured gene set coverage is judged
|
|
572
572
|
against — the same refusal the service would make. It used to be
|
|
573
573
|
optional, with the vendored gene table standing in; the table is the
|
|
574
|
-
service's now
|
|
574
|
+
service's now, and "coverage of every gene the family predicts"
|
|
575
575
|
was never a number the service refuses on.
|
|
576
576
|
resolve:
|
|
577
577
|
How a symbol becomes the Ensembl id a result is keyed by:
|
|
@@ -238,15 +238,11 @@ fetching a finished job's result is what that command is for — while
|
|
|
238
238
|
and holds both directions: a command declaring ``--wait`` with no row here, or
|
|
239
239
|
a row naming a flag that command does not declare, is a red build.
|
|
240
240
|
"""
|
|
241
|
-
|
|
242
241
|
OUTPUT_NEEDS_WAIT = f"needs --wait: {NEEDS_WAIT['submit']['output']}"
|
|
243
242
|
"""What ``submit --output``'s help says, DERIVED from the table the refusal
|
|
244
243
|
reads. It is the flag's own help, the sentence the command's description
|
|
245
244
|
carries, and what the refusal says — and those three drifted once already, which
|
|
246
|
-
is why it is one string.
|
|
247
|
-
the refusal moved into a table and the help did not: the existing test caught
|
|
248
|
-
it, which is the whole reason that test names the constant rather than a
|
|
249
|
-
phrase."""
|
|
245
|
+
is why it is one string."""
|
|
250
246
|
|
|
251
247
|
ACCESS_PENDING_CODE = "access_pending"
|
|
252
248
|
"""The mint refusal for an account that has asked and is waiting; the catalogue's code."""
|
|
@@ -427,7 +423,7 @@ def _calibration_check(args: argparse.Namespace, model_id: str) -> dict[str, Any
|
|
|
427
423
|
Every other check in ``doctor`` is cheap, so this one runs only when a
|
|
428
424
|
model is named: it loads an encoder and runs one forward pass. What it buys
|
|
429
425
|
is the answer to the question worth asking before a slide is embedded — will
|
|
430
|
-
the service accept
|
|
426
|
+
the service accept embeddings from this machine — asked once, on one image,
|
|
431
427
|
instead of after an hour of tiles.
|
|
432
428
|
|
|
433
429
|
Reported, never raised, like every check here: each way it cannot be
|
|
@@ -444,7 +440,7 @@ def _calibration_check(args: argparse.Namespace, model_id: str) -> dict[str, Any
|
|
|
444
440
|
cards = {str(card.get("id")): card for card in client.models()}
|
|
445
441
|
# Fetched in the same session as the cards, because both are the
|
|
446
442
|
# service's now: an encoder is a pinned repository and revision, and
|
|
447
|
-
# this package
|
|
443
|
+
# this package no longer carries a copy of that pin.
|
|
448
444
|
registry = client.encoders()
|
|
449
445
|
except Exception as exc: # noqa: BLE001 - a diagnostic reports, it does not raise
|
|
450
446
|
return verdict(False, str(exc))
|
|
@@ -455,7 +451,7 @@ def _calibration_check(args: argparse.Namespace, model_id: str) -> dict[str, Any
|
|
|
455
451
|
reads = accepted_observations(card)
|
|
456
452
|
locally = _embedded_locally()
|
|
457
453
|
if reads and locally and locally not in reads:
|
|
458
|
-
# This check asks whether
|
|
454
|
+
# This check asks whether embeddings from THIS machine would be accepted. A
|
|
459
455
|
# model that reads the tiles themselves is never handed one, so there is
|
|
460
456
|
# no question to ask — and asking it anyway loaded an encoder the caller
|
|
461
457
|
# on that route has every reason not to have installed, then reported
|
|
@@ -490,41 +486,108 @@ def _calibration_check(args: argparse.Namespace, model_id: str) -> dict[str, Any
|
|
|
490
486
|
|
|
491
487
|
try:
|
|
492
488
|
name = _encoder_name(spec, embed.describe_encoders(registry))
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
)
|
|
489
|
+
encoder = embed.resolve_encoder(registry, name)
|
|
490
|
+
run = embed.calibration_pass(encoder, allow_download=args.allow_download)
|
|
491
|
+
vector, stored = _as_a_file_would_hold_it(run.embedding, spec)
|
|
496
492
|
distance = embed.calibration_offset(vector, reference["reference"])
|
|
497
493
|
except Exception as exc: # noqa: BLE001
|
|
498
494
|
return verdict(False, str(exc))
|
|
499
495
|
within = distance <= tolerance
|
|
496
|
+
pace, projection = _pace(run, args.tiles)
|
|
500
497
|
return verdict(
|
|
501
498
|
within,
|
|
502
499
|
f"{name} on this machine is {distance:.3g} from {model_id}'s reference "
|
|
503
|
-
f"(tolerance {tolerance:.3g}
|
|
504
|
-
|
|
500
|
+
f"(tolerance {tolerance:.3g}, measured as {stored} — the element type a "
|
|
501
|
+
"file written here carries)"
|
|
502
|
+
+ ("" if within else "; embeddings from this machine would be refused")
|
|
503
|
+
+ f". {pace}",
|
|
505
504
|
model=model_id,
|
|
506
505
|
encoder=name,
|
|
507
506
|
offset=distance,
|
|
508
507
|
tolerance=tolerance,
|
|
508
|
+
dtype=stored,
|
|
509
|
+
device=run.device,
|
|
510
|
+
seconds_per_tile=run.seconds,
|
|
511
|
+
**projection,
|
|
509
512
|
)
|
|
510
513
|
|
|
511
514
|
|
|
512
|
-
def
|
|
513
|
-
"""The forward pass rounded
|
|
515
|
+
def _as_a_file_would_hold_it(vector: Any, spec: dict[str, Any]) -> tuple[Any, str]:
|
|
516
|
+
"""The forward pass rounded the way the file this machine writes will be.
|
|
517
|
+
|
|
518
|
+
The lane compares the numbers a submission STORES, so measuring the raw
|
|
519
|
+
forward pass here would make the client kinder than the service by exactly
|
|
520
|
+
that rounding, at the boundary the tolerance sits on. A machine told it
|
|
521
|
+
agrees and an hour of embedding refused afterwards is the failure this whole
|
|
522
|
+
check exists to prevent.
|
|
523
|
+
|
|
524
|
+
TWO things claim to say what a file stores, and they are not the same thing:
|
|
525
|
+
the card pins an ``input_spec.dtype``, and
|
|
526
|
+
[embed_tiles][auroraomics.embed.embed_tiles] writes at
|
|
527
|
+
[DEFAULT_DTYPE][auroraomics.embed.DEFAULT_DTYPE] — it is handed an encoder,
|
|
528
|
+
not a model, so it never reads the card. Today the two agree, by coincidence
|
|
529
|
+
of value rather than by construction, and a card that moved to the wider
|
|
530
|
+
type would leave this check measuring a rounding no file here would carry.
|
|
531
|
+
So it measures at whichever of them rounds HARDER, and names it: being
|
|
532
|
+
stricter than the lane costs a caller a second look, being kinder costs them
|
|
533
|
+
the afternoon.
|
|
534
|
+
|
|
535
|
+
A ``dtype`` the container does not allow is not one of the two: such a file
|
|
536
|
+
cannot be submitted at all, and it is its element type that would be
|
|
537
|
+
refused, not its calibration — so the writer's own is used alone.
|
|
538
|
+
"""
|
|
539
|
+
import numpy as np
|
|
540
|
+
|
|
541
|
+
from auroraomics import embed
|
|
514
542
|
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
543
|
+
allowed = contracts.embeddings_npz()["dtypes"]
|
|
544
|
+
pinned = str(spec.get("dtype") or "")
|
|
545
|
+
candidates = [embed.DEFAULT_DTYPE, *([pinned] if pinned in allowed else [])]
|
|
546
|
+
# Narrowest wins, by the element type's own width rather than by a list
|
|
547
|
+
# written here in an order someone has to keep right.
|
|
548
|
+
stored = min(candidates, key=lambda name: np.dtype(name).itemsize)
|
|
549
|
+
return vector.astype(stored), stored
|
|
521
550
|
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
551
|
+
|
|
552
|
+
def _pace(run: Any, tiles: int | None) -> tuple[str, dict[str, Any]]:
|
|
553
|
+
"""What a tile costs on THIS machine, and what that makes of a slide.
|
|
554
|
+
|
|
555
|
+
The one number `doctor` was already measuring and throwing away. The
|
|
556
|
+
published table for ``batch_size`` is a CUDA device's, and the lane most
|
|
557
|
+
callers run on is the CPU — a different order of magnitude on the same
|
|
558
|
+
encoder — so "minutes to hours" was a range a caller only resolved by
|
|
559
|
+
starting. A machine can be asked, in the pass that was happening anyway.
|
|
560
|
+
|
|
561
|
+
Stated as an upper bound, because that is what it is: one tile to a pass,
|
|
562
|
+
where a run feeds ``batch_size`` of them and pays less per tile.
|
|
525
563
|
"""
|
|
526
|
-
|
|
527
|
-
|
|
564
|
+
per = f"One tile takes {run.seconds * 1000:,.0f} ms on {run.device} here"
|
|
565
|
+
if not tiles:
|
|
566
|
+
return (
|
|
567
|
+
f"{per}, one tile to a pass — an upper bound, since a run feeds a "
|
|
568
|
+
"whole batch at a time. Pass --tiles to turn that into a slide's "
|
|
569
|
+
"wall clock.",
|
|
570
|
+
{},
|
|
571
|
+
)
|
|
572
|
+
total = run.seconds * tiles
|
|
573
|
+
return (
|
|
574
|
+
f"{per}, so {tiles:,} tiles is at most {_how_long(total)} — one tile to "
|
|
575
|
+
"a pass, and a run feeds a whole batch at a time for less per tile.",
|
|
576
|
+
{"tiles": int(tiles), "estimated_seconds": total},
|
|
577
|
+
)
|
|
578
|
+
|
|
579
|
+
|
|
580
|
+
def _how_long(seconds: float) -> str:
|
|
581
|
+
"""A duration in the unit a reader decides with, never in bare seconds.
|
|
582
|
+
|
|
583
|
+
"14,073 s" and "3.9 hours" are the same number and only one of them answers
|
|
584
|
+
"do I start this before lunch".
|
|
585
|
+
"""
|
|
586
|
+
if seconds < 90:
|
|
587
|
+
return f"{seconds:.0f} seconds"
|
|
588
|
+
if seconds < 90 * 60:
|
|
589
|
+
return f"{seconds / 60:.0f} minutes"
|
|
590
|
+
return f"{seconds / 3600:.1f} hours"
|
|
528
591
|
|
|
529
592
|
|
|
530
593
|
def _encoder_name(spec: dict[str, Any], known: dict[str, Any]) -> str:
|
|
@@ -550,7 +613,7 @@ def _encoder_name(spec: dict[str, Any], known: dict[str, Any]) -> str:
|
|
|
550
613
|
if differing:
|
|
551
614
|
raise errors.ClientError(
|
|
552
615
|
f"the card's {', '.join(differing)} differ(s) from what this release pins for "
|
|
553
|
-
f"{name}; upgrade the package rather than trust
|
|
616
|
+
f"{name}; upgrade the package rather than trust an embedding computed another way"
|
|
554
617
|
)
|
|
555
618
|
return name
|
|
556
619
|
raise errors.ClientError(
|
|
@@ -747,7 +810,7 @@ def kind_charged_in(unit: str) -> str:
|
|
|
747
810
|
"""The observation kind the registry charges by ``unit``.
|
|
748
811
|
|
|
749
812
|
The two routes differ in what a submission IS, and the registry already
|
|
750
|
-
says so: the tiles are charged per tile, the
|
|
813
|
+
says so: the tiles are charged per tile, the embeddings per row. So the
|
|
751
814
|
sentence that describes each route reads its kind from the contract instead
|
|
752
815
|
of naming it.
|
|
753
816
|
|
|
@@ -771,7 +834,7 @@ def _embedded_locally() -> str:
|
|
|
771
834
|
"""The observation kind a caller computes on their own machine, or ``""``.
|
|
772
835
|
|
|
773
836
|
The two routes differ in where tiles become numbers, and the registry
|
|
774
|
-
already says which is which by what it charges: the
|
|
837
|
+
already says which is which by what it charges: the embeddings are charged per
|
|
775
838
|
ROW. Read from there rather than from ``embed.KIND``, for the reason
|
|
776
839
|
[kind_charged_in][auroraomics.cli.kind_charged_in] gives — this runs inside
|
|
777
840
|
``doctor``, which is the command an install with no encoder runtime in it is
|
|
@@ -796,8 +859,8 @@ def _example_of(kind: str) -> str:
|
|
|
796
859
|
|
|
797
860
|
Mixing the two put the same example in both halves of one sentence, so the
|
|
798
861
|
submit description read "`embeddings=sample.npz` sends the tiles
|
|
799
|
-
themselves; `embeddings=sample.npz` sends the
|
|
800
|
-
and on the site.
|
|
862
|
+
themselves; `embeddings=sample.npz` sends the embeddings", in the
|
|
863
|
+
shipped help and on the site.
|
|
801
864
|
"""
|
|
802
865
|
try:
|
|
803
866
|
extensions = contracts.container_extensions(contracts.input_kind(kind)["container"])
|
|
@@ -806,13 +869,52 @@ def _example_of(kind: str) -> str:
|
|
|
806
869
|
return f"{kind}=sample{extensions[0]}" if extensions else f"{kind}=..."
|
|
807
870
|
|
|
808
871
|
|
|
872
|
+
# The covariate entry arrived with the served card's own declaration of it; the
|
|
873
|
+
# date and the change that carried it are in this file's history, and not in the
|
|
874
|
+
# docstring below, which mkdocstrings publishes.
|
|
875
|
+
CATALOGUED_KIND: dict[str, str] = {OBSERVATION_SLOT: "embeddings", COVARIATE_SLOT: "bulk_rna"}
|
|
876
|
+
"""The kind each slot's worked example names when no catalogue is in hand, as
|
|
877
|
+
the published catalogue read when this package was built.
|
|
878
|
+
|
|
879
|
+
DERIVED, and the derivation is [servable_kinds][auroraomics.cli.servable_kinds]
|
|
880
|
+
itself — the same rule applied to the same cards, run against the catalogue the
|
|
881
|
+
contracts publish rather than against one fetched now.
|
|
882
|
+
`test_the_recorded_kind_is_what_the_catalogue_derives_today` recomputes it and
|
|
883
|
+
names the answer it expected, so a card published for another kind turns a
|
|
884
|
+
build red instead of leaving the first journey a reader is shown quietly wrong.
|
|
885
|
+
|
|
886
|
+
RECORDED RATHER THAN FETCHED, because the caller here is help. `build_parser`
|
|
887
|
+
runs on ``--help``, on ``--version`` and on every usage refusal — always before
|
|
888
|
+
a credential is read, usually before one exists, and for a reader whose first
|
|
889
|
+
act is to type the command and see what it says. A value that took a request
|
|
890
|
+
would make that depend on a network, a key and a deployment being up; and the
|
|
891
|
+
answer on a bad day would be a fallback, which is the defect this replaces
|
|
892
|
+
rather than a fix for it. The catalogue a reader can actually submit against is
|
|
893
|
+
one command away and `models` answers it for the deployment they are on.
|
|
894
|
+
|
|
895
|
+
A SLOT WITH NO PUBLISHED KIND IS ABSENT, not guessed at — there is nothing to
|
|
896
|
+
derive from, and inventing something would be this same defect in the other
|
|
897
|
+
flag. Both slots have one now that a served card declares a covariate, so
|
|
898
|
+
``--covariate`` stops showing the registry's own first row and shows the kind a
|
|
899
|
+
reader can actually send. The guard above covers each slot separately, and
|
|
900
|
+
skips the one the catalogue has nothing for.
|
|
901
|
+
"""
|
|
902
|
+
|
|
903
|
+
|
|
809
904
|
def _example(slot: str) -> str:
|
|
810
|
-
"""A whole argument a reader can copy,
|
|
905
|
+
"""A whole argument a reader can copy, with a kind the catalogue accepts.
|
|
811
906
|
|
|
812
907
|
The kind and the file extension are both the contract's; only the stem is
|
|
813
908
|
invented, and a stem is the one part of an example that has to be.
|
|
909
|
+
|
|
910
|
+
Three sources, in order: the catalogue when the caller holds one, the kind
|
|
911
|
+
[CATALOGUED_KIND][auroraomics.cli.CATALOGUED_KIND] recorded for this slot,
|
|
912
|
+
and the registry's first row last. That last one is a tiebreak in a list,
|
|
913
|
+
not an answer about what is served — it printed `patches` for as long as it
|
|
914
|
+
decided this, while the only card served refused it.
|
|
814
915
|
"""
|
|
815
|
-
|
|
916
|
+
recorded = CATALOGUED_KIND.get(slot)
|
|
917
|
+
kinds = servable_kinds(slot) or ((recorded,) if recorded else ()) or kinds_in(slot)
|
|
816
918
|
if not kinds: # pragma: no cover - held by test_the_slots_are_the_registrys_own
|
|
817
919
|
return ""
|
|
818
920
|
first = kinds[0]
|
|
@@ -828,6 +930,8 @@ def _example(slot: str) -> str:
|
|
|
828
930
|
return f"{first}=sample{extensions[0]}" if extensions else f"{first}=..."
|
|
829
931
|
|
|
830
932
|
|
|
933
|
+
# The cards are not vendored into this wheel, which is why the fallback below
|
|
934
|
+
# is the registry and not a card.
|
|
831
935
|
def servable_kinds(slot: str, cards: Iterable[Mapping[str, Any]] = ()) -> tuple[str, ...]:
|
|
832
936
|
"""The kinds the models in ``cards`` will accept, in registry order.
|
|
833
937
|
|
|
@@ -840,11 +944,18 @@ def servable_kinds(slot: str, cards: Iterable[Mapping[str, Any]] = ()) -> tuple[
|
|
|
840
944
|
|
|
841
945
|
``cards`` is EMPTY on every path that builds help text, and that is not an
|
|
842
946
|
oversight. Help is rendered before a command runs and without a credential,
|
|
843
|
-
and the cards are the service's
|
|
947
|
+
and the cards are the service's — this used to read a card
|
|
844
948
|
vendored into the wheel, which meant a build could print the accepts of a
|
|
845
|
-
model the deployment it is pointed at does not serve.
|
|
846
|
-
|
|
847
|
-
|
|
949
|
+
model the deployment it is pointed at does not serve.
|
|
950
|
+
|
|
951
|
+
What an empty catalogue falls back to is
|
|
952
|
+
[CATALOGUED_KIND][auroraomics.cli.CATALOGUED_KIND], which is this function's
|
|
953
|
+
own answer over the published cards, recorded when the package was built and
|
|
954
|
+
held to them by a test. It fell back to the registry's first row until then,
|
|
955
|
+
and that row is a position in a list rather than a statement about what is
|
|
956
|
+
served: it named `patches`, no published card reads that kind, and the
|
|
957
|
+
journey the top-level help opens with sent a reader's whole archive before
|
|
958
|
+
the service could say so.
|
|
848
959
|
"""
|
|
849
960
|
accepted = {
|
|
850
961
|
kind
|
|
@@ -975,6 +1086,22 @@ def _described(text: str) -> str:
|
|
|
975
1086
|
return "\n\n".join(blocks)
|
|
976
1087
|
|
|
977
1088
|
|
|
1089
|
+
def _a_count(text: str) -> int:
|
|
1090
|
+
"""An argparse type for "how many of something", refused at the boundary.
|
|
1091
|
+
|
|
1092
|
+
Zero and below are not small runs, they are a caller who meant something
|
|
1093
|
+
else, and a projection built from either prints a confident nonsense. Caught
|
|
1094
|
+
here, argparse answers with its own usage line before any work starts.
|
|
1095
|
+
"""
|
|
1096
|
+
try:
|
|
1097
|
+
value = int(text)
|
|
1098
|
+
except ValueError:
|
|
1099
|
+
raise argparse.ArgumentTypeError(f"{text!r} is not a whole number") from None
|
|
1100
|
+
if value < 1:
|
|
1101
|
+
raise argparse.ArgumentTypeError(f"{value} is not a count; give at least 1")
|
|
1102
|
+
return value
|
|
1103
|
+
|
|
1104
|
+
|
|
978
1105
|
def build_parser() -> argparse.ArgumentParser:
|
|
979
1106
|
# These two options may be written on EITHER side of the subcommand, and the
|
|
980
1107
|
# top-level help advertises the left-hand side. Making that true takes two
|
|
@@ -1023,7 +1150,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
1023
1150
|
f" auroraomics submit --model ID --observation {_example(OBSERVATION_SLOT)} --wait\n"
|
|
1024
1151
|
" auroraomics download JOB_ID -o result.h5ad\n\n"
|
|
1025
1152
|
f"Build the sample with the Python package: auroraomics.embed_tiles "
|
|
1026
|
-
f"runs {EMBEDDING_COMPONENT} on your machine and sends
|
|
1153
|
+
f"runs {EMBEDDING_COMPONENT} on your machine and sends embeddings instead "
|
|
1027
1154
|
"of pixels, and auroraomics.pack_tiles sends the tiles themselves. "
|
|
1028
1155
|
"`auroraomics models` says which of the two a model takes. "
|
|
1029
1156
|
"`auroraomics doctor` says whether this machine is ready before you "
|
|
@@ -1063,7 +1190,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
1063
1190
|
# Positional, because this is the first command anyone runs and the address
|
|
1064
1191
|
# is the whole of it: `auroraomics login you@institution.edu` is what every
|
|
1065
1192
|
# page prints and what a person types. It was `--email` and the two
|
|
1066
|
-
# disagreed for as long as the pages existed
|
|
1193
|
+
# disagreed for as long as the pages existed.
|
|
1067
1194
|
login_parser.add_argument("email", metavar="EMAIL", help="the address of the account to sign in")
|
|
1068
1195
|
login_parser.add_argument(
|
|
1069
1196
|
"--code",
|
|
@@ -1095,7 +1222,13 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
1095
1222
|
"Checks the installed version, the contract documents shipped with "
|
|
1096
1223
|
"it, the stored credential and whether the deployment is taking "
|
|
1097
1224
|
"work. Each check prints ok or FAIL; the exit code is 0 only when "
|
|
1098
|
-
"every one passed
|
|
1225
|
+
"every one passed.\n\n"
|
|
1226
|
+
"--model answers the two questions worth asking before an "
|
|
1227
|
+
"hours-long run, in one forward pass: will this machine's numbers "
|
|
1228
|
+
"be accepted, and what does a tile cost here. The published "
|
|
1229
|
+
"throughput figures are a CUDA device's; a CPU is a different order "
|
|
1230
|
+
"of magnitude, and yours is the one that matters. Add --tiles to "
|
|
1231
|
+
"turn that into a wall clock for the slide."
|
|
1099
1232
|
),
|
|
1100
1233
|
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
1101
1234
|
)
|
|
@@ -1111,7 +1244,36 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
1111
1244
|
help=(
|
|
1112
1245
|
f"also run {EMBEDDING_COMPONENT} over the reference image shipped in this "
|
|
1113
1246
|
"package and compare the result with this model's reference, which "
|
|
1114
|
-
"answers whether
|
|
1247
|
+
"answers whether embeddings from this machine would be accepted — "
|
|
1248
|
+
"and time the pass, which answers how long the slide will take"
|
|
1249
|
+
),
|
|
1250
|
+
)
|
|
1251
|
+
doctor.add_argument(
|
|
1252
|
+
"--tiles",
|
|
1253
|
+
type=_a_count,
|
|
1254
|
+
default=None,
|
|
1255
|
+
metavar="N",
|
|
1256
|
+
help=(
|
|
1257
|
+
"how many tiles you mean to embed; turns the timed pass of --model "
|
|
1258
|
+
"into a wall clock for the whole run before you start it"
|
|
1259
|
+
),
|
|
1260
|
+
)
|
|
1261
|
+
# THE ADVICE THE FAILURE GIVES HAS TO BE TAKEABLE BY WHOEVER READS IT.
|
|
1262
|
+
# Without `--model` this check is not run at all and nothing fetches
|
|
1263
|
+
# anything; with it and the weights uncached, the refusal says to permit the
|
|
1264
|
+
# download — and said it as `allow_download=True`, a PYTHON keyword
|
|
1265
|
+
# argument, to a reader who is holding a shell. There was no CLI form of it
|
|
1266
|
+
# and no other subcommand that fetched weights either, so the one machine
|
|
1267
|
+
# this command exists to get ready could not be got ready from the command
|
|
1268
|
+
# line. Opt-in, and default off, for the reason `calibrate` states:
|
|
1269
|
+
# a diagnostic does not start a gigabyte download nobody asked for.
|
|
1270
|
+
doctor.add_argument(
|
|
1271
|
+
"--allow-download",
|
|
1272
|
+
action="store_true",
|
|
1273
|
+
help=(
|
|
1274
|
+
"let --model fetch the weights it needs if this machine does not "
|
|
1275
|
+
"already hold them; without it a missing encoder is reported, not "
|
|
1276
|
+
"downloaded"
|
|
1115
1277
|
),
|
|
1116
1278
|
)
|
|
1117
1279
|
doctor.set_defaults(handler=cmd_doctor)
|
|
@@ -1181,7 +1343,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
1181
1343
|
"Send one sample for prediction and print the job it created.\n\n"
|
|
1182
1344
|
"The sample is an --observation. "
|
|
1183
1345
|
f"`{_example_of(packed_kind)}` sends the tiles themselves; "
|
|
1184
|
-
f"`{_example_of(embedded_kind)}` sends the
|
|
1346
|
+
f"`{_example_of(embedded_kind)}` sends the embeddings "
|
|
1185
1347
|
f"{EMBEDDING_COMPONENT} computed from them on your machine, so the "
|
|
1186
1348
|
"pixels never leave it. Either way a path is uploaded first, and an "
|
|
1187
1349
|
"upload id you already hold is used as it stands.\n\n"
|
|
@@ -1233,7 +1395,13 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
1233
1395
|
submit.add_argument(
|
|
1234
1396
|
"--land-in-workspace",
|
|
1235
1397
|
action="store_true",
|
|
1236
|
-
|
|
1398
|
+
# This said "open the result in the web workspace", and nothing opens
|
|
1399
|
+
# anything: the result is offered to the same account's own workspace
|
|
1400
|
+
# inbox, which nobody else reads, and importing it into a workspace is
|
|
1401
|
+
# a separate step the account holder takes. A flag that moves data has
|
|
1402
|
+
# to say what it does, because the privacy page that describes the
|
|
1403
|
+
# destination and who can see it from there is written against this.
|
|
1404
|
+
help="offer the finished result to the same account's web workspace, to import there",
|
|
1237
1405
|
)
|
|
1238
1406
|
submit.add_argument(
|
|
1239
1407
|
"--idempotency-key",
|