auroraomics 0.1.0.dev1__tar.gz → 0.1.0.dev2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. {auroraomics-0.1.0.dev1/src/auroraomics.egg-info → auroraomics-0.1.0.dev2}/PKG-INFO +5 -4
  2. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/README.md +4 -3
  3. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/pyproject.toml +11 -11
  4. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/__init__.py +5 -5
  5. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_contracts/MANIFEST.json +1 -1
  6. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_contracts/public-api.tokens.json +8 -2
  7. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_text.py +1 -1
  8. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/bulk.py +7 -7
  9. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/cli.py +211 -43
  10. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/_http.py +4 -4
  11. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/api.py +64 -10
  12. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/credentials.py +14 -3
  13. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/contracts.py +26 -12
  14. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/embed.py +449 -98
  15. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/errors.py +1 -1
  16. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/genes.py +3 -3
  17. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/pack.py +125 -8
  18. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/postprocess.py +1 -1
  19. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/qc.py +4 -3
  20. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/runtimes/__init__.py +0 -8
  21. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/runtimes/deepspotm.py +210 -66
  22. auroraomics-0.1.0.dev2/src/auroraomics/slide.py +793 -0
  23. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/spatial.py +7 -0
  24. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2/src/auroraomics.egg-info}/PKG-INFO +5 -4
  25. auroraomics-0.1.0.dev1/src/auroraomics/slide.py +0 -325
  26. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/LICENSE +0 -0
  27. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/MANIFEST.in +0 -0
  28. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/setup.cfg +0 -0
  29. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_assets/ASSETS.json +0 -0
  30. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_assets/README.md +0 -0
  31. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_assets/calibration-tile.png +0 -0
  32. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_contracts/public-api-counters.tokens.json +0 -0
  33. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/_contracts/public-api-input-kinds.tokens.json +0 -0
  34. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/__init__.py +0 -0
  35. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/_generated.py +0 -0
  36. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/errors.py +0 -0
  37. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/jobs.py +0 -0
  38. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/client/uploads.py +0 -0
  39. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/h5ad.py +0 -0
  40. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/mcp/__init__.py +0 -0
  41. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/mcp/server.py +0 -0
  42. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/mcp/tools.py +0 -0
  43. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/py.typed +0 -0
  44. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics/subsample.py +0 -0
  45. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics.egg-info/SOURCES.txt +0 -0
  46. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics.egg-info/dependency_links.txt +0 -0
  47. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics.egg-info/entry_points.txt +0 -0
  48. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics.egg-info/requires.txt +0 -0
  49. {auroraomics-0.1.0.dev1 → auroraomics-0.1.0.dev2}/src/auroraomics.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: auroraomics
3
- Version: 0.1.0.dev1
3
+ Version: 0.1.0.dev2
4
4
  Summary: Virtual spatial transcriptomics from H&E histology: patch QC, tile packing and the .h5ad result contract.
5
5
  Author: Kalin Nonchev
6
6
  License-Expression: PolyForm-Noncommercial-1.0.0
@@ -205,10 +205,11 @@ its weights never leave it. What crosses between us is a vector from a model
205
205
  neither side owns.
206
206
 
207
207
  ```python
208
- from auroraomics.runtimes.deepspotm import available_models, check_archive
208
+ from auroraomics.runtimes.deepspotm import check_archive, describe_models
209
209
 
210
- available_models() # what the service will run
211
- check_archive("sample.zip", model=...) # are my tiles the right physical size?
210
+ cards = client.models() # the catalogue is the service's
211
+ describe_models(cards) # what it will run, and on what
212
+ check_archive("sample.zip", model=MODEL, cards=cards) # are my tiles the right size?
212
213
  ```
213
214
 
214
215
  Then submit with the API client and read the result back — the same `.h5ad`
@@ -162,10 +162,11 @@ its weights never leave it. What crosses between us is a vector from a model
162
162
  neither side owns.
163
163
 
164
164
  ```python
165
- from auroraomics.runtimes.deepspotm import available_models, check_archive
165
+ from auroraomics.runtimes.deepspotm import check_archive, describe_models
166
166
 
167
- available_models() # what the service will run
168
- check_archive("sample.zip", model=...) # are my tiles the right physical size?
167
+ cards = client.models() # the catalogue is the service's
168
+ describe_models(cards) # what it will run, and on what
169
+ check_archive("sample.zip", model=MODEL, cards=cards) # are my tiles the right size?
169
170
  ```
170
171
 
171
172
  Then submit with the API client and read the result back — the same `.h5ad`
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "auroraomics"
7
- version = "0.1.0.dev1"
7
+ version = "0.1.0.dev2"
8
8
  description = "Virtual spatial transcriptomics from H&E histology: patch QC, tile packing and the .h5ad result contract."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -31,7 +31,7 @@ classifiers = [
31
31
  # install line the quickstart prints is an install line that can do what the
32
32
  # quickstart says.
33
33
  #
34
- # It was four names until #691, with the API client and the slide reader behind
34
+ # It was four names once, with the API client and the slide reader behind
35
35
  # extras of their own. The comment on the extras below has always ended "the
36
36
  # line is at the ENCODER, not at the size of an install"; these lists did not
37
37
  # say it. They put a half-megabyte HTTP client behind the same ceremony as a
@@ -39,7 +39,7 @@ classifiers = [
39
39
  # auroraomics` could not then perform a single documented step. They say it now.
40
40
  #
41
41
  # WHAT IT COSTS, measured rather than estimated (importlib.metadata over the
42
- # transitive closure, on the interpreter this package targets, 2026-09-11): four
42
+ # transitive closure, on the interpreter this package targets): four
43
43
  # distributions and 179 MB became twenty-eight and 537 MB. Neither of the two
44
44
  # that dominate is named here — `imagecodecs` (142 MB) is what lets `tifffile`
45
45
  # decode a real slide, and `pandas` plus `scipy` (198 MB between them) arrive
@@ -79,7 +79,7 @@ dependencies = [
79
79
  # resolves the model, or the adapter and checkpoint stack that would carry one,
80
80
  # and `predict_local` refuses with a message that says so. What runs on a user's
81
81
  # own machine is cutting, filtering and EMBEDDING tiles — their pixels stay put,
82
- # and a vector from an open-weight encoder is what crosses. So the line is at the
82
+ # and a vector from the encoder is what crosses. So the line is at the
83
83
  # ENCODER, not at the size of an install: `embed` below does resolve a tensor
84
84
  # runtime, on purpose, and everything on the CLIENT side of the encoder — the
85
85
  # HTTP client, the slide reader — sits in the core above, where nobody has to
@@ -88,7 +88,7 @@ dependencies = [
88
88
  # encoder's four stay named, no shipped module imports either, and the release
89
89
  # guard refuses a wheel carrying a module that is not on its published list.
90
90
  #
91
- # `client` and `slide` were extras until #691 and are GONE rather than kept as
91
+ # `client` and `slide` were extras once and are GONE rather than kept as
92
92
  # empty aliases. The published wheel declares both, so `pip install
93
93
  # "auroraomics[client]"` exists in the wild and in our own older prose; pip
94
94
  # WARNS on an extra a distribution does not provide and installs anyway —
@@ -121,7 +121,7 @@ mcp = ["mcp>=2,<3"]
121
121
  #
122
122
  # anndata used to be listed here as a test-only dependency, with a note saying
123
123
  # that depending on it at run time would put pandas and its stack into every
124
- # install. That is now exactly what happens, knowingly (#691) — the note above
124
+ # install. That is now exactly what happens, knowingly — the note above
125
125
  # the core list records the price — so this extra no longer names anndata, nor
126
126
  # `tifffile` or `imagecodecs`, because the core already does. A second bound
127
127
  # here would be a test environment resolving what no user resolves. anndata is
@@ -141,7 +141,7 @@ dev = [
141
141
  # Coverage is measured on every run, with a floor, because this package is
142
142
  # the one that SHIPS: a user installs it and calls it, so a branch nothing
143
143
  # here exercises is a branch discovered in the field. The floor sat at
144
- # nothing until 2026-09-06, when the suite measured 94%.
144
+ # nothing until the suite first measured 94%.
145
145
  "pytest-cov>=5",
146
146
  # setuptools and wheel are TEST dependencies as well as build ones: the
147
147
  # release guard builds with --no-isolation so that it needs no network, and
@@ -223,8 +223,8 @@ source = ["src/auroraomics"]
223
223
  # when a real gap closes; never lower it to make a change pass.
224
224
  #
225
225
  # 92 rather than the 94.18 measured hours earlier, and the difference is not a
226
- # regression in anything tested here. #519 turned `predict_local` into a refusal
227
- # and moved the forward pass to the unpublished distribution — but left
226
+ # regression in anything tested here. Turning `predict_local` into a refusal
227
+ # moved the forward pass to the unpublished distribution — but left
228
228
  # `runtimes/deepspotm.py`'s `_panel_rows` and `_selection` behind, and their
229
229
  # ONLY callers are now in the unpublished inference distribution's runner. So
230
230
  # that file went from 271 statements at 99% to 169 at 73%: 46 statements this
@@ -233,8 +233,8 @@ source = ["src/auroraomics"]
233
233
  #
234
234
  # Moving those two helpers to the package that uses them takes the floor back
235
235
  # above 94 without writing a single test. Recorded as an issue rather than done
236
- # here, because it is a change to the boundary #519 drew and belongs with that
237
- # work, not with a coverage floor.
236
+ # here, because it is a change to the boundary that refusal drew and belongs
237
+ # with that work, not with a coverage floor.
238
238
  fail_under = 92
239
239
  # Two decimals, because `fail_under` is compared at this precision and rounding
240
240
  # 93.6 to 94 would pass a suite that had slipped.
@@ -1,7 +1,7 @@
1
1
  """Virtual spatial transcriptomics from H&E histology.
2
2
 
3
3
  **One journey, and one word chooses which.** Send the tiles, or send only the
4
- vectors computed from them; everything either side of that word is the same:
4
+ embeddings computed from them; everything either side of that word is the same:
5
5
 
6
6
  import auroraomics as ao
7
7
 
@@ -105,10 +105,10 @@ _EXPORTS: dict[str, str] = {
105
105
 
106
106
  Two properties, and the table buys both.
107
107
 
108
- ``resolve_encoder`` is here because since #694 it is on the journey rather than
109
- beside it: ``embed_tiles`` takes a resolved encoder and the registry is the
110
- service's, so ``resolve_encoder(client.encoders(), name)`` is a line every
111
- caller who embeds tiles writes by hand.
108
+ ``resolve_encoder`` is here because it is on the journey rather than beside it:
109
+ ``embed_tiles`` takes a resolved encoder and the registry is the service's, so
110
+ ``resolve_encoder(client.encoders(), name)`` is a line every caller who embeds
111
+ tiles writes by hand.
112
112
 
113
113
  **It is short.** Thirty-three names stood here while the two journeys through
114
114
  this package used two of them between them, because a module's whole public
@@ -3,6 +3,6 @@
3
3
  "files": {
4
4
  "public-api-counters.tokens.json": "4f5dd14ce76f7001d36940dfc1d1f49feac9ec1be2e7646d47b29ab21c124a8f",
5
5
  "public-api-input-kinds.tokens.json": "1c80e2637d27f2491ea178e2fa87e2a435640e5445791b2fd20adc53a574e54d",
6
- "public-api.tokens.json": "43ecf0f47c2d480464e7a936c0522041a74803c47a364432ba62f356a4a699d5"
6
+ "public-api.tokens.json": "a8d3253978aba375c1933aef3f3069bcb4b5e47c2edbf6062a08c95c0f255f4d"
7
7
  }
8
8
  }
@@ -354,8 +354,14 @@
354
354
  "uns_keys": [
355
355
  "model",
356
356
  "input_spec",
357
- "job"
358
- ]
357
+ "job",
358
+ "resolution"
359
+ ],
360
+ "resolution_checks": {
361
+ "slide": "slide_metadata",
362
+ "none": "nothing",
363
+ "unstated": "unstated"
364
+ }
359
365
  },
360
366
  "rateBuckets": {
361
367
  "read": {
@@ -38,7 +38,7 @@ def not_one_file(path, *, instead: str) -> str:
38
38
  return f"{path} is not a regular file; {instead}"
39
39
 
40
40
  EMBEDDING_COMPONENT = "DeepSpot-H"
41
- """The step that turns tiles into vectors, named as a reader meets it.
41
+ """The step that turns tiles into embeddings, named as a reader meets it.
42
42
 
43
43
  Typed once, and imported by every message that names it. It was written out
44
44
  five times across `cli.py` and `runtimes/deepspotm.py` — the defect the
@@ -157,10 +157,10 @@ Resolver = Callable[[Sequence[str]], Mapping[str, "str | None"]]
157
157
  [Client.resolve_keys][auroraomics.client.api.Client.resolve_keys] is one, and so is
158
158
  any callable a caller writes over their own mapping.
159
159
 
160
- An ARGUMENT since #694, because the gene table is the model's own output axis
161
- and the service holds it — this module used to read a 3.7 MB copy vendored into
162
- the wheel, which was frozen at build time and was a second implementation of the
163
- service's resolution order besides.
160
+ An ARGUMENT rather than a lookup this module performs, because the gene table is
161
+ the model's own output axis and the service holds it — this module used to read a
162
+ 3.7 MB copy vendored into the wheel, which was frozen at build time and was a
163
+ second implementation of the service's resolution order besides.
164
164
  """
165
165
 
166
166
 
@@ -446,7 +446,7 @@ def _asked_once(resolve: Resolver) -> Resolver:
446
446
  decides the layout by resolving ``header[1:]``, and then returns exactly
447
447
  those names as the pairs — which ``_resolve_all`` resolves again. Two
448
448
  identical passes over a transcriptome's worth of names, and since resolution
449
- became the service's (#694), two passes over the NETWORK: measured on a
449
+ became the service's, two passes over the NETWORK: measured on a
450
450
  2,000-column table, the resolver was called twice with 2.0x the header.
451
451
 
452
452
  Per CALL, not per process. A resolver is a live connection to one
@@ -513,7 +513,7 @@ def _coverage_reference(card: Mapping[str, Any]) -> tuple[frozenset[str], str]:
513
513
  """The gene set coverage is judged against: the model's measured set.
514
514
 
515
515
  The card's, always. It used to fall back to the whole vendored gene table for
516
- a caller who named no model, and that fallback went with the table (#694) —
516
+ a caller who named no model, and that fallback went with the table —
517
517
  which is no loss: coverage of "every gene the family can predict" was never
518
518
  the number the service refuses on.
519
519
  """
@@ -571,7 +571,7 @@ def bulk_rna(
571
571
  ``output.gene_mask`` is the measured gene set coverage is judged
572
572
  against — the same refusal the service would make. It used to be
573
573
  optional, with the vendored gene table standing in; the table is the
574
- service's now (#694), and "coverage of every gene the family predicts"
574
+ service's now, and "coverage of every gene the family predicts"
575
575
  was never a number the service refuses on.
576
576
  resolve:
577
577
  How a symbol becomes the Ensembl id a result is keyed by:
@@ -238,15 +238,11 @@ fetching a finished job's result is what that command is for — while
238
238
  and holds both directions: a command declaring ``--wait`` with no row here, or
239
239
  a row naming a flag that command does not declare, is a red build.
240
240
  """
241
-
242
241
  OUTPUT_NEEDS_WAIT = f"needs --wait: {NEEDS_WAIT['submit']['output']}"
243
242
  """What ``submit --output``'s help says, DERIVED from the table the refusal
244
243
  reads. It is the flag's own help, the sentence the command's description
245
244
  carries, and what the refusal says — and those three drifted once already, which
246
- is why it is one string. It stopped being one for an hour on 2026-09-10, when
247
- the refusal moved into a table and the help did not: the existing test caught
248
- it, which is the whole reason that test names the constant rather than a
249
- phrase."""
245
+ is why it is one string."""
250
246
 
251
247
  ACCESS_PENDING_CODE = "access_pending"
252
248
  """The mint refusal for an account that has asked and is waiting; the catalogue's code."""
@@ -427,7 +423,7 @@ def _calibration_check(args: argparse.Namespace, model_id: str) -> dict[str, Any
427
423
  Every other check in ``doctor`` is cheap, so this one runs only when a
428
424
  model is named: it loads an encoder and runs one forward pass. What it buys
429
425
  is the answer to the question worth asking before a slide is embedded — will
430
- the service accept vectors from this machine — asked once, on one image,
426
+ the service accept embeddings from this machine — asked once, on one image,
431
427
  instead of after an hour of tiles.
432
428
 
433
429
  Reported, never raised, like every check here: each way it cannot be
@@ -444,7 +440,7 @@ def _calibration_check(args: argparse.Namespace, model_id: str) -> dict[str, Any
444
440
  cards = {str(card.get("id")): card for card in client.models()}
445
441
  # Fetched in the same session as the cards, because both are the
446
442
  # service's now: an encoder is a pinned repository and revision, and
447
- # this package stopped carrying a copy of that pin with #694.
443
+ # this package no longer carries a copy of that pin.
448
444
  registry = client.encoders()
449
445
  except Exception as exc: # noqa: BLE001 - a diagnostic reports, it does not raise
450
446
  return verdict(False, str(exc))
@@ -455,7 +451,7 @@ def _calibration_check(args: argparse.Namespace, model_id: str) -> dict[str, Any
455
451
  reads = accepted_observations(card)
456
452
  locally = _embedded_locally()
457
453
  if reads and locally and locally not in reads:
458
- # This check asks whether vectors from THIS machine would be accepted. A
454
+ # This check asks whether embeddings from THIS machine would be accepted. A
459
455
  # model that reads the tiles themselves is never handed one, so there is
460
456
  # no question to ask — and asking it anyway loaded an encoder the caller
461
457
  # on that route has every reason not to have installed, then reported
@@ -490,41 +486,108 @@ def _calibration_check(args: argparse.Namespace, model_id: str) -> dict[str, Any
490
486
 
491
487
  try:
492
488
  name = _encoder_name(spec, embed.describe_encoders(registry))
493
- vector = _as_a_file_holds_it(
494
- embed.calibrate(embed.resolve_encoder(registry, name)), spec
495
- )
489
+ encoder = embed.resolve_encoder(registry, name)
490
+ run = embed.calibration_pass(encoder, allow_download=args.allow_download)
491
+ vector, stored = _as_a_file_would_hold_it(run.embedding, spec)
496
492
  distance = embed.calibration_offset(vector, reference["reference"])
497
493
  except Exception as exc: # noqa: BLE001
498
494
  return verdict(False, str(exc))
499
495
  within = distance <= tolerance
496
+ pace, projection = _pace(run, args.tiles)
500
497
  return verdict(
501
498
  within,
502
499
  f"{name} on this machine is {distance:.3g} from {model_id}'s reference "
503
- f"(tolerance {tolerance:.3g})"
504
- + ("" if within else "; vectors from this machine would be refused"),
500
+ f"(tolerance {tolerance:.3g}, measured as {stored} — the element type a "
501
+ "file written here carries)"
502
+ + ("" if within else "; embeddings from this machine would be refused")
503
+ + f". {pace}",
505
504
  model=model_id,
506
505
  encoder=name,
507
506
  offset=distance,
508
507
  tolerance=tolerance,
508
+ dtype=stored,
509
+ device=run.device,
510
+ seconds_per_tile=run.seconds,
511
+ **projection,
509
512
  )
510
513
 
511
514
 
512
- def _as_a_file_holds_it(vector: Any, spec: dict[str, Any]) -> Any:
513
- """The forward pass rounded into the element type the card pins.
515
+ def _as_a_file_would_hold_it(vector: Any, spec: dict[str, Any]) -> tuple[Any, str]:
516
+ """The forward pass rounded the way the file this machine writes will be.
517
+
518
+ The lane compares the numbers a submission STORES, so measuring the raw
519
+ forward pass here would make the client kinder than the service by exactly
520
+ that rounding, at the boundary the tolerance sits on. A machine told it
521
+ agrees and an hour of embedding refused afterwards is the failure this whole
522
+ check exists to prevent.
523
+
524
+ TWO things claim to say what a file stores, and they are not the same thing:
525
+ the card pins an ``input_spec.dtype``, and
526
+ [embed_tiles][auroraomics.embed.embed_tiles] writes at
527
+ [DEFAULT_DTYPE][auroraomics.embed.DEFAULT_DTYPE] — it is handed an encoder,
528
+ not a model, so it never reads the card. Today the two agree, by coincidence
529
+ of value rather than by construction, and a card that moved to the wider
530
+ type would leave this check measuring a rounding no file here would carry.
531
+ So it measures at whichever of them rounds HARDER, and names it: being
532
+ stricter than the lane costs a caller a second look, being kinder costs them
533
+ the afternoon.
534
+
535
+ A ``dtype`` the container does not allow is not one of the two: such a file
536
+ cannot be submitted at all, and it is its element type that would be
537
+ refused, not its calibration — so the writer's own is used alone.
538
+ """
539
+ import numpy as np
540
+
541
+ from auroraomics import embed
514
542
 
515
- The lane compares the numbers a submission STORES, and the writer stores
516
- this vector in the card's ``dtype`` — so measuring the raw forward pass
517
- here would make the client kinder than the service by that rounding, at
518
- exactly the boundary the tolerance sits on. A machine told it agrees and an
519
- hour of embedding refused afterwards is the failure this whole check exists
520
- to prevent.
543
+ allowed = contracts.embeddings_npz()["dtypes"]
544
+ pinned = str(spec.get("dtype") or "")
545
+ candidates = [embed.DEFAULT_DTYPE, *([pinned] if pinned in allowed else [])]
546
+ # Narrowest wins, by the element type's own width rather than by a list
547
+ # written here in an order someone has to keep right.
548
+ stored = min(candidates, key=lambda name: np.dtype(name).itemsize)
549
+ return vector.astype(stored), stored
521
550
 
522
- A ``dtype`` the container does not allow is left alone: such a file cannot
523
- be submitted at all, and it is its element type that would be refused, not
524
- its calibration.
551
+
552
+ def _pace(run: Any, tiles: int | None) -> tuple[str, dict[str, Any]]:
553
+ """What a tile costs on THIS machine, and what that makes of a slide.
554
+
555
+ The one number `doctor` was already measuring and throwing away. The
556
+ published table for ``batch_size`` is a CUDA device's, and the lane most
557
+ callers run on is the CPU — a different order of magnitude on the same
558
+ encoder — so "minutes to hours" was a range a caller only resolved by
559
+ starting. A machine can be asked, in the pass that was happening anyway.
560
+
561
+ Stated as an upper bound, because that is what it is: one tile to a pass,
562
+ where a run feeds ``batch_size`` of them and pays less per tile.
525
563
  """
526
- stored = str(spec.get("dtype") or "")
527
- return vector.astype(stored) if stored in contracts.embeddings_npz()["dtypes"] else vector
564
+ per = f"One tile takes {run.seconds * 1000:,.0f} ms on {run.device} here"
565
+ if not tiles:
566
+ return (
567
+ f"{per}, one tile to a pass — an upper bound, since a run feeds a "
568
+ "whole batch at a time. Pass --tiles to turn that into a slide's "
569
+ "wall clock.",
570
+ {},
571
+ )
572
+ total = run.seconds * tiles
573
+ return (
574
+ f"{per}, so {tiles:,} tiles is at most {_how_long(total)} — one tile to "
575
+ "a pass, and a run feeds a whole batch at a time for less per tile.",
576
+ {"tiles": int(tiles), "estimated_seconds": total},
577
+ )
578
+
579
+
580
+ def _how_long(seconds: float) -> str:
581
+ """A duration in the unit a reader decides with, never in bare seconds.
582
+
583
+ "14,073 s" and "3.9 hours" are the same number and only one of them answers
584
+ "do I start this before lunch".
585
+ """
586
+ if seconds < 90:
587
+ return f"{seconds:.0f} seconds"
588
+ if seconds < 90 * 60:
589
+ return f"{seconds / 60:.0f} minutes"
590
+ return f"{seconds / 3600:.1f} hours"
528
591
 
529
592
 
530
593
  def _encoder_name(spec: dict[str, Any], known: dict[str, Any]) -> str:
@@ -550,7 +613,7 @@ def _encoder_name(spec: dict[str, Any], known: dict[str, Any]) -> str:
550
613
  if differing:
551
614
  raise errors.ClientError(
552
615
  f"the card's {', '.join(differing)} differ(s) from what this release pins for "
553
- f"{name}; upgrade the package rather than trust a vector computed another way"
616
+ f"{name}; upgrade the package rather than trust an embedding computed another way"
554
617
  )
555
618
  return name
556
619
  raise errors.ClientError(
@@ -747,7 +810,7 @@ def kind_charged_in(unit: str) -> str:
747
810
  """The observation kind the registry charges by ``unit``.
748
811
 
749
812
  The two routes differ in what a submission IS, and the registry already
750
- says so: the tiles are charged per tile, the vectors per row. So the
813
+ says so: the tiles are charged per tile, the embeddings per row. So the
751
814
  sentence that describes each route reads its kind from the contract instead
752
815
  of naming it.
753
816
 
@@ -771,7 +834,7 @@ def _embedded_locally() -> str:
771
834
  """The observation kind a caller computes on their own machine, or ``""``.
772
835
 
773
836
  The two routes differ in where tiles become numbers, and the registry
774
- already says which is which by what it charges: the vectors are charged per
837
+ already says which is which by what it charges: the embeddings are charged per
775
838
  ROW. Read from there rather than from ``embed.KIND``, for the reason
776
839
  [kind_charged_in][auroraomics.cli.kind_charged_in] gives — this runs inside
777
840
  ``doctor``, which is the command an install with no encoder runtime in it is
@@ -796,8 +859,8 @@ def _example_of(kind: str) -> str:
796
859
 
797
860
  Mixing the two put the same example in both halves of one sentence, so the
798
861
  submit description read "`embeddings=sample.npz` sends the tiles
799
- themselves; `embeddings=sample.npz` sends the vectors", in the shipped help
800
- and on the site.
862
+ themselves; `embeddings=sample.npz` sends the embeddings", in the
863
+ shipped help and on the site.
801
864
  """
802
865
  try:
803
866
  extensions = contracts.container_extensions(contracts.input_kind(kind)["container"])
@@ -806,13 +869,52 @@ def _example_of(kind: str) -> str:
806
869
  return f"{kind}=sample{extensions[0]}" if extensions else f"{kind}=..."
807
870
 
808
871
 
872
+ # The covariate entry arrived with the served card's own declaration of it; the
873
+ # date and the change that carried it are in this file's history, and not in the
874
+ # docstring below, which mkdocstrings publishes.
875
+ CATALOGUED_KIND: dict[str, str] = {OBSERVATION_SLOT: "embeddings", COVARIATE_SLOT: "bulk_rna"}
876
+ """The kind each slot's worked example names when no catalogue is in hand, as
877
+ the published catalogue read when this package was built.
878
+
879
+ DERIVED, and the derivation is [servable_kinds][auroraomics.cli.servable_kinds]
880
+ itself — the same rule applied to the same cards, run against the catalogue the
881
+ contracts publish rather than against one fetched now.
882
+ `test_the_recorded_kind_is_what_the_catalogue_derives_today` recomputes it and
883
+ names the answer it expected, so a card published for another kind turns a
884
+ build red instead of leaving the first journey a reader is shown quietly wrong.
885
+
886
+ RECORDED RATHER THAN FETCHED, because the caller here is help. `build_parser`
887
+ runs on ``--help``, on ``--version`` and on every usage refusal — always before
888
+ a credential is read, usually before one exists, and for a reader whose first
889
+ act is to type the command and see what it says. A value that took a request
890
+ would make that depend on a network, a key and a deployment being up; and the
891
+ answer on a bad day would be a fallback, which is the defect this replaces
892
+ rather than a fix for it. The catalogue a reader can actually submit against is
893
+ one command away and `models` answers it for the deployment they are on.
894
+
895
+ A SLOT WITH NO PUBLISHED KIND IS ABSENT, not guessed at — there is nothing to
896
+ derive from, and inventing something would be this same defect in the other
897
+ flag. Both slots have one now that a served card declares a covariate, so
898
+ ``--covariate`` stops showing the registry's own first row and shows the kind a
899
+ reader can actually send. The guard above covers each slot separately, and
900
+ skips the one the catalogue has nothing for.
901
+ """
902
+
903
+
809
904
  def _example(slot: str) -> str:
810
- """A whole argument a reader can copy, built from the registry's first kind.
905
+ """A whole argument a reader can copy, with a kind the catalogue accepts.
811
906
 
812
907
  The kind and the file extension are both the contract's; only the stem is
813
908
  invented, and a stem is the one part of an example that has to be.
909
+
910
+ Three sources, in order: the catalogue when the caller holds one, the kind
911
+ [CATALOGUED_KIND][auroraomics.cli.CATALOGUED_KIND] recorded for this slot,
912
+ and the registry's first row last. That last one is a tiebreak in a list,
913
+ not an answer about what is served — it printed `patches` for as long as it
914
+ decided this, while the only card served refused it.
814
915
  """
815
- kinds = servable_kinds(slot) or kinds_in(slot)
916
+ recorded = CATALOGUED_KIND.get(slot)
917
+ kinds = servable_kinds(slot) or ((recorded,) if recorded else ()) or kinds_in(slot)
816
918
  if not kinds: # pragma: no cover - held by test_the_slots_are_the_registrys_own
817
919
  return ""
818
920
  first = kinds[0]
@@ -828,6 +930,8 @@ def _example(slot: str) -> str:
828
930
  return f"{first}=sample{extensions[0]}" if extensions else f"{first}=..."
829
931
 
830
932
 
933
+ # The cards are not vendored into this wheel, which is why the fallback below
934
+ # is the registry and not a card.
831
935
  def servable_kinds(slot: str, cards: Iterable[Mapping[str, Any]] = ()) -> tuple[str, ...]:
832
936
  """The kinds the models in ``cards`` will accept, in registry order.
833
937
 
@@ -840,11 +944,18 @@ def servable_kinds(slot: str, cards: Iterable[Mapping[str, Any]] = ()) -> tuple[
840
944
 
841
945
  ``cards`` is EMPTY on every path that builds help text, and that is not an
842
946
  oversight. Help is rendered before a command runs and without a credential,
843
- and the cards are the service's since #694 — this used to read a card
947
+ and the cards are the service's — this used to read a card
844
948
  vendored into the wheel, which meant a build could print the accepts of a
845
- model the deployment it is pointed at does not serve. An empty catalogue
846
- falls back to the registry: an example is better than none, and `models` is
847
- one command away and answers for the deployment the caller is actually on.
949
+ model the deployment it is pointed at does not serve.
950
+
951
+ What an empty catalogue falls back to is
952
+ [CATALOGUED_KIND][auroraomics.cli.CATALOGUED_KIND], which is this function's
953
+ own answer over the published cards, recorded when the package was built and
954
+ held to them by a test. It fell back to the registry's first row until then,
955
+ and that row is a position in a list rather than a statement about what is
956
+ served: it named `patches`, no published card reads that kind, and the
957
+ journey the top-level help opens with sent a reader's whole archive before
958
+ the service could say so.
848
959
  """
849
960
  accepted = {
850
961
  kind
@@ -975,6 +1086,22 @@ def _described(text: str) -> str:
975
1086
  return "\n\n".join(blocks)
976
1087
 
977
1088
 
1089
+ def _a_count(text: str) -> int:
1090
+ """An argparse type for "how many of something", refused at the boundary.
1091
+
1092
+ Zero and below are not small runs, they are a caller who meant something
1093
+ else, and a projection built from either prints a confident nonsense. Caught
1094
+ here, argparse answers with its own usage line before any work starts.
1095
+ """
1096
+ try:
1097
+ value = int(text)
1098
+ except ValueError:
1099
+ raise argparse.ArgumentTypeError(f"{text!r} is not a whole number") from None
1100
+ if value < 1:
1101
+ raise argparse.ArgumentTypeError(f"{value} is not a count; give at least 1")
1102
+ return value
1103
+
1104
+
978
1105
  def build_parser() -> argparse.ArgumentParser:
979
1106
  # These two options may be written on EITHER side of the subcommand, and the
980
1107
  # top-level help advertises the left-hand side. Making that true takes two
@@ -1023,7 +1150,7 @@ def build_parser() -> argparse.ArgumentParser:
1023
1150
  f" auroraomics submit --model ID --observation {_example(OBSERVATION_SLOT)} --wait\n"
1024
1151
  " auroraomics download JOB_ID -o result.h5ad\n\n"
1025
1152
  f"Build the sample with the Python package: auroraomics.embed_tiles "
1026
- f"runs {EMBEDDING_COMPONENT} on your machine and sends vectors instead "
1153
+ f"runs {EMBEDDING_COMPONENT} on your machine and sends embeddings instead "
1027
1154
  "of pixels, and auroraomics.pack_tiles sends the tiles themselves. "
1028
1155
  "`auroraomics models` says which of the two a model takes. "
1029
1156
  "`auroraomics doctor` says whether this machine is ready before you "
@@ -1063,7 +1190,7 @@ def build_parser() -> argparse.ArgumentParser:
1063
1190
  # Positional, because this is the first command anyone runs and the address
1064
1191
  # is the whole of it: `auroraomics login you@institution.edu` is what every
1065
1192
  # page prints and what a person types. It was `--email` and the two
1066
- # disagreed for as long as the pages existed (#636).
1193
+ # disagreed for as long as the pages existed.
1067
1194
  login_parser.add_argument("email", metavar="EMAIL", help="the address of the account to sign in")
1068
1195
  login_parser.add_argument(
1069
1196
  "--code",
@@ -1095,7 +1222,13 @@ def build_parser() -> argparse.ArgumentParser:
1095
1222
  "Checks the installed version, the contract documents shipped with "
1096
1223
  "it, the stored credential and whether the deployment is taking "
1097
1224
  "work. Each check prints ok or FAIL; the exit code is 0 only when "
1098
- "every one passed."
1225
+ "every one passed.\n\n"
1226
+ "--model answers the two questions worth asking before an "
1227
+ "hours-long run, in one forward pass: will this machine's numbers "
1228
+ "be accepted, and what does a tile cost here. The published "
1229
+ "throughput figures are a CUDA device's; a CPU is a different order "
1230
+ "of magnitude, and yours is the one that matters. Add --tiles to "
1231
+ "turn that into a wall clock for the slide."
1099
1232
  ),
1100
1233
  formatter_class=argparse.RawDescriptionHelpFormatter,
1101
1234
  )
@@ -1111,7 +1244,36 @@ def build_parser() -> argparse.ArgumentParser:
1111
1244
  help=(
1112
1245
  f"also run {EMBEDDING_COMPONENT} over the reference image shipped in this "
1113
1246
  "package and compare the result with this model's reference, which "
1114
- "answers whether vectors from this machine would be accepted"
1247
+ "answers whether embeddings from this machine would be accepted — "
1248
+ "and time the pass, which answers how long the slide will take"
1249
+ ),
1250
+ )
1251
+ doctor.add_argument(
1252
+ "--tiles",
1253
+ type=_a_count,
1254
+ default=None,
1255
+ metavar="N",
1256
+ help=(
1257
+ "how many tiles you mean to embed; turns the timed pass of --model "
1258
+ "into a wall clock for the whole run before you start it"
1259
+ ),
1260
+ )
1261
+ # THE ADVICE THE FAILURE GIVES HAS TO BE TAKEABLE BY WHOEVER READS IT.
1262
+ # Without `--model` this check is not run at all and nothing fetches
1263
+ # anything; with it and the weights uncached, the refusal says to permit the
1264
+ # download — and said it as `allow_download=True`, a PYTHON keyword
1265
+ # argument, to a reader who is holding a shell. There was no CLI form of it
1266
+ # and no other subcommand that fetched weights either, so the one machine
1267
+ # this command exists to get ready could not be got ready from the command
1268
+ # line. Opt-in, and default off, for the reason `calibrate` states:
1269
+ # a diagnostic does not start a gigabyte download nobody asked for.
1270
+ doctor.add_argument(
1271
+ "--allow-download",
1272
+ action="store_true",
1273
+ help=(
1274
+ "let --model fetch the weights it needs if this machine does not "
1275
+ "already hold them; without it a missing encoder is reported, not "
1276
+ "downloaded"
1115
1277
  ),
1116
1278
  )
1117
1279
  doctor.set_defaults(handler=cmd_doctor)
@@ -1181,7 +1343,7 @@ def build_parser() -> argparse.ArgumentParser:
1181
1343
  "Send one sample for prediction and print the job it created.\n\n"
1182
1344
  "The sample is an --observation. "
1183
1345
  f"`{_example_of(packed_kind)}` sends the tiles themselves; "
1184
- f"`{_example_of(embedded_kind)}` sends the vectors "
1346
+ f"`{_example_of(embedded_kind)}` sends the embeddings "
1185
1347
  f"{EMBEDDING_COMPONENT} computed from them on your machine, so the "
1186
1348
  "pixels never leave it. Either way a path is uploaded first, and an "
1187
1349
  "upload id you already hold is used as it stands.\n\n"
@@ -1233,7 +1395,13 @@ def build_parser() -> argparse.ArgumentParser:
1233
1395
  submit.add_argument(
1234
1396
  "--land-in-workspace",
1235
1397
  action="store_true",
1236
- help="also open the result in the web workspace when it finishes",
1398
+ # This said "open the result in the web workspace", and nothing opens
1399
+ # anything: the result is offered to the same account's own workspace
1400
+ # inbox, which nobody else reads, and importing it into a workspace is
1401
+ # a separate step the account holder takes. A flag that moves data has
1402
+ # to say what it does, because the privacy page that describes the
1403
+ # destination and who can see it from there is written against this.
1404
+ help="offer the finished result to the same account's web workspace, to import there",
1237
1405
  )
1238
1406
  submit.add_argument(
1239
1407
  "--idempotency-key",