auroraomics 0.1.0.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- auroraomics/__init__.py +74 -0
- auroraomics/_contracts/MANIFEST.json +11 -0
- auroraomics/_contracts/genes/gene-table.2026-09-05.json +19368 -0
- auroraomics/_contracts/public-api/golden-16-tiles.manifest.json +308 -0
- auroraomics/_contracts/public-api-counters.tokens.json +28 -0
- auroraomics/_contracts/public-api-input-kinds.tokens.json +71 -0
- auroraomics/_contracts/public-api.tokens.json +345 -0
- auroraomics/_contracts/qc-thresholds.tokens.json +10 -0
- auroraomics/contracts.py +155 -0
- auroraomics/h5ad.py +398 -0
- auroraomics/pack.py +667 -0
- auroraomics/py.typed +0 -0
- auroraomics/qc.py +288 -0
- auroraomics/subsample.py +140 -0
- auroraomics-0.1.0.dev0.dist-info/METADATA +134 -0
- auroraomics-0.1.0.dev0.dist-info/RECORD +19 -0
- auroraomics-0.1.0.dev0.dist-info/WHEEL +5 -0
- auroraomics-0.1.0.dev0.dist-info/licenses/LICENSE +131 -0
- auroraomics-0.1.0.dev0.dist-info/top_level.txt +1 -0
auroraomics/qc.py
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
1
|
+
"""Patch quality control: which tiles are worth predicting on.
|
|
2
|
+
|
|
3
|
+
A whole-slide image is mostly not tissue. Feeding the empty glass, the pen
|
|
4
|
+
marks and the out-of-focus edges to a model wastes the run and drags every
|
|
5
|
+
downstream summary toward the mean of nothing, so every candidate tile passes
|
|
6
|
+
three independent predicates first:
|
|
7
|
+
|
|
8
|
+
``foreground_mask``
|
|
9
|
+
How much of the tile is *not* slide background — the share of pixels darker
|
|
10
|
+
than the white level. Rejects empty glass.
|
|
11
|
+
``blur_filter``
|
|
12
|
+
The variance of the tile's Laplacian, the standard sharpness proxy: a flat
|
|
13
|
+
response means nothing in the tile has an edge. Rejects out-of-focus tiles.
|
|
14
|
+
``hsv_filter``
|
|
15
|
+
How much of the tile looks *stained* — pixels with enough saturation to
|
|
16
|
+
carry dye and not so bright that they are glare. Rejects grey artefacts,
|
|
17
|
+
which are dark (so they have foreground) and sharp (so they are not blurry)
|
|
18
|
+
but carry no stain.
|
|
19
|
+
|
|
20
|
+
All three are ratios or variances of the tile itself, so a tile can be judged
|
|
21
|
+
wherever it is decoded, with no slide reader and no model.
|
|
22
|
+
|
|
23
|
+
**Higher is stricter for every threshold**: each one is a floor the measured
|
|
24
|
+
value must reach, so raising it keeps fewer tiles.
|
|
25
|
+
|
|
26
|
+
The stained-tissue predicate is the one that has to agree across
|
|
27
|
+
implementations — the same rule decides which tiles a model sees and which
|
|
28
|
+
tiles a reviewer is shown, in more than one language — so its two constants are
|
|
29
|
+
named here and pinned by a shared vector fixture rather than restated in prose.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import math
|
|
35
|
+
from dataclasses import asdict, dataclass, fields
|
|
36
|
+
from typing import Any
|
|
37
|
+
|
|
38
|
+
import cv2
|
|
39
|
+
import numpy as np
|
|
40
|
+
from . import contracts
|
|
41
|
+
|
|
42
|
+
#: A pixel counts as stained tissue when its OpenCV-HSV saturation is strictly
|
|
43
|
+
#: above this and its value strictly below :data:`VALUE_MAX`. Both are in
|
|
44
|
+
#: OpenCV's 0-255 HSV units, not the 0-1 or 0-360 conventions.
|
|
45
|
+
SATURATION_MIN = 15
|
|
46
|
+
#: Above this value a pixel is glare or bare glass, however saturated it looks.
|
|
47
|
+
VALUE_MAX = 240
|
|
48
|
+
#: Grey level at or below which a pixel counts as foreground, in 0-255.
|
|
49
|
+
#:
|
|
50
|
+
#: Read from the shipped contract rather than written here, for the reason the
|
|
51
|
+
#: floors below are: it was a bare ``210`` in three places — here and in each
|
|
52
|
+
#: pipeline's ``patch_qc.py`` — and it is the other half of the foreground rule,
|
|
53
|
+
#: so a client cutting foreground at a different level filters a different set
|
|
54
|
+
#: of tiles than the service will.
|
|
55
|
+
WHITE_LEVEL = contracts.foreground_white_level()
|
|
56
|
+
|
|
57
|
+
#: The service's own floors, read from the contract shipped in this package.
|
|
58
|
+
_DEFAULTS = contracts.qc_thresholds()
|
|
59
|
+
|
|
60
|
+
@dataclass(frozen=True)
|
|
61
|
+
class QcThresholds:
|
|
62
|
+
"""The floors a tile must reach to be used.
|
|
63
|
+
|
|
64
|
+
The field names are the ones a model card's ``input_spec.qc`` block uses,
|
|
65
|
+
so a card's own thresholds go straight in:
|
|
66
|
+
|
|
67
|
+
>>> QcThresholds.from_input_spec(card["input_spec"]) # doctest: +SKIP
|
|
68
|
+
|
|
69
|
+
The defaults are the values a card carries when it has no reason to differ,
|
|
70
|
+
and they are what every predicate below falls back to.
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
#: The three floors, defaulting to the values the service itself runs. They
|
|
74
|
+
#: come from the vendored contract rather than being typed here: a client that
|
|
75
|
+
#: filters tiles differently from the server it sends them to would drop tiles
|
|
76
|
+
#: the service would have kept, or keep tiles it will discard, and every
|
|
77
|
+
#: number downstream would still look plausible.
|
|
78
|
+
foreground_ratio: float = _DEFAULTS["foreground_ratio"]
|
|
79
|
+
hsv_tissue_ratio: float = _DEFAULTS["hsv_tissue_ratio"]
|
|
80
|
+
blur_laplacian_var: float = _DEFAULTS["blur_laplacian_var"]
|
|
81
|
+
|
|
82
|
+
@classmethod
|
|
83
|
+
def field_names(cls) -> tuple[str, ...]:
|
|
84
|
+
"""The threshold keys, in declaration order."""
|
|
85
|
+
return tuple(f.name for f in fields(cls))
|
|
86
|
+
|
|
87
|
+
@classmethod
|
|
88
|
+
def from_input_spec(cls, input_spec: dict[str, Any]) -> QcThresholds:
|
|
89
|
+
"""Build from a model card's ``input_spec``.
|
|
90
|
+
|
|
91
|
+
Refuses a spec whose ``qc`` block names a threshold this version does
|
|
92
|
+
not know, rather than silently ignoring it: a threshold the caller
|
|
93
|
+
believes is being applied and is not is the one failure here that
|
|
94
|
+
produces plausible-looking results from the wrong tiles.
|
|
95
|
+
"""
|
|
96
|
+
qc = input_spec.get("qc")
|
|
97
|
+
if not isinstance(qc, dict):
|
|
98
|
+
raise ValueError("input_spec has no 'qc' block of thresholds")
|
|
99
|
+
unknown = sorted(set(qc) - set(cls.field_names()))
|
|
100
|
+
if unknown:
|
|
101
|
+
raise ValueError(
|
|
102
|
+
f"input_spec.qc names thresholds this version cannot apply: "
|
|
103
|
+
f"{', '.join(unknown)}"
|
|
104
|
+
)
|
|
105
|
+
return cls(**{k: float(v) for k, v in qc.items()})
|
|
106
|
+
|
|
107
|
+
def as_dict(self) -> dict[str, float]:
|
|
108
|
+
"""The thresholds as a plain mapping, for reports and provenance."""
|
|
109
|
+
return {k: float(v) for k, v in asdict(self).items()}
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
DEFAULT_THRESHOLDS = QcThresholds()
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
@dataclass(frozen=True)
|
|
116
|
+
class QcResult:
|
|
117
|
+
"""One tile's verdict and the three numbers behind it.
|
|
118
|
+
|
|
119
|
+
A statistic the cascade never reached is ``nan``, not zero: zero is a
|
|
120
|
+
measurement ("this tile has no stained pixels") and would be averaged into
|
|
121
|
+
per-slide summaries as one.
|
|
122
|
+
|
|
123
|
+
``rejected_by`` names the threshold that rejected the tile, using the
|
|
124
|
+
:class:`QcThresholds` field name, and is ``None`` when the tile passed.
|
|
125
|
+
``passes`` is derived from it rather than stored, because the two must
|
|
126
|
+
agree and a stored pair can be constructed disagreeing.
|
|
127
|
+
"""
|
|
128
|
+
|
|
129
|
+
foreground_ratio: float
|
|
130
|
+
hsv_ratio: float
|
|
131
|
+
laplacian_var: float
|
|
132
|
+
rejected_by: str | None = None
|
|
133
|
+
|
|
134
|
+
@property
|
|
135
|
+
def passes(self) -> bool:
|
|
136
|
+
"""No threshold rejected this tile."""
|
|
137
|
+
return self.rejected_by is None
|
|
138
|
+
|
|
139
|
+
@classmethod
|
|
140
|
+
def stat_names(cls) -> tuple[str, ...]:
|
|
141
|
+
"""The three measured statistics, in declaration order.
|
|
142
|
+
|
|
143
|
+
Callers that persist QC output — the tile manifest inside an archive,
|
|
144
|
+
the arrays beside an embedding — take their column names from here, so
|
|
145
|
+
a renamed statistic moves every writer at once.
|
|
146
|
+
"""
|
|
147
|
+
skip = {"rejected_by"}
|
|
148
|
+
return tuple(f.name for f in fields(cls) if f.name not in skip)
|
|
149
|
+
|
|
150
|
+
def stats(self) -> dict[str, float]:
|
|
151
|
+
"""The three measured statistics as a mapping."""
|
|
152
|
+
return {name: float(getattr(self, name)) for name in self.stat_names()}
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _as_rgb_uint8(tile: np.ndarray) -> np.ndarray:
|
|
156
|
+
"""Check a tile is what every predicate below assumes: HxWx3 RGB uint8.
|
|
157
|
+
|
|
158
|
+
Channel ORDER is the trap worth failing loudly on. OpenCV's own decoders
|
|
159
|
+
hand back BGR, and the conversions below are written against RGB, so a
|
|
160
|
+
caller who forgets the swap gets a red/blue exchange — which changes hue
|
|
161
|
+
and therefore saturation, which silently changes which tiles are called
|
|
162
|
+
tissue. Nothing in the array says which order it is in, so this can only be
|
|
163
|
+
documented and asserted in shape, never detected.
|
|
164
|
+
"""
|
|
165
|
+
arr = np.asarray(tile)
|
|
166
|
+
if arr.ndim != 3 or arr.shape[2] != 3:
|
|
167
|
+
raise ValueError(
|
|
168
|
+
f"a tile must be a height x width x 3 RGB array, got shape "
|
|
169
|
+
f"{arr.shape}"
|
|
170
|
+
)
|
|
171
|
+
if arr.dtype != np.uint8:
|
|
172
|
+
raise ValueError(
|
|
173
|
+
f"a tile must be uint8 (0-255), got dtype {arr.dtype}; converting "
|
|
174
|
+
"a float image without rescaling would change every threshold"
|
|
175
|
+
)
|
|
176
|
+
if arr.shape[0] == 0 or arr.shape[1] == 0:
|
|
177
|
+
raise ValueError("a tile must have a non-zero height and width")
|
|
178
|
+
return arr
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def foreground_mask(
|
|
182
|
+
tile: np.ndarray, threshold: float = DEFAULT_THRESHOLDS.foreground_ratio
|
|
183
|
+
) -> tuple[bool, float]:
|
|
184
|
+
"""Is at least ``threshold`` percent of the tile non-background?
|
|
185
|
+
|
|
186
|
+
Returns ``(passes, foreground_ratio)`` — the ratio is a percentage, 0-100.
|
|
187
|
+
"""
|
|
188
|
+
return _foreground_from_grey(_grey(tile), threshold)
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _grey(tile: np.ndarray) -> np.ndarray:
|
|
192
|
+
"""Validate the tile and convert it to greyscale, once.
|
|
193
|
+
|
|
194
|
+
Two of the three predicates work on greyscale, and :func:`qc_tile` runs
|
|
195
|
+
both on the same tile, so converting inside each one converts twice —
|
|
196
|
+
on a cascade whose whole ordering is justified by cost.
|
|
197
|
+
"""
|
|
198
|
+
return cv2.cvtColor(_as_rgb_uint8(tile), cv2.COLOR_RGB2GRAY)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _foreground_from_grey(grey: np.ndarray, threshold: float) -> tuple[bool, float]:
|
|
202
|
+
_, mask = cv2.threshold(grey, WHITE_LEVEL, 255, cv2.THRESH_BINARY_INV)
|
|
203
|
+
foreground_ratio = float(cv2.countNonZero(mask) / mask.size * 100)
|
|
204
|
+
return foreground_ratio >= threshold, foreground_ratio
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def hsv_filter(
|
|
208
|
+
tile: np.ndarray, required_ratio: float = DEFAULT_THRESHOLDS.hsv_tissue_ratio
|
|
209
|
+
) -> tuple[bool, float]:
|
|
210
|
+
"""Is at least ``required_ratio`` percent of the tile stained tissue?
|
|
211
|
+
|
|
212
|
+
A pixel is stained when its saturation is above :data:`SATURATION_MIN` and
|
|
213
|
+
its value below :data:`VALUE_MAX`; both bounds are exclusive.
|
|
214
|
+
|
|
215
|
+
Returns ``(passes, ratio)`` — the ratio is a percentage, 0-100.
|
|
216
|
+
"""
|
|
217
|
+
arr = _as_rgb_uint8(tile)
|
|
218
|
+
hsv = cv2.cvtColor(arr, cv2.COLOR_RGB2HSV)
|
|
219
|
+
mask = (hsv[:, :, 1] > SATURATION_MIN) & (hsv[:, :, 2] < VALUE_MAX)
|
|
220
|
+
ratio = float(np.sum(mask) / mask.size * 100)
|
|
221
|
+
return ratio >= required_ratio, ratio
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def blur_filter(
|
|
225
|
+
tile: np.ndarray, blur_threshold: float = DEFAULT_THRESHOLDS.blur_laplacian_var
|
|
226
|
+
) -> tuple[bool, float]:
|
|
227
|
+
"""Is the tile sharp enough, by Laplacian variance?
|
|
228
|
+
|
|
229
|
+
Returns ``(passes, laplacian_var)``. The variance is a bare number, not a
|
|
230
|
+
percentage: it scales with contrast, so a threshold tuned on one stain and
|
|
231
|
+
scanner does not transfer unexamined to another.
|
|
232
|
+
"""
|
|
233
|
+
return _blur_from_grey(_grey(tile), blur_threshold)
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def _blur_from_grey(grey: np.ndarray, blur_threshold: float) -> tuple[bool, float]:
|
|
237
|
+
laplacian_var = float(cv2.Laplacian(grey, cv2.CV_64F).var())
|
|
238
|
+
return laplacian_var >= blur_threshold, laplacian_var
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def qc_tile(
|
|
242
|
+
tile: np.ndarray, thresholds: QcThresholds | None = None
|
|
243
|
+
) -> QcResult:
|
|
244
|
+
"""Run the three predicates over one tile and return the verdict.
|
|
245
|
+
|
|
246
|
+
The order is foreground, then blur, then stain, and it SHORT-CIRCUITS: a
|
|
247
|
+
tile that fails an earlier predicate is not measured by the later ones, and
|
|
248
|
+
their statistics come back ``nan``. Two reasons, in this order:
|
|
249
|
+
|
|
250
|
+
* cost — the cascade runs on every candidate tile of a slide, which is
|
|
251
|
+
hundreds of thousands of them, and the cheapest predicate rejects the
|
|
252
|
+
most;
|
|
253
|
+
* meaning — the stain ratio of a tile that is 95 % empty glass is not a
|
|
254
|
+
property of any tissue, and recording it as a number invites it into an
|
|
255
|
+
average.
|
|
256
|
+
|
|
257
|
+
The order itself is part of the contract, because it decides which
|
|
258
|
+
statistics exist: swapping blur and stain would leave a different column
|
|
259
|
+
populated for exactly the same tiles.
|
|
260
|
+
"""
|
|
261
|
+
thresholds = thresholds or DEFAULT_THRESHOLDS
|
|
262
|
+
# nan, not zero: a statistic the cascade never reached was not measured.
|
|
263
|
+
foreground_ratio = hsv_ratio = laplacian_var = math.nan
|
|
264
|
+
|
|
265
|
+
def verdict(rejected_by: str | None = None) -> QcResult:
|
|
266
|
+
return QcResult(
|
|
267
|
+
foreground_ratio=foreground_ratio,
|
|
268
|
+
hsv_ratio=hsv_ratio,
|
|
269
|
+
laplacian_var=laplacian_var,
|
|
270
|
+
rejected_by=rejected_by,
|
|
271
|
+
)
|
|
272
|
+
|
|
273
|
+
# Both grey predicates read the same conversion.
|
|
274
|
+
grey = _grey(tile)
|
|
275
|
+
|
|
276
|
+
passed, foreground_ratio = _foreground_from_grey(grey, thresholds.foreground_ratio)
|
|
277
|
+
if not passed:
|
|
278
|
+
return verdict("foreground_ratio")
|
|
279
|
+
|
|
280
|
+
passed, laplacian_var = _blur_from_grey(grey, thresholds.blur_laplacian_var)
|
|
281
|
+
if not passed:
|
|
282
|
+
return verdict("blur_laplacian_var")
|
|
283
|
+
|
|
284
|
+
passed, hsv_ratio = hsv_filter(tile, required_ratio=thresholds.hsv_tissue_ratio)
|
|
285
|
+
if not passed:
|
|
286
|
+
return verdict("hsv_tissue_ratio")
|
|
287
|
+
|
|
288
|
+
return verdict()
|
auroraomics/subsample.py
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
"""Spend a limited tile budget on the densest part of the slide.
|
|
2
|
+
|
|
3
|
+
A run is allowed a maximum number of tiles. When a slide yields more, something
|
|
4
|
+
has to choose, and *which* tiles are dropped changes what the result can be
|
|
5
|
+
used for: taking the first N walks a raster line across the slide, and taking a
|
|
6
|
+
random N shreds the tissue into isolated spots, so every neighbourhood
|
|
7
|
+
statistic downstream is computed over holes.
|
|
8
|
+
|
|
9
|
+
So the choice is a contiguous region — the densest square that holds the budget
|
|
10
|
+
— and spatial structure survives inside it. What is lost is stated plainly: the
|
|
11
|
+
tissue outside that square is not predicted at all, and a result must say so
|
|
12
|
+
rather than look like a whole slide.
|
|
13
|
+
|
|
14
|
+
The implementation is an integral image: build an occupancy grid, prefix-sum
|
|
15
|
+
it, then binary-search for the smallest square window that holds the budget and
|
|
16
|
+
place it where the count is highest. That is O(n + G² log G) in the grid size,
|
|
17
|
+
against the O(n² · iterations) of sweeping windows over the points themselves —
|
|
18
|
+
the difference between milliseconds and minutes at 300k spots.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import numpy as np
|
|
24
|
+
|
|
25
|
+
#: Grid resolution cap per axis. 500 x 500 cells is 250k of them, which bounds
|
|
26
|
+
#: both the prefix sum and the binary search regardless of how many distinct
|
|
27
|
+
#: coordinates a slide has.
|
|
28
|
+
MAX_GRID = 500
|
|
29
|
+
|
|
30
|
+
#: Seed for the final trim. The densest square usually holds slightly MORE than
|
|
31
|
+
#: the budget, and the overflow has to go; a fixed seed makes which tiles go a
|
|
32
|
+
#: property of the input rather than of the day, so two runs of one slide
|
|
33
|
+
#: produce the same tiles.
|
|
34
|
+
SEED = 42
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def subsample_spatial_square(
|
|
38
|
+
x: np.ndarray, y: np.ndarray, max_items: int
|
|
39
|
+
) -> np.ndarray:
|
|
40
|
+
"""Choose at most ``max_items`` of the given points, as a boolean mask.
|
|
41
|
+
|
|
42
|
+
Parameters
|
|
43
|
+
----------
|
|
44
|
+
x, y:
|
|
45
|
+
Coordinates of each candidate spot, in the same units (pixels), one
|
|
46
|
+
entry per spot.
|
|
47
|
+
max_items:
|
|
48
|
+
The budget. When there are no more points than this, every point is
|
|
49
|
+
kept and no work is done.
|
|
50
|
+
|
|
51
|
+
Returns
|
|
52
|
+
-------
|
|
53
|
+
numpy.ndarray
|
|
54
|
+
A boolean mask over the input order: ``True`` for the points to keep.
|
|
55
|
+
Returning a mask rather than the selected coordinates lets the caller
|
|
56
|
+
apply the same choice to everything it holds per spot.
|
|
57
|
+
"""
|
|
58
|
+
x = np.asarray(x, dtype=np.float64).ravel()
|
|
59
|
+
y = np.asarray(y, dtype=np.float64).ravel()
|
|
60
|
+
if x.shape != y.shape:
|
|
61
|
+
raise ValueError(
|
|
62
|
+
f"x and y must describe the same points, got {x.shape} and {y.shape}"
|
|
63
|
+
)
|
|
64
|
+
if max_items <= 0:
|
|
65
|
+
raise ValueError(f"max_items must be positive, got {max_items}")
|
|
66
|
+
|
|
67
|
+
n_points = x.size
|
|
68
|
+
if n_points <= max_items:
|
|
69
|
+
return np.ones(n_points, dtype=bool)
|
|
70
|
+
|
|
71
|
+
# ── map coordinates onto a grid ─────────────────────────────────────────
|
|
72
|
+
# Distinct coordinates become grid cells directly while there are few
|
|
73
|
+
# enough of them, which keeps a regular spot lattice exact. Past the cap
|
|
74
|
+
# the axis is binned into equal-width cells instead.
|
|
75
|
+
column_index, columns = _axis_index(x)
|
|
76
|
+
row_index, rows = _axis_index(y)
|
|
77
|
+
|
|
78
|
+
# ── occupancy grid and its prefix sum ───────────────────────────────────
|
|
79
|
+
grid = np.zeros((rows, columns), dtype=np.int32)
|
|
80
|
+
np.add.at(grid, (row_index, column_index), 1)
|
|
81
|
+
prefix = np.zeros((rows + 1, columns + 1), dtype=np.int64)
|
|
82
|
+
prefix[1:, 1:] = np.cumsum(np.cumsum(grid, axis=0), axis=1)
|
|
83
|
+
|
|
84
|
+
def window_counts(size: int) -> np.ndarray:
|
|
85
|
+
"""Point count of every square window of ``size`` cells, vectorised."""
|
|
86
|
+
if size > rows or size > columns:
|
|
87
|
+
return np.zeros((1, 1), dtype=np.int64)
|
|
88
|
+
return (
|
|
89
|
+
prefix[size:, size:]
|
|
90
|
+
- prefix[size:, : columns - size + 1]
|
|
91
|
+
- prefix[: rows - size + 1, size:]
|
|
92
|
+
+ prefix[: rows - size + 1, : columns - size + 1]
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
# ── the smallest window that holds the budget ───────────────────────────
|
|
96
|
+
low, high = 1, max(rows, columns)
|
|
97
|
+
best_size = high
|
|
98
|
+
while low <= high:
|
|
99
|
+
middle = (low + high) // 2
|
|
100
|
+
counts = window_counts(middle)
|
|
101
|
+
if counts.size and counts.max() >= max_items:
|
|
102
|
+
best_size = middle
|
|
103
|
+
high = middle - 1
|
|
104
|
+
else:
|
|
105
|
+
low = middle + 1
|
|
106
|
+
|
|
107
|
+
# ── and the densest place to put it ─────────────────────────────────────
|
|
108
|
+
counts = window_counts(best_size)
|
|
109
|
+
best_row, best_column = np.unravel_index(int(np.argmax(counts)), counts.shape)
|
|
110
|
+
|
|
111
|
+
inside = (
|
|
112
|
+
(column_index >= best_column)
|
|
113
|
+
& (column_index < best_column + best_size)
|
|
114
|
+
& (row_index >= best_row)
|
|
115
|
+
& (row_index < best_row + best_size)
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
selected = int(inside.sum())
|
|
119
|
+
if selected <= max_items:
|
|
120
|
+
return inside
|
|
121
|
+
|
|
122
|
+
# The window is the smallest that HOLDS the budget, so it usually holds a
|
|
123
|
+
# few more; trim the overflow at random from inside it, seeded.
|
|
124
|
+
positions = np.flatnonzero(inside)
|
|
125
|
+
keep = np.random.RandomState(SEED).choice(
|
|
126
|
+
positions.size, size=max_items, replace=False
|
|
127
|
+
)
|
|
128
|
+
mask = np.zeros(inside.shape, dtype=bool)
|
|
129
|
+
mask[positions[keep]] = True
|
|
130
|
+
return mask
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _axis_index(values: np.ndarray) -> tuple[np.ndarray, int]:
|
|
134
|
+
"""Grid index of every coordinate on one axis, and the axis's cell count."""
|
|
135
|
+
unique = np.unique(values)
|
|
136
|
+
if unique.size > MAX_GRID:
|
|
137
|
+
edges = np.linspace(values.min(), values.max() + 1, MAX_GRID + 1)
|
|
138
|
+
index = np.clip(np.digitize(values, edges) - 1, 0, MAX_GRID - 1)
|
|
139
|
+
return index, MAX_GRID
|
|
140
|
+
return np.searchsorted(unique, values), int(unique.size)
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: auroraomics
|
|
3
|
+
Version: 0.1.0.dev0
|
|
4
|
+
Summary: Virtual spatial transcriptomics from H&E histology: patch QC, tile packing and the .h5ad result contract.
|
|
5
|
+
Author: Kalin Nonchev
|
|
6
|
+
License-Expression: PolyForm-Noncommercial-1.0.0
|
|
7
|
+
Keywords: histopathology,spatial-transcriptomics,h5ad,anndata,whole-slide-image
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Intended Audience :: Science/Research
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
12
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
13
|
+
Classifier: Typing :: Typed
|
|
14
|
+
Requires-Python: >=3.10
|
|
15
|
+
Description-Content-Type: text/markdown
|
|
16
|
+
License-File: LICENSE
|
|
17
|
+
Requires-Dist: numpy>=1.23
|
|
18
|
+
Requires-Dist: h5py>=3.9
|
|
19
|
+
Requires-Dist: opencv-python-headless>=4.5
|
|
20
|
+
Requires-Dist: pillow>=9
|
|
21
|
+
Provides-Extra: client
|
|
22
|
+
Requires-Dist: httpx>=0.27; extra == "client"
|
|
23
|
+
Requires-Dist: pydantic>=2; extra == "client"
|
|
24
|
+
Requires-Dist: anndata>=0.10; extra == "client"
|
|
25
|
+
Provides-Extra: deepspotm
|
|
26
|
+
Provides-Extra: slide
|
|
27
|
+
Requires-Dist: tifffile>=2023.7.10; extra == "slide"
|
|
28
|
+
Requires-Dist: imagecodecs>=2023.3.16; extra == "slide"
|
|
29
|
+
Provides-Extra: dev
|
|
30
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
31
|
+
Requires-Dist: setuptools>=77; extra == "dev"
|
|
32
|
+
Requires-Dist: wheel; extra == "dev"
|
|
33
|
+
Requires-Dist: anndata<0.12,>=0.10; extra == "dev"
|
|
34
|
+
Requires-Dist: tifffile>=2023.7.10; extra == "dev"
|
|
35
|
+
Requires-Dist: imagecodecs>=2023.3.16; extra == "dev"
|
|
36
|
+
Requires-Dist: build>=1.0; extra == "dev"
|
|
37
|
+
Requires-Dist: numpy<2; extra == "dev"
|
|
38
|
+
Dynamic: license-file
|
|
39
|
+
|
|
40
|
+
# auroraomics
|
|
41
|
+
|
|
42
|
+
Virtual spatial transcriptomics from H&E histology.
|
|
43
|
+
|
|
44
|
+
This package holds the pieces of that pipeline that are pure Python, so the
|
|
45
|
+
same code runs on your laptop, in a GPU container and on the service:
|
|
46
|
+
|
|
47
|
+
- **`auroraomics.qc`** — the three patch-quality predicates (foreground, blur,
|
|
48
|
+
stained tissue) applied to every candidate tile before a model sees it, and
|
|
49
|
+
the short-circuiting cascade that combines them.
|
|
50
|
+
- **`auroraomics.pack`** — build the `patches` archive a prediction takes as
|
|
51
|
+
input (`pack_tiles`), and validate one you have been handed (`read_archive`).
|
|
52
|
+
- **`auroraomics.h5ad`** — write the standard `.h5ad` result with `h5py`
|
|
53
|
+
alone, streaming the expression matrix batch by batch so peak memory is one
|
|
54
|
+
batch rather than the whole matrix.
|
|
55
|
+
- **`auroraomics.subsample`** — pick the densest contiguous square of spots
|
|
56
|
+
when a slide yields more tiles than a run is allowed to spend.
|
|
57
|
+
- **`auroraomics.contracts`** — the shared contract values (container layout,
|
|
58
|
+
input-kind caps, result layout) as data, so nothing here re-types a number
|
|
59
|
+
the service also reads.
|
|
60
|
+
|
|
61
|
+
## Install
|
|
62
|
+
|
|
63
|
+
```
|
|
64
|
+
pip install auroraomics
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
The core needs only numpy, h5py, OpenCV and Pillow. Extras add the API client
|
|
68
|
+
(`client`), a local model runtime (`deepspotm`) and pyramidal slide reading
|
|
69
|
+
(`slide`).
|
|
70
|
+
|
|
71
|
+
## Pack tiles, then look at the report
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
import auroraomics as ao
|
|
75
|
+
|
|
76
|
+
report = ao.pack_tiles(tiles, "sample.zip", mpp=0.499, thumbnail=thumb)
|
|
77
|
+
print(report.written, "tiles kept,", report.rejected, "dropped")
|
|
78
|
+
print(report.rejected_by_reason) # {'foreground_ratio': 12, ...}
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
`tiles` is any iterable of `ao.Tile(image, x, y)`, where `image` is an RGB
|
|
82
|
+
`uint8` array and `x`/`y` are the tile's top-left position in full-resolution
|
|
83
|
+
pixels. Tiles are quality-checked as they stream past, and only the ones that
|
|
84
|
+
pass are written, so an iterable that reads a slide lazily never has to hold
|
|
85
|
+
more than one tile in memory.
|
|
86
|
+
|
|
87
|
+
## Validate an archive before trusting it
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
archive = ao.read_archive("sample.zip") # raises ArchiveError on anything odd
|
|
91
|
+
for tile in archive.tiles(): # decoded one at a time
|
|
92
|
+
...
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
`read_archive` checks the member names, the manifest, the tile geometry and the
|
|
96
|
+
declared sizes *before* decoding a single pixel, and refuses an archive whose
|
|
97
|
+
members do not match the container contract.
|
|
98
|
+
|
|
99
|
+
## Write a result
|
|
100
|
+
|
|
101
|
+
```python
|
|
102
|
+
ao.write_result(
|
|
103
|
+
"result.h5ad",
|
|
104
|
+
obs=obs, # per-spot columns, as plain arrays
|
|
105
|
+
var=var, # per-gene columns, indexed by gene id
|
|
106
|
+
spatial=coords, # (n_spots, 2) array -> obsm["spatial"]
|
|
107
|
+
x=batches, # an array, or an iterable of row batches
|
|
108
|
+
uns={"model": {"id": "..."}},
|
|
109
|
+
layers={"image_only": other_batches},
|
|
110
|
+
)
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
The file reads back as an ordinary `AnnData` in anndata 0.10 and 0.11. Passing
|
|
114
|
+
an iterable for `x` streams it: each batch is compressed into the file as it
|
|
115
|
+
arrives and then dropped, which is what makes a matrix larger than memory
|
|
116
|
+
writable.
|
|
117
|
+
|
|
118
|
+
## Typing
|
|
119
|
+
|
|
120
|
+
The package ships `py.typed`, so annotations are visible to type checkers in
|
|
121
|
+
your project.
|
|
122
|
+
|
|
123
|
+
## Licence
|
|
124
|
+
|
|
125
|
+
The code in this package is licensed under
|
|
126
|
+
[PolyForm Noncommercial 1.0.0](https://polyformproject.org/licenses/noncommercial/1.0.0),
|
|
127
|
+
which permits use for any purpose that is not commercial. It is the same licence the
|
|
128
|
+
model package this client is built for carries, so installing both puts you under one
|
|
129
|
+
rule rather than two.
|
|
130
|
+
|
|
131
|
+
The model weights are licensed separately by whoever publishes them, and access to them
|
|
132
|
+
may be gated. Read those terms before you use a model: they are not this licence, and a
|
|
133
|
+
permission granted here is not a permission granted there.
|
|
134
|
+
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
auroraomics/__init__.py,sha256=JkWGXnQTUFu362kVcQJpcybwZo6ycIYqgAOfwIt3XO4,2109
|
|
2
|
+
auroraomics/contracts.py,sha256=YGn1FT4NFGDwm4a9SKETCd59DPUcWEJMqbsdaoqDzf8,5972
|
|
3
|
+
auroraomics/h5ad.py,sha256=7Oq19u8vVyLBhvi4Io--hKBrUoFemq8ZBoNBJ5Ivnww,15939
|
|
4
|
+
auroraomics/pack.py,sha256=ToxFmpfeC8TJOtrVkwbrOyZy-LfEezd_tz3S9bnpt7c,25422
|
|
5
|
+
auroraomics/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
6
|
+
auroraomics/qc.py,sha256=f60ebJDhsc22OcoO3qTDMK3iSZuVR8rCqob-YVGrjzM,11773
|
|
7
|
+
auroraomics/subsample.py,sha256=lojo6iJrq1CY2SMTyC6ifAUwD_Q2GRL6UVwXqVO22F0,5880
|
|
8
|
+
auroraomics/_contracts/MANIFEST.json,sha256=IseYJJm-_Q8shHZvodOfFcsNOO1RVC3FEcxxEc_KbpA,782
|
|
9
|
+
auroraomics/_contracts/public-api-counters.tokens.json,sha256=glRNScqQiXFyrVBk2q5FAIF1_HFSuOrWQRtJQI35l50,849
|
|
10
|
+
auroraomics/_contracts/public-api-input-kinds.tokens.json,sha256=ZUtF90BIWsbne9P8j9DpG3RT2jjERbN69GC9ooA1EVg,2188
|
|
11
|
+
auroraomics/_contracts/public-api.tokens.json,sha256=p35ItqrtjuQCD7_Zl5bb-23YGMpvsByeIQK09KB6POc,7499
|
|
12
|
+
auroraomics/_contracts/qc-thresholds.tokens.json,sha256=1l3jwwbXWrZYijoT_t_i5WLIThKSHr7gj4EM2LC5rkc,1617
|
|
13
|
+
auroraomics/_contracts/genes/gene-table.2026-09-05.json,sha256=Db-X9ps35P2xoT5cBQiMcX-PbOx_sExJpJh_jGSVpck,3827986
|
|
14
|
+
auroraomics/_contracts/public-api/golden-16-tiles.manifest.json,sha256=LZP7_SqkVCzHXQwt6jTxmcbPh_USMgH5UrWN5LfRQJQ,7897
|
|
15
|
+
auroraomics-0.1.0.dev0.dist-info/licenses/LICENSE,sha256=i_s2EEVLTyPBklHvoeFFrtXRyxN1A6u_SKohEXlDkYc,4595
|
|
16
|
+
auroraomics-0.1.0.dev0.dist-info/METADATA,sha256=SUKCRgZZRICGGf52ex5kK-8mZbmumjCF2eBrCZKiBuk,5257
|
|
17
|
+
auroraomics-0.1.0.dev0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
18
|
+
auroraomics-0.1.0.dev0.dist-info/top_level.txt,sha256=3Q7rafAR7MW37pdCm6DADeEIO_QkLMB-b3tAK_V1kjU,12
|
|
19
|
+
auroraomics-0.1.0.dev0.dist-info/RECORD,,
|