slide2vec 5.3.0__tar.gz → 5.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {slide2vec-5.3.0 → slide2vec-5.5.0}/PKG-INFO +3 -3
- {slide2vec-5.3.0 → slide2vec-5.5.0}/pyproject.toml +4 -4
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/__init__.py +20 -2
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/api.py +387 -4
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/artifacts.py +291 -4
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/configs/default.yaml +1 -0
- slide2vec-5.5.0/slide2vec/data/dataset.py +142 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/data/tile_reader.py +88 -12
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/distributed/__init__.py +34 -66
- slide2vec-5.5.0/slide2vec/distributed/dense_image_worker.py +78 -0
- slide2vec-5.5.0/slide2vec/distributed/dense_worker.py +80 -0
- slide2vec-5.5.0/slide2vec/distributed/direct_embed_worker.py +192 -0
- slide2vec-5.5.0/slide2vec/distributed/image_worker.py +73 -0
- slide2vec-5.5.0/slide2vec/distributed/pipeline_worker.py +109 -0
- slide2vec-5.5.0/slide2vec/distributed/worker_entry.py +87 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/base.py +15 -13
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/__init__.py +2 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/conch.py +4 -2
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/dinov2.py +1 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/genbio.py +3 -2
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/gigapath.py +14 -2
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/gpfm.py +1 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/hibou.py +3 -1
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/hoptimus.py +19 -1
- slide2vec-5.5.0/slide2vec/encoders/models/isight.py +305 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/lunit.py +1 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/midnight.py +2 -1
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/mstar.py +1 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/musk.py +2 -1
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/phikon.py +4 -2
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/prost40m.py +2 -1
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/uni.py +2 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/virchow.py +34 -5
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/registry.py +34 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/inference.py +79 -15
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/progress.py +41 -4
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/batching.py +120 -206
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/cpu_budget.py +1 -1
- slide2vec-5.5.0/slide2vec/runtime/dense_encoder_input.py +119 -0
- slide2vec-5.5.0/slide2vec/runtime/dense_image_shard.py +209 -0
- slide2vec-5.5.0/slide2vec/runtime/dense_image_stage.py +164 -0
- slide2vec-5.5.0/slide2vec/runtime/dense_regions.py +447 -0
- slide2vec-5.5.0/slide2vec/runtime/dense_shard.py +295 -0
- slide2vec-5.5.0/slide2vec/runtime/dense_stage.py +306 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/distributed.py +15 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/distributed_stage.py +8 -1
- slide2vec-5.5.0/slide2vec/runtime/effective_encoder_input.py +124 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/embedding.py +21 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/embedding_persist.py +8 -3
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/embedding_pipeline.py +18 -6
- slide2vec-5.5.0/slide2vec/runtime/encoder_input_contract.py +122 -0
- slide2vec-5.5.0/slide2vec/runtime/image_shard.py +222 -0
- slide2vec-5.5.0/slide2vec/runtime/image_specs.py +70 -0
- slide2vec-5.5.0/slide2vec/runtime/image_stage.py +148 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/patient_pipeline.py +4 -1
- slide2vec-5.5.0/slide2vec/runtime/pooled_encoder_input.py +85 -0
- slide2vec-5.5.0/slide2vec/runtime/preprocessing.py +196 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/process_list.py +2 -2
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/progress_bridge.py +1 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/serialization.py +97 -1
- slide2vec-5.5.0/slide2vec/runtime/sharding.py +34 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/tiling.py +7 -4
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/types.py +1 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/utils/config.py +9 -9
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/utils/tiling_io.py +11 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec.egg-info/PKG-INFO +3 -3
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec.egg-info/SOURCES.txt +31 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec.egg-info/requires.txt +2 -2
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_architecture_runtime_split.py +1 -0
- slide2vec-5.5.0/tests/test_dense_encoder_input.py +317 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_dense_extraction.py +7 -7
- slide2vec-5.5.0/tests/test_dense_image_shard.py +407 -0
- slide2vec-5.5.0/tests/test_dense_image_stage.py +338 -0
- slide2vec-5.5.0/tests/test_dense_regions.py +450 -0
- slide2vec-5.5.0/tests/test_dense_shard.py +389 -0
- slide2vec-5.5.0/tests/test_dense_stage.py +286 -0
- slide2vec-5.5.0/tests/test_dense_worker.py +106 -0
- slide2vec-5.5.0/tests/test_encoder_input_contract.py +350 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_encoder_registry.py +1 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_hs2p_package_cutover.py +32 -3
- slide2vec-5.5.0/tests/test_image_shard.py +313 -0
- slide2vec-5.5.0/tests/test_image_stage.py +306 -0
- slide2vec-5.5.0/tests/test_isight.py +239 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_output_consistency.py +12 -65
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_patch_size_metadata.py +33 -6
- slide2vec-5.5.0/tests/test_pooled_encoder_input.py +518 -0
- slide2vec-5.5.0/tests/test_pooled_geometry.py +279 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_progress.py +174 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_regression_core.py +62 -3
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_regression_inference.py +154 -69
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_regression_models.py +5 -5
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_runtime_batching.py +1 -1
- slide2vec-5.5.0/tests/test_sharding.py +58 -0
- slide2vec-5.3.0/slide2vec/data/dataset.py +0 -47
- slide2vec-5.3.0/slide2vec/distributed/direct_embed_worker.py +0 -192
- slide2vec-5.3.0/slide2vec/distributed/pipeline_worker.py +0 -113
- slide2vec-5.3.0/slide2vec/runtime/dense_regions.py +0 -301
- slide2vec-5.3.0/tests/test_dense_regions.py +0 -264
- {slide2vec-5.3.0 → slide2vec-5.5.0}/LICENSE +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/README.md +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/setup.cfg +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/__main__.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/cli.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/configs/__init__.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/configs/resources.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/data/__init__.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/data/tile_store.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/__init__.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/moozy/__init__.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/moozy/blocks.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/moozy/case.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/moozy/loading.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/moozy/slide.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/moozy/types.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/prism.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/models/titan.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/encoders/validation.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/__init__.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/artifacts_collect.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/dense_sliding.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/hierarchical.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/manifest.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/model_settings.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/persist_callbacks.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/persistence.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/registry.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/slide_encode.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/tiling_pipeline.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/runtime/worker_io.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/utils/__init__.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/utils/coordinates.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/utils/log_utils.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec/utils/utils.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec.egg-info/dependency_links.txt +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec.egg-info/entry_points.txt +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec.egg-info/not-zip-safe +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/slide2vec.egg-info/top_level.txt +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_attention_extraction.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_dense_sliding.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_dinov2_natimage.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_gpfm_genbio_heavy.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_tile_store.py +0 -0
- {slide2vec-5.3.0 → slide2vec-5.5.0}/tests/test_tiling_pipeline.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: slide2vec
|
|
3
|
-
Version: 5.
|
|
3
|
+
Version: 5.5.0
|
|
4
4
|
Summary: Embedding of whole slide images with Foundation Models
|
|
5
5
|
Author-email: Clément Grisi <clement.grisi@radboudumc.nl>
|
|
6
6
|
License-Expression: Apache-2.0
|
|
@@ -15,7 +15,7 @@ Classifier: Programming Language :: Python :: 3.13
|
|
|
15
15
|
Requires-Python: >=3.10
|
|
16
16
|
Description-Content-Type: text/markdown
|
|
17
17
|
License-File: LICENSE
|
|
18
|
-
Requires-Dist: hs2p[asap,cucim,openslide,sam2,vips]>=4.
|
|
18
|
+
Requires-Dist: hs2p[asap,cucim,openslide,sam2,vips]>=4.3.0
|
|
19
19
|
Requires-Dist: omegaconf
|
|
20
20
|
Requires-Dist: matplotlib
|
|
21
21
|
Requires-Dist: numpy<2
|
|
@@ -65,7 +65,7 @@ Requires-Dist: numpy<2; extra == "fm"
|
|
|
65
65
|
Requires-Dist: pandas; extra == "fm"
|
|
66
66
|
Requires-Dist: pillow; extra == "fm"
|
|
67
67
|
Requires-Dist: rich; extra == "fm"
|
|
68
|
-
Requires-Dist: hs2p[asap,cucim,openslide,sam2,vips]>=4.
|
|
68
|
+
Requires-Dist: hs2p[asap,cucim,openslide,sam2,vips]>=4.3.0; extra == "fm"
|
|
69
69
|
Requires-Dist: wandb; extra == "fm"
|
|
70
70
|
Requires-Dist: torch<2.8,>=2.3; extra == "fm"
|
|
71
71
|
Requires-Dist: torchvision>=0.18.0; extra == "fm"
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "slide2vec"
|
|
7
|
-
version = "5.
|
|
7
|
+
version = "5.5.0"
|
|
8
8
|
description = "Embedding of whole slide images with Foundation Models"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -21,7 +21,7 @@ classifiers = [
|
|
|
21
21
|
"Programming Language :: Python :: 3.13",
|
|
22
22
|
]
|
|
23
23
|
dependencies = [
|
|
24
|
-
"hs2p[asap,cucim,openslide,sam2,vips]>=4.
|
|
24
|
+
"hs2p[asap,cucim,openslide,sam2,vips]>=4.3.0",
|
|
25
25
|
"omegaconf",
|
|
26
26
|
"matplotlib",
|
|
27
27
|
"numpy<2",
|
|
@@ -88,7 +88,7 @@ fm = [
|
|
|
88
88
|
"pandas",
|
|
89
89
|
"pillow",
|
|
90
90
|
"rich",
|
|
91
|
-
"hs2p[asap,cucim,openslide,sam2,vips]>=4.
|
|
91
|
+
"hs2p[asap,cucim,openslide,sam2,vips]>=4.3.0",
|
|
92
92
|
"wandb",
|
|
93
93
|
"torch>=2.3,<2.8",
|
|
94
94
|
"torchvision>=0.18.0",
|
|
@@ -167,7 +167,7 @@ no_implicit_reexport = true
|
|
|
167
167
|
max-line-length = 160
|
|
168
168
|
|
|
169
169
|
[tool.bumpver]
|
|
170
|
-
current_version = "5.
|
|
170
|
+
current_version = "5.5.0"
|
|
171
171
|
version_pattern = "MAJOR.MINOR.PATCH"
|
|
172
172
|
commit = false # We do version bumping in CI, not as a commit
|
|
173
173
|
tag = false # Git tag already exists — we don't auto-tag
|
|
@@ -1,23 +1,38 @@
|
|
|
1
1
|
from slide2vec.api import (
|
|
2
|
+
DenseImageOptions,
|
|
3
|
+
DenseOptions,
|
|
2
4
|
EmbeddedPatient,
|
|
3
5
|
EmbeddedSlide,
|
|
4
6
|
ExecutionOptions,
|
|
7
|
+
ImageSpec,
|
|
5
8
|
Model,
|
|
6
9
|
Pipeline,
|
|
7
10
|
PreprocessingConfig,
|
|
8
11
|
RunResult,
|
|
12
|
+
SlideRegions,
|
|
9
13
|
list_models,
|
|
10
14
|
)
|
|
11
|
-
from slide2vec.artifacts import
|
|
15
|
+
from slide2vec.artifacts import (
|
|
16
|
+
DenseImageArtifact,
|
|
17
|
+
DenseRegionArtifact,
|
|
18
|
+
HierarchicalEmbeddingArtifact,
|
|
19
|
+
ImageEmbeddingArtifact,
|
|
20
|
+
SlideEmbeddingArtifact,
|
|
21
|
+
TileEmbeddingArtifact,
|
|
22
|
+
)
|
|
12
23
|
|
|
13
24
|
|
|
14
|
-
__version__ = "5.
|
|
25
|
+
__version__ = "5.5.0"
|
|
15
26
|
|
|
16
27
|
__all__ = [
|
|
17
28
|
"Model",
|
|
18
29
|
"list_models",
|
|
19
30
|
"Pipeline",
|
|
20
31
|
"PreprocessingConfig",
|
|
32
|
+
"DenseOptions",
|
|
33
|
+
"DenseImageOptions",
|
|
34
|
+
"SlideRegions",
|
|
35
|
+
"ImageSpec",
|
|
21
36
|
"ExecutionOptions",
|
|
22
37
|
"RunResult",
|
|
23
38
|
"EmbeddedPatient",
|
|
@@ -25,5 +40,8 @@ __all__ = [
|
|
|
25
40
|
"SlideEmbeddingArtifact",
|
|
26
41
|
"HierarchicalEmbeddingArtifact",
|
|
27
42
|
"TileEmbeddingArtifact",
|
|
43
|
+
"DenseRegionArtifact",
|
|
44
|
+
"DenseImageArtifact",
|
|
45
|
+
"ImageEmbeddingArtifact",
|
|
28
46
|
"__version__",
|
|
29
47
|
]
|
|
@@ -11,7 +11,10 @@ import torch
|
|
|
11
11
|
from hs2p import SlideSpec
|
|
12
12
|
|
|
13
13
|
from slide2vec.artifacts import (
|
|
14
|
+
DenseImageArtifact,
|
|
15
|
+
DenseRegionArtifact,
|
|
14
16
|
HierarchicalEmbeddingArtifact,
|
|
17
|
+
ImageEmbeddingArtifact,
|
|
15
18
|
PatientEmbeddingArtifact,
|
|
16
19
|
SlideEmbeddingArtifact,
|
|
17
20
|
TileEmbeddingArtifact,
|
|
@@ -28,6 +31,8 @@ from slide2vec.runtime.model_settings import (
|
|
|
28
31
|
)
|
|
29
32
|
from slide2vec.progress import emit_progress
|
|
30
33
|
from slide2vec.runtime.types import LoadedModel
|
|
34
|
+
from slide2vec.runtime.effective_encoder_input import format_input_size
|
|
35
|
+
from slide2vec.runtime.encoder_input_contract import EncoderInputContract
|
|
31
36
|
from slide2vec.utils.utils import cpu_worker_limit, slurm_cpu_limit
|
|
32
37
|
|
|
33
38
|
PathLike = str | Path
|
|
@@ -101,6 +106,13 @@ class PreprocessingConfig:
|
|
|
101
106
|
#: Slide reading backend. ``"auto"`` tries cucim → openslide → vips in order.
|
|
102
107
|
#: Explicit choices: ``"cucim"``, ``"openslide"``, ``"vips"``, ``"asap"``.
|
|
103
108
|
backend: str = "auto"
|
|
109
|
+
#: Source-mask reading backend, resolved independently from the *mask* path
|
|
110
|
+
#: (hs2p ≥ 4.3.0). ``"auto"`` probes openability just like :attr:`backend`. Set this
|
|
111
|
+
#: explicitly (e.g. ``"openslide"``) when a precomputed tissue or annotation mask needs
|
|
112
|
+
#: a different decoder than its slide — hs2p no longer silently falls back to another
|
|
113
|
+
#: reader, so a mask the slide backend cannot decode fails unless overridden here.
|
|
114
|
+
#: Accepts the same values as :attr:`backend`; ignored for slides with no source mask.
|
|
115
|
+
mask_backend: str = "auto"
|
|
104
116
|
#: Target spacing in µm/px. Resolved from the model preset when ``None``.
|
|
105
117
|
requested_spacing_um: float | None = None
|
|
106
118
|
#: Tile side length in pixels at *requested_spacing_um*.
|
|
@@ -181,6 +193,7 @@ class PreprocessingConfig:
|
|
|
181
193
|
region_tile_multiple = getattr(tiling.params, "region_tile_multiple", None)
|
|
182
194
|
return cls(
|
|
183
195
|
backend=tiling.backend,
|
|
196
|
+
mask_backend=getattr(tiling, "mask_backend", "auto"),
|
|
184
197
|
requested_spacing_um=float(tiling.params.requested_spacing_um),
|
|
185
198
|
requested_tile_size_px=int(tiling.params.requested_tile_size_px),
|
|
186
199
|
requested_region_size_px=int(region_size_px) if region_size_px is not None else None,
|
|
@@ -208,6 +221,9 @@ class PreprocessingConfig:
|
|
|
208
221
|
def with_backend(self, backend: str) -> "PreprocessingConfig":
|
|
209
222
|
return replace(self, backend=backend)
|
|
210
223
|
|
|
224
|
+
def with_mask_backend(self, mask_backend: str) -> "PreprocessingConfig":
|
|
225
|
+
return replace(self, mask_backend=mask_backend)
|
|
226
|
+
|
|
211
227
|
|
|
212
228
|
|
|
213
229
|
@dataclass(frozen=True, kw_only=True)
|
|
@@ -312,6 +328,124 @@ class ExecutionOptions:
|
|
|
312
328
|
return replace(self, output_dir=Path(output_dir))
|
|
313
329
|
|
|
314
330
|
|
|
331
|
+
@dataclass(frozen=True, kw_only=True)
|
|
332
|
+
class DenseOptions:
|
|
333
|
+
"""Dense ``(d, gh, gw)`` grid extraction settings (issue #217).
|
|
334
|
+
|
|
335
|
+
The dense counterpart of the pooled :class:`PreprocessingConfig`: it names the
|
|
336
|
+
extraction geometry (spacing → level, supervision ``target_size``, padding) and the
|
|
337
|
+
dense encode knobs (whole-tile vs sliding-window, patch grid vs CLS-attention). Unlike
|
|
338
|
+
the pooled path there is no tiling — the caller supplies ROI coordinates directly (see
|
|
339
|
+
:class:`SlideRegions`) — so a ``DenseOptions`` carries only what slide2vec needs to read
|
|
340
|
+
and encode each ROI. ``ExecutionOptions`` is reused unchanged for output/precision/GPUs.
|
|
341
|
+
"""
|
|
342
|
+
|
|
343
|
+
#: Target spacing in µm/px the ROI is read at (resolved to a pyramid level per slide).
|
|
344
|
+
spacing_um: float
|
|
345
|
+
#: Supervision tile side length in pixels at *spacing_um* (the dense grid registers to it).
|
|
346
|
+
target_size: int
|
|
347
|
+
#: Relative spacing tolerance for pyramid level selection.
|
|
348
|
+
tolerance: float = 0.05
|
|
349
|
+
#: Slide reading backend. ``"auto"`` resolves per slide (cucim → openslide → vips).
|
|
350
|
+
backend: str = "auto"
|
|
351
|
+
#: Padding mode used to pad the tile up to the encoder's patch multiple.
|
|
352
|
+
#: One of ``"reflect"`` / ``"replicate"`` / ``"constant"`` / ``"zero"``.
|
|
353
|
+
pad_mode: str = "reflect"
|
|
354
|
+
#: Constant fill value for ``pad_mode in {"constant", "zero"}`` (ignored otherwise).
|
|
355
|
+
image_pad_value: float | None = None
|
|
356
|
+
#: Encoder field-of-view chunk fed through the backbone per forward. ``None`` (default)
|
|
357
|
+
#: is one whole-tile forward; a smaller value slides the encoder and blends token grids.
|
|
358
|
+
#: Together with ``target_size`` this fixes the *effective encoder input* — the geometry
|
|
359
|
+
#: handed to ``encode_tiles_dense`` — from which the encoder's variable-input constructor
|
|
360
|
+
#: settings are derived; hence no ``dynamic_img_size`` knob here.
|
|
361
|
+
window_size: int | None = None
|
|
362
|
+
#: Fractional window overlap in ``[0, 1)`` for the sliding path (ignored when
|
|
363
|
+
#: ``window_size is None``).
|
|
364
|
+
overlap: float = 0.0
|
|
365
|
+
#: ``"patch_features"`` (the patch-token grid) or ``"cls_attention"`` (CLS/register
|
|
366
|
+
#: self-attention grid).
|
|
367
|
+
feature_kind: str = "patch_features"
|
|
368
|
+
#: Transformer blocks whose CLS attention is read (``cls_attention`` only).
|
|
369
|
+
attention_blocks: tuple[int, ...] = (-1,)
|
|
370
|
+
#: Include register-token query rows as extra attention channels (``cls_attention`` only).
|
|
371
|
+
attention_include_registers: bool = False
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
@dataclass(frozen=True, kw_only=True)
|
|
375
|
+
class DenseImageOptions:
|
|
376
|
+
"""Dense ``(d, gh, gw)`` grid extraction over pre-cropped images (issue #235).
|
|
377
|
+
|
|
378
|
+
:class:`DenseOptions` minus everything that only a slide has: there is no spacing, no
|
|
379
|
+
tolerance and no reading backend here, because the image *is* the region — it is read
|
|
380
|
+
from disk at the size it was written. What remains is the same supervision geometry and
|
|
381
|
+
the same dense encode knobs, so a run migrating from ROIs to image/mask pairs keeps its
|
|
382
|
+
recipe.
|
|
383
|
+
|
|
384
|
+
``target_size`` is a **declaration**, not a resize: the dense transform is
|
|
385
|
+
normalization-only, so every image must already be exactly this size and one that is not
|
|
386
|
+
is an error rather than a silent rescale. Declaring it up front is what lets the
|
|
387
|
+
effective encoder input be validated (and the encoder's variable-input constructor
|
|
388
|
+
settings resolved) before a single image is decoded. A run whose images are not all the
|
|
389
|
+
same size is therefore several runs, one per geometry — which is also the only way their
|
|
390
|
+
grids could be batched downstream.
|
|
391
|
+
"""
|
|
392
|
+
|
|
393
|
+
#: Supervision geometry in pixels the dense grid registers to: a square side length, or
|
|
394
|
+
#: an explicit ``(height, width)`` for non-square images.
|
|
395
|
+
target_size: int | tuple[int, int]
|
|
396
|
+
#: Padding mode used to pad the image up to the encoder's patch multiple.
|
|
397
|
+
#: One of ``"reflect"`` / ``"replicate"`` / ``"constant"`` / ``"zero"``.
|
|
398
|
+
pad_mode: str = "reflect"
|
|
399
|
+
#: Constant fill value for ``pad_mode in {"constant", "zero"}`` (ignored otherwise).
|
|
400
|
+
image_pad_value: float | None = None
|
|
401
|
+
#: Encoder field-of-view chunk fed through the backbone per forward. ``None`` (default)
|
|
402
|
+
#: is one whole-image forward; a smaller value slides the encoder and blends token grids.
|
|
403
|
+
window_size: int | None = None
|
|
404
|
+
#: Fractional window overlap in ``[0, 1)`` for the sliding path (ignored when
|
|
405
|
+
#: ``window_size is None``).
|
|
406
|
+
overlap: float = 0.0
|
|
407
|
+
#: ``"patch_features"`` (the patch-token grid) or ``"cls_attention"`` (CLS/register
|
|
408
|
+
#: self-attention grid).
|
|
409
|
+
feature_kind: str = "patch_features"
|
|
410
|
+
#: Transformer blocks whose CLS attention is read (``cls_attention`` only).
|
|
411
|
+
attention_blocks: tuple[int, ...] = (-1,)
|
|
412
|
+
#: Include register-token query rows as extra attention channels (``cls_attention`` only).
|
|
413
|
+
attention_include_registers: bool = False
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
@dataclass(frozen=True, kw_only=True)
|
|
417
|
+
class SlideRegions:
|
|
418
|
+
"""One slide's ROIs for dense extraction: ``(sample_id, image_path, coordinates, annotation)``.
|
|
419
|
+
|
|
420
|
+
The dense input unit soma's slide-manifest path hands to
|
|
421
|
+
:meth:`Model.embed_regions_dense`. ``coordinates`` is an ``(N, 2)`` array of level-0
|
|
422
|
+
top-left ``(x, y)`` pixel coordinates; each ROI is read + encoded into one persisted
|
|
423
|
+
``(d, gh, gw)`` grid named ``<x>_<y>.pt``. ``annotation`` namespaces the output under a
|
|
424
|
+
per-class subdirectory (reusing the pooled convention); ``None`` is the flat layout.
|
|
425
|
+
"""
|
|
426
|
+
|
|
427
|
+
sample_id: str
|
|
428
|
+
image_path: PathLike
|
|
429
|
+
coordinates: Any
|
|
430
|
+
annotation: str | None = None
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
@dataclass(frozen=True, kw_only=True)
|
|
434
|
+
class ImageSpec:
|
|
435
|
+
"""One pre-cropped image to embed: ``(sample_id, image_path)``.
|
|
436
|
+
|
|
437
|
+
The input unit of :meth:`Model.embed_images` — the Given-geometry counterpart of
|
|
438
|
+
:class:`SlideRegions`. There is no slide, no coordinate and no spacing here: the caller
|
|
439
|
+
holds an image file it never asked slide2vec to produce (a public patch benchmark
|
|
440
|
+
sample), and names it. ``sample_id`` is the artifact's whole identity, so it must be
|
|
441
|
+
unique within a run and a valid filename component; slide2vec never derives it from the
|
|
442
|
+
path, because two directories can hold the same filename.
|
|
443
|
+
"""
|
|
444
|
+
|
|
445
|
+
sample_id: str
|
|
446
|
+
image_path: PathLike
|
|
447
|
+
|
|
448
|
+
|
|
315
449
|
@dataclass(frozen=True, kw_only=True)
|
|
316
450
|
class RunResult:
|
|
317
451
|
"""Return value of :meth:`Pipeline.run`."""
|
|
@@ -374,6 +508,8 @@ class EmbeddedSlide:
|
|
|
374
508
|
tiling_preview_path: Path | None = None
|
|
375
509
|
#: Encoder latent representations when available; ``None`` otherwise.
|
|
376
510
|
latents: Any | None = None
|
|
511
|
+
#: Factual square tensor side length immediately before tile encoding.
|
|
512
|
+
encoder_input_size_px: int | None = None
|
|
377
513
|
|
|
378
514
|
|
|
379
515
|
class Model:
|
|
@@ -391,6 +527,13 @@ class Model:
|
|
|
391
527
|
self.allow_non_recommended_settings = bool(allow_non_recommended_settings)
|
|
392
528
|
self._output_variant = output_variant
|
|
393
529
|
self._backend: LoadedModel | None = None
|
|
530
|
+
# Unset, deliberately: a Model has no encoder-input contract until a route
|
|
531
|
+
# declares one. There is no initial Given contract, because an initial value is
|
|
532
|
+
# a default by another name — it would silently hand the shipped transform to
|
|
533
|
+
# any route that forgot to declare, which is the confusion this contract exists
|
|
534
|
+
# to delete. ``_load_backend`` refuses to load until this is set.
|
|
535
|
+
self._encoder_input: EncoderInputContract | None = None
|
|
536
|
+
self._backend_encoder_input: EncoderInputContract | None = None
|
|
394
537
|
|
|
395
538
|
@classmethod
|
|
396
539
|
def from_preset(
|
|
@@ -410,11 +553,13 @@ class Model:
|
|
|
410
553
|
|
|
411
554
|
@property
|
|
412
555
|
def device(self) -> Any:
|
|
413
|
-
|
|
556
|
+
# Construction fact, not an encode: see _load_backend_without_transform.
|
|
557
|
+
return self._load_backend_without_transform().device
|
|
414
558
|
|
|
415
559
|
@property
|
|
416
560
|
def feature_dim(self) -> int:
|
|
417
|
-
|
|
561
|
+
# Construction fact, not an encode: see _load_backend_without_transform.
|
|
562
|
+
return int(self._load_backend_without_transform().feature_dim)
|
|
418
563
|
|
|
419
564
|
def embed_tiles(
|
|
420
565
|
self,
|
|
@@ -588,16 +733,254 @@ class Model:
|
|
|
588
733
|
execution=resolved,
|
|
589
734
|
)
|
|
590
735
|
|
|
736
|
+
def embed_regions_dense(
|
|
737
|
+
self,
|
|
738
|
+
regions: "Sequence[SlideRegions]",
|
|
739
|
+
*,
|
|
740
|
+
dense: "DenseOptions",
|
|
741
|
+
execution: ExecutionOptions | None = None,
|
|
742
|
+
) -> list[DenseRegionArtifact]:
|
|
743
|
+
"""Extract + persist a dense ``(d, gh, gw)`` grid per caller-supplied ROI.
|
|
744
|
+
|
|
745
|
+
The dense counterpart of the pooled coordinate path: each ``SlideRegions`` names a
|
|
746
|
+
slide + a set of level-0 ROI coordinates, and every ROI is read, encoded through the
|
|
747
|
+
dense transform, and written to ``dense_embeddings/[<class>/]<sample_id>/<x>_<y>.pt``
|
|
748
|
+
plus a geometry sidecar. The run splits its ROIs across all visible GPUs
|
|
749
|
+
(``execution.num_gpus``); ``num_gpus=1`` encodes fully in-process. Resume is
|
|
750
|
+
automatic — ROIs whose sidecar already exists are skipped. Returns one
|
|
751
|
+
:class:`~slide2vec.artifacts.DenseRegionArtifact` per input ROI.
|
|
752
|
+
|
|
753
|
+
The effective encoder input — the padded ROI for a whole-tile run, one
|
|
754
|
+
patch-aligned window for a sliding one — is declared before any region is read, so
|
|
755
|
+
a geometry the encoder cannot accept raises here rather than at the first forward
|
|
756
|
+
pass. Variable-input capable encoders get their registry-declared constructor
|
|
757
|
+
settings applied automatically; there is nothing for the caller to pass.
|
|
758
|
+
"""
|
|
759
|
+
from slide2vec.runtime.dense_stage import embed_regions_dense
|
|
760
|
+
|
|
761
|
+
resolved = _coerce_execution_options(execution, model=self)
|
|
762
|
+
_require_output_dir_for_persistence(resolved, method_name="Model.embed_regions_dense(...)")
|
|
763
|
+
with _auto_progress_reporting(output_dir=resolved.output_dir):
|
|
764
|
+
return embed_regions_dense(self, regions, dense=dense, execution=resolved)
|
|
765
|
+
|
|
766
|
+
def embed_images_dense(
|
|
767
|
+
self,
|
|
768
|
+
images: "Sequence[ImageSpec]",
|
|
769
|
+
*,
|
|
770
|
+
dense: "DenseImageOptions",
|
|
771
|
+
execution: ExecutionOptions | None = None,
|
|
772
|
+
) -> list[DenseImageArtifact]:
|
|
773
|
+
"""Extract + persist a dense ``(d, gh, gw)`` grid per caller-supplied image.
|
|
774
|
+
|
|
775
|
+
The image-sourced counterpart of :meth:`embed_regions_dense`, for consumers whose
|
|
776
|
+
supervision arrives as image/mask pairs rather than as slides (segmentation,
|
|
777
|
+
detection): each :class:`ImageSpec` is decoded, run through the encoder's
|
|
778
|
+
**normalization-only** transform, padded up to the encoder's patch multiple, encoded
|
|
779
|
+
— whole-image, or by sliding the encoder's native field and blending the token grids
|
|
780
|
+
— and written to ``dense_image_embeddings/<sample_id>.pt`` plus a geometry sidecar.
|
|
781
|
+
The run splits its images across all visible GPUs (``execution.num_gpus``);
|
|
782
|
+
``num_gpus=1`` encodes fully in-process. Resume is automatic — images whose sidecar
|
|
783
|
+
already exists are skipped. Returns one
|
|
784
|
+
:class:`~slide2vec.artifacts.DenseImageArtifact` per input image, in input order.
|
|
785
|
+
|
|
786
|
+
It differs from :meth:`embed_regions_dense` in exactly one respect: there is no
|
|
787
|
+
slide, no coordinate and no spacing→level plan, because the image *is* the region.
|
|
788
|
+
Everything else is shared, including the effective encoder input — the padded image
|
|
789
|
+
for a whole-image run, one patch-aligned window for a sliding one — which is declared
|
|
790
|
+
before any image is decoded, so a geometry the encoder cannot accept raises here
|
|
791
|
+
rather than on a torchrun rank's first forward pass.
|
|
792
|
+
|
|
793
|
+
``dense.target_size`` is a declaration, not a resize request: dense extraction never
|
|
794
|
+
rescales, so every image must already be that size (a non-square ``(h, w)`` is fine).
|
|
795
|
+
"""
|
|
796
|
+
from slide2vec.runtime.dense_image_stage import embed_images_dense
|
|
797
|
+
|
|
798
|
+
resolved = _coerce_execution_options(execution, model=self)
|
|
799
|
+
_require_output_dir_for_persistence(resolved, method_name="Model.embed_images_dense(...)")
|
|
800
|
+
with _auto_progress_reporting(output_dir=resolved.output_dir):
|
|
801
|
+
return embed_images_dense(self, images, dense=dense, execution=resolved)
|
|
802
|
+
|
|
803
|
+
def embed_images(
|
|
804
|
+
self,
|
|
805
|
+
images: "Sequence[ImageSpec]",
|
|
806
|
+
*,
|
|
807
|
+
execution: ExecutionOptions | None = None,
|
|
808
|
+
) -> list[ImageEmbeddingArtifact]:
|
|
809
|
+
"""Embed + persist one embedding per caller-supplied image.
|
|
810
|
+
|
|
811
|
+
The Given-geometry entry point: the caller already holds pre-cropped images — a
|
|
812
|
+
public patch benchmark (BACH, CRC, Gleason, BreakHis, MHIST, PCam), an exported ROI
|
|
813
|
+
set — and slide2vec neither tiles nor reads a slide. Each :class:`ImageSpec` is
|
|
814
|
+
decoded, preprocessed with the encoder's **shipped** transform, encoded, and written
|
|
815
|
+
to ``image_embeddings/<sample_id>.pt`` plus a provenance sidecar. The run splits its
|
|
816
|
+
images across all visible GPUs (``execution.num_gpus``); ``num_gpus=1`` encodes
|
|
817
|
+
fully in-process. Resume is automatic — images whose sidecar already exists are
|
|
818
|
+
skipped. Returns one :class:`~slide2vec.artifacts.ImageEmbeddingArtifact` per input
|
|
819
|
+
image, in input order.
|
|
820
|
+
|
|
821
|
+
Unlike the pooled and dense paths there is no geometry to declare: the images are
|
|
822
|
+
heterogeneously sized (2048x1536 beside 96x96) and were never requested, so the
|
|
823
|
+
encoder's shipped transform is the contract and slide2vec *records* the resulting
|
|
824
|
+
encoder input size as run provenance rather than validating it. That also means
|
|
825
|
+
preprocessing runs itemwise in the loader workers — differently sized images cannot
|
|
826
|
+
be stacked before they are resized.
|
|
827
|
+
"""
|
|
828
|
+
from slide2vec.runtime.image_stage import embed_images
|
|
829
|
+
|
|
830
|
+
resolved = _coerce_execution_options(execution, model=self)
|
|
831
|
+
_require_output_dir_for_persistence(resolved, method_name="Model.embed_images(...)")
|
|
832
|
+
with _auto_progress_reporting(output_dir=resolved.output_dir):
|
|
833
|
+
return embed_images(self, images, execution=resolved)
|
|
834
|
+
|
|
835
|
+
def _declare_encoder_input(
|
|
836
|
+
self,
|
|
837
|
+
preprocessing: PreprocessingConfig,
|
|
838
|
+
*,
|
|
839
|
+
emit_run_info: bool,
|
|
840
|
+
) -> EncoderInputContract:
|
|
841
|
+
"""Declare the pooled encoder input geometry this run requested, or raise.
|
|
842
|
+
|
|
843
|
+
Idempotent: resolving the same preprocessing twice yields an equal contract, so
|
|
844
|
+
every layer that reaches the encoder may declare for itself rather than trust
|
|
845
|
+
the layer above to have done it.
|
|
846
|
+
"""
|
|
847
|
+
if preprocessing.requested_tile_size_px is None:
|
|
848
|
+
raise ValueError(
|
|
849
|
+
"requested_tile_size_px must be resolved before declaring the encoder "
|
|
850
|
+
"input geometry; a pooled run reads tiles at a size it requested."
|
|
851
|
+
)
|
|
852
|
+
contract = EncoderInputContract.declared_pooled(
|
|
853
|
+
self.name,
|
|
854
|
+
requested_tile_size_px=int(preprocessing.requested_tile_size_px),
|
|
855
|
+
allow_non_recommended_settings=self.allow_non_recommended_settings,
|
|
856
|
+
)
|
|
857
|
+
self._encoder_input = contract
|
|
858
|
+
plan = contract.plan
|
|
859
|
+
if emit_run_info and plan.requires_variable_model_input:
|
|
860
|
+
logging.getLogger("slide2vec").info(
|
|
861
|
+
"Pooled encoder input for '%s': preset %dpx, requested %dpx, "
|
|
862
|
+
"exact encoder input %dpx; using normalization-only preprocessing.",
|
|
863
|
+
self.name,
|
|
864
|
+
plan.preset_input_size_px,
|
|
865
|
+
plan.requested_tile_size_px,
|
|
866
|
+
plan.expected_encoder_input_size_px,
|
|
867
|
+
)
|
|
868
|
+
return contract
|
|
869
|
+
|
|
870
|
+
def _declare_dense_encoder_input(
|
|
871
|
+
self,
|
|
872
|
+
dense: "DenseOptions",
|
|
873
|
+
*,
|
|
874
|
+
emit_run_info: bool,
|
|
875
|
+
) -> EncoderInputContract:
|
|
876
|
+
"""Declare the dense encoder input geometry this run requested, or raise.
|
|
877
|
+
|
|
878
|
+
Dense states a supervision geometry (``target_size``, optional ``window_size``)
|
|
879
|
+
rather than an encoder input; the contract derives the tensor the backbone will
|
|
880
|
+
actually see and validates it exactly as the pooled path's is validated. Like the
|
|
881
|
+
pooled declaration this is idempotent, so each layer that reaches the encoder — the
|
|
882
|
+
parent stage and every torchrun rank — declares for itself.
|
|
883
|
+
|
|
884
|
+
*dense* is a :class:`DenseOptions` (ROIs on a slide) or a :class:`DenseImageOptions`
|
|
885
|
+
(pre-cropped images); only the supervision geometry is read here, and the two state
|
|
886
|
+
it the same way.
|
|
887
|
+
"""
|
|
888
|
+
contract = EncoderInputContract.declared_dense(
|
|
889
|
+
self.name,
|
|
890
|
+
target_size_px=dense.target_size,
|
|
891
|
+
window_size=None if dense.window_size is None else int(dense.window_size),
|
|
892
|
+
)
|
|
893
|
+
self._encoder_input = contract
|
|
894
|
+
plan = contract.plan
|
|
895
|
+
if emit_run_info and plan.requires_variable_model_input:
|
|
896
|
+
logging.getLogger("slide2vec").info(
|
|
897
|
+
"Dense encoder input for '%s': native %dpx, effective encoder input %s "
|
|
898
|
+
"(target_size=%s, window_size=%s); enabling variable input size via %s.",
|
|
899
|
+
self.name,
|
|
900
|
+
plan.preset_input_size_px,
|
|
901
|
+
format_input_size(plan.effective_encoder_input_size_px),
|
|
902
|
+
format_input_size(plan.target_size_px),
|
|
903
|
+
plan.window_size_px,
|
|
904
|
+
plan.model_construction_kwargs or "no constructor setting",
|
|
905
|
+
)
|
|
906
|
+
return contract
|
|
907
|
+
|
|
908
|
+
def _declare_given_encoder_input(self, *, emit_run_info: bool) -> EncoderInputContract:
|
|
909
|
+
"""Declare that this run's encoder input is whatever the caller handed over.
|
|
910
|
+
|
|
911
|
+
The Given regime's affirmative statement. It is deliberately not the same thing as
|
|
912
|
+
leaving ``_encoder_input`` unset: an absent contract means "this route forgot", and
|
|
913
|
+
the contract refuses to guess between the two. Like the declared variants this is
|
|
914
|
+
idempotent, so the parent stage and every torchrun rank declare for themselves.
|
|
915
|
+
"""
|
|
916
|
+
contract = EncoderInputContract.given()
|
|
917
|
+
self._encoder_input = contract
|
|
918
|
+
if emit_run_info:
|
|
919
|
+
logging.getLogger("slide2vec").info(
|
|
920
|
+
"Given encoder input for '%s': using the encoder's shipped preprocessing; "
|
|
921
|
+
"the observed encoder input size is recorded per artifact, not validated.",
|
|
922
|
+
self.name,
|
|
923
|
+
)
|
|
924
|
+
return contract
|
|
925
|
+
|
|
591
926
|
def _load_backend(self) -> LoadedModel:
|
|
592
|
-
|
|
927
|
+
"""Load the backend under this run's declared encoder-input contract.
|
|
928
|
+
|
|
929
|
+
Every caller that reads ``loaded.transforms`` — i.e. everything that turns
|
|
930
|
+
pixels into features — must come through here, and must therefore have
|
|
931
|
+
declared its geometry first.
|
|
932
|
+
"""
|
|
933
|
+
if self._encoder_input is None:
|
|
934
|
+
raise ValueError(
|
|
935
|
+
f"No encoder-input contract has been declared for model '{self.name}'. "
|
|
936
|
+
"A route that encodes pixels must state its geometry before the "
|
|
937
|
+
"backend is loaded: call _declare_encoder_input(preprocessing, ...) "
|
|
938
|
+
"for a pooled run, or _declare_dense_encoder_input(dense, ...) for a "
|
|
939
|
+
"dense one. Callers that never read loaded.transforms use "
|
|
940
|
+
"_load_backend_without_transform() instead."
|
|
941
|
+
)
|
|
942
|
+
return self._load_backend_under(self._encoder_input)
|
|
943
|
+
|
|
944
|
+
def _load_backend_without_transform(self) -> LoadedModel:
|
|
945
|
+
"""Load the backend for callers that never read ``loaded.transforms``.
|
|
946
|
+
|
|
947
|
+
Two kinds of caller need the constructed encoder module without ever selecting a
|
|
948
|
+
tile transform: the ``device``/``feature_dim`` properties (pure construction
|
|
949
|
+
facts) and tile→slide/patient aggregation (``encode_slide`` / ``encode_patient``
|
|
950
|
+
consume already-computed features). They cannot observe, let alone encode
|
|
951
|
+
through, the transform the backend happens to carry.
|
|
952
|
+
|
|
953
|
+
Dense extraction is deliberately NOT in this set. It builds its own normalization
|
|
954
|
+
transform and never reads ``loaded.transforms``, but it does need the
|
|
955
|
+
variable-input constructor settings its geometry implies — which is exactly what
|
|
956
|
+
an encoder-input contract carries — so it declares (see
|
|
957
|
+
``_declare_dense_encoder_input``) and loads through ``_load_backend``.
|
|
958
|
+
|
|
959
|
+
A declared contract is honored when one exists so the cached backend is shared;
|
|
960
|
+
otherwise an explicit Given contract is used for this load only. This never
|
|
961
|
+
assigns ``_encoder_input``: an embed route still has to declare, and
|
|
962
|
+
``_load_backend`` reloads when the declared contract differs from the one the
|
|
963
|
+
cached backend was built under.
|
|
964
|
+
"""
|
|
965
|
+
return self._load_backend_under(
|
|
966
|
+
self._encoder_input
|
|
967
|
+
if self._encoder_input is not None
|
|
968
|
+
else EncoderInputContract.given()
|
|
969
|
+
)
|
|
970
|
+
|
|
971
|
+
def _load_backend_under(self, encoder_input: EncoderInputContract) -> LoadedModel:
|
|
972
|
+
if self._backend is None or self._backend_encoder_input != encoder_input:
|
|
593
973
|
from slide2vec.inference import load_model
|
|
594
974
|
|
|
595
975
|
emit_progress("model.loading", model_name=self.name)
|
|
596
976
|
self._backend = load_model(
|
|
597
977
|
name=self.name,
|
|
978
|
+
encoder_input=encoder_input,
|
|
598
979
|
device=self._requested_device,
|
|
599
980
|
output_variant=self._output_variant,
|
|
981
|
+
allow_non_recommended_settings=self.allow_non_recommended_settings,
|
|
600
982
|
)
|
|
983
|
+
self._backend_encoder_input = encoder_input
|
|
601
984
|
emit_progress("model.ready", model_name=self.name, device=str(self._backend.device))
|
|
602
985
|
return self._backend
|
|
603
986
|
|
|
@@ -838,12 +1221,12 @@ def _validate_model_config(
|
|
|
838
1221
|
info = encoder_registry.info(name)
|
|
839
1222
|
if info["level"] != "tile":
|
|
840
1223
|
raise ValueError("Hierarchical preprocessing is only supported for tile encoders")
|
|
1224
|
+
model._declare_encoder_input(preprocessing, emit_run_info=True)
|
|
841
1225
|
# Skip precision validation for CPU execution (fp32 is always valid on CPU).
|
|
842
1226
|
on_cpu = model._requested_device == "cpu"
|
|
843
1227
|
precision = None if on_cpu or execution is None else execution.precision
|
|
844
1228
|
validate_encoder_config(
|
|
845
1229
|
name,
|
|
846
|
-
requested_tile_size_px=preprocessing.requested_tile_size_px,
|
|
847
1230
|
requested_spacing_um=preprocessing.requested_spacing_um,
|
|
848
1231
|
precision=precision,
|
|
849
1232
|
output_variant=model._output_variant,
|