slide2vec 5.5.0__tar.gz → 5.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {slide2vec-5.5.0 → slide2vec-5.6.0}/PKG-INFO +3 -3
- {slide2vec-5.5.0 → slide2vec-5.6.0}/pyproject.toml +6 -5
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/__init__.py +1 -1
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/api.py +172 -61
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/artifacts.py +44 -24
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/configs/default.yaml +3 -3
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/data/dataset.py +20 -5
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/distributed/dense_image_worker.py +9 -1
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/distributed/image_worker.py +1 -1
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/registry.py +30 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/validation.py +8 -1
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/inference.py +13 -23
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/artifacts_collect.py +13 -16
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/batching.py +11 -3
- slide2vec-5.6.0/slide2vec/runtime/dense_image_reading.py +260 -0
- slide2vec-5.6.0/slide2vec/runtime/dense_image_recipe.py +290 -0
- slide2vec-5.6.0/slide2vec/runtime/dense_image_shard.py +341 -0
- slide2vec-5.6.0/slide2vec/runtime/dense_image_stage.py +303 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/dense_regions.py +52 -8
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/dense_shard.py +5 -2
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/distributed.py +4 -12
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/embedding.py +6 -4
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/embedding_pipeline.py +10 -20
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/image_shard.py +3 -2
- slide2vec-5.6.0/slide2vec/runtime/image_specs.py +218 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/image_stage.py +10 -2
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/persist_callbacks.py +2 -4
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/persistence.py +51 -47
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/serialization.py +22 -1
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/slide_encode.py +5 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/utils/config.py +47 -27
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/utils/tiling_io.py +40 -10
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec.egg-info/PKG-INFO +3 -3
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec.egg-info/SOURCES.txt +7 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec.egg-info/requires.txt +2 -2
- slide2vec-5.6.0/tests/test_dense_image_reading.py +408 -0
- slide2vec-5.6.0/tests/test_dense_image_resume.py +431 -0
- slide2vec-5.6.0/tests/test_dense_image_shard.py +870 -0
- slide2vec-5.6.0/tests/test_dense_image_stage.py +1052 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dense_shard.py +36 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dense_stage.py +8 -1
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_hs2p_package_cutover.py +78 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_image_stage.py +51 -2
- slide2vec-5.6.0/tests/test_patient_manifest.py +108 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_regression_core.py +453 -12
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_regression_inference.py +276 -25
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_regression_models.py +64 -2
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_runtime_batching.py +20 -0
- slide2vec-5.6.0/tests/test_slide_coordinate_preparation.py +263 -0
- slide2vec-5.6.0/tests/test_soma_migration.py +557 -0
- slide2vec-5.5.0/slide2vec/runtime/dense_image_shard.py +0 -209
- slide2vec-5.5.0/slide2vec/runtime/dense_image_stage.py +0 -164
- slide2vec-5.5.0/slide2vec/runtime/image_specs.py +0 -70
- slide2vec-5.5.0/tests/test_dense_image_shard.py +0 -407
- slide2vec-5.5.0/tests/test_dense_image_stage.py +0 -338
- {slide2vec-5.5.0 → slide2vec-5.6.0}/LICENSE +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/README.md +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/setup.cfg +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/__main__.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/cli.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/configs/__init__.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/configs/resources.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/data/__init__.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/data/tile_reader.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/data/tile_store.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/distributed/__init__.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/distributed/dense_worker.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/distributed/direct_embed_worker.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/distributed/pipeline_worker.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/distributed/worker_entry.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/__init__.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/base.py +10 -10
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/__init__.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/conch.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/dinov2.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/genbio.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/gigapath.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/gpfm.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/hibou.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/hoptimus.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/isight.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/lunit.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/midnight.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/moozy/__init__.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/moozy/blocks.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/moozy/case.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/moozy/loading.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/moozy/slide.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/moozy/types.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/mstar.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/musk.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/phikon.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/prism.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/prost40m.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/titan.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/uni.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/virchow.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/progress.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/__init__.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/cpu_budget.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/dense_encoder_input.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/dense_sliding.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/dense_stage.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/distributed_stage.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/effective_encoder_input.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/embedding_persist.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/encoder_input_contract.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/hierarchical.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/manifest.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/model_settings.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/patient_pipeline.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/pooled_encoder_input.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/preprocessing.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/process_list.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/progress_bridge.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/registry.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/sharding.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/tiling.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/tiling_pipeline.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/types.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/worker_io.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/utils/__init__.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/utils/coordinates.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/utils/log_utils.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/utils/utils.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec.egg-info/dependency_links.txt +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec.egg-info/entry_points.txt +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec.egg-info/not-zip-safe +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec.egg-info/top_level.txt +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_architecture_runtime_split.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_attention_extraction.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dense_encoder_input.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dense_extraction.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dense_regions.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dense_sliding.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dense_worker.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dinov2_natimage.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_encoder_input_contract.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_encoder_registry.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_gpfm_genbio_heavy.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_image_shard.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_isight.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_output_consistency.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_patch_size_metadata.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_pooled_encoder_input.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_pooled_geometry.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_progress.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_sharding.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_tile_store.py +0 -0
- {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_tiling_pipeline.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: slide2vec
|
|
3
|
-
Version: 5.
|
|
3
|
+
Version: 5.6.0
|
|
4
4
|
Summary: Embedding of whole slide images with Foundation Models
|
|
5
5
|
Author-email: Clément Grisi <clement.grisi@radboudumc.nl>
|
|
6
6
|
License-Expression: Apache-2.0
|
|
@@ -15,7 +15,7 @@ Classifier: Programming Language :: Python :: 3.13
|
|
|
15
15
|
Requires-Python: >=3.10
|
|
16
16
|
Description-Content-Type: text/markdown
|
|
17
17
|
License-File: LICENSE
|
|
18
|
-
Requires-Dist: hs2p[asap,cucim,openslide,sam2,vips]>=4.
|
|
18
|
+
Requires-Dist: hs2p[asap,cucim,openslide,sam2,vips]>=4.4.0
|
|
19
19
|
Requires-Dist: omegaconf
|
|
20
20
|
Requires-Dist: matplotlib
|
|
21
21
|
Requires-Dist: numpy<2
|
|
@@ -65,7 +65,7 @@ Requires-Dist: numpy<2; extra == "fm"
|
|
|
65
65
|
Requires-Dist: pandas; extra == "fm"
|
|
66
66
|
Requires-Dist: pillow; extra == "fm"
|
|
67
67
|
Requires-Dist: rich; extra == "fm"
|
|
68
|
-
Requires-Dist: hs2p[asap,cucim,openslide,sam2,vips]>=4.
|
|
68
|
+
Requires-Dist: hs2p[asap,cucim,openslide,sam2,vips]>=4.4.0; extra == "fm"
|
|
69
69
|
Requires-Dist: wandb; extra == "fm"
|
|
70
70
|
Requires-Dist: torch<2.8,>=2.3; extra == "fm"
|
|
71
71
|
Requires-Dist: torchvision>=0.18.0; extra == "fm"
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "slide2vec"
|
|
7
|
-
version = "5.
|
|
7
|
+
version = "5.6.0"
|
|
8
8
|
description = "Embedding of whole slide images with Foundation Models"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -21,7 +21,7 @@ classifiers = [
|
|
|
21
21
|
"Programming Language :: Python :: 3.13",
|
|
22
22
|
]
|
|
23
23
|
dependencies = [
|
|
24
|
-
"hs2p[asap,cucim,openslide,sam2,vips]>=4.
|
|
24
|
+
"hs2p[asap,cucim,openslide,sam2,vips]>=4.4.0",
|
|
25
25
|
"omegaconf",
|
|
26
26
|
"matplotlib",
|
|
27
27
|
"numpy<2",
|
|
@@ -88,7 +88,7 @@ fm = [
|
|
|
88
88
|
"pandas",
|
|
89
89
|
"pillow",
|
|
90
90
|
"rich",
|
|
91
|
-
"hs2p[asap,cucim,openslide,sam2,vips]>=4.
|
|
91
|
+
"hs2p[asap,cucim,openslide,sam2,vips]>=4.4.0",
|
|
92
92
|
"wandb",
|
|
93
93
|
"torch>=2.3,<2.8",
|
|
94
94
|
"torchvision>=0.18.0",
|
|
@@ -141,12 +141,13 @@ slide2vec = ["py.typed"]
|
|
|
141
141
|
"slide2vec.configs" = ["*.yaml", "models/*.yaml", "preprocessing/*.yaml"]
|
|
142
142
|
|
|
143
143
|
[tool.pytest.ini_options]
|
|
144
|
-
addopts = "--cov=slide2vec"
|
|
144
|
+
addopts = "--cov=slide2vec -m 'not gpu_integration'"
|
|
145
145
|
testpaths = [
|
|
146
146
|
"tests",
|
|
147
147
|
]
|
|
148
148
|
markers = [
|
|
149
149
|
"heavy: real-weight foundation-model inference on CPU; minutes per test. Excluded from the PR suite via `-m 'not heavy'`; run on the scheduled/manual heavy workflow (.github/workflows/nightly-heavy.yaml).",
|
|
150
|
+
"gpu_integration: real one-GPU versus multi-GPU parity; requires at least two visible CUDA devices. Run explicitly with `CUDA_VISIBLE_DEVICES=0,1 python -m pytest -m gpu_integration --no-cov`.",
|
|
150
151
|
]
|
|
151
152
|
|
|
152
153
|
[tool.mypy]
|
|
@@ -167,7 +168,7 @@ no_implicit_reexport = true
|
|
|
167
168
|
max-line-length = 160
|
|
168
169
|
|
|
169
170
|
[tool.bumpver]
|
|
170
|
-
current_version = "5.
|
|
171
|
+
current_version = "5.6.0"
|
|
171
172
|
version_pattern = "MAJOR.MINOR.PATCH"
|
|
172
173
|
commit = false # We do version bumping in CI, not as a commit
|
|
173
174
|
tag = false # Git tag already exists — we don't auto-tag
|
|
@@ -19,9 +19,10 @@ from slide2vec.artifacts import (
|
|
|
19
19
|
SlideEmbeddingArtifact,
|
|
20
20
|
TileEmbeddingArtifact,
|
|
21
21
|
)
|
|
22
|
+
from slide2vec.configs.resources import load_config
|
|
22
23
|
from slide2vec.encoders.registry import (
|
|
23
24
|
encoder_registry,
|
|
24
|
-
|
|
25
|
+
resolve_preprocessing_fields,
|
|
25
26
|
)
|
|
26
27
|
from slide2vec.encoders.validation import validate_encoder_config
|
|
27
28
|
from slide2vec.runtime.model_settings import (
|
|
@@ -63,14 +64,42 @@ DEFAULT_MASKS: dict[str, Any] = {
|
|
|
63
64
|
"min_coverage": {"background": None, "tissue": 0.01},
|
|
64
65
|
}
|
|
65
66
|
|
|
67
|
+
_REQUESTED_TILE_SIZE_INTERPOLATION = "${tiling.params.requested_tile_size_px}"
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _load_default_preprocessing() -> dict[str, dict[str, Any]]:
|
|
71
|
+
"""Read the public nested defaults from the package's canonical YAML config."""
|
|
72
|
+
from omegaconf import OmegaConf
|
|
73
|
+
|
|
74
|
+
tiling = load_config("default").tiling
|
|
75
|
+
defaults: dict[str, dict[str, Any]] = {}
|
|
76
|
+
for public_name, config_name in (
|
|
77
|
+
("segmentation", "seg_params"),
|
|
78
|
+
("filtering", "filter_params"),
|
|
79
|
+
("preview", "preview"),
|
|
80
|
+
):
|
|
81
|
+
section = OmegaConf.to_container(getattr(tiling, config_name), resolve=False)
|
|
82
|
+
if not isinstance(section, dict):
|
|
83
|
+
raise TypeError(f"tiling.{config_name} must be a mapping")
|
|
84
|
+
defaults[public_name] = section
|
|
85
|
+
defaults["preview"]["tissue_contour_color"] = tuple(
|
|
86
|
+
defaults["preview"]["tissue_contour_color"]
|
|
87
|
+
)
|
|
88
|
+
return defaults
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
#: Complete defaults for the nested public preprocessing sections, loaded from
|
|
92
|
+
#: ``configs/default.yaml`` so Python and YAML entry points share one source.
|
|
93
|
+
DEFAULT_PREPROCESSING = _load_default_preprocessing()
|
|
66
94
|
|
|
67
|
-
|
|
95
|
+
|
|
96
|
+
def _deep_merge_dicts(base: Mapping[str, Any], override: Mapping[str, Any]) -> dict[str, Any]:
|
|
68
97
|
"""Deep-merge *override* onto a copy of *base* (nested dicts merge key-by-key)."""
|
|
69
98
|
merged = copy.deepcopy(dict(base))
|
|
70
99
|
for key, value in override.items():
|
|
71
100
|
existing = merged.get(key)
|
|
72
101
|
if isinstance(value, Mapping) and isinstance(existing, dict):
|
|
73
|
-
merged[key] =
|
|
102
|
+
merged[key] = _deep_merge_dicts(existing, value)
|
|
74
103
|
else:
|
|
75
104
|
merged[key] = copy.deepcopy(value)
|
|
76
105
|
return merged
|
|
@@ -80,7 +109,7 @@ def resolve_masks(masks: Mapping[str, Any] | None) -> dict[str, Any]:
|
|
|
80
109
|
"""Complete a (possibly partial) ``masks`` mapping by merging it over :data:`DEFAULT_MASKS`."""
|
|
81
110
|
if not masks:
|
|
82
111
|
return copy.deepcopy(DEFAULT_MASKS)
|
|
83
|
-
return
|
|
112
|
+
return _deep_merge_dicts(DEFAULT_MASKS, masks)
|
|
84
113
|
|
|
85
114
|
|
|
86
115
|
def _masks_to_plain_dict(node: Any) -> dict[str, Any]:
|
|
@@ -103,7 +132,7 @@ def _masks_to_plain_dict(node: Any) -> dict[str, Any]:
|
|
|
103
132
|
class PreprocessingConfig:
|
|
104
133
|
"""Configuration for slide tiling and preprocessing."""
|
|
105
134
|
|
|
106
|
-
#: Slide reading backend. ``"auto"`` tries cucim →
|
|
135
|
+
#: Slide reading backend. ``"auto"`` tries cucim → vips → openslide → asap.
|
|
107
136
|
#: Explicit choices: ``"cucim"``, ``"openslide"``, ``"vips"``, ``"asap"``.
|
|
108
137
|
backend: str = "auto"
|
|
109
138
|
#: Source-mask reading backend, resolved independently from the *mask* path
|
|
@@ -140,19 +169,23 @@ class PreprocessingConfig:
|
|
|
140
169
|
adaptive_batching: bool = False
|
|
141
170
|
#: Group adjacent tiles into supertile batches for faster I/O.
|
|
142
171
|
use_supertiles: bool = True
|
|
143
|
-
#: JPEG
|
|
144
|
-
|
|
172
|
+
#: JPEG encoder for extracted tile archives — portable ``"pil"`` (default) or
|
|
173
|
+
#: explicitly requested ``"turbojpeg"``.
|
|
174
|
+
jpeg_backend: str = "pil"
|
|
145
175
|
#: Number of CuCIM reader threads.
|
|
146
176
|
num_cucim_workers: int = 4
|
|
147
177
|
#: Skip slides already present in the output directory when ``True``.
|
|
148
178
|
resume: bool = False
|
|
149
|
-
#:
|
|
150
|
-
#: ``downsample``, ``sam2_device``.
|
|
179
|
+
#: Partial override forwarded to hs2p segmentation config. Supported keys:
|
|
180
|
+
#: ``method``, ``downsample``, ``sam2_device``. Omitted keys retain the
|
|
181
|
+
#: standard configuration defaults. See :doc:`preprocessing` for details.
|
|
151
182
|
segmentation: dict[str, Any] = field(default_factory=dict)
|
|
152
|
-
#:
|
|
183
|
+
#: Partial override forwarded to hs2p tile-filtering config. Omitted keys
|
|
184
|
+
#: retain the standard configuration defaults.
|
|
153
185
|
filtering: dict[str, Any] = field(default_factory=dict)
|
|
154
|
-
#:
|
|
155
|
-
#: Keys: ``save_mask_preview``, ``save_tiling_preview``,
|
|
186
|
+
#: Partial override controlling whether hs2p writes mask and tiling preview
|
|
187
|
+
#: images. Keys: ``save_mask_preview``, ``save_tiling_preview``,
|
|
188
|
+
#: ``downsample``. Omitted keys retain the standard configuration defaults.
|
|
156
189
|
preview: dict[str, Any] = field(default_factory=dict)
|
|
157
190
|
#: Annotation-mask vocabulary forwarded to hs2p's sampling resolver. Keys:
|
|
158
191
|
#: ``output_mode``, ``pixel_mapping``, ``colors``, ``min_coverage``. A partial
|
|
@@ -166,6 +199,36 @@ class PreprocessingConfig:
|
|
|
166
199
|
independent_sampling: bool = True
|
|
167
200
|
|
|
168
201
|
def __post_init__(self) -> None:
|
|
202
|
+
filtering_defaults = DEFAULT_PREPROCESSING["filtering"]
|
|
203
|
+
filtering_override = self.filtering
|
|
204
|
+
if self.requested_tile_size_px is not None:
|
|
205
|
+
filtering_defaults = {
|
|
206
|
+
**filtering_defaults,
|
|
207
|
+
"ref_tile_size": int(self.requested_tile_size_px),
|
|
208
|
+
}
|
|
209
|
+
if (
|
|
210
|
+
filtering_override.get("ref_tile_size")
|
|
211
|
+
== _REQUESTED_TILE_SIZE_INTERPOLATION
|
|
212
|
+
):
|
|
213
|
+
filtering_override = {
|
|
214
|
+
**filtering_override,
|
|
215
|
+
"ref_tile_size": int(self.requested_tile_size_px),
|
|
216
|
+
}
|
|
217
|
+
object.__setattr__(
|
|
218
|
+
self,
|
|
219
|
+
"segmentation",
|
|
220
|
+
_deep_merge_dicts(DEFAULT_PREPROCESSING["segmentation"], self.segmentation),
|
|
221
|
+
)
|
|
222
|
+
object.__setattr__(
|
|
223
|
+
self,
|
|
224
|
+
"filtering",
|
|
225
|
+
_deep_merge_dicts(filtering_defaults, filtering_override),
|
|
226
|
+
)
|
|
227
|
+
object.__setattr__(
|
|
228
|
+
self,
|
|
229
|
+
"preview",
|
|
230
|
+
_deep_merge_dicts(DEFAULT_PREPROCESSING["preview"], self.preview),
|
|
231
|
+
)
|
|
169
232
|
# Complete a (possibly partial) masks mapping against the shipped default.
|
|
170
233
|
object.__setattr__(self, "masks", resolve_masks(self.masks))
|
|
171
234
|
|
|
@@ -238,6 +301,7 @@ class ExecutionOptions:
|
|
|
238
301
|
batch_size: int = 32
|
|
239
302
|
#: DataLoader worker count per GPU rank. ``None`` means auto
|
|
240
303
|
#: (capped by CPU / SLURM limit, then split across the resolved GPU count).
|
|
304
|
+
#: Image-only routes safely use zero when auto selection happens after model loading.
|
|
241
305
|
num_workers_per_gpu: int | None = None
|
|
242
306
|
#: Tiling worker count. ``None`` means auto (capped by CPU / SLURM limit).
|
|
243
307
|
num_preprocessing_workers: int | None = None
|
|
@@ -322,6 +386,12 @@ class ExecutionOptions:
|
|
|
322
386
|
return self.num_workers_per_gpu
|
|
323
387
|
return max(1, cpu_worker_limit() // self.num_gpus)
|
|
324
388
|
|
|
389
|
+
def resolved_image_num_workers_per_gpu(self) -> int:
|
|
390
|
+
"""Resolve safe post-model-load image-transform workers for this rank."""
|
|
391
|
+
if self.num_workers_per_gpu is None:
|
|
392
|
+
return 0
|
|
393
|
+
return self.resolved_num_workers_per_gpu()
|
|
394
|
+
|
|
325
395
|
def with_output_dir(self, output_dir: PathLike | None) -> "ExecutionOptions":
|
|
326
396
|
if output_dir is None:
|
|
327
397
|
return self
|
|
@@ -346,7 +416,8 @@ class DenseOptions:
|
|
|
346
416
|
target_size: int
|
|
347
417
|
#: Relative spacing tolerance for pyramid level selection.
|
|
348
418
|
tolerance: float = 0.05
|
|
349
|
-
#: Slide reading backend. ``"auto"`` resolves per slide
|
|
419
|
+
#: Slide reading backend. ``"auto"`` resolves per slide
|
|
420
|
+
#: (cucim → vips → openslide → asap).
|
|
350
421
|
backend: str = "auto"
|
|
351
422
|
#: Padding mode used to pad the tile up to the encoder's patch multiple.
|
|
352
423
|
#: One of ``"reflect"`` / ``"replicate"`` / ``"constant"`` / ``"zero"``.
|
|
@@ -373,26 +444,37 @@ class DenseOptions:
|
|
|
373
444
|
|
|
374
445
|
@dataclass(frozen=True, kw_only=True)
|
|
375
446
|
class DenseImageOptions:
|
|
376
|
-
"""Dense ``(d, gh, gw)``
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
447
|
+
"""Dense ``(d, gh, gw)`` extraction over pre-cropped images.
|
|
448
|
+
|
|
449
|
+
One run has one reader regime. ``.png``, ``.jpg``, and ``.jpeg`` inputs
|
|
450
|
+
(case-insensitive) use Pillow and treat numeric ``spacing_um`` as an assertion about
|
|
451
|
+
unchanged pixels. hs2p-supported WSI formats form the spacing-readable regime: hs2p
|
|
452
|
+
resolves source spacing, backend, pyramid level, tolerance, complete-extent read, and
|
|
453
|
+
area downsampling at the one requested run-level ``spacing_um``. Omitted spacing resolves
|
|
454
|
+
the encoder's single registry default for spacing-readable inputs and requires an
|
|
455
|
+
explicit value when no default is available.
|
|
456
|
+
|
|
457
|
+
``ImageSpec.spacing_at_level_0`` overrides level-0 metadata only for spacing-readable
|
|
458
|
+
sources. ``target_size`` is always a strict post-read declaration, never a fit-to-size
|
|
459
|
+
request: each final pixel array must already be exactly this size. Declaring it up front
|
|
460
|
+
lets the effective encoder input be validated (and variable-input constructor settings
|
|
461
|
+
resolved) before model loading or pixel decoding. Differing final geometries therefore
|
|
462
|
+
require separate runs.
|
|
391
463
|
"""
|
|
392
464
|
|
|
393
465
|
#: Supervision geometry in pixels the dense grid registers to: a square side length, or
|
|
394
466
|
#: an explicit ``(height, width)`` for non-square images.
|
|
395
467
|
target_size: int | tuple[int, int]
|
|
468
|
+
#: Positive, finite run-level spacing in µm/px: asserted for raster pixels and requested
|
|
469
|
+
#: for spacing-readable reads. ``None`` is unknown for raster and resolves the encoder's
|
|
470
|
+
#: single registry default for spacing-readable inputs.
|
|
471
|
+
spacing_um: float | None = None
|
|
472
|
+
#: Relative spacing tolerance used by hs2p for spacing-readable level selection; raster
|
|
473
|
+
#: reads have no tolerance result.
|
|
474
|
+
tolerance: float = 0.05
|
|
475
|
+
#: Requested hs2p backend for spacing-readable inputs. ``"auto"`` is resolved in the
|
|
476
|
+
#: parent; raster images always resolve to Pillow.
|
|
477
|
+
backend: str = "auto"
|
|
396
478
|
#: Padding mode used to pad the image up to the encoder's patch multiple.
|
|
397
479
|
#: One of ``"reflect"`` / ``"replicate"`` / ``"constant"`` / ``"zero"``.
|
|
398
480
|
pad_mode: str = "reflect"
|
|
@@ -432,18 +514,21 @@ class SlideRegions:
|
|
|
432
514
|
|
|
433
515
|
@dataclass(frozen=True, kw_only=True)
|
|
434
516
|
class ImageSpec:
|
|
435
|
-
"""One
|
|
517
|
+
"""One named image source: ``(sample_id, image_path, spacing_at_level_0)``.
|
|
436
518
|
|
|
437
519
|
The input unit of :meth:`Model.embed_images` — the Given-geometry counterpart of
|
|
438
|
-
:class:`SlideRegions`.
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
unique within a run and a valid filename
|
|
442
|
-
|
|
520
|
+
:class:`SlideRegions`. ``spacing_at_level_0`` represents caller metadata for
|
|
521
|
+
spacing-readable image sources. The current raster paths reject a non-null value: a
|
|
522
|
+
pre-cropped PNG/JPEG has no slide pyramid or level-0 read plan to override. ``sample_id``
|
|
523
|
+
is the artifact's whole identity, so it must be unique within a run and a valid filename
|
|
524
|
+
component.
|
|
443
525
|
"""
|
|
444
526
|
|
|
445
527
|
sample_id: str
|
|
446
528
|
image_path: PathLike
|
|
529
|
+
#: Optional caller override for the source's level-0 spacing. Raster image paths reject
|
|
530
|
+
#: non-null overrides because they have no slide pyramid or level-0 read plan.
|
|
531
|
+
spacing_at_level_0: float | None = None
|
|
447
532
|
|
|
448
533
|
|
|
449
534
|
@dataclass(frozen=True, kw_only=True)
|
|
@@ -779,19 +864,31 @@ class Model:
|
|
|
779
864
|
— whole-image, or by sliding the encoder's native field and blending the token grids
|
|
780
865
|
— and written to ``dense_image_embeddings/<sample_id>.pt`` plus a geometry sidecar.
|
|
781
866
|
The run splits its images across all visible GPUs (``execution.num_gpus``);
|
|
782
|
-
``num_gpus=1`` encodes fully in-process. Resume is automatic
|
|
783
|
-
|
|
867
|
+
``num_gpus=1`` encodes fully in-process. Resume is automatic: an image is skipped
|
|
868
|
+
only when its payload exists and its sidecar records the same normalized source
|
|
869
|
+
identity and complete extraction recipe. Returns one
|
|
784
870
|
:class:`~slide2vec.artifacts.DenseImageArtifact` per input image, in input order.
|
|
785
871
|
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
872
|
+
One run uses one reader regime. Raster inputs are exactly ``.png``, ``.jpg``, and
|
|
873
|
+
``.jpeg`` (case-insensitive) and always use Pillow's RGB decoder.
|
|
874
|
+
``dense.spacing_um`` asserts the scale those unchanged pixels already have; ``None``
|
|
875
|
+
records unknown spacing. hs2p-supported WSI inputs are spacing-readable:
|
|
876
|
+
``dense.spacing_um`` requests one physical read scale, or ``None`` resolves the
|
|
877
|
+
encoder's single registry default. The parent resolves each source's metadata,
|
|
878
|
+
concrete backend, native level, tolerance result, and final geometry before resume;
|
|
879
|
+
hs2p reads that complete level and area-downsamples when required, but never
|
|
880
|
+
upsamples. ``ImageSpec.spacing_at_level_0`` overrides source metadata only in this
|
|
881
|
+
spacing-readable regime and is rejected for raster inputs.
|
|
882
|
+
|
|
883
|
+
Everything after reading is shared with the slide path, including the effective encoder input — the padded image
|
|
789
884
|
for a whole-image run, one patch-aligned window for a sliding one — which is declared
|
|
790
885
|
before any image is decoded, so a geometry the encoder cannot accept raises here
|
|
791
886
|
rather than on a torchrun rank's first forward pass.
|
|
792
887
|
|
|
793
|
-
``dense.target_size`` is a declaration, not a
|
|
794
|
-
|
|
888
|
+
``dense.target_size`` is a strict post-read declaration, not a fit-to-size request:
|
|
889
|
+
every final image must already be that size (a non-square ``(h, w)`` is fine).
|
|
890
|
+
Spacing-driven area downsampling establishes physical scale; it never repairs a
|
|
891
|
+
mismatch with the declared geometry.
|
|
795
892
|
"""
|
|
796
893
|
from slide2vec.runtime.dense_image_stage import embed_images_dense
|
|
797
894
|
|
|
@@ -822,8 +919,11 @@ class Model:
|
|
|
822
919
|
heterogeneously sized (2048x1536 beside 96x96) and were never requested, so the
|
|
823
920
|
encoder's shipped transform is the contract and slide2vec *records* the resulting
|
|
824
921
|
encoder input size as run provenance rather than validating it. That also means
|
|
825
|
-
preprocessing runs itemwise
|
|
826
|
-
|
|
922
|
+
preprocessing runs itemwise before stacking — in-process by default, or in spawned
|
|
923
|
+
loader workers when ``num_workers_per_gpu`` is explicit — because differently sized
|
|
924
|
+
images cannot be stacked before they are resized. ``ImageSpec.spacing_at_level_0``
|
|
925
|
+
is rejected here rather than ignored because this path has no slide level-0 read
|
|
926
|
+
plan.
|
|
827
927
|
"""
|
|
828
928
|
from slide2vec.runtime.image_stage import embed_images
|
|
829
929
|
|
|
@@ -1163,32 +1263,27 @@ def _resolve_direct_api_preprocessing(
|
|
|
1163
1263
|
preprocessing: PreprocessingConfig | None,
|
|
1164
1264
|
) -> PreprocessingConfig:
|
|
1165
1265
|
name = model.name
|
|
1166
|
-
defaults = None
|
|
1167
|
-
|
|
1168
|
-
def ensure_defaults() -> tuple[int, float]:
|
|
1169
|
-
nonlocal defaults
|
|
1170
|
-
if defaults is None:
|
|
1171
|
-
defaults = _default_preprocessing_from_registry(name)
|
|
1172
|
-
return defaults
|
|
1173
1266
|
|
|
1174
1267
|
if preprocessing is None:
|
|
1175
|
-
|
|
1268
|
+
default_tile_size_px, default_spacing_um = _default_preprocessing_from_registry(name)
|
|
1176
1269
|
return _resolve_hierarchical_preprocessing(
|
|
1177
1270
|
PreprocessingConfig(
|
|
1178
1271
|
backend="auto",
|
|
1179
|
-
requested_spacing_um=
|
|
1180
|
-
requested_tile_size_px=
|
|
1272
|
+
requested_spacing_um=default_spacing_um,
|
|
1273
|
+
requested_tile_size_px=default_tile_size_px,
|
|
1181
1274
|
)
|
|
1182
1275
|
)
|
|
1183
1276
|
|
|
1184
1277
|
requested_spacing_um = preprocessing.requested_spacing_um
|
|
1185
1278
|
requested_tile_size_px = preprocessing.requested_tile_size_px
|
|
1186
1279
|
if requested_spacing_um is None or requested_tile_size_px is None:
|
|
1187
|
-
|
|
1188
|
-
|
|
1189
|
-
requested_spacing_um
|
|
1190
|
-
|
|
1191
|
-
|
|
1280
|
+
resolved_fields = _resolve_registered_preprocessing_fields(
|
|
1281
|
+
name,
|
|
1282
|
+
requested_spacing_um=requested_spacing_um,
|
|
1283
|
+
requested_tile_size_px=requested_tile_size_px,
|
|
1284
|
+
)
|
|
1285
|
+
requested_spacing_um = float(resolved_fields["spacing_um"])
|
|
1286
|
+
requested_tile_size_px = int(resolved_fields["tile_size_px"])
|
|
1192
1287
|
return _resolve_hierarchical_preprocessing(
|
|
1193
1288
|
replace(
|
|
1194
1289
|
preprocessing,
|
|
@@ -1199,14 +1294,30 @@ def _resolve_direct_api_preprocessing(
|
|
|
1199
1294
|
|
|
1200
1295
|
|
|
1201
1296
|
def _default_preprocessing_from_registry(name: str | None) -> tuple[int, float]:
|
|
1297
|
+
resolved_fields = _resolve_registered_preprocessing_fields(
|
|
1298
|
+
name,
|
|
1299
|
+
requested_spacing_um=None,
|
|
1300
|
+
requested_tile_size_px=None,
|
|
1301
|
+
)
|
|
1302
|
+
return int(resolved_fields["tile_size_px"]), float(resolved_fields["spacing_um"])
|
|
1303
|
+
|
|
1304
|
+
|
|
1305
|
+
def _resolve_registered_preprocessing_fields(
|
|
1306
|
+
name: str | None,
|
|
1307
|
+
*,
|
|
1308
|
+
requested_spacing_um: float | None,
|
|
1309
|
+
requested_tile_size_px: int | None,
|
|
1310
|
+
) -> dict[str, Any]:
|
|
1202
1311
|
if not name or name not in encoder_registry:
|
|
1203
1312
|
raise ValueError(
|
|
1204
1313
|
"Cannot infer preprocessing defaults without a registered model. "
|
|
1205
1314
|
"Pass preprocessing.requested_spacing_um and preprocessing.requested_tile_size_px explicitly."
|
|
1206
1315
|
)
|
|
1207
|
-
|
|
1208
|
-
|
|
1209
|
-
|
|
1316
|
+
return resolve_preprocessing_fields(
|
|
1317
|
+
name,
|
|
1318
|
+
requested_spacing_um=requested_spacing_um,
|
|
1319
|
+
requested_tile_size_px=requested_tile_size_px,
|
|
1320
|
+
)
|
|
1210
1321
|
|
|
1211
1322
|
|
|
1212
1323
|
def _validate_model_config(
|
|
@@ -7,7 +7,6 @@ from uuid import uuid4
|
|
|
7
7
|
|
|
8
8
|
import numpy as np
|
|
9
9
|
import torch
|
|
10
|
-
from hs2p.fileops import is_flattened_annotation
|
|
11
10
|
|
|
12
11
|
from slide2vec.runtime.model_settings import output_torch_dtype
|
|
13
12
|
|
|
@@ -20,6 +19,7 @@ class TileEmbeddingArtifact:
|
|
|
20
19
|
format: str
|
|
21
20
|
feature_dim: int
|
|
22
21
|
num_tiles: int
|
|
22
|
+
annotation: str | None = None
|
|
23
23
|
|
|
24
24
|
@property
|
|
25
25
|
def metadata(self) -> dict[str, Any]:
|
|
@@ -208,15 +208,32 @@ def _write_metadata(path: Path, metadata: dict[str, Any]) -> None:
|
|
|
208
208
|
)
|
|
209
209
|
|
|
210
210
|
|
|
211
|
+
def normalize_artifact_annotation(annotation: str | None) -> str | None:
|
|
212
|
+
"""Return the annotation component used for slide2vec artifact placement.
|
|
213
|
+
|
|
214
|
+
``"merged"`` is a process-list sentinel, not an hs2p 4.4 artifact annotation:
|
|
215
|
+
structural merged output is ``annotation=None`` plus ``output_mode="merged"``.
|
|
216
|
+
Accepting the sentinel here keeps 4.3 process lists reusable without delegating
|
|
217
|
+
process identity to hs2p's artifact-annotation helper.
|
|
218
|
+
"""
|
|
219
|
+
if annotation in (None, "tissue", "merged"):
|
|
220
|
+
return None
|
|
221
|
+
return str(annotation)
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def structural_artifact_annotation(annotation: str | None) -> str | None:
|
|
225
|
+
"""Translate only the structural merged process sentinel to artifact identity."""
|
|
226
|
+
return None if annotation == "merged" else annotation
|
|
227
|
+
|
|
228
|
+
|
|
211
229
|
def tile_embeddings_subdir(annotation: str | None) -> str:
|
|
212
230
|
"""Namespace the ``tile_embeddings`` output dir per annotation class.
|
|
213
231
|
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
path is byte-for-byte unchanged; any real class label gets its own
|
|
217
|
-
``tile_embeddings/<class>`` subdirectory.
|
|
232
|
+
Structural ``None`` and the process sentinels ``"tissue"``/``"merged"`` collapse to
|
|
233
|
+
the flat root; a genuine class gets ``tile_embeddings/<class>``.
|
|
218
234
|
"""
|
|
219
|
-
|
|
235
|
+
annotation = normalize_artifact_annotation(annotation)
|
|
236
|
+
if annotation is None:
|
|
220
237
|
return "tile_embeddings"
|
|
221
238
|
return f"tile_embeddings/{annotation}"
|
|
222
239
|
|
|
@@ -224,19 +241,18 @@ def tile_embeddings_subdir(annotation: str | None) -> str:
|
|
|
224
241
|
def slide_embeddings_subdir(annotation: str | None) -> str:
|
|
225
242
|
"""Namespace the ``slide_embeddings`` output dir per annotation class.
|
|
226
243
|
|
|
227
|
-
|
|
228
|
-
:func:`tile_embeddings_subdir`): ``None`` and the sentinel ``"tissue"`` collapse to the
|
|
229
|
-
flat ``slide_embeddings`` root, so the default tissue-only path is byte-for-byte
|
|
230
|
-
unchanged; any real class label gets its own ``slide_embeddings/<class>`` subdirectory.
|
|
244
|
+
Uses the same structural/process identity rule as :func:`tile_embeddings_subdir`.
|
|
231
245
|
"""
|
|
232
|
-
|
|
246
|
+
annotation = normalize_artifact_annotation(annotation)
|
|
247
|
+
if annotation is None:
|
|
233
248
|
return "slide_embeddings"
|
|
234
249
|
return f"slide_embeddings/{annotation}"
|
|
235
250
|
|
|
236
251
|
|
|
237
252
|
def slide_latents_subdir(annotation: str | None) -> str:
|
|
238
253
|
"""Namespace the ``slide_latents`` output dir per annotation class (mirrors slide embeddings)."""
|
|
239
|
-
|
|
254
|
+
annotation = normalize_artifact_annotation(annotation)
|
|
255
|
+
if annotation is None:
|
|
240
256
|
return "slide_latents"
|
|
241
257
|
return f"slide_latents/{annotation}"
|
|
242
258
|
|
|
@@ -244,13 +260,10 @@ def slide_latents_subdir(annotation: str | None) -> str:
|
|
|
244
260
|
def hierarchical_embeddings_subdir(annotation: str | None) -> str:
|
|
245
261
|
"""Namespace the ``hierarchical_embeddings`` output dir per annotation class.
|
|
246
262
|
|
|
247
|
-
|
|
248
|
-
:func:`tile_embeddings_subdir` and :func:`slide_embeddings_subdir`): ``None`` and the
|
|
249
|
-
sentinel ``"tissue"`` collapse to the flat ``hierarchical_embeddings`` root, so the
|
|
250
|
-
default tissue-only path is byte-for-byte unchanged; any real class label gets its own
|
|
251
|
-
``hierarchical_embeddings/<class>`` subdirectory.
|
|
263
|
+
Uses the same structural/process identity rule as :func:`tile_embeddings_subdir`.
|
|
252
264
|
"""
|
|
253
|
-
|
|
265
|
+
annotation = normalize_artifact_annotation(annotation)
|
|
266
|
+
if annotation is None:
|
|
254
267
|
return "hierarchical_embeddings"
|
|
255
268
|
return f"hierarchical_embeddings/{annotation}"
|
|
256
269
|
|
|
@@ -272,12 +285,10 @@ def _validate_path_component(value: str, *, field: str) -> str:
|
|
|
272
285
|
def dense_embeddings_subdir(annotation: str | None) -> str:
|
|
273
286
|
"""Namespace the ``dense_embeddings`` output dir per annotation class.
|
|
274
287
|
|
|
275
|
-
|
|
276
|
-
:func:`tile_embeddings_subdir` and the other pooled subdir helpers): ``None`` and the
|
|
277
|
-
sentinel ``"tissue"`` collapse to the flat ``dense_embeddings`` root; any real class
|
|
278
|
-
label gets its own ``dense_embeddings/<class>`` subdirectory.
|
|
288
|
+
Uses the same structural/process identity rule as :func:`tile_embeddings_subdir`.
|
|
279
289
|
"""
|
|
280
|
-
|
|
290
|
+
annotation = normalize_artifact_annotation(annotation)
|
|
291
|
+
if annotation is None:
|
|
281
292
|
return "dense_embeddings"
|
|
282
293
|
annotation_component = _validate_path_component(annotation, field="annotation")
|
|
283
294
|
return f"dense_embeddings/{annotation_component}"
|
|
@@ -329,6 +340,7 @@ def write_dense_region(
|
|
|
329
340
|
) -> DenseRegionArtifact:
|
|
330
341
|
"""Persist one ROI's ``(d, gh, gw)`` grid + its geometry sidecar (see
|
|
331
342
|
:func:`_write_dense_grid` for the write order this shares with every dense artifact)."""
|
|
343
|
+
annotation = structural_artifact_annotation(annotation)
|
|
332
344
|
payload_path, metadata_path = region_dense_paths(
|
|
333
345
|
output_dir, sample_id=sample_id, annotation=annotation, x=x, y=y
|
|
334
346
|
)
|
|
@@ -353,7 +365,8 @@ def dense_image_paths(output_dir: str | Path, *, sample_id: str) -> tuple[Path,
|
|
|
353
365
|
``dense_image_embeddings/<sample_id>.pt`` plus ``<sample_id>.meta.json``. Flat, like the
|
|
354
366
|
pooled image layout and unlike the per-slide dense one: a pre-cropped image has no slide
|
|
355
367
|
directory to live under and no ``(x, y)`` to be named by, so the caller's ``sample_id``
|
|
356
|
-
|
|
368
|
+
determines the paths. Resume additionally validates the payload and the complete
|
|
369
|
+
compatibility record in the sidecar.
|
|
357
370
|
"""
|
|
358
371
|
output_root = Path(output_dir).expanduser().resolve()
|
|
359
372
|
sample_component = _validate_path_component(sample_id, field="sample_id")
|
|
@@ -463,6 +476,7 @@ def _build_tile_embedding_metadata(
|
|
|
463
476
|
output_format: str,
|
|
464
477
|
feature_dim: int | None,
|
|
465
478
|
num_tiles: int,
|
|
479
|
+
annotation: str | None,
|
|
466
480
|
metadata: dict[str, Any] | None = None,
|
|
467
481
|
) -> dict[str, Any]:
|
|
468
482
|
tile_metadata = {
|
|
@@ -474,6 +488,7 @@ def _build_tile_embedding_metadata(
|
|
|
474
488
|
}
|
|
475
489
|
if metadata:
|
|
476
490
|
tile_metadata.update(metadata)
|
|
491
|
+
tile_metadata["annotation"] = normalize_artifact_annotation(annotation)
|
|
477
492
|
return tile_metadata
|
|
478
493
|
|
|
479
494
|
|
|
@@ -521,6 +536,7 @@ def write_tile_embeddings(
|
|
|
521
536
|
output_format=output_format,
|
|
522
537
|
feature_dim=int(feature_array.shape[-1]) if feature_array.ndim else 1,
|
|
523
538
|
num_tiles=int(feature_array.shape[0]) if feature_array.ndim else 1,
|
|
539
|
+
annotation=annotation,
|
|
524
540
|
metadata=metadata,
|
|
525
541
|
)
|
|
526
542
|
_write_metadata(metadata_path, tile_metadata)
|
|
@@ -531,6 +547,7 @@ def write_tile_embeddings(
|
|
|
531
547
|
format=output_format,
|
|
532
548
|
feature_dim=tile_metadata["feature_dim"],
|
|
533
549
|
num_tiles=tile_metadata["num_tiles"],
|
|
550
|
+
annotation=tile_metadata["annotation"],
|
|
534
551
|
)
|
|
535
552
|
|
|
536
553
|
|
|
@@ -553,6 +570,7 @@ def write_tile_embedding_metadata(
|
|
|
553
570
|
output_format=output_format,
|
|
554
571
|
feature_dim=feature_dim,
|
|
555
572
|
num_tiles=num_tiles,
|
|
573
|
+
annotation=annotation,
|
|
556
574
|
metadata=metadata,
|
|
557
575
|
)
|
|
558
576
|
_write_metadata(metadata_path, tile_metadata)
|
|
@@ -570,6 +588,7 @@ def write_slide_embeddings(
|
|
|
570
588
|
annotation: str | None = None,
|
|
571
589
|
) -> SlideEmbeddingArtifact:
|
|
572
590
|
output_format = _validate_output_format(output_format)
|
|
591
|
+
annotation = structural_artifact_annotation(annotation)
|
|
573
592
|
artifact_path, metadata_path = _setup_artifact_paths(
|
|
574
593
|
output_dir, slide_embeddings_subdir(annotation), sample_id, output_format
|
|
575
594
|
)
|
|
@@ -657,6 +676,7 @@ def write_hierarchical_embeddings(
|
|
|
657
676
|
annotation: str | None = None,
|
|
658
677
|
) -> HierarchicalEmbeddingArtifact:
|
|
659
678
|
output_format = _validate_output_format(output_format)
|
|
679
|
+
annotation = structural_artifact_annotation(annotation)
|
|
660
680
|
artifact_path, metadata_path = _setup_artifact_paths(
|
|
661
681
|
output_dir, hierarchical_embeddings_subdir(annotation), sample_id, output_format
|
|
662
682
|
)
|