slide2vec 5.5.0__tar.gz → 5.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (150) hide show
  1. {slide2vec-5.5.0 → slide2vec-5.6.0}/PKG-INFO +3 -3
  2. {slide2vec-5.5.0 → slide2vec-5.6.0}/pyproject.toml +6 -5
  3. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/__init__.py +1 -1
  4. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/api.py +172 -61
  5. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/artifacts.py +44 -24
  6. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/configs/default.yaml +3 -3
  7. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/data/dataset.py +20 -5
  8. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/distributed/dense_image_worker.py +9 -1
  9. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/distributed/image_worker.py +1 -1
  10. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/registry.py +30 -0
  11. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/validation.py +8 -1
  12. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/inference.py +13 -23
  13. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/artifacts_collect.py +13 -16
  14. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/batching.py +11 -3
  15. slide2vec-5.6.0/slide2vec/runtime/dense_image_reading.py +260 -0
  16. slide2vec-5.6.0/slide2vec/runtime/dense_image_recipe.py +290 -0
  17. slide2vec-5.6.0/slide2vec/runtime/dense_image_shard.py +341 -0
  18. slide2vec-5.6.0/slide2vec/runtime/dense_image_stage.py +303 -0
  19. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/dense_regions.py +52 -8
  20. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/dense_shard.py +5 -2
  21. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/distributed.py +4 -12
  22. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/embedding.py +6 -4
  23. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/embedding_pipeline.py +10 -20
  24. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/image_shard.py +3 -2
  25. slide2vec-5.6.0/slide2vec/runtime/image_specs.py +218 -0
  26. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/image_stage.py +10 -2
  27. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/persist_callbacks.py +2 -4
  28. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/persistence.py +51 -47
  29. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/serialization.py +22 -1
  30. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/slide_encode.py +5 -0
  31. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/utils/config.py +47 -27
  32. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/utils/tiling_io.py +40 -10
  33. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec.egg-info/PKG-INFO +3 -3
  34. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec.egg-info/SOURCES.txt +7 -0
  35. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec.egg-info/requires.txt +2 -2
  36. slide2vec-5.6.0/tests/test_dense_image_reading.py +408 -0
  37. slide2vec-5.6.0/tests/test_dense_image_resume.py +431 -0
  38. slide2vec-5.6.0/tests/test_dense_image_shard.py +870 -0
  39. slide2vec-5.6.0/tests/test_dense_image_stage.py +1052 -0
  40. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dense_shard.py +36 -0
  41. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dense_stage.py +8 -1
  42. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_hs2p_package_cutover.py +78 -0
  43. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_image_stage.py +51 -2
  44. slide2vec-5.6.0/tests/test_patient_manifest.py +108 -0
  45. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_regression_core.py +453 -12
  46. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_regression_inference.py +276 -25
  47. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_regression_models.py +64 -2
  48. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_runtime_batching.py +20 -0
  49. slide2vec-5.6.0/tests/test_slide_coordinate_preparation.py +263 -0
  50. slide2vec-5.6.0/tests/test_soma_migration.py +557 -0
  51. slide2vec-5.5.0/slide2vec/runtime/dense_image_shard.py +0 -209
  52. slide2vec-5.5.0/slide2vec/runtime/dense_image_stage.py +0 -164
  53. slide2vec-5.5.0/slide2vec/runtime/image_specs.py +0 -70
  54. slide2vec-5.5.0/tests/test_dense_image_shard.py +0 -407
  55. slide2vec-5.5.0/tests/test_dense_image_stage.py +0 -338
  56. {slide2vec-5.5.0 → slide2vec-5.6.0}/LICENSE +0 -0
  57. {slide2vec-5.5.0 → slide2vec-5.6.0}/README.md +0 -0
  58. {slide2vec-5.5.0 → slide2vec-5.6.0}/setup.cfg +0 -0
  59. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/__main__.py +0 -0
  60. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/cli.py +0 -0
  61. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/configs/__init__.py +0 -0
  62. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/configs/resources.py +0 -0
  63. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/data/__init__.py +0 -0
  64. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/data/tile_reader.py +0 -0
  65. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/data/tile_store.py +0 -0
  66. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/distributed/__init__.py +0 -0
  67. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/distributed/dense_worker.py +0 -0
  68. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/distributed/direct_embed_worker.py +0 -0
  69. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/distributed/pipeline_worker.py +0 -0
  70. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/distributed/worker_entry.py +0 -0
  71. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/__init__.py +0 -0
  72. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/base.py +10 -10
  73. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/__init__.py +0 -0
  74. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/conch.py +0 -0
  75. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/dinov2.py +0 -0
  76. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/genbio.py +0 -0
  77. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/gigapath.py +0 -0
  78. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/gpfm.py +0 -0
  79. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/hibou.py +0 -0
  80. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/hoptimus.py +0 -0
  81. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/isight.py +0 -0
  82. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/lunit.py +0 -0
  83. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/midnight.py +0 -0
  84. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/moozy/__init__.py +0 -0
  85. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/moozy/blocks.py +0 -0
  86. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/moozy/case.py +0 -0
  87. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/moozy/loading.py +0 -0
  88. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/moozy/slide.py +0 -0
  89. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/moozy/types.py +0 -0
  90. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/mstar.py +0 -0
  91. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/musk.py +0 -0
  92. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/phikon.py +0 -0
  93. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/prism.py +0 -0
  94. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/prost40m.py +0 -0
  95. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/titan.py +0 -0
  96. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/uni.py +0 -0
  97. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/encoders/models/virchow.py +0 -0
  98. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/progress.py +0 -0
  99. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/__init__.py +0 -0
  100. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/cpu_budget.py +0 -0
  101. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/dense_encoder_input.py +0 -0
  102. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/dense_sliding.py +0 -0
  103. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/dense_stage.py +0 -0
  104. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/distributed_stage.py +0 -0
  105. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/effective_encoder_input.py +0 -0
  106. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/embedding_persist.py +0 -0
  107. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/encoder_input_contract.py +0 -0
  108. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/hierarchical.py +0 -0
  109. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/manifest.py +0 -0
  110. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/model_settings.py +0 -0
  111. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/patient_pipeline.py +0 -0
  112. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/pooled_encoder_input.py +0 -0
  113. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/preprocessing.py +0 -0
  114. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/process_list.py +0 -0
  115. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/progress_bridge.py +0 -0
  116. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/registry.py +0 -0
  117. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/sharding.py +0 -0
  118. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/tiling.py +0 -0
  119. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/tiling_pipeline.py +0 -0
  120. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/types.py +0 -0
  121. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/runtime/worker_io.py +0 -0
  122. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/utils/__init__.py +0 -0
  123. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/utils/coordinates.py +0 -0
  124. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/utils/log_utils.py +0 -0
  125. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec/utils/utils.py +0 -0
  126. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec.egg-info/dependency_links.txt +0 -0
  127. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec.egg-info/entry_points.txt +0 -0
  128. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec.egg-info/not-zip-safe +0 -0
  129. {slide2vec-5.5.0 → slide2vec-5.6.0}/slide2vec.egg-info/top_level.txt +0 -0
  130. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_architecture_runtime_split.py +0 -0
  131. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_attention_extraction.py +0 -0
  132. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dense_encoder_input.py +0 -0
  133. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dense_extraction.py +0 -0
  134. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dense_regions.py +0 -0
  135. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dense_sliding.py +0 -0
  136. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dense_worker.py +0 -0
  137. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_dinov2_natimage.py +0 -0
  138. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_encoder_input_contract.py +0 -0
  139. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_encoder_registry.py +0 -0
  140. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_gpfm_genbio_heavy.py +0 -0
  141. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_image_shard.py +0 -0
  142. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_isight.py +0 -0
  143. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_output_consistency.py +0 -0
  144. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_patch_size_metadata.py +0 -0
  145. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_pooled_encoder_input.py +0 -0
  146. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_pooled_geometry.py +0 -0
  147. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_progress.py +0 -0
  148. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_sharding.py +0 -0
  149. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_tile_store.py +0 -0
  150. {slide2vec-5.5.0 → slide2vec-5.6.0}/tests/test_tiling_pipeline.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: slide2vec
3
- Version: 5.5.0
3
+ Version: 5.6.0
4
4
  Summary: Embedding of whole slide images with Foundation Models
5
5
  Author-email: Clément Grisi <clement.grisi@radboudumc.nl>
6
6
  License-Expression: Apache-2.0
@@ -15,7 +15,7 @@ Classifier: Programming Language :: Python :: 3.13
15
15
  Requires-Python: >=3.10
16
16
  Description-Content-Type: text/markdown
17
17
  License-File: LICENSE
18
- Requires-Dist: hs2p[asap,cucim,openslide,sam2,vips]>=4.3.0
18
+ Requires-Dist: hs2p[asap,cucim,openslide,sam2,vips]>=4.4.0
19
19
  Requires-Dist: omegaconf
20
20
  Requires-Dist: matplotlib
21
21
  Requires-Dist: numpy<2
@@ -65,7 +65,7 @@ Requires-Dist: numpy<2; extra == "fm"
65
65
  Requires-Dist: pandas; extra == "fm"
66
66
  Requires-Dist: pillow; extra == "fm"
67
67
  Requires-Dist: rich; extra == "fm"
68
- Requires-Dist: hs2p[asap,cucim,openslide,sam2,vips]>=4.3.0; extra == "fm"
68
+ Requires-Dist: hs2p[asap,cucim,openslide,sam2,vips]>=4.4.0; extra == "fm"
69
69
  Requires-Dist: wandb; extra == "fm"
70
70
  Requires-Dist: torch<2.8,>=2.3; extra == "fm"
71
71
  Requires-Dist: torchvision>=0.18.0; extra == "fm"
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "slide2vec"
7
- version = "5.5.0"
7
+ version = "5.6.0"
8
8
  description = "Embedding of whole slide images with Foundation Models"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -21,7 +21,7 @@ classifiers = [
21
21
  "Programming Language :: Python :: 3.13",
22
22
  ]
23
23
  dependencies = [
24
- "hs2p[asap,cucim,openslide,sam2,vips]>=4.3.0",
24
+ "hs2p[asap,cucim,openslide,sam2,vips]>=4.4.0",
25
25
  "omegaconf",
26
26
  "matplotlib",
27
27
  "numpy<2",
@@ -88,7 +88,7 @@ fm = [
88
88
  "pandas",
89
89
  "pillow",
90
90
  "rich",
91
- "hs2p[asap,cucim,openslide,sam2,vips]>=4.3.0",
91
+ "hs2p[asap,cucim,openslide,sam2,vips]>=4.4.0",
92
92
  "wandb",
93
93
  "torch>=2.3,<2.8",
94
94
  "torchvision>=0.18.0",
@@ -141,12 +141,13 @@ slide2vec = ["py.typed"]
141
141
  "slide2vec.configs" = ["*.yaml", "models/*.yaml", "preprocessing/*.yaml"]
142
142
 
143
143
  [tool.pytest.ini_options]
144
- addopts = "--cov=slide2vec"
144
+ addopts = "--cov=slide2vec -m 'not gpu_integration'"
145
145
  testpaths = [
146
146
  "tests",
147
147
  ]
148
148
  markers = [
149
149
  "heavy: real-weight foundation-model inference on CPU; minutes per test. Excluded from the PR suite via `-m 'not heavy'`; run on the scheduled/manual heavy workflow (.github/workflows/nightly-heavy.yaml).",
150
+ "gpu_integration: real one-GPU versus multi-GPU parity; requires at least two visible CUDA devices. Run explicitly with `CUDA_VISIBLE_DEVICES=0,1 python -m pytest -m gpu_integration --no-cov`.",
150
151
  ]
151
152
 
152
153
  [tool.mypy]
@@ -167,7 +168,7 @@ no_implicit_reexport = true
167
168
  max-line-length = 160
168
169
 
169
170
  [tool.bumpver]
170
- current_version = "5.5.0"
171
+ current_version = "5.6.0"
171
172
  version_pattern = "MAJOR.MINOR.PATCH"
172
173
  commit = false # We do version bumping in CI, not as a commit
173
174
  tag = false # Git tag already exists — we don't auto-tag
@@ -22,7 +22,7 @@ from slide2vec.artifacts import (
22
22
  )
23
23
 
24
24
 
25
- __version__ = "5.5.0"
25
+ __version__ = "5.6.0"
26
26
 
27
27
  __all__ = [
28
28
  "Model",
@@ -19,9 +19,10 @@ from slide2vec.artifacts import (
19
19
  SlideEmbeddingArtifact,
20
20
  TileEmbeddingArtifact,
21
21
  )
22
+ from slide2vec.configs.resources import load_config
22
23
  from slide2vec.encoders.registry import (
23
24
  encoder_registry,
24
- resolve_preprocessing_defaults,
25
+ resolve_preprocessing_fields,
25
26
  )
26
27
  from slide2vec.encoders.validation import validate_encoder_config
27
28
  from slide2vec.runtime.model_settings import (
@@ -63,14 +64,42 @@ DEFAULT_MASKS: dict[str, Any] = {
63
64
  "min_coverage": {"background": None, "tissue": 0.01},
64
65
  }
65
66
 
67
+ _REQUESTED_TILE_SIZE_INTERPOLATION = "${tiling.params.requested_tile_size_px}"
68
+
69
+
70
+ def _load_default_preprocessing() -> dict[str, dict[str, Any]]:
71
+ """Read the public nested defaults from the package's canonical YAML config."""
72
+ from omegaconf import OmegaConf
73
+
74
+ tiling = load_config("default").tiling
75
+ defaults: dict[str, dict[str, Any]] = {}
76
+ for public_name, config_name in (
77
+ ("segmentation", "seg_params"),
78
+ ("filtering", "filter_params"),
79
+ ("preview", "preview"),
80
+ ):
81
+ section = OmegaConf.to_container(getattr(tiling, config_name), resolve=False)
82
+ if not isinstance(section, dict):
83
+ raise TypeError(f"tiling.{config_name} must be a mapping")
84
+ defaults[public_name] = section
85
+ defaults["preview"]["tissue_contour_color"] = tuple(
86
+ defaults["preview"]["tissue_contour_color"]
87
+ )
88
+ return defaults
89
+
90
+
91
+ #: Complete defaults for the nested public preprocessing sections, loaded from
92
+ #: ``configs/default.yaml`` so Python and YAML entry points share one source.
93
+ DEFAULT_PREPROCESSING = _load_default_preprocessing()
66
94
 
67
- def _deep_merge_masks(base: Mapping[str, Any], override: Mapping[str, Any]) -> dict[str, Any]:
95
+
96
+ def _deep_merge_dicts(base: Mapping[str, Any], override: Mapping[str, Any]) -> dict[str, Any]:
68
97
  """Deep-merge *override* onto a copy of *base* (nested dicts merge key-by-key)."""
69
98
  merged = copy.deepcopy(dict(base))
70
99
  for key, value in override.items():
71
100
  existing = merged.get(key)
72
101
  if isinstance(value, Mapping) and isinstance(existing, dict):
73
- merged[key] = _deep_merge_masks(existing, value)
102
+ merged[key] = _deep_merge_dicts(existing, value)
74
103
  else:
75
104
  merged[key] = copy.deepcopy(value)
76
105
  return merged
@@ -80,7 +109,7 @@ def resolve_masks(masks: Mapping[str, Any] | None) -> dict[str, Any]:
80
109
  """Complete a (possibly partial) ``masks`` mapping by merging it over :data:`DEFAULT_MASKS`."""
81
110
  if not masks:
82
111
  return copy.deepcopy(DEFAULT_MASKS)
83
- return _deep_merge_masks(DEFAULT_MASKS, masks)
112
+ return _deep_merge_dicts(DEFAULT_MASKS, masks)
84
113
 
85
114
 
86
115
  def _masks_to_plain_dict(node: Any) -> dict[str, Any]:
@@ -103,7 +132,7 @@ def _masks_to_plain_dict(node: Any) -> dict[str, Any]:
103
132
  class PreprocessingConfig:
104
133
  """Configuration for slide tiling and preprocessing."""
105
134
 
106
- #: Slide reading backend. ``"auto"`` tries cucim → openslide → vips in order.
135
+ #: Slide reading backend. ``"auto"`` tries cucim → vips → openslide → asap.
107
136
  #: Explicit choices: ``"cucim"``, ``"openslide"``, ``"vips"``, ``"asap"``.
108
137
  backend: str = "auto"
109
138
  #: Source-mask reading backend, resolved independently from the *mask* path
@@ -140,19 +169,23 @@ class PreprocessingConfig:
140
169
  adaptive_batching: bool = False
141
170
  #: Group adjacent tiles into supertile batches for faster I/O.
142
171
  use_supertiles: bool = True
143
- #: JPEG decode library — ``"turbojpeg"`` (default) or ``"pillow"``.
144
- jpeg_backend: str = "turbojpeg"
172
+ #: JPEG encoder for extracted tile archives — portable ``"pil"`` (default) or
173
+ #: explicitly requested ``"turbojpeg"``.
174
+ jpeg_backend: str = "pil"
145
175
  #: Number of CuCIM reader threads.
146
176
  num_cucim_workers: int = 4
147
177
  #: Skip slides already present in the output directory when ``True``.
148
178
  resume: bool = False
149
- #: Forwarded to hs2p segmentation config. Supported keys: ``method``,
150
- #: ``downsample``, ``sam2_device``. See :doc:`preprocessing` for details.
179
+ #: Partial override forwarded to hs2p segmentation config. Supported keys:
180
+ #: ``method``, ``downsample``, ``sam2_device``. Omitted keys retain the
181
+ #: standard configuration defaults. See :doc:`preprocessing` for details.
151
182
  segmentation: dict[str, Any] = field(default_factory=dict)
152
- #: Forwarded to hs2p tile-filtering config.
183
+ #: Partial override forwarded to hs2p tile-filtering config. Omitted keys
184
+ #: retain the standard configuration defaults.
153
185
  filtering: dict[str, Any] = field(default_factory=dict)
154
- #: Controls whether hs2p writes mask and tiling preview images.
155
- #: Keys: ``save_mask_preview``, ``save_tiling_preview``, ``downsample``.
186
+ #: Partial override controlling whether hs2p writes mask and tiling preview
187
+ #: images. Keys: ``save_mask_preview``, ``save_tiling_preview``,
188
+ #: ``downsample``. Omitted keys retain the standard configuration defaults.
156
189
  preview: dict[str, Any] = field(default_factory=dict)
157
190
  #: Annotation-mask vocabulary forwarded to hs2p's sampling resolver. Keys:
158
191
  #: ``output_mode``, ``pixel_mapping``, ``colors``, ``min_coverage``. A partial
@@ -166,6 +199,36 @@ class PreprocessingConfig:
166
199
  independent_sampling: bool = True
167
200
 
168
201
  def __post_init__(self) -> None:
202
+ filtering_defaults = DEFAULT_PREPROCESSING["filtering"]
203
+ filtering_override = self.filtering
204
+ if self.requested_tile_size_px is not None:
205
+ filtering_defaults = {
206
+ **filtering_defaults,
207
+ "ref_tile_size": int(self.requested_tile_size_px),
208
+ }
209
+ if (
210
+ filtering_override.get("ref_tile_size")
211
+ == _REQUESTED_TILE_SIZE_INTERPOLATION
212
+ ):
213
+ filtering_override = {
214
+ **filtering_override,
215
+ "ref_tile_size": int(self.requested_tile_size_px),
216
+ }
217
+ object.__setattr__(
218
+ self,
219
+ "segmentation",
220
+ _deep_merge_dicts(DEFAULT_PREPROCESSING["segmentation"], self.segmentation),
221
+ )
222
+ object.__setattr__(
223
+ self,
224
+ "filtering",
225
+ _deep_merge_dicts(filtering_defaults, filtering_override),
226
+ )
227
+ object.__setattr__(
228
+ self,
229
+ "preview",
230
+ _deep_merge_dicts(DEFAULT_PREPROCESSING["preview"], self.preview),
231
+ )
169
232
  # Complete a (possibly partial) masks mapping against the shipped default.
170
233
  object.__setattr__(self, "masks", resolve_masks(self.masks))
171
234
 
@@ -238,6 +301,7 @@ class ExecutionOptions:
238
301
  batch_size: int = 32
239
302
  #: DataLoader worker count per GPU rank. ``None`` means auto
240
303
  #: (capped by CPU / SLURM limit, then split across the resolved GPU count).
304
+ #: Image-only routes safely use zero when auto selection happens after model loading.
241
305
  num_workers_per_gpu: int | None = None
242
306
  #: Tiling worker count. ``None`` means auto (capped by CPU / SLURM limit).
243
307
  num_preprocessing_workers: int | None = None
@@ -322,6 +386,12 @@ class ExecutionOptions:
322
386
  return self.num_workers_per_gpu
323
387
  return max(1, cpu_worker_limit() // self.num_gpus)
324
388
 
389
+ def resolved_image_num_workers_per_gpu(self) -> int:
390
+ """Resolve safe post-model-load image-transform workers for this rank."""
391
+ if self.num_workers_per_gpu is None:
392
+ return 0
393
+ return self.resolved_num_workers_per_gpu()
394
+
325
395
  def with_output_dir(self, output_dir: PathLike | None) -> "ExecutionOptions":
326
396
  if output_dir is None:
327
397
  return self
@@ -346,7 +416,8 @@ class DenseOptions:
346
416
  target_size: int
347
417
  #: Relative spacing tolerance for pyramid level selection.
348
418
  tolerance: float = 0.05
349
- #: Slide reading backend. ``"auto"`` resolves per slide (cucim → openslide → vips).
419
+ #: Slide reading backend. ``"auto"`` resolves per slide
420
+ #: (cucim → vips → openslide → asap).
350
421
  backend: str = "auto"
351
422
  #: Padding mode used to pad the tile up to the encoder's patch multiple.
352
423
  #: One of ``"reflect"`` / ``"replicate"`` / ``"constant"`` / ``"zero"``.
@@ -373,26 +444,37 @@ class DenseOptions:
373
444
 
374
445
  @dataclass(frozen=True, kw_only=True)
375
446
  class DenseImageOptions:
376
- """Dense ``(d, gh, gw)`` grid extraction over pre-cropped images (issue #235).
377
-
378
- :class:`DenseOptions` minus everything that only a slide has: there is no spacing, no
379
- tolerance and no reading backend here, because the image *is* the region — it is read
380
- from disk at the size it was written. What remains is the same supervision geometry and
381
- the same dense encode knobs, so a run migrating from ROIs to image/mask pairs keeps its
382
- recipe.
383
-
384
- ``target_size`` is a **declaration**, not a resize: the dense transform is
385
- normalization-only, so every image must already be exactly this size and one that is not
386
- is an error rather than a silent rescale. Declaring it up front is what lets the
387
- effective encoder input be validated (and the encoder's variable-input constructor
388
- settings resolved) before a single image is decoded. A run whose images are not all the
389
- same size is therefore several runs, one per geometry — which is also the only way their
390
- grids could be batched downstream.
447
+ """Dense ``(d, gh, gw)`` extraction over pre-cropped images.
448
+
449
+ One run has one reader regime. ``.png``, ``.jpg``, and ``.jpeg`` inputs
450
+ (case-insensitive) use Pillow and treat numeric ``spacing_um`` as an assertion about
451
+ unchanged pixels. hs2p-supported WSI formats form the spacing-readable regime: hs2p
452
+ resolves source spacing, backend, pyramid level, tolerance, complete-extent read, and
453
+ area downsampling at the one requested run-level ``spacing_um``. Omitted spacing resolves
454
+ the encoder's single registry default for spacing-readable inputs and requires an
455
+ explicit value when no default is available.
456
+
457
+ ``ImageSpec.spacing_at_level_0`` overrides level-0 metadata only for spacing-readable
458
+ sources. ``target_size`` is always a strict post-read declaration, never a fit-to-size
459
+ request: each final pixel array must already be exactly this size. Declaring it up front
460
+ lets the effective encoder input be validated (and variable-input constructor settings
461
+ resolved) before model loading or pixel decoding. Differing final geometries therefore
462
+ require separate runs.
391
463
  """
392
464
 
393
465
  #: Supervision geometry in pixels the dense grid registers to: a square side length, or
394
466
  #: an explicit ``(height, width)`` for non-square images.
395
467
  target_size: int | tuple[int, int]
468
+ #: Positive, finite run-level spacing in µm/px: asserted for raster pixels and requested
469
+ #: for spacing-readable reads. ``None`` is unknown for raster and resolves the encoder's
470
+ #: single registry default for spacing-readable inputs.
471
+ spacing_um: float | None = None
472
+ #: Relative spacing tolerance used by hs2p for spacing-readable level selection; raster
473
+ #: reads have no tolerance result.
474
+ tolerance: float = 0.05
475
+ #: Requested hs2p backend for spacing-readable inputs. ``"auto"`` is resolved in the
476
+ #: parent; raster images always resolve to Pillow.
477
+ backend: str = "auto"
396
478
  #: Padding mode used to pad the image up to the encoder's patch multiple.
397
479
  #: One of ``"reflect"`` / ``"replicate"`` / ``"constant"`` / ``"zero"``.
398
480
  pad_mode: str = "reflect"
@@ -432,18 +514,21 @@ class SlideRegions:
432
514
 
433
515
  @dataclass(frozen=True, kw_only=True)
434
516
  class ImageSpec:
435
- """One pre-cropped image to embed: ``(sample_id, image_path)``.
517
+ """One named image source: ``(sample_id, image_path, spacing_at_level_0)``.
436
518
 
437
519
  The input unit of :meth:`Model.embed_images` — the Given-geometry counterpart of
438
- :class:`SlideRegions`. There is no slide, no coordinate and no spacing here: the caller
439
- holds an image file it never asked slide2vec to produce (a public patch benchmark
440
- sample), and names it. ``sample_id`` is the artifact's whole identity, so it must be
441
- unique within a run and a valid filename component; slide2vec never derives it from the
442
- path, because two directories can hold the same filename.
520
+ :class:`SlideRegions`. ``spacing_at_level_0`` represents caller metadata for
521
+ spacing-readable image sources. The current raster paths reject a non-null value: a
522
+ pre-cropped PNG/JPEG has no slide pyramid or level-0 read plan to override. ``sample_id``
523
+ is the artifact's whole identity, so it must be unique within a run and a valid filename
524
+ component.
443
525
  """
444
526
 
445
527
  sample_id: str
446
528
  image_path: PathLike
529
+ #: Optional caller override for the source's level-0 spacing. Raster image paths reject
530
+ #: non-null overrides because they have no slide pyramid or level-0 read plan.
531
+ spacing_at_level_0: float | None = None
447
532
 
448
533
 
449
534
  @dataclass(frozen=True, kw_only=True)
@@ -779,19 +864,31 @@ class Model:
779
864
  — whole-image, or by sliding the encoder's native field and blending the token grids
780
865
  — and written to ``dense_image_embeddings/<sample_id>.pt`` plus a geometry sidecar.
781
866
  The run splits its images across all visible GPUs (``execution.num_gpus``);
782
- ``num_gpus=1`` encodes fully in-process. Resume is automatic — images whose sidecar
783
- already exists are skipped. Returns one
867
+ ``num_gpus=1`` encodes fully in-process. Resume is automatic: an image is skipped
868
+ only when its payload exists and its sidecar records the same normalized source
869
+ identity and complete extraction recipe. Returns one
784
870
  :class:`~slide2vec.artifacts.DenseImageArtifact` per input image, in input order.
785
871
 
786
- It differs from :meth:`embed_regions_dense` in exactly one respect: there is no
787
- slide, no coordinate and no spacing→level plan, because the image *is* the region.
788
- Everything else is shared, including the effective encoder input — the padded image
872
+ One run uses one reader regime. Raster inputs are exactly ``.png``, ``.jpg``, and
873
+ ``.jpeg`` (case-insensitive) and always use Pillow's RGB decoder.
874
+ ``dense.spacing_um`` asserts the scale those unchanged pixels already have; ``None``
875
+ records unknown spacing. hs2p-supported WSI inputs are spacing-readable:
876
+ ``dense.spacing_um`` requests one physical read scale, or ``None`` resolves the
877
+ encoder's single registry default. The parent resolves each source's metadata,
878
+ concrete backend, native level, tolerance result, and final geometry before resume;
879
+ hs2p reads that complete level and area-downsamples when required, but never
880
+ upsamples. ``ImageSpec.spacing_at_level_0`` overrides source metadata only in this
881
+ spacing-readable regime and is rejected for raster inputs.
882
+
883
+ Everything after reading is shared with the slide path, including the effective encoder input — the padded image
789
884
  for a whole-image run, one patch-aligned window for a sliding one — which is declared
790
885
  before any image is decoded, so a geometry the encoder cannot accept raises here
791
886
  rather than on a torchrun rank's first forward pass.
792
887
 
793
- ``dense.target_size`` is a declaration, not a resize request: dense extraction never
794
- rescales, so every image must already be that size (a non-square ``(h, w)`` is fine).
888
+ ``dense.target_size`` is a strict post-read declaration, not a fit-to-size request:
889
+ every final image must already be that size (a non-square ``(h, w)`` is fine).
890
+ Spacing-driven area downsampling establishes physical scale; it never repairs a
891
+ mismatch with the declared geometry.
795
892
  """
796
893
  from slide2vec.runtime.dense_image_stage import embed_images_dense
797
894
 
@@ -822,8 +919,11 @@ class Model:
822
919
  heterogeneously sized (2048x1536 beside 96x96) and were never requested, so the
823
920
  encoder's shipped transform is the contract and slide2vec *records* the resulting
824
921
  encoder input size as run provenance rather than validating it. That also means
825
- preprocessing runs itemwise in the loader workers — differently sized images cannot
826
- be stacked before they are resized.
922
+ preprocessing runs itemwise before stacking — in-process by default, or in spawned
923
+ loader workers when ``num_workers_per_gpu`` is explicit — because differently sized
924
+ images cannot be stacked before they are resized. ``ImageSpec.spacing_at_level_0``
925
+ is rejected here rather than ignored because this path has no slide level-0 read
926
+ plan.
827
927
  """
828
928
  from slide2vec.runtime.image_stage import embed_images
829
929
 
@@ -1163,32 +1263,27 @@ def _resolve_direct_api_preprocessing(
1163
1263
  preprocessing: PreprocessingConfig | None,
1164
1264
  ) -> PreprocessingConfig:
1165
1265
  name = model.name
1166
- defaults = None
1167
-
1168
- def ensure_defaults() -> tuple[int, float]:
1169
- nonlocal defaults
1170
- if defaults is None:
1171
- defaults = _default_preprocessing_from_registry(name)
1172
- return defaults
1173
1266
 
1174
1267
  if preprocessing is None:
1175
- requested_tile_size_px, requested_spacing_um = ensure_defaults()
1268
+ default_tile_size_px, default_spacing_um = _default_preprocessing_from_registry(name)
1176
1269
  return _resolve_hierarchical_preprocessing(
1177
1270
  PreprocessingConfig(
1178
1271
  backend="auto",
1179
- requested_spacing_um=requested_spacing_um,
1180
- requested_tile_size_px=requested_tile_size_px,
1272
+ requested_spacing_um=default_spacing_um,
1273
+ requested_tile_size_px=default_tile_size_px,
1181
1274
  )
1182
1275
  )
1183
1276
 
1184
1277
  requested_spacing_um = preprocessing.requested_spacing_um
1185
1278
  requested_tile_size_px = preprocessing.requested_tile_size_px
1186
1279
  if requested_spacing_um is None or requested_tile_size_px is None:
1187
- default_tile_size_px, default_spacing_um = ensure_defaults()
1188
- if requested_spacing_um is None:
1189
- requested_spacing_um = default_spacing_um
1190
- if requested_tile_size_px is None:
1191
- requested_tile_size_px = default_tile_size_px
1280
+ resolved_fields = _resolve_registered_preprocessing_fields(
1281
+ name,
1282
+ requested_spacing_um=requested_spacing_um,
1283
+ requested_tile_size_px=requested_tile_size_px,
1284
+ )
1285
+ requested_spacing_um = float(resolved_fields["spacing_um"])
1286
+ requested_tile_size_px = int(resolved_fields["tile_size_px"])
1192
1287
  return _resolve_hierarchical_preprocessing(
1193
1288
  replace(
1194
1289
  preprocessing,
@@ -1199,14 +1294,30 @@ def _resolve_direct_api_preprocessing(
1199
1294
 
1200
1295
 
1201
1296
  def _default_preprocessing_from_registry(name: str | None) -> tuple[int, float]:
1297
+ resolved_fields = _resolve_registered_preprocessing_fields(
1298
+ name,
1299
+ requested_spacing_um=None,
1300
+ requested_tile_size_px=None,
1301
+ )
1302
+ return int(resolved_fields["tile_size_px"]), float(resolved_fields["spacing_um"])
1303
+
1304
+
1305
+ def _resolve_registered_preprocessing_fields(
1306
+ name: str | None,
1307
+ *,
1308
+ requested_spacing_um: float | None,
1309
+ requested_tile_size_px: int | None,
1310
+ ) -> dict[str, Any]:
1202
1311
  if not name or name not in encoder_registry:
1203
1312
  raise ValueError(
1204
1313
  "Cannot infer preprocessing defaults without a registered model. "
1205
1314
  "Pass preprocessing.requested_spacing_um and preprocessing.requested_tile_size_px explicitly."
1206
1315
  )
1207
-
1208
- defaults = resolve_preprocessing_defaults(name)
1209
- return int(defaults["tile_size_px"]), float(defaults["spacing_um"])
1316
+ return resolve_preprocessing_fields(
1317
+ name,
1318
+ requested_spacing_um=requested_spacing_um,
1319
+ requested_tile_size_px=requested_tile_size_px,
1320
+ )
1210
1321
 
1211
1322
 
1212
1323
  def _validate_model_config(
@@ -7,7 +7,6 @@ from uuid import uuid4
7
7
 
8
8
  import numpy as np
9
9
  import torch
10
- from hs2p.fileops import is_flattened_annotation
11
10
 
12
11
  from slide2vec.runtime.model_settings import output_torch_dtype
13
12
 
@@ -20,6 +19,7 @@ class TileEmbeddingArtifact:
20
19
  format: str
21
20
  feature_dim: int
22
21
  num_tiles: int
22
+ annotation: str | None = None
23
23
 
24
24
  @property
25
25
  def metadata(self) -> dict[str, Any]:
@@ -208,15 +208,32 @@ def _write_metadata(path: Path, metadata: dict[str, Any]) -> None:
208
208
  )
209
209
 
210
210
 
211
+ def normalize_artifact_annotation(annotation: str | None) -> str | None:
212
+ """Return the annotation component used for slide2vec artifact placement.
213
+
214
+ ``"merged"`` is a process-list sentinel, not an hs2p 4.4 artifact annotation:
215
+ structural merged output is ``annotation=None`` plus ``output_mode="merged"``.
216
+ Accepting the sentinel here keeps 4.3 process lists reusable without delegating
217
+ process identity to hs2p's artifact-annotation helper.
218
+ """
219
+ if annotation in (None, "tissue", "merged"):
220
+ return None
221
+ return str(annotation)
222
+
223
+
224
+ def structural_artifact_annotation(annotation: str | None) -> str | None:
225
+ """Translate only the structural merged process sentinel to artifact identity."""
226
+ return None if annotation == "merged" else annotation
227
+
228
+
211
229
  def tile_embeddings_subdir(annotation: str | None) -> str:
212
230
  """Namespace the ``tile_embeddings`` output dir per annotation class.
213
231
 
214
- Reuses hs2p's flatten rule (the single source of truth): ``None`` and the sentinel
215
- ``"tissue"`` collapse to the flat ``tile_embeddings`` root, so the default tissue-only
216
- path is byte-for-byte unchanged; any real class label gets its own
217
- ``tile_embeddings/<class>`` subdirectory.
232
+ Structural ``None`` and the process sentinels ``"tissue"``/``"merged"`` collapse to
233
+ the flat root; a genuine class gets ``tile_embeddings/<class>``.
218
234
  """
219
- if is_flattened_annotation(annotation):
235
+ annotation = normalize_artifact_annotation(annotation)
236
+ if annotation is None:
220
237
  return "tile_embeddings"
221
238
  return f"tile_embeddings/{annotation}"
222
239
 
@@ -224,19 +241,18 @@ def tile_embeddings_subdir(annotation: str | None) -> str:
224
241
  def slide_embeddings_subdir(annotation: str | None) -> str:
225
242
  """Namespace the ``slide_embeddings`` output dir per annotation class.
226
243
 
227
- Reuses hs2p's flatten rule (the single source of truth, shared with
228
- :func:`tile_embeddings_subdir`): ``None`` and the sentinel ``"tissue"`` collapse to the
229
- flat ``slide_embeddings`` root, so the default tissue-only path is byte-for-byte
230
- unchanged; any real class label gets its own ``slide_embeddings/<class>`` subdirectory.
244
+ Uses the same structural/process identity rule as :func:`tile_embeddings_subdir`.
231
245
  """
232
- if is_flattened_annotation(annotation):
246
+ annotation = normalize_artifact_annotation(annotation)
247
+ if annotation is None:
233
248
  return "slide_embeddings"
234
249
  return f"slide_embeddings/{annotation}"
235
250
 
236
251
 
237
252
  def slide_latents_subdir(annotation: str | None) -> str:
238
253
  """Namespace the ``slide_latents`` output dir per annotation class (mirrors slide embeddings)."""
239
- if is_flattened_annotation(annotation):
254
+ annotation = normalize_artifact_annotation(annotation)
255
+ if annotation is None:
240
256
  return "slide_latents"
241
257
  return f"slide_latents/{annotation}"
242
258
 
@@ -244,13 +260,10 @@ def slide_latents_subdir(annotation: str | None) -> str:
244
260
  def hierarchical_embeddings_subdir(annotation: str | None) -> str:
245
261
  """Namespace the ``hierarchical_embeddings`` output dir per annotation class.
246
262
 
247
- Reuses hs2p's flatten rule (the single source of truth, shared with
248
- :func:`tile_embeddings_subdir` and :func:`slide_embeddings_subdir`): ``None`` and the
249
- sentinel ``"tissue"`` collapse to the flat ``hierarchical_embeddings`` root, so the
250
- default tissue-only path is byte-for-byte unchanged; any real class label gets its own
251
- ``hierarchical_embeddings/<class>`` subdirectory.
263
+ Uses the same structural/process identity rule as :func:`tile_embeddings_subdir`.
252
264
  """
253
- if is_flattened_annotation(annotation):
265
+ annotation = normalize_artifact_annotation(annotation)
266
+ if annotation is None:
254
267
  return "hierarchical_embeddings"
255
268
  return f"hierarchical_embeddings/{annotation}"
256
269
 
@@ -272,12 +285,10 @@ def _validate_path_component(value: str, *, field: str) -> str:
272
285
  def dense_embeddings_subdir(annotation: str | None) -> str:
273
286
  """Namespace the ``dense_embeddings`` output dir per annotation class.
274
287
 
275
- Reuses hs2p's flatten rule (the single source of truth, shared with
276
- :func:`tile_embeddings_subdir` and the other pooled subdir helpers): ``None`` and the
277
- sentinel ``"tissue"`` collapse to the flat ``dense_embeddings`` root; any real class
278
- label gets its own ``dense_embeddings/<class>`` subdirectory.
288
+ Uses the same structural/process identity rule as :func:`tile_embeddings_subdir`.
279
289
  """
280
- if is_flattened_annotation(annotation):
290
+ annotation = normalize_artifact_annotation(annotation)
291
+ if annotation is None:
281
292
  return "dense_embeddings"
282
293
  annotation_component = _validate_path_component(annotation, field="annotation")
283
294
  return f"dense_embeddings/{annotation_component}"
@@ -329,6 +340,7 @@ def write_dense_region(
329
340
  ) -> DenseRegionArtifact:
330
341
  """Persist one ROI's ``(d, gh, gw)`` grid + its geometry sidecar (see
331
342
  :func:`_write_dense_grid` for the write order this shares with every dense artifact)."""
343
+ annotation = structural_artifact_annotation(annotation)
332
344
  payload_path, metadata_path = region_dense_paths(
333
345
  output_dir, sample_id=sample_id, annotation=annotation, x=x, y=y
334
346
  )
@@ -353,7 +365,8 @@ def dense_image_paths(output_dir: str | Path, *, sample_id: str) -> tuple[Path,
353
365
  ``dense_image_embeddings/<sample_id>.pt`` plus ``<sample_id>.meta.json``. Flat, like the
354
366
  pooled image layout and unlike the per-slide dense one: a pre-cropped image has no slide
355
367
  directory to live under and no ``(x, y)`` to be named by, so the caller's ``sample_id``
356
- is the whole identity and the resume check needs nothing else.
368
+ determines the paths. Resume additionally validates the payload and the complete
369
+ compatibility record in the sidecar.
357
370
  """
358
371
  output_root = Path(output_dir).expanduser().resolve()
359
372
  sample_component = _validate_path_component(sample_id, field="sample_id")
@@ -463,6 +476,7 @@ def _build_tile_embedding_metadata(
463
476
  output_format: str,
464
477
  feature_dim: int | None,
465
478
  num_tiles: int,
479
+ annotation: str | None,
466
480
  metadata: dict[str, Any] | None = None,
467
481
  ) -> dict[str, Any]:
468
482
  tile_metadata = {
@@ -474,6 +488,7 @@ def _build_tile_embedding_metadata(
474
488
  }
475
489
  if metadata:
476
490
  tile_metadata.update(metadata)
491
+ tile_metadata["annotation"] = normalize_artifact_annotation(annotation)
477
492
  return tile_metadata
478
493
 
479
494
 
@@ -521,6 +536,7 @@ def write_tile_embeddings(
521
536
  output_format=output_format,
522
537
  feature_dim=int(feature_array.shape[-1]) if feature_array.ndim else 1,
523
538
  num_tiles=int(feature_array.shape[0]) if feature_array.ndim else 1,
539
+ annotation=annotation,
524
540
  metadata=metadata,
525
541
  )
526
542
  _write_metadata(metadata_path, tile_metadata)
@@ -531,6 +547,7 @@ def write_tile_embeddings(
531
547
  format=output_format,
532
548
  feature_dim=tile_metadata["feature_dim"],
533
549
  num_tiles=tile_metadata["num_tiles"],
550
+ annotation=tile_metadata["annotation"],
534
551
  )
535
552
 
536
553
 
@@ -553,6 +570,7 @@ def write_tile_embedding_metadata(
553
570
  output_format=output_format,
554
571
  feature_dim=feature_dim,
555
572
  num_tiles=num_tiles,
573
+ annotation=annotation,
556
574
  metadata=metadata,
557
575
  )
558
576
  _write_metadata(metadata_path, tile_metadata)
@@ -570,6 +588,7 @@ def write_slide_embeddings(
570
588
  annotation: str | None = None,
571
589
  ) -> SlideEmbeddingArtifact:
572
590
  output_format = _validate_output_format(output_format)
591
+ annotation = structural_artifact_annotation(annotation)
573
592
  artifact_path, metadata_path = _setup_artifact_paths(
574
593
  output_dir, slide_embeddings_subdir(annotation), sample_id, output_format
575
594
  )
@@ -657,6 +676,7 @@ def write_hierarchical_embeddings(
657
676
  annotation: str | None = None,
658
677
  ) -> HierarchicalEmbeddingArtifact:
659
678
  output_format = _validate_output_format(output_format)
679
+ annotation = structural_artifact_annotation(annotation)
660
680
  artifact_path, metadata_path = _setup_artifact_paths(
661
681
  output_dir, hierarchical_embeddings_subdir(annotation), sample_id, output_format
662
682
  )