slide2vec 5.4.0__tar.gz → 5.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (141) hide show
  1. {slide2vec-5.4.0 → slide2vec-5.5.0}/PKG-INFO +1 -1
  2. {slide2vec-5.4.0 → slide2vec-5.5.0}/pyproject.toml +2 -2
  3. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/__init__.py +9 -1
  4. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/api.py +273 -17
  5. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/artifacts.py +190 -20
  6. slide2vec-5.5.0/slide2vec/data/dataset.py +142 -0
  7. slide2vec-5.5.0/slide2vec/distributed/dense_image_worker.py +78 -0
  8. slide2vec-5.5.0/slide2vec/distributed/dense_worker.py +80 -0
  9. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/distributed/direct_embed_worker.py +1 -3
  10. slide2vec-5.5.0/slide2vec/distributed/image_worker.py +73 -0
  11. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/distributed/pipeline_worker.py +1 -3
  12. slide2vec-5.5.0/slide2vec/distributed/worker_entry.py +87 -0
  13. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/__init__.py +2 -0
  14. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/gigapath.py +7 -1
  15. slide2vec-5.5.0/slide2vec/encoders/models/isight.py +305 -0
  16. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/prost40m.py +1 -1
  17. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/inference.py +58 -33
  18. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/batching.py +101 -213
  19. slide2vec-5.5.0/slide2vec/runtime/dense_encoder_input.py +119 -0
  20. slide2vec-5.5.0/slide2vec/runtime/dense_image_shard.py +209 -0
  21. slide2vec-5.5.0/slide2vec/runtime/dense_image_stage.py +164 -0
  22. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/dense_regions.py +156 -47
  23. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/dense_shard.py +7 -26
  24. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/dense_stage.py +8 -1
  25. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/distributed_stage.py +3 -1
  26. slide2vec-5.5.0/slide2vec/runtime/effective_encoder_input.py +124 -0
  27. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/embedding_pipeline.py +6 -0
  28. slide2vec-5.5.0/slide2vec/runtime/encoder_input_contract.py +122 -0
  29. slide2vec-5.5.0/slide2vec/runtime/image_shard.py +222 -0
  30. slide2vec-5.5.0/slide2vec/runtime/image_specs.py +70 -0
  31. slide2vec-5.5.0/slide2vec/runtime/image_stage.py +148 -0
  32. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/patient_pipeline.py +2 -0
  33. slide2vec-5.5.0/slide2vec/runtime/pooled_encoder_input.py +85 -0
  34. slide2vec-5.5.0/slide2vec/runtime/preprocessing.py +196 -0
  35. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/serialization.py +48 -1
  36. slide2vec-5.5.0/slide2vec/runtime/sharding.py +34 -0
  37. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec.egg-info/PKG-INFO +1 -1
  38. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec.egg-info/SOURCES.txt +22 -0
  39. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_architecture_runtime_split.py +1 -0
  40. slide2vec-5.5.0/tests/test_dense_encoder_input.py +317 -0
  41. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_dense_extraction.py +1 -1
  42. slide2vec-5.5.0/tests/test_dense_image_shard.py +407 -0
  43. slide2vec-5.5.0/tests/test_dense_image_stage.py +338 -0
  44. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_dense_shard.py +10 -74
  45. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_dense_stage.py +15 -3
  46. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_dense_worker.py +14 -2
  47. slide2vec-5.5.0/tests/test_encoder_input_contract.py +350 -0
  48. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_encoder_registry.py +1 -0
  49. slide2vec-5.5.0/tests/test_image_shard.py +313 -0
  50. slide2vec-5.5.0/tests/test_image_stage.py +306 -0
  51. slide2vec-5.5.0/tests/test_isight.py +239 -0
  52. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_patch_size_metadata.py +33 -6
  53. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_pooled_encoder_input.py +44 -39
  54. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_progress.py +6 -0
  55. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_regression_core.py +1 -0
  56. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_regression_inference.py +67 -12
  57. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_runtime_batching.py +1 -1
  58. slide2vec-5.5.0/tests/test_sharding.py +58 -0
  59. slide2vec-5.4.0/slide2vec/data/dataset.py +0 -47
  60. slide2vec-5.4.0/slide2vec/distributed/dense_worker.py +0 -98
  61. slide2vec-5.4.0/slide2vec/runtime/pooled_encoder_input.py +0 -102
  62. {slide2vec-5.4.0 → slide2vec-5.5.0}/LICENSE +0 -0
  63. {slide2vec-5.4.0 → slide2vec-5.5.0}/README.md +0 -0
  64. {slide2vec-5.4.0 → slide2vec-5.5.0}/setup.cfg +0 -0
  65. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/__main__.py +0 -0
  66. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/cli.py +0 -0
  67. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/configs/__init__.py +0 -0
  68. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/configs/default.yaml +0 -0
  69. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/configs/resources.py +0 -0
  70. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/data/__init__.py +0 -0
  71. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/data/tile_reader.py +0 -0
  72. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/data/tile_store.py +0 -0
  73. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/distributed/__init__.py +0 -0
  74. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/__init__.py +0 -0
  75. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/base.py +0 -0
  76. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/conch.py +0 -0
  77. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/dinov2.py +0 -0
  78. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/genbio.py +0 -0
  79. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/gpfm.py +0 -0
  80. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/hibou.py +0 -0
  81. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/hoptimus.py +0 -0
  82. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/lunit.py +0 -0
  83. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/midnight.py +0 -0
  84. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/moozy/__init__.py +0 -0
  85. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/moozy/blocks.py +0 -0
  86. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/moozy/case.py +0 -0
  87. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/moozy/loading.py +0 -0
  88. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/moozy/slide.py +0 -0
  89. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/moozy/types.py +0 -0
  90. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/mstar.py +0 -0
  91. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/musk.py +0 -0
  92. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/phikon.py +0 -0
  93. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/prism.py +0 -0
  94. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/titan.py +0 -0
  95. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/uni.py +0 -0
  96. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/models/virchow.py +0 -0
  97. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/registry.py +0 -0
  98. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/encoders/validation.py +0 -0
  99. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/progress.py +0 -0
  100. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/__init__.py +0 -0
  101. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/artifacts_collect.py +0 -0
  102. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/cpu_budget.py +0 -0
  103. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/dense_sliding.py +0 -0
  104. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/distributed.py +0 -0
  105. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/embedding.py +0 -0
  106. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/embedding_persist.py +0 -0
  107. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/hierarchical.py +0 -0
  108. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/manifest.py +0 -0
  109. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/model_settings.py +0 -0
  110. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/persist_callbacks.py +0 -0
  111. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/persistence.py +0 -0
  112. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/process_list.py +0 -0
  113. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/progress_bridge.py +0 -0
  114. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/registry.py +0 -0
  115. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/slide_encode.py +0 -0
  116. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/tiling.py +0 -0
  117. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/tiling_pipeline.py +0 -0
  118. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/types.py +0 -0
  119. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/runtime/worker_io.py +0 -0
  120. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/utils/__init__.py +0 -0
  121. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/utils/config.py +0 -0
  122. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/utils/coordinates.py +0 -0
  123. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/utils/log_utils.py +0 -0
  124. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/utils/tiling_io.py +0 -0
  125. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec/utils/utils.py +0 -0
  126. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec.egg-info/dependency_links.txt +0 -0
  127. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec.egg-info/entry_points.txt +0 -0
  128. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec.egg-info/not-zip-safe +0 -0
  129. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec.egg-info/requires.txt +0 -0
  130. {slide2vec-5.4.0 → slide2vec-5.5.0}/slide2vec.egg-info/top_level.txt +0 -0
  131. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_attention_extraction.py +0 -0
  132. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_dense_regions.py +0 -0
  133. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_dense_sliding.py +0 -0
  134. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_dinov2_natimage.py +0 -0
  135. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_gpfm_genbio_heavy.py +0 -0
  136. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_hs2p_package_cutover.py +0 -0
  137. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_output_consistency.py +0 -0
  138. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_pooled_geometry.py +0 -0
  139. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_regression_models.py +0 -0
  140. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_tile_store.py +0 -0
  141. {slide2vec-5.4.0 → slide2vec-5.5.0}/tests/test_tiling_pipeline.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: slide2vec
3
- Version: 5.4.0
3
+ Version: 5.5.0
4
4
  Summary: Embedding of whole slide images with Foundation Models
5
5
  Author-email: Clément Grisi <clement.grisi@radboudumc.nl>
6
6
  License-Expression: Apache-2.0
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "slide2vec"
7
- version = "5.4.0"
7
+ version = "5.5.0"
8
8
  description = "Embedding of whole slide images with Foundation Models"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -167,7 +167,7 @@ no_implicit_reexport = true
167
167
  max-line-length = 160
168
168
 
169
169
  [tool.bumpver]
170
- current_version = "5.4.0"
170
+ current_version = "5.5.0"
171
171
  version_pattern = "MAJOR.MINOR.PATCH"
172
172
  commit = false # We do version bumping in CI, not as a commit
173
173
  tag = false # Git tag already exists — we don't auto-tag
@@ -1,8 +1,10 @@
1
1
  from slide2vec.api import (
2
+ DenseImageOptions,
2
3
  DenseOptions,
3
4
  EmbeddedPatient,
4
5
  EmbeddedSlide,
5
6
  ExecutionOptions,
7
+ ImageSpec,
6
8
  Model,
7
9
  Pipeline,
8
10
  PreprocessingConfig,
@@ -11,14 +13,16 @@ from slide2vec.api import (
11
13
  list_models,
12
14
  )
13
15
  from slide2vec.artifacts import (
16
+ DenseImageArtifact,
14
17
  DenseRegionArtifact,
15
18
  HierarchicalEmbeddingArtifact,
19
+ ImageEmbeddingArtifact,
16
20
  SlideEmbeddingArtifact,
17
21
  TileEmbeddingArtifact,
18
22
  )
19
23
 
20
24
 
21
- __version__ = "5.4.0"
25
+ __version__ = "5.5.0"
22
26
 
23
27
  __all__ = [
24
28
  "Model",
@@ -26,7 +30,9 @@ __all__ = [
26
30
  "Pipeline",
27
31
  "PreprocessingConfig",
28
32
  "DenseOptions",
33
+ "DenseImageOptions",
29
34
  "SlideRegions",
35
+ "ImageSpec",
30
36
  "ExecutionOptions",
31
37
  "RunResult",
32
38
  "EmbeddedPatient",
@@ -35,5 +41,7 @@ __all__ = [
35
41
  "HierarchicalEmbeddingArtifact",
36
42
  "TileEmbeddingArtifact",
37
43
  "DenseRegionArtifact",
44
+ "DenseImageArtifact",
45
+ "ImageEmbeddingArtifact",
38
46
  "__version__",
39
47
  ]
@@ -11,8 +11,10 @@ import torch
11
11
  from hs2p import SlideSpec
12
12
 
13
13
  from slide2vec.artifacts import (
14
+ DenseImageArtifact,
14
15
  DenseRegionArtifact,
15
16
  HierarchicalEmbeddingArtifact,
17
+ ImageEmbeddingArtifact,
16
18
  PatientEmbeddingArtifact,
17
19
  SlideEmbeddingArtifact,
18
20
  TileEmbeddingArtifact,
@@ -29,7 +31,8 @@ from slide2vec.runtime.model_settings import (
29
31
  )
30
32
  from slide2vec.progress import emit_progress
31
33
  from slide2vec.runtime.types import LoadedModel
32
- from slide2vec.runtime.pooled_encoder_input import PooledEncoderInputPlan
34
+ from slide2vec.runtime.effective_encoder_input import format_input_size
35
+ from slide2vec.runtime.encoder_input_contract import EncoderInputContract
33
36
  from slide2vec.utils.utils import cpu_worker_limit, slurm_cpu_limit
34
37
 
35
38
  PathLike = str | Path
@@ -352,6 +355,51 @@ class DenseOptions:
352
355
  image_pad_value: float | None = None
353
356
  #: Encoder field-of-view chunk fed through the backbone per forward. ``None`` (default)
354
357
  #: is one whole-tile forward; a smaller value slides the encoder and blends token grids.
358
+ #: Together with ``target_size`` this fixes the *effective encoder input* — the geometry
359
+ #: handed to ``encode_tiles_dense`` — from which the encoder's variable-input constructor
360
+ #: settings are derived; hence no ``dynamic_img_size`` knob here.
361
+ window_size: int | None = None
362
+ #: Fractional window overlap in ``[0, 1)`` for the sliding path (ignored when
363
+ #: ``window_size is None``).
364
+ overlap: float = 0.0
365
+ #: ``"patch_features"`` (the patch-token grid) or ``"cls_attention"`` (CLS/register
366
+ #: self-attention grid).
367
+ feature_kind: str = "patch_features"
368
+ #: Transformer blocks whose CLS attention is read (``cls_attention`` only).
369
+ attention_blocks: tuple[int, ...] = (-1,)
370
+ #: Include register-token query rows as extra attention channels (``cls_attention`` only).
371
+ attention_include_registers: bool = False
372
+
373
+
374
+ @dataclass(frozen=True, kw_only=True)
375
+ class DenseImageOptions:
376
+ """Dense ``(d, gh, gw)`` grid extraction over pre-cropped images (issue #235).
377
+
378
+ :class:`DenseOptions` minus everything that only a slide has: there is no spacing, no
379
+ tolerance and no reading backend here, because the image *is* the region — it is read
380
+ from disk at the size it was written. What remains is the same supervision geometry and
381
+ the same dense encode knobs, so a run migrating from ROIs to image/mask pairs keeps its
382
+ recipe.
383
+
384
+ ``target_size`` is a **declaration**, not a resize: the dense transform is
385
+ normalization-only, so every image must already be exactly this size and one that is not
386
+ is an error rather than a silent rescale. Declaring it up front is what lets the
387
+ effective encoder input be validated (and the encoder's variable-input constructor
388
+ settings resolved) before a single image is decoded. A run whose images are not all the
389
+ same size is therefore several runs, one per geometry — which is also the only way their
390
+ grids could be batched downstream.
391
+ """
392
+
393
+ #: Supervision geometry in pixels the dense grid registers to: a square side length, or
394
+ #: an explicit ``(height, width)`` for non-square images.
395
+ target_size: int | tuple[int, int]
396
+ #: Padding mode used to pad the image up to the encoder's patch multiple.
397
+ #: One of ``"reflect"`` / ``"replicate"`` / ``"constant"`` / ``"zero"``.
398
+ pad_mode: str = "reflect"
399
+ #: Constant fill value for ``pad_mode in {"constant", "zero"}`` (ignored otherwise).
400
+ image_pad_value: float | None = None
401
+ #: Encoder field-of-view chunk fed through the backbone per forward. ``None`` (default)
402
+ #: is one whole-image forward; a smaller value slides the encoder and blends token grids.
355
403
  window_size: int | None = None
356
404
  #: Fractional window overlap in ``[0, 1)`` for the sliding path (ignored when
357
405
  #: ``window_size is None``).
@@ -382,6 +430,22 @@ class SlideRegions:
382
430
  annotation: str | None = None
383
431
 
384
432
 
433
+ @dataclass(frozen=True, kw_only=True)
434
+ class ImageSpec:
435
+ """One pre-cropped image to embed: ``(sample_id, image_path)``.
436
+
437
+ The input unit of :meth:`Model.embed_images` — the Given-geometry counterpart of
438
+ :class:`SlideRegions`. There is no slide, no coordinate and no spacing here: the caller
439
+ holds an image file it never asked slide2vec to produce (a public patch benchmark
440
+ sample), and names it. ``sample_id`` is the artifact's whole identity, so it must be
441
+ unique within a run and a valid filename component; slide2vec never derives it from the
442
+ path, because two directories can hold the same filename.
443
+ """
444
+
445
+ sample_id: str
446
+ image_path: PathLike
447
+
448
+
385
449
  @dataclass(frozen=True, kw_only=True)
386
450
  class RunResult:
387
451
  """Return value of :meth:`Pipeline.run`."""
@@ -463,8 +527,13 @@ class Model:
463
527
  self.allow_non_recommended_settings = bool(allow_non_recommended_settings)
464
528
  self._output_variant = output_variant
465
529
  self._backend: LoadedModel | None = None
466
- self._pooled_input_plan: PooledEncoderInputPlan | None = None
467
- self._backend_pooled_input_plan: PooledEncoderInputPlan | None = None
530
+ # Unset, deliberately: a Model has no encoder-input contract until a route
531
+ # declares one. There is no initial Given contract, because an initial value is
532
+ # a default by another name — it would silently hand the shipped transform to
533
+ # any route that forgot to declare, which is the confusion this contract exists
534
+ # to delete. ``_load_backend`` refuses to load until this is set.
535
+ self._encoder_input: EncoderInputContract | None = None
536
+ self._backend_encoder_input: EncoderInputContract | None = None
468
537
 
469
538
  @classmethod
470
539
  def from_preset(
@@ -484,11 +553,13 @@ class Model:
484
553
 
485
554
  @property
486
555
  def device(self) -> Any:
487
- return self._load_backend().device
556
+ # Construction fact, not an encode: see _load_backend_without_transform.
557
+ return self._load_backend_without_transform().device
488
558
 
489
559
  @property
490
560
  def feature_dim(self) -> int:
491
- return int(self._load_backend().feature_dim)
561
+ # Construction fact, not an encode: see _load_backend_without_transform.
562
+ return int(self._load_backend_without_transform().feature_dim)
492
563
 
493
564
  def embed_tiles(
494
565
  self,
@@ -678,6 +749,12 @@ class Model:
678
749
  (``execution.num_gpus``); ``num_gpus=1`` encodes fully in-process. Resume is
679
750
  automatic — ROIs whose sidecar already exists are skipped. Returns one
680
751
  :class:`~slide2vec.artifacts.DenseRegionArtifact` per input ROI.
752
+
753
+ The effective encoder input — the padded ROI for a whole-tile run, one
754
+ patch-aligned window for a sliding one — is declared before any region is read, so
755
+ a geometry the encoder cannot accept raises here rather than at the first forward
756
+ pass. Variable-input capable encoders get their registry-declared constructor
757
+ settings applied automatically; there is nothing for the caller to pass.
681
758
  """
682
759
  from slide2vec.runtime.dense_stage import embed_regions_dense
683
760
 
@@ -686,18 +763,99 @@ class Model:
686
763
  with _auto_progress_reporting(output_dir=resolved.output_dir):
687
764
  return embed_regions_dense(self, regions, dense=dense, execution=resolved)
688
765
 
689
- def _prepare_pooled_input(
766
+ def embed_images_dense(
767
+ self,
768
+ images: "Sequence[ImageSpec]",
769
+ *,
770
+ dense: "DenseImageOptions",
771
+ execution: ExecutionOptions | None = None,
772
+ ) -> list[DenseImageArtifact]:
773
+ """Extract + persist a dense ``(d, gh, gw)`` grid per caller-supplied image.
774
+
775
+ The image-sourced counterpart of :meth:`embed_regions_dense`, for consumers whose
776
+ supervision arrives as image/mask pairs rather than as slides (segmentation,
777
+ detection): each :class:`ImageSpec` is decoded, run through the encoder's
778
+ **normalization-only** transform, padded up to the encoder's patch multiple, encoded
779
+ — whole-image, or by sliding the encoder's native field and blending the token grids
780
+ — and written to ``dense_image_embeddings/<sample_id>.pt`` plus a geometry sidecar.
781
+ The run splits its images across all visible GPUs (``execution.num_gpus``);
782
+ ``num_gpus=1`` encodes fully in-process. Resume is automatic — images whose sidecar
783
+ already exists are skipped. Returns one
784
+ :class:`~slide2vec.artifacts.DenseImageArtifact` per input image, in input order.
785
+
786
+ It differs from :meth:`embed_regions_dense` in exactly one respect: there is no
787
+ slide, no coordinate and no spacing→level plan, because the image *is* the region.
788
+ Everything else is shared, including the effective encoder input — the padded image
789
+ for a whole-image run, one patch-aligned window for a sliding one — which is declared
790
+ before any image is decoded, so a geometry the encoder cannot accept raises here
791
+ rather than on a torchrun rank's first forward pass.
792
+
793
+ ``dense.target_size`` is a declaration, not a resize request: dense extraction never
794
+ rescales, so every image must already be that size (a non-square ``(h, w)`` is fine).
795
+ """
796
+ from slide2vec.runtime.dense_image_stage import embed_images_dense
797
+
798
+ resolved = _coerce_execution_options(execution, model=self)
799
+ _require_output_dir_for_persistence(resolved, method_name="Model.embed_images_dense(...)")
800
+ with _auto_progress_reporting(output_dir=resolved.output_dir):
801
+ return embed_images_dense(self, images, dense=dense, execution=resolved)
802
+
803
+ def embed_images(
804
+ self,
805
+ images: "Sequence[ImageSpec]",
806
+ *,
807
+ execution: ExecutionOptions | None = None,
808
+ ) -> list[ImageEmbeddingArtifact]:
809
+ """Embed + persist one embedding per caller-supplied image.
810
+
811
+ The Given-geometry entry point: the caller already holds pre-cropped images — a
812
+ public patch benchmark (BACH, CRC, Gleason, BreakHis, MHIST, PCam), an exported ROI
813
+ set — and slide2vec neither tiles nor reads a slide. Each :class:`ImageSpec` is
814
+ decoded, preprocessed with the encoder's **shipped** transform, encoded, and written
815
+ to ``image_embeddings/<sample_id>.pt`` plus a provenance sidecar. The run splits its
816
+ images across all visible GPUs (``execution.num_gpus``); ``num_gpus=1`` encodes
817
+ fully in-process. Resume is automatic — images whose sidecar already exists are
818
+ skipped. Returns one :class:`~slide2vec.artifacts.ImageEmbeddingArtifact` per input
819
+ image, in input order.
820
+
821
+ Unlike the pooled and dense paths there is no geometry to declare: the images are
822
+ heterogeneously sized (2048x1536 beside 96x96) and were never requested, so the
823
+ encoder's shipped transform is the contract and slide2vec *records* the resulting
824
+ encoder input size as run provenance rather than validating it. That also means
825
+ preprocessing runs itemwise in the loader workers — differently sized images cannot
826
+ be stacked before they are resized.
827
+ """
828
+ from slide2vec.runtime.image_stage import embed_images
829
+
830
+ resolved = _coerce_execution_options(execution, model=self)
831
+ _require_output_dir_for_persistence(resolved, method_name="Model.embed_images(...)")
832
+ with _auto_progress_reporting(output_dir=resolved.output_dir):
833
+ return embed_images(self, images, execution=resolved)
834
+
835
+ def _declare_encoder_input(
690
836
  self,
691
837
  preprocessing: PreprocessingConfig,
692
838
  *,
693
839
  emit_run_info: bool,
694
- ) -> PooledEncoderInputPlan:
695
- plan = PooledEncoderInputPlan.resolve(
840
+ ) -> EncoderInputContract:
841
+ """Declare the pooled encoder input geometry this run requested, or raise.
842
+
843
+ Idempotent: resolving the same preprocessing twice yields an equal contract, so
844
+ every layer that reaches the encoder may declare for itself rather than trust
845
+ the layer above to have done it.
846
+ """
847
+ if preprocessing.requested_tile_size_px is None:
848
+ raise ValueError(
849
+ "requested_tile_size_px must be resolved before declaring the encoder "
850
+ "input geometry; a pooled run reads tiles at a size it requested."
851
+ )
852
+ contract = EncoderInputContract.declared_pooled(
696
853
  self.name,
697
854
  requested_tile_size_px=int(preprocessing.requested_tile_size_px),
698
855
  allow_non_recommended_settings=self.allow_non_recommended_settings,
699
856
  )
700
- self._pooled_input_plan = plan
857
+ self._encoder_input = contract
858
+ plan = contract.plan
701
859
  if emit_run_info and plan.requires_variable_model_input:
702
860
  logging.getLogger("slide2vec").info(
703
861
  "Pooled encoder input for '%s': preset %dpx, requested %dpx, "
@@ -707,24 +865,122 @@ class Model:
707
865
  plan.requested_tile_size_px,
708
866
  plan.expected_encoder_input_size_px,
709
867
  )
710
- return plan
868
+ return contract
869
+
870
+ def _declare_dense_encoder_input(
871
+ self,
872
+ dense: "DenseOptions",
873
+ *,
874
+ emit_run_info: bool,
875
+ ) -> EncoderInputContract:
876
+ """Declare the dense encoder input geometry this run requested, or raise.
877
+
878
+ Dense states a supervision geometry (``target_size``, optional ``window_size``)
879
+ rather than an encoder input; the contract derives the tensor the backbone will
880
+ actually see and validates it exactly as the pooled path's is validated. Like the
881
+ pooled declaration this is idempotent, so each layer that reaches the encoder — the
882
+ parent stage and every torchrun rank — declares for itself.
883
+
884
+ *dense* is a :class:`DenseOptions` (ROIs on a slide) or a :class:`DenseImageOptions`
885
+ (pre-cropped images); only the supervision geometry is read here, and the two state
886
+ it the same way.
887
+ """
888
+ contract = EncoderInputContract.declared_dense(
889
+ self.name,
890
+ target_size_px=dense.target_size,
891
+ window_size=None if dense.window_size is None else int(dense.window_size),
892
+ )
893
+ self._encoder_input = contract
894
+ plan = contract.plan
895
+ if emit_run_info and plan.requires_variable_model_input:
896
+ logging.getLogger("slide2vec").info(
897
+ "Dense encoder input for '%s': native %dpx, effective encoder input %s "
898
+ "(target_size=%s, window_size=%s); enabling variable input size via %s.",
899
+ self.name,
900
+ plan.preset_input_size_px,
901
+ format_input_size(plan.effective_encoder_input_size_px),
902
+ format_input_size(plan.target_size_px),
903
+ plan.window_size_px,
904
+ plan.model_construction_kwargs or "no constructor setting",
905
+ )
906
+ return contract
907
+
908
+ def _declare_given_encoder_input(self, *, emit_run_info: bool) -> EncoderInputContract:
909
+ """Declare that this run's encoder input is whatever the caller handed over.
910
+
911
+ The Given regime's affirmative statement. It is deliberately not the same thing as
912
+ leaving ``_encoder_input`` unset: an absent contract means "this route forgot", and
913
+ the contract refuses to guess between the two. Like the declared variants this is
914
+ idempotent, so the parent stage and every torchrun rank declare for themselves.
915
+ """
916
+ contract = EncoderInputContract.given()
917
+ self._encoder_input = contract
918
+ if emit_run_info:
919
+ logging.getLogger("slide2vec").info(
920
+ "Given encoder input for '%s': using the encoder's shipped preprocessing; "
921
+ "the observed encoder input size is recorded per artifact, not validated.",
922
+ self.name,
923
+ )
924
+ return contract
711
925
 
712
926
  def _load_backend(self) -> LoadedModel:
713
- if (
714
- self._backend is None
715
- or self._backend_pooled_input_plan != self._pooled_input_plan
716
- ):
927
+ """Load the backend under this run's declared encoder-input contract.
928
+
929
+ Every caller that reads ``loaded.transforms`` — i.e. everything that turns
930
+ pixels into features — must come through here, and must therefore have
931
+ declared its geometry first.
932
+ """
933
+ if self._encoder_input is None:
934
+ raise ValueError(
935
+ f"No encoder-input contract has been declared for model '{self.name}'. "
936
+ "A route that encodes pixels must state its geometry before the "
937
+ "backend is loaded: call _declare_encoder_input(preprocessing, ...) "
938
+ "for a pooled run, or _declare_dense_encoder_input(dense, ...) for a "
939
+ "dense one. Callers that never read loaded.transforms use "
940
+ "_load_backend_without_transform() instead."
941
+ )
942
+ return self._load_backend_under(self._encoder_input)
943
+
944
+ def _load_backend_without_transform(self) -> LoadedModel:
945
+ """Load the backend for callers that never read ``loaded.transforms``.
946
+
947
+ Two kinds of caller need the constructed encoder module without ever selecting a
948
+ tile transform: the ``device``/``feature_dim`` properties (pure construction
949
+ facts) and tile→slide/patient aggregation (``encode_slide`` / ``encode_patient``
950
+ consume already-computed features). They cannot observe, let alone encode
951
+ through, the transform the backend happens to carry.
952
+
953
+ Dense extraction is deliberately NOT in this set. It builds its own normalization
954
+ transform and never reads ``loaded.transforms``, but it does need the
955
+ variable-input constructor settings its geometry implies — which is exactly what
956
+ an encoder-input contract carries — so it declares (see
957
+ ``_declare_dense_encoder_input``) and loads through ``_load_backend``.
958
+
959
+ A declared contract is honored when one exists so the cached backend is shared;
960
+ otherwise an explicit Given contract is used for this load only. This never
961
+ assigns ``_encoder_input``: an embed route still has to declare, and
962
+ ``_load_backend`` reloads when the declared contract differs from the one the
963
+ cached backend was built under.
964
+ """
965
+ return self._load_backend_under(
966
+ self._encoder_input
967
+ if self._encoder_input is not None
968
+ else EncoderInputContract.given()
969
+ )
970
+
971
+ def _load_backend_under(self, encoder_input: EncoderInputContract) -> LoadedModel:
972
+ if self._backend is None or self._backend_encoder_input != encoder_input:
717
973
  from slide2vec.inference import load_model
718
974
 
719
975
  emit_progress("model.loading", model_name=self.name)
720
976
  self._backend = load_model(
721
977
  name=self.name,
978
+ encoder_input=encoder_input,
722
979
  device=self._requested_device,
723
980
  output_variant=self._output_variant,
724
981
  allow_non_recommended_settings=self.allow_non_recommended_settings,
725
- pooled_input_plan=self._pooled_input_plan,
726
982
  )
727
- self._backend_pooled_input_plan = self._pooled_input_plan
983
+ self._backend_encoder_input = encoder_input
728
984
  emit_progress("model.ready", model_name=self.name, device=str(self._backend.device))
729
985
  return self._backend
730
986
 
@@ -965,7 +1221,7 @@ def _validate_model_config(
965
1221
  info = encoder_registry.info(name)
966
1222
  if info["level"] != "tile":
967
1223
  raise ValueError("Hierarchical preprocessing is only supported for tile encoders")
968
- model._prepare_pooled_input(preprocessing, emit_run_info=True)
1224
+ model._declare_encoder_input(preprocessing, emit_run_info=True)
969
1225
  # Skip precision validation for CPU execution (fp32 is always valid on CPU).
970
1226
  on_cpu = model._requested_device == "cpu"
971
1227
  precision = None if on_cpu or execution is None else execution.precision