patchworks 2.6.4__tar.gz → 2.6.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. {patchworks-2.6.4 → patchworks-2.6.6}/PKG-INFO +1 -1
  2. {patchworks-2.6.4 → patchworks-2.6.6}/docs/guide/label_relations.md +15 -1
  3. {patchworks-2.6.4 → patchworks-2.6.6}/docs/guide/snakemake.md +16 -5
  4. {patchworks-2.6.4 → patchworks-2.6.6}/tests/test_run_multi.py +131 -11
  5. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/scripts/relate.py +35 -0
  6. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/scripts/run_multi.py +119 -15
  7. {patchworks-2.6.4 → patchworks-2.6.6}/.github/workflows/docs.yml +0 -0
  8. {patchworks-2.6.4 → patchworks-2.6.6}/.github/workflows/lint.yml +0 -0
  9. {patchworks-2.6.4 → patchworks-2.6.6}/.github/workflows/release.yml +0 -0
  10. {patchworks-2.6.4 → patchworks-2.6.6}/.gitignore +0 -0
  11. {patchworks-2.6.4 → patchworks-2.6.6}/.markdownlint-cli2.yaml +0 -0
  12. {patchworks-2.6.4 → patchworks-2.6.6}/LICENSE +0 -0
  13. {patchworks-2.6.4 → patchworks-2.6.6}/README.md +0 -0
  14. {patchworks-2.6.4 → patchworks-2.6.6}/cliff.toml +0 -0
  15. {patchworks-2.6.4 → patchworks-2.6.6}/docs/api/chunks.md +0 -0
  16. {patchworks-2.6.4 → patchworks-2.6.6}/docs/api/cluster.md +0 -0
  17. {patchworks-2.6.4 → patchworks-2.6.6}/docs/api/io.md +0 -0
  18. {patchworks-2.6.4 → patchworks-2.6.6}/docs/api/merge_tile_labels.md +0 -0
  19. {patchworks-2.6.4 → patchworks-2.6.6}/docs/api/plugins/cellpose.md +0 -0
  20. {patchworks-2.6.4 → patchworks-2.6.6}/docs/api/plugins/dog.md +0 -0
  21. {patchworks-2.6.4 → patchworks-2.6.6}/docs/api/plugins/napari.md +0 -0
  22. {patchworks-2.6.4 → patchworks-2.6.6}/docs/api/plugins/ome_zarr.md +0 -0
  23. {patchworks-2.6.4 → patchworks-2.6.6}/docs/api/postprocess.md +0 -0
  24. {patchworks-2.6.4 → patchworks-2.6.6}/docs/api/relabel.md +0 -0
  25. {patchworks-2.6.4 → patchworks-2.6.6}/docs/api/tile_process.md +0 -0
  26. {patchworks-2.6.4 → patchworks-2.6.6}/docs/assets/logo.png +0 -0
  27. {patchworks-2.6.4 → patchworks-2.6.6}/docs/examples/cellpose_2d.md +0 -0
  28. {patchworks-2.6.4 → patchworks-2.6.6}/docs/examples/cellpose_2d.py +0 -0
  29. {patchworks-2.6.4 → patchworks-2.6.6}/docs/examples/cellpose_3d.md +0 -0
  30. {patchworks-2.6.4 → patchworks-2.6.6}/docs/examples/cellpose_3d.py +0 -0
  31. {patchworks-2.6.4 → patchworks-2.6.6}/docs/examples/custom.md +0 -0
  32. {patchworks-2.6.4 → patchworks-2.6.6}/docs/examples/custom_method.py +0 -0
  33. {patchworks-2.6.4 → patchworks-2.6.6}/docs/examples/dog.md +0 -0
  34. {patchworks-2.6.4 → patchworks-2.6.6}/docs/examples/dog.py +0 -0
  35. {patchworks-2.6.4 → patchworks-2.6.6}/docs/examples/standalone_merge.md +0 -0
  36. {patchworks-2.6.4 → patchworks-2.6.6}/docs/examples/stardist.md +0 -0
  37. {patchworks-2.6.4 → patchworks-2.6.6}/docs/examples/stardist_2d.py +0 -0
  38. {patchworks-2.6.4 → patchworks-2.6.6}/docs/getting_started.md +0 -0
  39. {patchworks-2.6.4 → patchworks-2.6.6}/docs/guide/custom_segmentation.md +0 -0
  40. {patchworks-2.6.4 → patchworks-2.6.6}/docs/guide/gpu_distributed.md +0 -0
  41. {patchworks-2.6.4 → patchworks-2.6.6}/docs/guide/measurements.md +0 -0
  42. {patchworks-2.6.4 → patchworks-2.6.6}/docs/guide/merging.md +0 -0
  43. {patchworks-2.6.4 → patchworks-2.6.6}/docs/guide/ome_zarr_napari.md +0 -0
  44. {patchworks-2.6.4 → patchworks-2.6.6}/docs/guide/performance.md +0 -0
  45. {patchworks-2.6.4 → patchworks-2.6.6}/docs/guide/pitfalls.md +0 -0
  46. {patchworks-2.6.4 → patchworks-2.6.6}/docs/guide/skip_empty.md +0 -0
  47. {patchworks-2.6.4 → patchworks-2.6.6}/docs/guide/tiling.md +0 -0
  48. {patchworks-2.6.4 → patchworks-2.6.6}/docs/index.md +0 -0
  49. {patchworks-2.6.4 → patchworks-2.6.6}/mkdocs.yml +0 -0
  50. {patchworks-2.6.4 → patchworks-2.6.6}/pyproject.toml +0 -0
  51. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/__init__.py +0 -0
  52. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/_chunks.py +0 -0
  53. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/_cluster.py +0 -0
  54. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/_core.py +0 -0
  55. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/_distributed.py +0 -0
  56. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/_gpu.py +0 -0
  57. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/_io.py +0 -0
  58. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/_merge.py +0 -0
  59. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/_notify.py +0 -0
  60. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/_occupancy.py +0 -0
  61. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/_postprocess.py +0 -0
  62. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/_progress.py +0 -0
  63. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/_relabel.py +0 -0
  64. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/_relations.py +0 -0
  65. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/plugins/__init__.py +0 -0
  66. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/plugins/cellpose.py +0 -0
  67. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/plugins/dog.py +0 -0
  68. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/plugins/napari.py +0 -0
  69. {patchworks-2.6.4 → patchworks-2.6.6}/src/patchworks/plugins/ome_zarr.py +0 -0
  70. {patchworks-2.6.4 → patchworks-2.6.6}/tests/test_allocation.py +0 -0
  71. {patchworks-2.6.4 → patchworks-2.6.6}/tests/test_core.py +0 -0
  72. {patchworks-2.6.4 → patchworks-2.6.6}/tests/test_distributed.py +0 -0
  73. {patchworks-2.6.4 → patchworks-2.6.6}/tests/test_dog.py +0 -0
  74. {patchworks-2.6.4 → patchworks-2.6.6}/tests/test_gpu.py +0 -0
  75. {patchworks-2.6.4 → patchworks-2.6.6}/tests/test_napari.py +0 -0
  76. {patchworks-2.6.4 → patchworks-2.6.6}/tests/test_notify.py +0 -0
  77. {patchworks-2.6.4 → patchworks-2.6.6}/tests/test_occupancy.py +0 -0
  78. {patchworks-2.6.4 → patchworks-2.6.6}/tests/test_ome_zarr.py +0 -0
  79. {patchworks-2.6.4 → patchworks-2.6.6}/tests/test_postprocess.py +0 -0
  80. {patchworks-2.6.4 → patchworks-2.6.6}/tests/test_progress.py +0 -0
  81. {patchworks-2.6.4 → patchworks-2.6.6}/tests/test_relations.py +0 -0
  82. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/README.md +0 -0
  83. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/Snakefile +0 -0
  84. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/config/common.yaml +0 -0
  85. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/config/config.yaml +0 -0
  86. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/config/config_cilia.yaml +0 -0
  87. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/config/config_cyto.yaml +0 -0
  88. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/config/config_nuclei.yaml +0 -0
  89. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/config/multi.yaml +0 -0
  90. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/pixi.toml +0 -0
  91. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/profile/slurm/config.yaml +0 -0
  92. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/rules/common.smk +0 -0
  93. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/rules/convert.smk +0 -0
  94. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/rules/merge.smk +0 -0
  95. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/rules/segment.smk +0 -0
  96. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/scripts/_pw.py +0 -0
  97. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/scripts/build_occupancy.py +0 -0
  98. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/scripts/convert.py +0 -0
  99. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/scripts/fetch_model.py +0 -0
  100. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/scripts/merge.py +0 -0
  101. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/scripts/prepare_tiles.py +0 -0
  102. {patchworks-2.6.4 → patchworks-2.6.6}/workflow/scripts/segment_tile.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: patchworks
3
- Version: 2.6.4
3
+ Version: 2.6.6
4
4
  Summary: Tiled processing of arbitrarily large images with globally consistent labels
5
5
  Project-URL: Homepage, https://github.com/imcf/patchworks
6
6
  Project-URL: Issues, https://github.com/imcf/patchworks/issues
@@ -6,7 +6,21 @@ the other — by streaming both arrays chunk by chunk, so it scales to
6
6
  hundreds of thousands of objects without loading anything fully into RAM.
7
7
 
8
8
  Both label arrays must share the exact same chunk layout — same
9
- `tile_shape`/pyramid `level` when they were produced.
9
+ `tile_shape`/pyramid `level` when they were produced. `label_relations()`
10
+ raises rather than silently doing something slower if they don't; rechunk
11
+ one side to match first (same shape, different chunking is a normal dask
12
+ op — some extra I/O reading across misaligned source chunks, not a
13
+ correctness issue):
14
+
15
+ ```python
16
+ if nuclei.chunks != cells.chunks:
17
+ cells = cells.rechunk(nuclei.chunks)
18
+ ```
19
+
20
+ (The cluster workflow's `run_multi.py`/`relate.py` does this automatically
21
+ — see below — so two configs segmented at different `tile_shape` still
22
+ relate correctly; you only need to do it by hand when calling
23
+ `label_relations()` directly.)
10
24
 
11
25
  ```python
12
26
  import dask.array as da
@@ -398,11 +398,22 @@ results/image.zarr/labels/nuclei_labels/
398
398
  results/image.zarr/labels/cyto_labels/
399
399
  ```
400
400
 
401
- !!! tip "Keep `tile_shape` (and `level`) identical across configs"
402
- Different segmentations of the same image can use different `channel` and
403
- `cellpose:` settings freely, but keep `tile_shape`/`level` the same across
404
- configs — the label arrays then share the exact same chunk layout, which
405
- [`label_relations()`](label_relations.md) requires.
401
+ !!! tip "Keep `level` identical across configs; `tile_shape` doesn't have to match anymore"
402
+ Different segmentations of the same image can use different `channel`
403
+ and `cellpose:` settings freely, but keep `level` the same across
404
+ configs so the label arrays cover the same voxel grid.
405
+
406
+ `tile_shape` itself no longer has to match. Two configs naturally
407
+ produce different on-disk chunk layouts when `tile_shape: "auto"`
408
+ charges a `nuclei_channel` config for a smaller tile than a
409
+ single-channel one, or when one config was segmented before the other's
410
+ settings changed, or a cheaper method (e.g. a DoG detector) sized its
411
+ own tile differently — `relate.py` detects a chunk mismatch per pair and
412
+ rechunks the finer side to match before relating, so this is no longer
413
+ something you need to plan around. (`run_multi.py` *also* auto-pins
414
+ every config under `"auto"` to one shared, smallest-computed tile
415
+ up front — mainly useful to keep segmentation itself GPU-efficient
416
+ across configs, not required for the relate step to work.)
406
417
 
407
418
  See [Relating labels across segmentations](label_relations.md) for what
408
419
  `label_relations()` returns and how to save it yourself — the cluster
@@ -8,6 +8,8 @@ sys.path.insert(
8
8
  0, str(Path(__file__).resolve().parents[1] / "workflow" / "scripts")
9
9
  )
10
10
 
11
+ import numpy as np # noqa: E402
12
+ import openpyxl # noqa: E402
11
13
  import pytest # noqa: E402
12
14
  import yaml # noqa: E402
13
15
 
@@ -186,28 +188,146 @@ def test_relate_script_has_the_real_bookkeeping():
186
188
  assert "openpyxl" in src
187
189
 
188
190
 
189
- def test_auto_tile_shape_with_a_lone_nuclei_channel_is_refused():
190
- """Matching `tile_shape` *values* are not enough when one config is 2-ch.
191
+ def test_relate_rechunks_mismatched_label_arrays(tmp_path):
192
+ """A chunk-layout mismatch must be rechunked away, not require a re-run.
191
193
 
192
- "auto" == "auto" passes the plain equality check, but the sizer charges
193
- per channel, so the nuclei_channel config gets a smaller tile. The label
194
- groups then disagree on chunk layout and label_relations raises -- after
195
- every segmentation has already run, which is the expensive way to find out.
194
+ Two configs are free to have segmented at different tile_shape (one
195
+ published before the other's config changed, or a cheaper method sized
196
+ its own tile differently) -- label_relations() itself refuses mismatched
197
+ chunks by design, but that only means the caller has to rechunk one side
198
+ first, not that the whole segmentation needs redoing.
199
+ """
200
+ import zarr
201
+
202
+ from relate import run_relations
203
+
204
+ image_store = str(tmp_path / "image.zarr")
205
+
206
+ # a: labels 1 and 2, split at x=5. b: a single label 10 covering all of
207
+ # a's label 1 and none of label 2 -- built with a *different* chunking.
208
+ a_data = np.zeros((1, 10), dtype=np.int32)
209
+ a_data[0, :5] = 1
210
+ a_data[0, 5:] = 2
211
+ b_data = np.zeros((1, 10), dtype=np.int32)
212
+ b_data[0, :5] = 10
213
+
214
+ root = zarr.open_group(image_store, mode="w")
215
+ labels = root.require_group("labels")
216
+ a_grp = labels.require_group("nuclei_labels")
217
+ a_arr = a_grp.create_array(
218
+ name="0", shape=a_data.shape, chunks=(1, 2), dtype=np.int32
219
+ )
220
+ a_arr[:] = a_data
221
+ a_grp.attrs["sequential_labels"] = True
222
+ a_grp.attrs["n_objects"] = 2
223
+
224
+ b_grp = labels.require_group("cyto_labels")
225
+ b_arr = b_grp.create_array(
226
+ name="0", shape=b_data.shape, chunks=(1, 5), dtype=np.int32
227
+ )
228
+ b_arr[:] = b_data
229
+ b_grp.attrs["sequential_labels"] = True
230
+ b_grp.attrs["n_objects"] = 1
231
+
232
+ out_dir = tmp_path / "work"
233
+ out_dir.mkdir()
234
+ run_relations(
235
+ str(out_dir),
236
+ image_store,
237
+ [{"a": "nuclei_labels", "b": "cyto_labels", "output": "rel.xlsx"}],
238
+ )
239
+
240
+ wb = openpyxl.load_workbook(out_dir / "rel.xlsx")
241
+ rows = {
242
+ row[0]: (row[1], row[2], row[3])
243
+ for row in wb["nuclei_labels"].iter_rows(min_row=2, values_only=True)
244
+ }
245
+ assert rows[1] == (10, 5, 1.0) # label 1 fully inside b's label 10
246
+ assert rows[2] == (None, 0, 0) # label 2 touches nothing in b
247
+
248
+
249
+ def test_mixed_nuclei_channel_auto_passes_validation():
250
+ """A channel-count mismatch under `tile_shape: "auto"` is no longer
251
+
252
+ refused at validation time -- it's resolved automatically instead (see
253
+ `_resolve_shared_tile_shape`), which needs the converted image's real
254
+ shape/dtype and so can only run after phase A, not from
255
+ `_validate_configs()`. This used to `sys.exit` here; asserting that
256
+ would now be testing the wrong layer.
196
257
  """
197
258
  paths = [Path("a.yaml"), Path("b.yaml")]
198
259
  base = {"work_dir": "/w", "tile_shape": "auto", "level": 0}
199
260
 
200
- bad = [
261
+ mixed = [
201
262
  {**base, "label_name": "a", "channel": 0, "nuclei_channel": 1},
202
263
  {**base, "label_name": "b", "channel": 2},
203
264
  ]
204
- with pytest.raises(SystemExit):
205
- _validate_configs(paths, bad)
265
+ assert _validate_configs(paths, mixed) == "/w"
206
266
 
207
267
  # Same pair with one explicit shape is fine: both get that tile.
208
- pinned = [{**c, "tile_shape": [16, 512, 512]} for c in bad]
268
+ pinned = [{**c, "tile_shape": [16, 512, 512]} for c in mixed]
209
269
  assert _validate_configs(paths, pinned) == "/w"
210
270
 
211
271
  # And "auto" is fine when every config carries the same channel count.
212
- both = [{**bad[0]}, {**bad[1], "nuclei_channel": 3}]
272
+ both = [{**mixed[0]}, {**mixed[1], "nuclei_channel": 3}]
213
273
  assert _validate_configs(paths, both) == "/w"
274
+
275
+
276
+ def test_resolve_shared_tile_shape_pins_the_smallest_candidate(
277
+ tmp_path, monkeypatch
278
+ ):
279
+ """The shared tile must be the tightest of every config's own budget.
280
+
281
+ A larger tile than some config's own sizer output would ask that config
282
+ for more memory than its settings were judged to need -- only the
283
+ smallest candidate is safe for every config at once.
284
+ """
285
+ import run_multi
286
+
287
+ class _FakeImage:
288
+ shape = (10, 100, 100)
289
+ dtype = "uint16"
290
+
291
+ calls = []
292
+
293
+ def _fake_load_ome_zarr(store, *, channel, level):
294
+ calls.append((store, channel, level))
295
+ return _FakeImage()
296
+
297
+ # One tile per config, matched up by call order (channel 0 then 1).
298
+ fake_tiles = [(8, 64, 64), (4, 32, 32)]
299
+
300
+ def _fake_sizer_cellpose(shape, dtype, **kwargs):
301
+ return fake_tiles[len(calls) - 1]
302
+
303
+ monkeypatch.setattr(
304
+ "patchworks.load_ome_zarr", _fake_load_ome_zarr, raising=False
305
+ )
306
+ monkeypatch.setattr(
307
+ "patchworks.auto_tile_shape_cellpose",
308
+ _fake_sizer_cellpose,
309
+ raising=False,
310
+ )
311
+
312
+ seg_cfgs = [
313
+ {
314
+ "channel": 0,
315
+ "level": 0,
316
+ "method": "cellpose",
317
+ "cellpose": {"do_3D": True, "gpu": True},
318
+ },
319
+ {
320
+ "channel": 1,
321
+ "level": 0,
322
+ "nuclei_channel": 2,
323
+ "method": "cellpose",
324
+ "cellpose": {"do_3D": True, "gpu": True},
325
+ },
326
+ ]
327
+ out = run_multi._resolve_shared_tile_shape(
328
+ seg_cfgs, "/w/image.zarr", str(tmp_path)
329
+ )
330
+ assert out == tmp_path / ".multi_tile_shape.generated.yaml"
331
+ written = yaml.safe_load(out.read_text())
332
+ assert written == {"tile_shape": [4, 32, 32]} # the smaller candidate
333
+ assert len(calls) == 2
@@ -16,9 +16,15 @@ from __future__ import annotations
16
16
 
17
17
  import argparse
18
18
  import json
19
+ import math
19
20
  from pathlib import Path
20
21
 
21
22
 
23
+ def _num_chunks(arr: "da.Array") -> int: # noqa: F821 - dask imported lazily by callers
24
+ """Total chunk count of a dask array, for picking the coarser side to rechunk to."""
25
+ return math.prod(len(c) for c in arr.chunks)
26
+
27
+
22
28
  def _label_ids(image_store: str, name: str) -> list[int]:
23
29
  """Ids present in a label image, without scanning the volume.
24
30
 
@@ -72,6 +78,35 @@ def run_relations(
72
78
  print(f"[relate] relating {a_name} -> {b_name} …", flush=True)
73
79
  a = da.from_zarr(image_store, component=f"labels/{a_name}/0")
74
80
  b = da.from_zarr(image_store, component=f"labels/{b_name}/0")
81
+
82
+ # label_relations() requires matching chunk layouts (it walks both
83
+ # arrays block-by-block at the same index) but two configs are free
84
+ # to have segmented at different tile_shape -- e.g. one already
85
+ # published before the other's config changed, or a cheaper method
86
+ # naturally sized its tile differently. Same shape, different
87
+ # chunking is a normal dask op (extra I/O reading across misaligned
88
+ # source chunks, not a correctness issue), so rechunk the finer side
89
+ # to the coarser one here rather than require identical tile_shape
90
+ # across every config up front.
91
+ if a.chunks != b.chunks:
92
+ a_n, b_n = _num_chunks(a), _num_chunks(b)
93
+ if a_n <= b_n:
94
+ print(
95
+ f"[relate] {a_name} chunks {a.chunks} != {b_name} "
96
+ f"chunks {b.chunks}; rechunking {b_name} to match "
97
+ f"{a_name} (fewer chunks)",
98
+ flush=True,
99
+ )
100
+ b = b.rechunk(a.chunks)
101
+ else:
102
+ print(
103
+ f"[relate] {a_name} chunks {a.chunks} != {b_name} "
104
+ f"chunks {b.chunks}; rechunking {a_name} to match "
105
+ f"{b_name} (fewer chunks)",
106
+ flush=True,
107
+ )
108
+ a = a.rechunk(b.chunks)
109
+
75
110
  table = label_relations(a, b)
76
111
 
77
112
  # label_relations() only returns a-objects that touch a b-object.
@@ -54,6 +54,7 @@ def _snakemake_cmd(
54
54
  extra: list[str] | None = None,
55
55
  jobname_prefix: str | None = None,
56
56
  common: Path | None = None,
57
+ extra_configfiles: list[Path] | None = None,
57
58
  ) -> list[str]:
58
59
  """Build one snakemake invocation.
59
60
 
@@ -66,10 +67,17 @@ def _snakemake_cmd(
66
67
  settings every config shares -- the input, the work_dir, everything
67
68
  ``convert`` reads -- live in one file and the per-config file carries only
68
69
  what actually differs.
70
+
71
+ *extra_configfiles*, when given, are appended after *configfile* and so
72
+ win over both it and *common* -- used to pin a driver-computed value
73
+ (e.g. a resolved ``tile_shape``) across every config without editing any
74
+ config file on disk.
69
75
  """
70
76
  configfiles = [str(configfile.resolve())]
71
77
  if common is not None:
72
78
  configfiles.insert(0, str(common.resolve()))
79
+ if extra_configfiles:
80
+ configfiles += [str(p.resolve()) for p in extra_configfiles]
73
81
  cmd = [
74
82
  "snakemake",
75
83
  "-s",
@@ -234,21 +242,12 @@ def _validate_configs(paths: list[Path], cfgs: list[dict]) -> str:
234
242
 
235
243
  # `tile_shape: "auto"` is identical as a *value* across configs while
236
244
  # producing different tiles: the sizer charges per channel, so a config
237
- # with nuclei_channel gets a smaller one. The label groups then disagree on
238
- # chunk layout and label_relations raises -- after every segmentation has
239
- # run. Matching values are not enough here, so check the inputs that feed
240
- # the sizer instead.
241
- if {repr(cfg.get("tile_shape", "auto")) for cfg in cfgs} == {repr("auto")}:
242
- if len({cfg.get("nuclei_channel") is not None for cfg in cfgs}) != 1:
243
- problems.append(
244
- 'tile_shape: "auto" sizes a nuclei_channel config smaller '
245
- "(a tile carries two channels), so the label groups would "
246
- "not share a chunk layout and label_relations would fail "
247
- f"after every segmentation had run; got "
248
- f"{_spread('nuclei_channel')}. Set one explicit tile_shape in "
249
- "the file `common:` points at, sized for the two-channel "
250
- "config."
251
- )
245
+ # with nuclei_channel gets a smaller one. The label groups would then
246
+ # disagree on chunk layout and label_relations would raise -- after every
247
+ # segmentation had run. That used to be a hard error asking for a manual
248
+ # explicit tile_shape; main() now resolves and pins one automatically
249
+ # (see _resolve_shared_tile_shape), once the converted image exists to
250
+ # size against, so there is nothing to check here anymore.
252
251
 
253
252
  # Phase A converts once, from the first config. Anything `convert` reads
254
253
  # out of a later config is therefore silently ignored -- someone setting
@@ -292,6 +291,92 @@ def _validate_configs(paths: list[Path], cfgs: list[dict]) -> str:
292
291
  return work_dirs.pop()
293
292
 
294
293
 
294
+ def _resolve_shared_tile_shape(
295
+ seg_cfgs: list[dict], image_store: str, work_dir: str
296
+ ) -> Path:
297
+ """Auto-size ``tile_shape`` once, shared across every config.
298
+
299
+ ``tile_shape: "auto"`` resolves differently per config when
300
+ ``nuclei_channel`` differs -- the sizer charges per channel, so a
301
+ two-channel config gets a smaller tile. Left alone, the label groups
302
+ would end up with different chunk layouts and ``label_relations`` would
303
+ raise, after every segmentation had already run.
304
+
305
+ Computes what each config's own settings (channel count, ``do_3D``,
306
+ ``diameter``, GPU budget) would actually produce, then pins every config
307
+ to the *smallest* of them by voxel count -- the most memory-constrained
308
+ case, and safe for every config since a smaller tile only ever asks for
309
+ less memory than that config's own budget allows, never more. Configs
310
+ can therefore use different ``do_3D``/``diameter``/channel-count
311
+ settings and still end up with one shared, valid ``tile_shape``.
312
+
313
+ Only called once the converted image exists (the sizer needs its real
314
+ shape/dtype), so this runs from ``main()`` after phase A, not from
315
+ ``_validate_configs()``.
316
+
317
+ Returns
318
+ -------
319
+ Path
320
+ A generated one-key YAML file (``tile_shape: [...]``), meant to be
321
+ passed as an ``extra_configfiles`` entry to ``_snakemake_cmd`` so it
322
+ overrides every config's own (or common's) ``tile_shape`` value.
323
+ """
324
+ from functools import partial
325
+
326
+ import numpy as np
327
+ from patchworks import (
328
+ auto_tile_shape,
329
+ auto_tile_shape_cellpose,
330
+ load_ome_zarr,
331
+ )
332
+
333
+ candidates = []
334
+ for cfg in seg_cfgs:
335
+ image = load_ome_zarr(
336
+ image_store, channel=cfg["channel"], level=int(cfg.get("level", 0))
337
+ )
338
+ gpu_gb = cfg.get("gpu_memory_gb")
339
+ gpu_bytes = int(gpu_gb * 1024**3) if gpu_gb else None
340
+ n_channels = 2 if cfg.get("nuclei_channel") is not None else 1
341
+ method = cfg.get("method", "cellpose")
342
+ if method == "cellpose":
343
+ cp = cfg["cellpose"]
344
+ sizer = partial(
345
+ auto_tile_shape_cellpose,
346
+ do_3D=cp.get("do_3D", False),
347
+ use_gpu=cp.get("gpu", True),
348
+ diameter=cp.get("diameter"),
349
+ gpu_memory=gpu_bytes,
350
+ n_channels=n_channels,
351
+ )
352
+ else:
353
+ sizer = partial(
354
+ auto_tile_shape,
355
+ use_gpu=gpu_bytes is not None,
356
+ gpu_memory=gpu_bytes,
357
+ n_channels=n_channels,
358
+ )
359
+ candidates.append(tuple(int(x) for x in sizer(image.shape, image.dtype)))
360
+
361
+ tile_shape = min(candidates, key=lambda t: int(np.prod(t)))
362
+ print(
363
+ f'[run_multi] tile_shape: "auto" resolves differently across '
364
+ f"configs (nuclei_channel differs); pinning every config to the "
365
+ f"smallest computed tile {list(tile_shape)} so the label arrays "
366
+ f"share a chunk layout. Candidates were {[list(c) for c in candidates]}.",
367
+ flush=True,
368
+ )
369
+
370
+ override_path = Path(work_dir) / ".multi_tile_shape.generated.yaml"
371
+ override_path.write_text(
372
+ "# Generated by run_multi.py -- pins tile_shape across configs so\n"
373
+ "# label_relations sees matching chunk layouts. Safe to delete; it\n"
374
+ "# is regenerated on the next multi-config run.\n"
375
+ f"tile_shape: {list(tile_shape)}\n"
376
+ )
377
+ return override_path
378
+
379
+
295
380
  def _resolve(workflow_dir: Path, path_str: str) -> Path:
296
381
  path = Path(path_str)
297
382
  return path if path.is_absolute() else workflow_dir / path
@@ -453,6 +538,24 @@ def main() -> None:
453
538
  )
454
539
  sys.exit(rc)
455
540
 
541
+ # tile_shape: "auto" resolves differently per config when nuclei_channel
542
+ # differs (the sizer charges per channel) -- pin every config to one
543
+ # shared, computed value so the label arrays end up with matching chunk
544
+ # layouts. Needs the just-converted image, so this can only happen here,
545
+ # not in _validate_configs(). Dry runs never reach a real image.zarr.
546
+ tile_override = None
547
+ if not args.dry_run and {
548
+ cfg.get("tile_shape", "auto") for cfg in seg_cfgs
549
+ } == {"auto"}:
550
+ channel_counts = {
551
+ 2 if cfg.get("nuclei_channel") is not None else 1
552
+ for cfg in seg_cfgs
553
+ }
554
+ if len(channel_counts) > 1:
555
+ tile_override = _resolve_shared_tile_shape(
556
+ seg_cfgs, image_store, work_dir
557
+ )
558
+
456
559
  # Phase B: the segmentations touch disjoint files under
457
560
  # work_dir/<label_name>/, so run them together and let the GPU partition
458
561
  # stay busy instead of idling through each config's prepare and merge.
@@ -470,6 +573,7 @@ def main() -> None:
470
573
  # Names the config in squeue, so concurrent runs are tellable apart.
471
574
  jobname_prefix=slurm_jobname_prefix(cfg["label_name"]),
472
575
  common=common_path,
576
+ extra_configfiles=[tile_override] if tile_override else None,
473
577
  )
474
578
  print(f"[run_multi] $ {' '.join(cmd)}", flush=True)
475
579
  procs.append((cfg_path.name, subprocess.Popen(cmd, cwd=workflow_dir)))
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes