patchworks 2.6.10__tar.gz → 2.6.11__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. {patchworks-2.6.10 → patchworks-2.6.11}/PKG-INFO +1 -1
  2. {patchworks-2.6.10 → patchworks-2.6.11}/docs/examples/dog.md +15 -0
  3. {patchworks-2.6.10 → patchworks-2.6.11}/docs/guide/snakemake.md +31 -13
  4. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/plugins/dog.py +20 -15
  5. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_dog.py +18 -6
  6. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_run_multi.py +165 -0
  7. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/scripts/relate.py +39 -0
  8. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/scripts/run_multi.py +125 -30
  9. {patchworks-2.6.10 → patchworks-2.6.11}/.github/workflows/docs.yml +0 -0
  10. {patchworks-2.6.10 → patchworks-2.6.11}/.github/workflows/lint.yml +0 -0
  11. {patchworks-2.6.10 → patchworks-2.6.11}/.github/workflows/release.yml +0 -0
  12. {patchworks-2.6.10 → patchworks-2.6.11}/.gitignore +0 -0
  13. {patchworks-2.6.10 → patchworks-2.6.11}/.markdownlint-cli2.yaml +0 -0
  14. {patchworks-2.6.10 → patchworks-2.6.11}/LICENSE +0 -0
  15. {patchworks-2.6.10 → patchworks-2.6.11}/README.md +0 -0
  16. {patchworks-2.6.10 → patchworks-2.6.11}/cliff.toml +0 -0
  17. {patchworks-2.6.10 → patchworks-2.6.11}/docs/api/chunks.md +0 -0
  18. {patchworks-2.6.10 → patchworks-2.6.11}/docs/api/cluster.md +0 -0
  19. {patchworks-2.6.10 → patchworks-2.6.11}/docs/api/io.md +0 -0
  20. {patchworks-2.6.10 → patchworks-2.6.11}/docs/api/merge_tile_labels.md +0 -0
  21. {patchworks-2.6.10 → patchworks-2.6.11}/docs/api/plugins/cellpose.md +0 -0
  22. {patchworks-2.6.10 → patchworks-2.6.11}/docs/api/plugins/dog.md +0 -0
  23. {patchworks-2.6.10 → patchworks-2.6.11}/docs/api/plugins/napari.md +0 -0
  24. {patchworks-2.6.10 → patchworks-2.6.11}/docs/api/plugins/ome_zarr.md +0 -0
  25. {patchworks-2.6.10 → patchworks-2.6.11}/docs/api/postprocess.md +0 -0
  26. {patchworks-2.6.10 → patchworks-2.6.11}/docs/api/relabel.md +0 -0
  27. {patchworks-2.6.10 → patchworks-2.6.11}/docs/api/tile_process.md +0 -0
  28. {patchworks-2.6.10 → patchworks-2.6.11}/docs/api/volume_filter.md +0 -0
  29. {patchworks-2.6.10 → patchworks-2.6.11}/docs/assets/logo.png +0 -0
  30. {patchworks-2.6.10 → patchworks-2.6.11}/docs/examples/cellpose_2d.md +0 -0
  31. {patchworks-2.6.10 → patchworks-2.6.11}/docs/examples/cellpose_2d.py +0 -0
  32. {patchworks-2.6.10 → patchworks-2.6.11}/docs/examples/cellpose_3d.md +0 -0
  33. {patchworks-2.6.10 → patchworks-2.6.11}/docs/examples/cellpose_3d.py +0 -0
  34. {patchworks-2.6.10 → patchworks-2.6.11}/docs/examples/custom.md +0 -0
  35. {patchworks-2.6.10 → patchworks-2.6.11}/docs/examples/custom_method.py +0 -0
  36. {patchworks-2.6.10 → patchworks-2.6.11}/docs/examples/dog.py +0 -0
  37. {patchworks-2.6.10 → patchworks-2.6.11}/docs/examples/standalone_merge.md +0 -0
  38. {patchworks-2.6.10 → patchworks-2.6.11}/docs/examples/stardist.md +0 -0
  39. {patchworks-2.6.10 → patchworks-2.6.11}/docs/examples/stardist_2d.py +0 -0
  40. {patchworks-2.6.10 → patchworks-2.6.11}/docs/getting_started.md +0 -0
  41. {patchworks-2.6.10 → patchworks-2.6.11}/docs/guide/custom_segmentation.md +0 -0
  42. {patchworks-2.6.10 → patchworks-2.6.11}/docs/guide/gpu_distributed.md +0 -0
  43. {patchworks-2.6.10 → patchworks-2.6.11}/docs/guide/label_relations.md +0 -0
  44. {patchworks-2.6.10 → patchworks-2.6.11}/docs/guide/measurements.md +0 -0
  45. {patchworks-2.6.10 → patchworks-2.6.11}/docs/guide/merging.md +0 -0
  46. {patchworks-2.6.10 → patchworks-2.6.11}/docs/guide/ome_zarr_napari.md +0 -0
  47. {patchworks-2.6.10 → patchworks-2.6.11}/docs/guide/performance.md +0 -0
  48. {patchworks-2.6.10 → patchworks-2.6.11}/docs/guide/pitfalls.md +0 -0
  49. {patchworks-2.6.10 → patchworks-2.6.11}/docs/guide/skip_empty.md +0 -0
  50. {patchworks-2.6.10 → patchworks-2.6.11}/docs/guide/tiling.md +0 -0
  51. {patchworks-2.6.10 → patchworks-2.6.11}/docs/index.md +0 -0
  52. {patchworks-2.6.10 → patchworks-2.6.11}/mkdocs.yml +0 -0
  53. {patchworks-2.6.10 → patchworks-2.6.11}/pyproject.toml +0 -0
  54. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/__init__.py +0 -0
  55. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/_chunks.py +0 -0
  56. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/_cluster.py +0 -0
  57. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/_core.py +0 -0
  58. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/_distributed.py +0 -0
  59. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/_gpu.py +0 -0
  60. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/_io.py +0 -0
  61. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/_merge.py +0 -0
  62. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/_notify.py +0 -0
  63. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/_occupancy.py +0 -0
  64. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/_postprocess.py +0 -0
  65. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/_progress.py +0 -0
  66. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/_relabel.py +0 -0
  67. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/_relations.py +0 -0
  68. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/_volume_filter.py +0 -0
  69. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/plugins/__init__.py +0 -0
  70. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/plugins/cellpose.py +0 -0
  71. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/plugins/napari.py +0 -0
  72. {patchworks-2.6.10 → patchworks-2.6.11}/src/patchworks/plugins/ome_zarr.py +0 -0
  73. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_allocation.py +0 -0
  74. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_cellpose.py +0 -0
  75. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_core.py +0 -0
  76. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_distributed.py +0 -0
  77. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_gpu.py +0 -0
  78. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_napari.py +0 -0
  79. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_notify.py +0 -0
  80. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_occupancy.py +0 -0
  81. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_ome_zarr.py +0 -0
  82. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_postprocess.py +0 -0
  83. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_progress.py +0 -0
  84. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_pw.py +0 -0
  85. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_relations.py +0 -0
  86. {patchworks-2.6.10 → patchworks-2.6.11}/tests/test_volume_filter.py +0 -0
  87. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/README.md +0 -0
  88. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/Snakefile +0 -0
  89. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/config/common.yaml +0 -0
  90. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/config/config.yaml +0 -0
  91. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/config/config_cilia.yaml +0 -0
  92. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/config/config_cyto.yaml +0 -0
  93. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/config/config_nuclei.yaml +0 -0
  94. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/config/multi.yaml +0 -0
  95. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/pixi.toml +0 -0
  96. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/profile/slurm/config.yaml +0 -0
  97. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/rules/common.smk +0 -0
  98. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/rules/convert.smk +0 -0
  99. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/rules/merge.smk +0 -0
  100. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/rules/segment.smk +0 -0
  101. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/scripts/_pw.py +0 -0
  102. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/scripts/build_occupancy.py +0 -0
  103. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/scripts/convert.py +0 -0
  104. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/scripts/fetch_model.py +0 -0
  105. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/scripts/merge.py +0 -0
  106. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/scripts/prepare_tiles.py +0 -0
  107. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/scripts/segment_tile.py +0 -0
  108. {patchworks-2.6.10 → patchworks-2.6.11}/workflow/scripts/view.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: patchworks
3
- Version: 2.6.10
3
+ Version: 2.6.11
4
4
  Summary: Tiled processing of arbitrarily large images with globally consistent labels
5
5
  Project-URL: Homepage, https://github.com/imcf/patchworks
6
6
  Project-URL: Issues, https://github.com/imcf/patchworks/issues
@@ -102,6 +102,21 @@ decon_kwargs=dict(psf=psf, dxpsf=0.05, dzpsf=0.1, wavelength=525, ...)
102
102
  so edge tiles keep enough context (a plain intensity/threshold halo is
103
103
  too thin).
104
104
 
105
+ !!! note "cudaDecon can return a smaller volume than it was given"
106
+ It rounds each axis down to an FFT-efficient length — e.g. a
107
+ `(32, 1084, 1084)` tile comes back `(32, 1080, 1080)`, because
108
+ `1080 = 2³·3³·5` while `1084 = 4·271` — and trims the excess off the
109
+ **high end**, leaving voxel `(0, 0, 0)` where it was. patchworks restores
110
+ the input shape before the DoG step (one label per input voxel is
111
+ required) anchored at that origin, and logs a WARNING with both shapes.
112
+
113
+ Anchoring matters: restoring it *centred* instead moves every voxel by
114
+ `excess // 2` — a 2 px y/x shift for the tile above, identical on every
115
+ tile. That is invisible on a cell tens of voxels wide and obvious on a
116
+ cilium a few voxels wide, which is how it was eventually caught. If the
117
+ logged difference is more than a few voxels, the PSF or the voxel sizes
118
+ are wrong.
119
+
105
120
  ## Growing the labels afterwards
106
121
 
107
122
  DoG spots/threads are often thin — grow each label by a few pixels with
@@ -444,10 +444,14 @@ results/image.zarr/labels/cyto_labels/
444
444
 
445
445
  !!! tip "The relate step's log"
446
446
  Unlike `prepare`/`segment`/`merge`, the relate step isn't a Snakemake
447
- rule, so it doesn't get a `log:` directive for free. It writes its own
448
- log to `<work_dir>/logs/relate.log` (override with `relate.py --log`),
449
- the same tee-to-file-and-stdout behaviour as the other steps — check
450
- there instead of scrolling back through `srun`'s live output.
447
+ rule, so it doesn't get a `log:` directive for free. Standalone (or
448
+ under plain `multi`), it writes to `<work_dir>/logs/relate.log`
449
+ (override with `relate.py --log`), the same tee-to-file-and-stdout
450
+ behaviour as the other steps. Under `multi-slurm`, where each pair is
451
+ its own concurrent job, `run_multi.py` points each one at its own file
452
+ instead — `<work_dir>/logs/relate/<a>_to_<b>.log` — so concurrent pairs
453
+ don't interleave into one log; check there instead of scrolling back
454
+ through `srun`'s live output.
451
455
 
452
456
  See [Relating labels across segmentations](label_relations.md) for what
453
457
  `label_relations()` returns and how to save it yourself — the cluster
@@ -498,15 +502,29 @@ abort the others; you get a per-config status and a non-zero exit.
498
502
 
499
503
  !!! tip "The relate step runs on the cluster too, under `multi-slurm`"
500
504
  `label_relations()` streams every chunk of two full-resolution label
501
- volumes — real CPU/IO work, not orchestration. Under `multi-slurm` it is
502
- submitted as its own `srun` job (`scripts/relate.py`) instead of running
503
- in the driver process on the login node, the same fix already applied to
504
- the occupancy map. Tune its allocation with `--relate-partition`,
505
- `--relate-mem`, `--relate-cpus` and `--relate-time` (defaults: `scicore`,
506
- `32G`, `8`, `180` minutes) — these are wide-margin guesses, not measured
507
- numbers, so raise them for a very large or very object-dense pair. Under
508
- plain `multi` (no `--profile`), it still runs locally, in-process, as
509
- before.
505
+ volumes — real CPU/IO work, not orchestration. Under `multi-slurm`,
506
+ **each pair in `relations:` is submitted as its own `srun` job**
507
+ (`scripts/relate.py`) instead of running in the driver process on the
508
+ login node, the same fix already applied to the occupancy map. Tune the
509
+ allocation with `--relate-partition`, `--relate-mem`, `--relate-cpus`
510
+ and `--relate-time` (defaults: `scicore`, `32G`, `8`, `180` minutes,
511
+ **per relation**) — these are wide-margin guesses, not measured numbers,
512
+ so raise them for a very large or very object-dense pair; a pair needing
513
+ a chunk-layout rechunk first (see below) is the usual reason one runs
514
+ long. Set `--relate-qos` if your account's default QOS for the partition
515
+ caps the wall time below `--relate-time` — `srun` fails immediately with
516
+ `QOSMaxWallDurationPerJobLimit` when that happens; `sacctmgr -p show
517
+ assoc user=$USER` and `sacctmgr -p show qos` list what's available and
518
+ each one's `MaxWall`. Under plain `multi` (no `--profile`), relations
519
+ still run locally, in-process, one after another, as before.
520
+
521
+ Because every pair gets its own job, one running long no longer starves
522
+ the others out of a shared time budget, and a pair that gets killed no
523
+ longer takes an already-finished sibling's workbook down with it.
524
+ `relate.py` also skips a pair whose `.xlsx` is already newer than both
525
+ labels' merge marker, so **re-running the exact same `multi-slurm`
526
+ command only recomputes what's still missing or stale** — delete a
527
+ specific `.xlsx` yourself to force just that one to recompute.
510
528
 
511
529
  !!! tip "After a killed run"
512
530
  Snakemake only releases its lock on a clean exit, so a run that was killed
@@ -239,18 +239,23 @@ def _run(block: np.ndarray, dog_dict: dict[str, Any]) -> np.ndarray:
239
239
 
240
240
 
241
241
  def _restore_shape(arr: np.ndarray, shape: tuple[int, ...]) -> np.ndarray:
242
- """Centre *arr* back into an array of *shape*, cropping or edge-padding.
242
+ """Restore *arr* to *shape*, anchored at the origin, cropping or padding.
243
243
 
244
244
  Deconvolution must not change the field of view: patchworks writes the
245
245
  result into a destination slice derived from the tile's geometry, so one
246
246
  label per input voxel is required.
247
247
 
248
- Centring is the right correction for a symmetric crop, which is what
249
- apodisation produces. The discrepancies observed are small (a voxel in z,
250
- a few in x/y) and land inside the halo, which is discarded anyway -- so
251
- the labels that survive the trim are unaffected. It is logged at WARNING
252
- with the exact shapes so a larger, non-symmetric crop cannot pass
253
- silently.
248
+ The alignment is **origin-anchored**, not centred: cudaDecon rounds each
249
+ axis down to an FFT-efficient length (e.g. 1084 -> 1080, since
250
+ 1080 = 2**3 * 3**3 * 5 while 1084 = 4 * 271) and trims the excess off the
251
+ high end, leaving voxel (0, 0, 0) where it was. Re-centring content that
252
+ was never centred shifts every voxel by ``excess // 2`` -- measured at 2
253
+ px in y and x on a real (32, 1084, 1084) tile, in the same direction on
254
+ every tile. That is invisible on a cell tens of voxels across and glaring
255
+ on a cilium a few voxels across, which is exactly how it was found.
256
+
257
+ A mismatch is still logged at WARNING with the exact shapes, since a
258
+ large one means the PSF or voxel sizes are wrong.
254
259
 
255
260
  Parameters
256
261
  ----------
@@ -271,22 +276,22 @@ def _restore_shape(arr: np.ndarray, shape: tuple[int, ...]) -> np.ndarray:
271
276
  (14, 1024)
272
277
  """
273
278
  logger.warning(
274
- "deconvolution returned %s for a %s input; re-centring to the input "
275
- "shape. patchworks needs one label per input voxel. A large or "
276
- "asymmetric difference here would shift labels -- check the PSF and "
277
- "voxel sizes if this is more than a few voxels.",
279
+ "deconvolution returned %s for a %s input; restoring the input shape "
280
+ "from the origin. patchworks needs one label per input voxel. A large "
281
+ "difference here means the PSF or voxel sizes are wrong -- check them "
282
+ "if this is more than a few voxels.",
278
283
  arr.shape,
279
284
  shape,
280
285
  )
281
286
  # Crop first, so an axis that grew is handled before padding the rest.
287
+ # Both keep voxel 0 where it is: cudaDecon trims off the high end (see the
288
+ # docstring), so the low corner is the one landmark known to be unmoved.
282
289
  crop = tuple(
283
- slice((a - s) // 2, (a - s) // 2 + s) if a > s else slice(None)
284
- for a, s in zip(arr.shape, shape)
290
+ slice(0, s) if a > s else slice(None) for a, s in zip(arr.shape, shape)
285
291
  )
286
292
  arr = arr[crop]
287
293
  pad = tuple(
288
- ((s - a) // 2, s - a - (s - a) // 2) if a < s else (0, 0)
289
- for a, s in zip(arr.shape, shape)
294
+ (0, s - a) if a < s else (0, 0) for a, s in zip(arr.shape, shape)
290
295
  )
291
296
  if any(lo or hi for lo, hi in pad):
292
297
  arr = np.pad(arr, pad, mode="edge")
@@ -109,20 +109,32 @@ def test_explicit_decon_kwargs_win_over_the_calibration(monkeypatch):
109
109
  assert captured["dzpsf"] == 0.2
110
110
 
111
111
 
112
- def test_restore_shape_recentres_a_cropped_decon():
112
+ def test_restore_shape_anchors_a_cropped_decon_at_the_origin():
113
113
  """cudaDecon can hand back a smaller volume than it was given.
114
114
 
115
- Observed on a real edge tile: (14, 1024, 1024) in, (13, 1020, 1020) out.
116
- patchworks needs one label per input voxel, so the field of view has to be
117
- restored before the DoG step.
115
+ Observed on a real tile: (32, 1084, 1084) in, (32, 1080, 1080) out --
116
+ each axis rounded down to an FFT-efficient length, with the excess taken
117
+ off the high end. Restoring it *centred* (what this used to do) moved
118
+ every voxel by excess // 2, measured as a 2 px y/x shift on real data:
119
+ invisible on a cell, glaring on a cilium a few voxels across.
118
120
  """
119
121
  from patchworks.plugins.dog import _restore_shape
120
122
 
121
123
  arr = np.arange(13 * 1020 * 1020, dtype="float32").reshape(13, 1020, 1020)
122
124
  out = _restore_shape(arr, (14, 1024, 1024))
123
125
  assert out.shape == (14, 1024, 1024)
124
- # The original content is preserved, centred, not resampled.
125
- assert np.array_equal(out[0:13, 2:1022, 2:1022], arr)
126
+ # Content keeps its original indices -- voxel 0 stays voxel 0.
127
+ assert np.array_equal(out[0:13, 0:1020, 0:1020], arr)
128
+
129
+
130
+ def test_restore_shape_crops_from_the_high_end():
131
+ """The mirror case: an axis that came back too long keeps its low corner."""
132
+ from patchworks.plugins.dog import _restore_shape
133
+
134
+ arr = np.arange(6 * 12, dtype="float32").reshape(6, 12)
135
+ out = _restore_shape(arr, (4, 8))
136
+ assert out.shape == (4, 8)
137
+ assert np.array_equal(out, arr[0:4, 0:8])
126
138
 
127
139
 
128
140
  def test_restore_shape_handles_growth_and_exact_fit():
@@ -1,7 +1,10 @@
1
1
  """Tests for the multi-config driver's SLURM-facing behaviour."""
2
2
 
3
+ import json
4
+ import os
3
5
  import re
4
6
  import sys
7
+ import time
5
8
  from pathlib import Path
6
9
 
7
10
  sys.path.insert(
@@ -15,6 +18,7 @@ import yaml # noqa: E402
15
18
 
16
19
  from run_multi import ( # noqa: E402
17
20
  _CONVERT_KEYS,
21
+ _relate_cmd,
18
22
  _snakemake_cmd,
19
23
  _validate_configs,
20
24
  slurm_jobname_prefix,
@@ -176,6 +180,85 @@ def test_relate_is_submitted_via_slurm_under_profile():
176
180
  assert "label_relations(" not in src
177
181
 
178
182
 
183
+ def _relate_kwargs(**overrides):
184
+ kwargs = dict(
185
+ work_dir="/w",
186
+ image_store="/w/image.zarr",
187
+ workflow_dir=Path("/workflow"),
188
+ relate_partition="scicore",
189
+ relate_mem="32G",
190
+ relate_cpus=8,
191
+ relate_time=180,
192
+ relate_qos=None,
193
+ )
194
+ kwargs.update(overrides)
195
+ return kwargs
196
+
197
+
198
+ def test_relate_cmd_is_one_job_per_relation_pair():
199
+ """A killed shared job used to lose every relation still queued behind
200
+
201
+ the one that was running -- one srun per pair means a slow or failing
202
+ pair can no longer starve, or take down, its siblings' time budget.
203
+ """
204
+ relations = [
205
+ {"a": "nuclei_labels", "b": "cyto_labels", "output": "n2c.xlsx"},
206
+ {"a": "cilia_labels", "b": "cyto_labels", "output": "c2c.xlsx"},
207
+ ]
208
+ cmds = [_relate_cmd(rel, **_relate_kwargs()) for rel in relations]
209
+
210
+ assert len(cmds) == 2
211
+ for cmd, rel in zip(cmds, relations):
212
+ assert cmd[0] == "srun"
213
+ assert "--relations" in cmd
214
+ # Each job's payload is *only* its own pair, not the whole list.
215
+ payload = json.loads(cmd[cmd.index("--relations") + 1])
216
+ assert payload == [rel]
217
+
218
+
219
+ def test_relate_cmd_job_name_identifies_the_pair():
220
+ cmd = _relate_cmd(
221
+ {"a": "cilia_labels", "b": "cyto_labels"}, **_relate_kwargs()
222
+ )
223
+ name = cmd[cmd.index("--job-name") + 1]
224
+ assert _EXECUTOR_RULE.match(name)
225
+ assert "cilia_labels" in name
226
+ assert "cyto_labels" in name
227
+
228
+
229
+ def test_relate_cmd_gives_each_pair_its_own_log():
230
+ """Concurrent per-pair jobs sharing one relate.log would interleave --
231
+
232
+ each pair's --log must be a distinct file, or the whole point of
233
+ splitting the log the way segment/<batch>.log already does is lost.
234
+ """
235
+ cmds = [
236
+ _relate_cmd(rel, **_relate_kwargs())
237
+ for rel in (
238
+ {"a": "nuclei_labels", "b": "cyto_labels"},
239
+ {"a": "cilia_labels", "b": "cyto_labels"},
240
+ )
241
+ ]
242
+ logs = [cmd[cmd.index("--log") + 1] for cmd in cmds]
243
+ assert len(set(logs)) == 2
244
+ assert all(log.startswith("/w/logs/relate/") for log in logs)
245
+
246
+
247
+ def test_relate_cmd_omits_qos_by_default():
248
+ cmd = _relate_cmd({"a": "a", "b": "b"}, **_relate_kwargs())
249
+ assert "--qos" not in cmd
250
+
251
+
252
+ def test_relate_cmd_passes_qos_when_set():
253
+ """A default QOS whose MaxWall is shorter than --relate-time is exactly
254
+
255
+ what killed a real run (QOSMaxWallDurationPerJobLimit) -- --relate-qos
256
+ lets a longer one be requested explicitly instead of guessed at.
257
+ """
258
+ cmd = _relate_cmd({"a": "a", "b": "b"}, **_relate_kwargs(relate_qos="1day"))
259
+ assert cmd[cmd.index("--qos") + 1] == "1day"
260
+
261
+
179
262
  def test_relate_script_has_the_real_bookkeeping():
180
263
  """relate.py must be the actual implementation, not a stub.
181
264
 
@@ -285,6 +368,88 @@ def test_relate_rechunks_mismatched_label_arrays(tmp_path):
285
368
  assert rows[2] == (None, 0, 0) # label 2 touches nothing in b
286
369
 
287
370
 
371
+ def test_relation_up_to_date_missing_output_is_false(tmp_path):
372
+ from relate import _relation_up_to_date
373
+
374
+ assert not _relation_up_to_date(
375
+ str(tmp_path), "a", "b", tmp_path / "nope.xlsx"
376
+ )
377
+
378
+
379
+ def test_relation_up_to_date_missing_marker_is_false(tmp_path):
380
+ """No labels.done for a label means its state can't be judged -- treat
381
+
382
+ that as "recompute", not as "trust the existing workbook".
383
+ """
384
+ from relate import _relation_up_to_date
385
+
386
+ out = tmp_path / "rel.xlsx"
387
+ out.write_text("x")
388
+ assert not _relation_up_to_date(str(tmp_path), "a", "b", out)
389
+
390
+
391
+ def test_relation_up_to_date_true_only_when_newer_than_both_markers(
392
+ tmp_path,
393
+ ):
394
+ from relate import _relation_up_to_date
395
+
396
+ (tmp_path / "a").mkdir()
397
+ (tmp_path / "b").mkdir()
398
+ (tmp_path / "a" / "labels.done").touch()
399
+ (tmp_path / "b" / "labels.done").touch()
400
+ out = tmp_path / "rel.xlsx"
401
+ out.write_text("x")
402
+
403
+ now = time.time()
404
+ os.utime(tmp_path / "a" / "labels.done", (now, now))
405
+ os.utime(tmp_path / "b" / "labels.done", (now, now))
406
+
407
+ # older than both markers -> stale
408
+ os.utime(out, (now - 10, now - 10))
409
+ assert not _relation_up_to_date(str(tmp_path), "a", "b", out)
410
+
411
+ # newer than both markers -> up to date
412
+ os.utime(out, (now + 10, now + 10))
413
+ assert _relation_up_to_date(str(tmp_path), "a", "b", out)
414
+
415
+ # b re-merged after the workbook was written -> stale again
416
+ os.utime(tmp_path / "b" / "labels.done", (now + 20, now + 20))
417
+ assert not _relation_up_to_date(str(tmp_path), "a", "b", out)
418
+
419
+
420
+ def test_relate_skips_a_relation_whose_workbook_is_up_to_date(tmp_path):
421
+ """A retry must not recompute what already finished -- only what a
422
+
423
+ shared, now-split-per-pair job left missing after a partial failure.
424
+ Proven by planting sentinel content no real relation would produce: if
425
+ run_relations() recomputed anyway, the sentinel would be gone.
426
+ """
427
+ from relate import run_relations
428
+
429
+ work_dir = tmp_path / "work"
430
+ work_dir.mkdir()
431
+ for name in ("a_labels", "b_labels"):
432
+ (work_dir / name).mkdir()
433
+ (work_dir / name / "labels.done").touch()
434
+
435
+ out_path = work_dir / "rel.xlsx"
436
+ sentinel = openpyxl.Workbook()
437
+ sentinel.active.append(["sentinel"])
438
+ sentinel.save(out_path)
439
+ os.utime(out_path, (time.time() + 60, time.time() + 60))
440
+
441
+ # No image_store/labels at all -- if this weren't skipped, run_relations
442
+ # would raise trying to open them, not just produce the wrong content.
443
+ run_relations(
444
+ str(work_dir),
445
+ str(tmp_path / "image.zarr"),
446
+ [{"a": "a_labels", "b": "b_labels", "output": "rel.xlsx"}],
447
+ )
448
+
449
+ wb = openpyxl.load_workbook(out_path)
450
+ assert wb.active["A1"].value == "sentinel"
451
+
452
+
288
453
  def test_mixed_nuclei_channel_auto_passes_validation():
289
454
  """A channel-count mismatch under `tile_shape: "auto"` is no longer
290
455
 
@@ -54,11 +54,43 @@ def _label_ids(image_store: str, name: str) -> list[int]:
54
54
  return sorted(int(x) for x in da.unique(arr[arr > 0]).compute())
55
55
 
56
56
 
57
+ def _relation_up_to_date(
58
+ work_dir: str, a_name: str, b_name: str, out_path: Path
59
+ ) -> bool:
60
+ """Whether *out_path* already reflects the current *a_name*/*b_name* labels.
61
+
62
+ Mirrors the workflow's existing "delete to force" convention (see
63
+ ``image.zarr`` not being reconverted once it exists): a relation is
64
+ considered current when its workbook is newer than both labels'
65
+ ``labels.done`` merge marker, and stale (or never computed) otherwise.
66
+ Missing markers -- a label group written before merge started recording
67
+ one, or a nonstandard ``image_store`` layout -- are treated as "unknown,
68
+ recompute" rather than raise, since staleness can't be judged without them.
69
+ """
70
+ if not out_path.exists():
71
+ return False
72
+ try:
73
+ out_mtime = out_path.stat().st_mtime
74
+ for name in (a_name, b_name):
75
+ marker = Path(work_dir) / name / "labels.done"
76
+ if not marker.exists() or marker.stat().st_mtime > out_mtime:
77
+ return False
78
+ except OSError:
79
+ return False
80
+ return True
81
+
82
+
57
83
  def run_relations(
58
84
  work_dir: str, image_store: str, relations: list[dict]
59
85
  ) -> None:
60
86
  """Compute and write every configured relation pair as an .xlsx workbook.
61
87
 
88
+ A relation already reflected by an up-to-date workbook (see
89
+ :func:`_relation_up_to_date`) is skipped rather than recomputed -- so
90
+ retrying a partially-failed run (this step has no Snakemake rule of its
91
+ own to track that for it) only redoes what's actually missing or stale.
92
+ Delete the ``.xlsx`` yourself to force a specific relation to recompute.
93
+
62
94
  Parameters
63
95
  ----------
64
96
  work_dir : str
@@ -81,6 +113,13 @@ def run_relations(
81
113
  out_path = Path(work_dir) / rel.get(
82
114
  "output", f"{a_name}_to_{b_name}.xlsx"
83
115
  )
116
+ if _relation_up_to_date(work_dir, a_name, b_name, out_path):
117
+ print(
118
+ f"[relate] {out_path} is already up to date with "
119
+ f"{a_name}/{b_name}; skipping",
120
+ flush=True,
121
+ )
122
+ continue
84
123
  print(f"[relate] relating {a_name} -> {b_name} …", flush=True)
85
124
  a = da.from_zarr(image_store, component=f"labels/{a_name}/0")
86
125
  b = da.from_zarr(image_store, component=f"labels/{b_name}/0")
@@ -129,6 +129,11 @@ def slurm_jobname_prefix(label: str) -> str:
129
129
  return f"pw-{safe}"[:50]
130
130
 
131
131
 
132
+ def _safe_filename(name: str) -> str:
133
+ """Sanitise a label name for use as (part of) a log filename."""
134
+ return re.sub(r"[^A-Za-z0-9_-]", "-", name)
135
+
136
+
132
137
  def _test_email(cfg: dict) -> int:
133
138
  """Send one test notification and report the outcome. Returns an exit code.
134
139
 
@@ -186,6 +191,67 @@ def _test_email(cfg: dict) -> int:
186
191
  return 1
187
192
 
188
193
 
194
+ def _relate_cmd(
195
+ rel: dict,
196
+ *,
197
+ work_dir: str,
198
+ image_store: str,
199
+ workflow_dir: Path,
200
+ relate_partition: str,
201
+ relate_mem: str,
202
+ relate_cpus: int,
203
+ relate_time: int,
204
+ relate_qos: str | None,
205
+ ) -> list[str]:
206
+ """Build one ``srun`` invocation of ``relate.py`` for a single relation pair.
207
+
208
+ One job per pair, not one job for the whole ``relations:`` list: a
209
+ single shared ``srun`` time budget lets a slow pair (e.g. one needing a
210
+ chunk-layout rechunk first) starve the others out of a fixed
211
+ ``--relate-time``, and a kill that way loses everything not yet written
212
+ even though earlier pairs already finished. Separate jobs also run
213
+ concurrently instead of one after another, and ``relate.py`` itself
214
+ skips a pair whose workbook is already up to date, so retrying with the
215
+ same relations only redoes what actually failed.
216
+ """
217
+ a_name, b_name = rel["a"], rel["b"]
218
+ cmd = [
219
+ "srun",
220
+ "--partition",
221
+ relate_partition,
222
+ ]
223
+ if relate_qos:
224
+ cmd += ["--qos", relate_qos]
225
+ cmd += [
226
+ "--mem",
227
+ relate_mem,
228
+ "--cpus-per-task",
229
+ str(relate_cpus),
230
+ "--time",
231
+ str(relate_time),
232
+ "--job-name",
233
+ slurm_jobname_prefix(f"relate-{a_name}-to-{b_name}"),
234
+ sys.executable,
235
+ str(workflow_dir / "scripts" / "relate.py"),
236
+ "--work-dir",
237
+ work_dir,
238
+ "--image-store",
239
+ image_store,
240
+ "--relations",
241
+ json.dumps([rel]),
242
+ # Concurrent per-pair jobs sharing the default <work_dir>/logs/
243
+ # relate.log would interleave -- give each pair its own file.
244
+ "--log",
245
+ str(
246
+ Path(work_dir)
247
+ / "logs"
248
+ / "relate"
249
+ / f"{_safe_filename(a_name)}_to_{_safe_filename(b_name)}.log"
250
+ ),
251
+ ]
252
+ return cmd
253
+
254
+
189
255
  def _run(cmd: list[str], workflow_dir: Path) -> int:
190
256
  print(f"[run_multi] $ {' '.join(cmd)}", flush=True)
191
257
  return subprocess.run(cmd, cwd=workflow_dir).returncode
@@ -448,7 +514,22 @@ def main() -> None:
448
514
  "--relate-time",
449
515
  type=int,
450
516
  default=180,
451
- help="srun --time in minutes for the relate step under --profile (default: 180)",
517
+ help=(
518
+ "srun --time in minutes for the relate step under --profile "
519
+ "(default: 180). Each relation pair is its own SLURM job, so "
520
+ "this bounds one relation, not the whole relations: list."
521
+ ),
522
+ )
523
+ parser.add_argument(
524
+ "--relate-qos",
525
+ default=None,
526
+ help=(
527
+ "srun --qos for the relate step under --profile. Omit to let "
528
+ "SLURM pick your account's default QOS for the partition -- set "
529
+ "this explicitly if that default's MaxWall is shorter than "
530
+ "--relate-time (sacctmgr -p show assoc/qos shows what's "
531
+ "available, e.g. '1day', '1week')."
532
+ ),
452
533
  )
453
534
  args = parser.parse_args()
454
535
 
@@ -603,38 +684,52 @@ def main() -> None:
603
684
  # Real CPU/IO work -- tens of thousands of zarr chunk reads for a
604
685
  # full-resolution label volume -- not orchestration, so (like the
605
686
  # occupancy map) it does not belong in this driver process on the
606
- # login node. Submit it as its own job instead.
607
- cmd = [
608
- "srun",
609
- "--partition",
610
- args.relate_partition,
611
- "--mem",
612
- args.relate_mem,
613
- "--cpus-per-task",
614
- str(args.relate_cpus),
615
- "--time",
616
- str(args.relate_time),
617
- "--job-name",
618
- "pw-relate",
619
- sys.executable,
620
- str(workflow_dir / "scripts" / "relate.py"),
621
- "--work-dir",
622
- work_dir,
623
- "--image-store",
624
- image_store,
625
- "--relations",
626
- json.dumps(relations),
627
- ]
628
- rc = _run(cmd, workflow_dir)
629
- if rc != 0:
687
+ # login node. Submit it as its own job.
688
+ #
689
+ # One job *per relation pair*, not one job for the whole list: a
690
+ # single shared srun budget lets a slow pair (e.g. one needing a
691
+ # chunk-layout rechunk first) starve the others' time out of a fixed
692
+ # --relate-time, and killed that way loses everything not yet
693
+ # written even though earlier pairs already finished. Separate jobs
694
+ # also run concurrently rather than one after another, and
695
+ # relate.py itself now skips a pair whose workbook is already
696
+ # up to date, so retrying this exact command only redoes what
697
+ # actually failed.
698
+ procs = []
699
+ for rel in relations:
700
+ cmd = _relate_cmd(
701
+ rel,
702
+ work_dir=work_dir,
703
+ image_store=image_store,
704
+ workflow_dir=workflow_dir,
705
+ relate_partition=args.relate_partition,
706
+ relate_mem=args.relate_mem,
707
+ relate_cpus=args.relate_cpus,
708
+ relate_time=args.relate_time,
709
+ relate_qos=args.relate_qos,
710
+ )
711
+ print(f"[run_multi] $ {' '.join(cmd)}", flush=True)
712
+ procs.append(
713
+ (
714
+ f"{rel['a']} -> {rel['b']}",
715
+ subprocess.Popen(cmd, cwd=workflow_dir),
716
+ )
717
+ )
718
+
719
+ failed = [name for name, p in procs if p.wait() != 0]
720
+ for name, p in procs:
721
+ status = "FAILED" if p.returncode else "ok"
722
+ print(f"[run_multi] relate {name}: {status}", flush=True)
723
+ if failed:
630
724
  print(
631
- f"[run_multi] ERROR: relate step failed (exit {rc}). "
632
- "Segmentations already succeeded -- only the relation "
633
- "workbook(s) are missing. Re-run with the same --config to "
634
- "retry just this step.",
725
+ f"[run_multi] ERROR: {len(failed)} relation(s) failed: "
726
+ f"{', '.join(failed)}. Segmentations already succeeded, and "
727
+ "any relation that did finish is written -- re-run with the "
728
+ "same --config to retry just the ones still missing (already "
729
+ "up-to-date workbooks are skipped, not recomputed).",
635
730
  file=sys.stderr,
636
731
  )
637
- sys.exit(rc)
732
+ sys.exit(1)
638
733
  return
639
734
 
640
735
  from relate import run_relations
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes