patchworks 2.7.0__tar.gz → 2.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (121) hide show
  1. {patchworks-2.7.0 → patchworks-2.8.0}/PKG-INFO +1 -1
  2. {patchworks-2.7.0 → patchworks-2.8.0}/docs/guide/custom_segmentation.md +5 -2
  3. {patchworks-2.7.0 → patchworks-2.8.0}/docs/guide/snakemake.md +211 -8
  4. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/_io.py +90 -2
  5. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/_relations.py +26 -2
  6. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/plugins/napari.py +7 -8
  7. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/plugins/ome_zarr.py +26 -4
  8. {patchworks-2.7.0 → patchworks-2.8.0}/tests/test_napari.py +75 -0
  9. {patchworks-2.7.0 → patchworks-2.8.0}/tests/test_ome_zarr.py +53 -0
  10. patchworks-2.8.0/tests/test_relations.py +104 -0
  11. {patchworks-2.7.0 → patchworks-2.8.0}/tests/test_run_multi.py +510 -2
  12. patchworks-2.8.0/workflow/config/multi.yaml +75 -0
  13. patchworks-2.8.0/workflow/pixi.toml +126 -0
  14. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/profile/slurm/config.yaml +28 -0
  15. patchworks-2.8.0/workflow/scripts/export_iso.py +314 -0
  16. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/scripts/merge.py +45 -1
  17. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/scripts/relate.py +35 -0
  18. patchworks-2.8.0/workflow/scripts/reshard_store.py +196 -0
  19. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/scripts/run_multi.py +267 -18
  20. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/scripts/view.py +10 -1
  21. patchworks-2.7.0/tests/test_relations.py +0 -47
  22. patchworks-2.7.0/workflow/config/multi.yaml +0 -36
  23. patchworks-2.7.0/workflow/pixi.toml +0 -66
  24. {patchworks-2.7.0 → patchworks-2.8.0}/.github/workflows/docs.yml +0 -0
  25. {patchworks-2.7.0 → patchworks-2.8.0}/.github/workflows/lint.yml +0 -0
  26. {patchworks-2.7.0 → patchworks-2.8.0}/.github/workflows/release.yml +0 -0
  27. {patchworks-2.7.0 → patchworks-2.8.0}/.gitignore +0 -0
  28. {patchworks-2.7.0 → patchworks-2.8.0}/.markdownlint-cli2.yaml +0 -0
  29. {patchworks-2.7.0 → patchworks-2.8.0}/LICENSE +0 -0
  30. {patchworks-2.7.0 → patchworks-2.8.0}/README.md +0 -0
  31. {patchworks-2.7.0 → patchworks-2.8.0}/cliff.toml +0 -0
  32. {patchworks-2.7.0 → patchworks-2.8.0}/docs/api/chunks.md +0 -0
  33. {patchworks-2.7.0 → patchworks-2.8.0}/docs/api/cluster.md +0 -0
  34. {patchworks-2.7.0 → patchworks-2.8.0}/docs/api/io.md +0 -0
  35. {patchworks-2.7.0 → patchworks-2.8.0}/docs/api/merge_tile_labels.md +0 -0
  36. {patchworks-2.7.0 → patchworks-2.8.0}/docs/api/plugins/cellpose.md +0 -0
  37. {patchworks-2.7.0 → patchworks-2.8.0}/docs/api/plugins/dog.md +0 -0
  38. {patchworks-2.7.0 → patchworks-2.8.0}/docs/api/plugins/napari.md +0 -0
  39. {patchworks-2.7.0 → patchworks-2.8.0}/docs/api/plugins/ome_zarr.md +0 -0
  40. {patchworks-2.7.0 → patchworks-2.8.0}/docs/api/postprocess.md +0 -0
  41. {patchworks-2.7.0 → patchworks-2.8.0}/docs/api/relabel.md +0 -0
  42. {patchworks-2.7.0 → patchworks-2.8.0}/docs/api/tile_process.md +0 -0
  43. {patchworks-2.7.0 → patchworks-2.8.0}/docs/api/volume_filter.md +0 -0
  44. {patchworks-2.7.0 → patchworks-2.8.0}/docs/assets/logo.png +0 -0
  45. {patchworks-2.7.0 → patchworks-2.8.0}/docs/examples/cellpose_2d.md +0 -0
  46. {patchworks-2.7.0 → patchworks-2.8.0}/docs/examples/cellpose_2d.py +0 -0
  47. {patchworks-2.7.0 → patchworks-2.8.0}/docs/examples/cellpose_3d.md +0 -0
  48. {patchworks-2.7.0 → patchworks-2.8.0}/docs/examples/cellpose_3d.py +0 -0
  49. {patchworks-2.7.0 → patchworks-2.8.0}/docs/examples/custom.md +0 -0
  50. {patchworks-2.7.0 → patchworks-2.8.0}/docs/examples/custom_method.py +0 -0
  51. {patchworks-2.7.0 → patchworks-2.8.0}/docs/examples/dog.md +0 -0
  52. {patchworks-2.7.0 → patchworks-2.8.0}/docs/examples/dog.py +0 -0
  53. {patchworks-2.7.0 → patchworks-2.8.0}/docs/examples/standalone_merge.md +0 -0
  54. {patchworks-2.7.0 → patchworks-2.8.0}/docs/examples/stardist.md +0 -0
  55. {patchworks-2.7.0 → patchworks-2.8.0}/docs/examples/stardist_2d.py +0 -0
  56. {patchworks-2.7.0 → patchworks-2.8.0}/docs/getting_started.md +0 -0
  57. {patchworks-2.7.0 → patchworks-2.8.0}/docs/guide/gpu_distributed.md +0 -0
  58. {patchworks-2.7.0 → patchworks-2.8.0}/docs/guide/label_relations.md +0 -0
  59. {patchworks-2.7.0 → patchworks-2.8.0}/docs/guide/measurements.md +0 -0
  60. {patchworks-2.7.0 → patchworks-2.8.0}/docs/guide/merging.md +0 -0
  61. {patchworks-2.7.0 → patchworks-2.8.0}/docs/guide/ome_zarr_napari.md +0 -0
  62. {patchworks-2.7.0 → patchworks-2.8.0}/docs/guide/performance.md +0 -0
  63. {patchworks-2.7.0 → patchworks-2.8.0}/docs/guide/pitfalls.md +0 -0
  64. {patchworks-2.7.0 → patchworks-2.8.0}/docs/guide/skip_empty.md +0 -0
  65. {patchworks-2.7.0 → patchworks-2.8.0}/docs/guide/tiling.md +0 -0
  66. {patchworks-2.7.0 → patchworks-2.8.0}/docs/index.md +0 -0
  67. {patchworks-2.7.0 → patchworks-2.8.0}/mkdocs.yml +0 -0
  68. {patchworks-2.7.0 → patchworks-2.8.0}/pyproject.toml +0 -0
  69. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/__init__.py +0 -0
  70. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/_chunks.py +0 -0
  71. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/_cluster.py +0 -0
  72. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/_core.py +0 -0
  73. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/_distributed.py +0 -0
  74. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/_gpu.py +0 -0
  75. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/_merge.py +0 -0
  76. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/_notify.py +0 -0
  77. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/_occupancy.py +0 -0
  78. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/_postprocess.py +0 -0
  79. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/_progress.py +0 -0
  80. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/_relabel.py +0 -0
  81. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/_volume_filter.py +0 -0
  82. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/plugins/__init__.py +0 -0
  83. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/plugins/cellpose.py +0 -0
  84. {patchworks-2.7.0 → patchworks-2.8.0}/src/patchworks/plugins/dog.py +0 -0
  85. {patchworks-2.7.0 → patchworks-2.8.0}/tests/ngff_schemas/0.4/image.schema +0 -0
  86. {patchworks-2.7.0 → patchworks-2.8.0}/tests/ngff_schemas/0.4/label.schema +0 -0
  87. {patchworks-2.7.0 → patchworks-2.8.0}/tests/ngff_schemas/0.4/ome.schema +0 -0
  88. {patchworks-2.7.0 → patchworks-2.8.0}/tests/ngff_schemas/0.5/_version.schema +0 -0
  89. {patchworks-2.7.0 → patchworks-2.8.0}/tests/ngff_schemas/0.5/image.schema +0 -0
  90. {patchworks-2.7.0 → patchworks-2.8.0}/tests/ngff_schemas/0.5/label.schema +0 -0
  91. {patchworks-2.7.0 → patchworks-2.8.0}/tests/ngff_schemas/0.5/ome.schema +0 -0
  92. {patchworks-2.7.0 → patchworks-2.8.0}/tests/ngff_schemas/README.md +0 -0
  93. {patchworks-2.7.0 → patchworks-2.8.0}/tests/test_allocation.py +0 -0
  94. {patchworks-2.7.0 → patchworks-2.8.0}/tests/test_cellpose.py +0 -0
  95. {patchworks-2.7.0 → patchworks-2.8.0}/tests/test_core.py +0 -0
  96. {patchworks-2.7.0 → patchworks-2.8.0}/tests/test_distributed.py +0 -0
  97. {patchworks-2.7.0 → patchworks-2.8.0}/tests/test_dog.py +0 -0
  98. {patchworks-2.7.0 → patchworks-2.8.0}/tests/test_gpu.py +0 -0
  99. {patchworks-2.7.0 → patchworks-2.8.0}/tests/test_notify.py +0 -0
  100. {patchworks-2.7.0 → patchworks-2.8.0}/tests/test_occupancy.py +0 -0
  101. {patchworks-2.7.0 → patchworks-2.8.0}/tests/test_postprocess.py +0 -0
  102. {patchworks-2.7.0 → patchworks-2.8.0}/tests/test_progress.py +0 -0
  103. {patchworks-2.7.0 → patchworks-2.8.0}/tests/test_pw.py +0 -0
  104. {patchworks-2.7.0 → patchworks-2.8.0}/tests/test_volume_filter.py +0 -0
  105. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/README.md +0 -0
  106. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/Snakefile +0 -0
  107. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/config/common.yaml +0 -0
  108. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/config/config.yaml +0 -0
  109. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/config/config_cilia.yaml +0 -0
  110. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/config/config_cyto.yaml +0 -0
  111. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/config/config_nuclei.yaml +0 -0
  112. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/rules/common.smk +0 -0
  113. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/rules/convert.smk +0 -0
  114. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/rules/merge.smk +0 -0
  115. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/rules/segment.smk +0 -0
  116. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/scripts/_pw.py +0 -0
  117. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/scripts/build_occupancy.py +0 -0
  118. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/scripts/convert.py +0 -0
  119. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/scripts/fetch_model.py +0 -0
  120. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/scripts/prepare_tiles.py +0 -0
  121. {patchworks-2.7.0 → patchworks-2.8.0}/workflow/scripts/segment_tile.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: patchworks
3
- Version: 2.7.0
3
+ Version: 2.8.0
4
4
  Summary: Tiled processing of arbitrarily large images with globally consistent labels
5
5
  Project-URL: Homepage, https://github.com/imcf/patchworks
6
6
  Project-URL: Issues, https://github.com/imcf/patchworks/issues
@@ -81,8 +81,11 @@ backend `fn` used — pass `use_gpu=True` to dilate via cupy instead:
81
81
  fn = dilate_labels(fn, iterations=2, use_gpu=True)
82
82
  ```
83
83
 
84
- Needs `cupy` installed **manually**, matching your CUDA version (e.g.
85
- `pip install cupy-cuda12x`) — it's never installed automatically by
84
+ Needs `cupy`, matching your CUDA version. Under the Snakemake workflow, use
85
+ the environment that already carries it — `pixi install -e cuda12` (or
86
+ `-e cuda13`), then `pixi run -e cuda12 <task>`; outside it, install the
87
+ matching wheel yourself (e.g. `pip install cupy-cuda12x`). It's never
88
+ installed automatically by
86
89
  patchworks, unlike Cellpose's GPU support (which comes for free via
87
90
  PyTorch's self-contained CUDA wheels); cupy ships one wheel per CUDA major
88
91
  version, so there's no single generic pin that works everywhere. On the
@@ -93,10 +93,21 @@ shard_labels: false # true → also reshard label level 0 after the
93
93
  `dilate: N` grows every label by `N` pixels once segmentation finishes,
94
94
  regardless of `method`. `0` (default) disables it. Runs on CPU (scipy)
95
95
  by default; set `dilate_gpu: true` to dilate via cupy instead — that
96
- needs `cupy` installed in the segment job's environment (matching your
97
- CUDA version, e.g. `pip install cupy-cuda12x`) and a GPU allocated for
96
+ needs `cupy` in the segment job's environment and a GPU allocated for
98
97
  that job (`set-resources: segment:` in `profile/slurm/config.yaml`,
99
- same as for a GPU `method`). It's independent of whatever `method`
98
+ same as for a GPU `method`). cupy ships one wheel per CUDA major
99
+ version, so pick the environment that matches yours rather than editing
100
+ `pixi.toml`:
101
+
102
+ ```bash
103
+ nvidia-smi # read the CUDA version off a GPU node
104
+ pixi install -e cuda12 # or -e cuda13
105
+ pixi run -e cuda12 multi-slurm # every task works in these too
106
+ ```
107
+
108
+ A pixi environment includes the default feature as well, so `-e cuda12`
109
+ is "everything the default has, plus cupy". `cellpose4-cuda12` and
110
+ `cellpose4-cuda13` combine it with the pinned Cellpose. It's independent of whatever `method`
100
111
  itself runs on — you can dilate on GPU even with `method: "threshold"`
101
112
  (CPU), or on CPU even with `method: "cellpose"` (GPU). See [Growing
102
113
  labels afterwards](custom_segmentation.md#growing-labels-afterwards-dilation)
@@ -127,10 +138,29 @@ shard_labels: false # true → also reshard label level 0 after the
127
138
  `true` reuses whatever spec `shard` carries; a list overrides it.
128
139
  `shard: false` does not veto it — an unsharded image with sharded labels
129
140
  is a valid combination. It is opt-in because it costs one extra full
130
- read+write of level 0, and because it is the level with the most chunks
131
- it is also the one worth paying for. The array's attributes (including
132
- the merge's own completion marker) are carried across, so a later rerun
133
- still sees the merge as done.
141
+ read+write of level 0. The array's attributes (including the merge's own
142
+ completion marker) are carried across, so a later rerun still sees the
143
+ merge as done.
144
+
145
+ **Set both keys**, but expect level 0 to dominate. `shard: true` sends
146
+ the pyramid down the dask path, where each level is rechunked to the
147
+ `(16, 1024, 1024)` cap — so levels 1..N shrink fourfold each, the way
148
+ you would expect. Level 0 does not: its chunks come from `tile_shape`,
149
+ it is the full-resolution level, and it is the one `shard` cannot reach.
150
+
151
+ Measured on a real `(126, 45961, 42072)` label group whose tiles are 32%
152
+ occupied (empty chunks are never written):
153
+
154
+ | level | files, `shard` only | with `shard_labels` too |
155
+ |---|---|---|
156
+ | 0 | ~10,656 | ~666 |
157
+ | 1–4 | ~462 | ~462 |
158
+ | **total** | **~11,100** | **~1,130** |
159
+
160
+ So level 0 is ~96% of a label group here, and `shard` alone barely moves
161
+ the file count. The two keys are not interchangeable and neither is
162
+ redundant — `shard` handles the pyramid, `shard_labels` handles the level
163
+ that actually holds the files.
134
164
 
135
165
  Check whether the file count is actually a problem for your data first —
136
166
  a `(126, 34000, 28500)` image at the `(16, 1024, 1024)` label chunk cap
@@ -173,6 +203,123 @@ shard_labels: false # true → also reshard label level 0 after the
173
203
  to 0.5. Setting `ngff_version: "0.6"` is rejected with that explanation
174
204
  rather than writing a store you could not open.
175
205
 
206
+ !!! tip "Repacking a store that was written unsharded (`pixi run reshard`)"
207
+ Sharding normally has to be chosen before a store is written, because
208
+ concurrent writers cannot share a shard file. Once the store is finished
209
+ nothing is writing it, so a single pass can repack it in place — same
210
+ data, same chunking, same metadata, far fewer files, no re-conversion and
211
+ no re-segmentation:
212
+
213
+ ```bash
214
+ pixi run reshard --store /path/to/image.zarr --dry-run # report only
215
+ pixi run reshard --store /path/to/image.zarr --labels-only
216
+ ```
217
+
218
+ It skips anything already sharded, so re-running it is a no-op, and it
219
+ carries each array's attributes across — including the merge's own
220
+ completion marker, without which a later re-run would merge already-merged
221
+ ids together.
222
+
223
+ Submit it rather than running it on a login node: it reads and writes
224
+ every level. Mind the QOS ceiling (see the warning in section 5b) —
225
+ `sbatch --qos=1day --time=12:00:00 --cpus-per-task=8 --mem=64G`.
226
+
227
+ !!! tip "Exporting a store as a single file (`.zip` or `.iso`)"
228
+ A zarr store is tens of thousands of small files, which copies slowly
229
+ everywhere and badly to Windows. Two ways to make it one file:
230
+
231
+ ```bash
232
+ pixi run zip --store /path/to/image.zarr # needs nothing extra
233
+ pixi run iso --store /path/to/image.zarr # needs an ISO builder
234
+ ```
235
+
236
+ **`.zip` is the one that always works.** It needs nothing beyond Python,
237
+ and zarr reads a store straight out of it *without unpacking*:
238
+
239
+ ```python
240
+ import zarr
241
+ store = zarr.storage.ZipStore("image.zarr.zip", mode="r")
242
+ group = zarr.open_group(store, path="image.zarr", mode="r")
243
+ ```
244
+
245
+ Windows Explorer opens it natively, and unzipping gives the store back
246
+ byte for byte. patchworks reads a bundle wherever it takes a store path,
247
+ so the viewer works on one directly — same layers, same calibration,
248
+ same auto-loaded label groups:
249
+
250
+ ```bash
251
+ pixi run -e viewer napari /path/to/image.zarr.zip
252
+ ``` It is written `ZIP_STORED` — the chunks are already
253
+ zstd-compressed, so deflating them again would cost a full pass to save
254
+ almost nothing — and entry by entry, so memory stays flat.
255
+
256
+ **Automatically, at the end of a run.** Add a `bundle:` block to the
257
+ multi config and `pixi run multi-slurm` packs the store itself, once
258
+ every segmentation *and* relation has succeeded:
259
+
260
+ ```yaml
261
+ # config/multi.yaml
262
+ bundle:
263
+ format: "zip" # or "iso"; omit the block for no bundle
264
+ qos: "1day" # `time` must stay under this QOS's MaxWall
265
+ ```
266
+
267
+ Or `--bundle zip` for a one-off. Under `--profile` it is submitted as
268
+ its own job, for the same reason the occupancy and relate steps are. A
269
+ failed relation skips it deliberately: a bundle of a half-finished run
270
+ would look complete while missing workbooks. If the packing itself
271
+ fails, nothing is lost — the store is complete on disk and
272
+ `pixi run zip` retries just that step.
273
+
274
+ **`.iso` mounts as a read-only drive**, which `.zip` does not, so the
275
+ store can be opened in place by anything that takes a path. The cost is
276
+ that it needs `xorriso`, `genisoimage` or `mkisofs` on the system, and
277
+ none of them is on conda-forge, so on a cluster without one this option
278
+ is simply unavailable.
279
+
280
+ !!! tip "Details of the `.iso` format"
281
+ A zarr store is tens of thousands of small files, which copies slowly
282
+ everywhere and badly to Windows. `pixi run iso` packs a finished store
283
+ into one image that mounts read-only with a double-click:
284
+
285
+ ```bash
286
+ pixi run iso --store /path/to/image.zarr --dry-run # report, write nothing
287
+ pixi run iso --store /path/to/image.zarr
288
+ ```
289
+
290
+ Mount it: **Windows** right-click → Mount; **macOS** double-click or
291
+ `hdiutil attach`; **Linux** `sudo mount -o loop <iso> /mnt/point`. The
292
+ store inside opens with napari/patchworks unchanged — same bytes, just
293
+ packaged.
294
+
295
+ **Submit it rather than running it on a login node**: packing a store
296
+ reads every file and writes the whole image, and a shared login node
297
+ kills a process that large with no message — the run just returns to the
298
+ prompt partway through, leaving no usable `.iso`. Same QOS ceiling as
299
+ everything else:
300
+
301
+ ```bash
302
+ sbatch --qos=1day --time=12:00:00 --cpus-per-task=4 --mem=8G \
303
+ --wrap "cd $PWD && pixi run iso --store /path/to/image.zarr"
304
+ ```
305
+
306
+ It needs **`xorriso`, `genisoimage` or `mkisofs`** from the system —
307
+ whichever is present. None of them is on conda-forge (it carries no
308
+ ISO-building C tool), so this is deliberately *not* a pixi dependency:
309
+ declaring one makes `pixi install` unsolvable and takes every
310
+ environment down with it. Check with
311
+ `which xorriso genisoimage mkisofs`, and try `module avail` before
312
+ asking an admin.
313
+
314
+ Do not substitute a pure-Python ISO builder such as `pycdlib`: those
315
+ assemble the whole image in RAM and die partway through a store with
316
+ tens of thousands of files. The C tools stream straight to the output
317
+ file, so the file count costs no memory at all. The
318
+ image is ISO-9660 level 3 with Rock Ridge *and* Joliet and deep-directory
319
+ relocation disabled, because a zarr v3 chunk path nests deeper than
320
+ ISO-9660's 8 levels — without that, Windows sees a tree flattened into
321
+ `RR_MOVED` that still looks like it copied correctly.
322
+
176
323
  !!! tip "Dropping objects by size with `min_volume`/`max_volume`"
177
324
  `min_volume: N` drops any object smaller than `N` µm³; `max_volume: N`
178
325
  drops any object larger than `N` µm³ (e.g. several objects merged into
@@ -333,6 +480,41 @@ set-resources:
333
480
  runtime: 240
334
481
  ```
335
482
 
483
+ !!! warning "`runtime` is capped by the QOS, and sbatch rejects — it does not truncate"
484
+ Every `runtime:` above is bounded by the QOS the job lands in. Ask for
485
+ more and **`sbatch` refuses the job**, so it never starts:
486
+
487
+ ```
488
+ sbatch: error: QOSMaxWallDurationPerJobLimit
489
+ sbatch: error: Batch job submission failed: Job violates accounting/QOS policy
490
+ ```
491
+
492
+ The giveaway is an **empty log file**: the rule's `logs/<rule>.log` is
493
+ created by Snakemake but nothing ever writes to it, because the script
494
+ never ran. The reason appears only in the submission error, not in the
495
+ log.
496
+
497
+ Raising `runtime` past the cap therefore does not work on its own — you
498
+ have to request a QOS that allows it, per rule:
499
+
500
+ ```yaml
501
+ set-resources:
502
+ merge:
503
+ qos: "1day" # a QOS your account may use on that partition
504
+ runtime: 720
505
+ ```
506
+
507
+ List what you may ask for, and each one's ceiling:
508
+
509
+ ```bash
510
+ sacctmgr show assoc user=$USER format=partition,qos%40
511
+ sacctmgr show qos format=name,maxwall
512
+ ```
513
+
514
+ On scicore the default QOS allows 6h, which is why the shipped profile
515
+ keeps every CPU rule at or below `runtime: 360` and gives the long
516
+ `segment` rule an explicit `qos:`.
517
+
336
518
  Then launch (from a login node — Snakemake submits and watches the jobs):
337
519
 
338
520
  ```bash
@@ -646,9 +828,30 @@ abort the others; you get a per-config status and a non-zero exit.
646
828
  caps the wall time below `--relate-time` — `srun` fails immediately with
647
829
  `QOSMaxWallDurationPerJobLimit` when that happens; `sacctmgr -p show
648
830
  assoc user=$USER` and `sacctmgr -p show qos` list what's available and
649
- each one's `MaxWall`. Under plain `multi` (no `--profile`), relations
831
+ each one's `MaxWall`.
832
+
833
+ The same settings can live in the multi config as a `relate:` block, so
834
+ the shipped `pixi run multi-slurm` task keeps working without extra
835
+ flags:
836
+
837
+ ```yaml
838
+ # config/multi.yaml
839
+ relate:
840
+ qos: "1day"
841
+ time: 720 # minutes, per pair; must stay under that QOS's MaxWall
842
+ ```
843
+
844
+ A `--relate-*` flag overrides the block for that one key; anything the
845
+ block does not set keeps its default. An unknown key there is an error
846
+ rather than silently ignored, since a typo would otherwise run with the
847
+ default you meant to replace. Under plain `multi` (no `--profile`), relations
650
848
  still run locally, in-process, one after another, as before.
651
849
 
850
+ Each pair logs its shape, chunk count and object count before it starts,
851
+ then a progress line roughly once a minute (`label_relations: 412/3,600
852
+ (11%) after 7m, ~55m left`), so a long relation is distinguishable from a
853
+ hung one in `logs/relate/<a>_to_<b>.log`.
854
+
652
855
  Because every pair gets its own job, one running long no longer starves
653
856
  the others out of a shared time budget, and a pair that gets killed no
654
857
  longer takes an already-finished sibling's workbook down with it.
@@ -55,6 +55,91 @@ def zarr_compressor_kwargs(zarr_format: int = 3) -> dict:
55
55
  return {}
56
56
 
57
57
 
58
+ def open_zarr_source(
59
+ store_path: Union[str, Path],
60
+ ) -> tuple[Union[str, "zarr.storage.StoreLike"], str]:
61
+ """Resolve a store path, transparently opening a ``.zip`` bundle.
62
+
63
+ A store packed by ``pixi run zip`` is one file holding
64
+ ``<name>.zarr/...``. zarr reads it in place through a ``ZipStore``, so
65
+ nothing has to be unpacked first -- but a plain path string does not,
66
+ and every reader here takes a path. This returns what zarr and dask
67
+ should actually be handed, plus the prefix to prepend to a component.
68
+
69
+ Parameters
70
+ ----------
71
+ store_path : str or Path
72
+ A ``.zarr`` directory, or a ``.zip`` bundle containing one.
73
+
74
+ Returns
75
+ -------
76
+ tuple
77
+ ``(source, prefix)``. For a directory, the path and ``""``. For a
78
+ bundle, an open read-only ``ZipStore`` and the store's name inside
79
+ it, so a component is addressed as ``f"{prefix}/{component}"``.
80
+
81
+ Raises
82
+ ------
83
+ ValueError
84
+ If a ``.zip`` does not hold exactly one top-level store.
85
+ """
86
+ text = str(store_path)
87
+ if ".zip" not in text:
88
+ return text, ""
89
+
90
+ import zipfile
91
+
92
+ # The bundle may be addressed with a group path after it, e.g.
93
+ # "scan.zarr.zip/labels/cells" -- callers build those by string-joining.
94
+ head, _, tail = text.partition(".zip")
95
+ archive_path = head + ".zip"
96
+ with zipfile.ZipFile(archive_path) as archive:
97
+ tops = {
98
+ name.split("/", 1)[0] for name in archive.namelist() if "/" in name
99
+ }
100
+ if len(tops) != 1:
101
+ raise ValueError(
102
+ f"{archive_path} must contain exactly one top-level store; found "
103
+ f"{sorted(tops) or 'nothing'}. Bundles written by "
104
+ "`pixi run zip` always do."
105
+ )
106
+ prefix = tops.pop()
107
+ inner = tail.strip("/")
108
+ if inner:
109
+ prefix = f"{prefix}/{inner}"
110
+ return zarr.storage.ZipStore(archive_path, mode="r"), prefix
111
+
112
+
113
+ def open_group_any(path: Union[str, Path], mode: str = "r"):
114
+ """``zarr.open_group`` that also accepts a path *inside* a .zip bundle.
115
+
116
+ Callers build group paths by string-joining (``f"{store}/labels"``),
117
+ which a bundle breaks: the archive is a file, not a directory. Split on
118
+ the ``.zip`` instead, so ``bundle.zip/labels/cells`` resolves to the
119
+ right group inside it.
120
+ """
121
+ source, prefix = open_zarr_source(path)
122
+ return zarr.open_group(source, path=prefix, mode=mode)
123
+
124
+
125
+ def from_zarr_any(path: Union[str, Path], component: str | None = None):
126
+ """``dask.array.from_zarr`` that also accepts a .zip bundle."""
127
+ import dask.array as _da
128
+
129
+ source, prefix = open_zarr_source(path)
130
+ inner = _component(prefix, component) if component else prefix
131
+ return (
132
+ _da.from_zarr(source, component=inner)
133
+ if inner
134
+ else _da.from_zarr(source)
135
+ )
136
+
137
+
138
+ def _component(prefix: str, name: str) -> str:
139
+ """Join a bundle prefix and a component, tolerating an empty prefix."""
140
+ return f"{prefix}/{name}" if prefix else name
141
+
142
+
58
143
  def load_ome_zarr(
59
144
  store_path: Union[str, Path],
60
145
  channel: int | None = 0,
@@ -86,7 +171,8 @@ def load_ome_zarr(
86
171
  >>> arr.shape
87
172
  (128, 2048, 2048)
88
173
  """
89
- root = zarr.open_group(str(store_path), mode="r")
174
+ source, prefix = open_zarr_source(store_path)
175
+ root = zarr.open_group(source, path=prefix, mode="r")
90
176
  # OME-ZARR 0.5 nests under "ome" key; older stores use "multiscales" directly
91
177
  _attrs = dict(root.attrs)
92
178
  _ms = _attrs.get("multiscales") or _attrs.get("ome", {}).get("multiscales")
@@ -104,7 +190,9 @@ def load_ome_zarr(
104
190
  if zarr_ndim > len(chunks):
105
191
  zarr_chunks = (1,) * (zarr_ndim - len(chunks)) + tuple(chunks)
106
192
 
107
- arr = da.from_zarr(str(store_path), component=path, chunks=zarr_chunks)
193
+ arr = da.from_zarr(
194
+ source, component=_component(prefix, path), chunks=zarr_chunks
195
+ )
108
196
  if channel is not None:
109
197
  arr = _select_channel(arr, channel, _ms[0], store_path)
110
198
  return arr
@@ -3,7 +3,8 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  import logging
6
- from concurrent.futures import ThreadPoolExecutor
6
+ import time as _time
7
+ from concurrent.futures import ThreadPoolExecutor, as_completed
7
8
  from pathlib import Path
8
9
  from typing import Union
9
10
 
@@ -11,6 +12,8 @@ import dask.array as da
11
12
  import numpy as np
12
13
 
13
14
  from ._chunks import cpu_allocation
15
+ from ._progress import PROGRESS_INTERVAL_S as _PROGRESS_INTERVAL_S
16
+ from ._progress import log_progress
14
17
 
15
18
  logger = logging.getLogger(__name__)
16
19
 
@@ -97,14 +100,35 @@ def label_relations(
97
100
  total = int(np.prod(n_blocks))
98
101
  nw = n_workers if n_workers is not None else min(4, cpu_allocation())
99
102
 
103
+ logger.info(
104
+ "label_relations: scanning %d chunk(s) of %s with %d worker(s)",
105
+ total,
106
+ "x".join(str(n) for n in a.shape),
107
+ nw,
108
+ )
109
+
100
110
  def _one(flat_idx: int) -> np.ndarray:
101
111
  idx = np.unravel_index(flat_idx, n_blocks)
102
112
  return _chunk_pairs(
103
113
  np.asarray(a.blocks[idx]), np.asarray(b.blocks[idx])
104
114
  )
105
115
 
116
+ # as_completed, not ex.map: map returns an iterator that yields in
117
+ # submission order, so one slow early chunk withholds every later result
118
+ # and the log stays silent however many have actually finished. This step
119
+ # runs for hours in a batch job where the only question the log has to
120
+ # answer is "working, or hung?".
121
+ started = _time.monotonic()
122
+ last = started
123
+ parts = []
106
124
  with ThreadPoolExecutor(max_workers=nw) as ex:
107
- parts = list(ex.map(_one, range(total)))
125
+ futures = {ex.submit(_one, i): i for i in range(total)}
126
+ for done, future in enumerate(as_completed(futures), start=1):
127
+ parts.append(future.result())
128
+ now = _time.monotonic()
129
+ if now - last >= _PROGRESS_INTERVAL_S or done == total:
130
+ log_progress("label_relations", done, total, started)
131
+ last = now
108
132
 
109
133
  rows = [p for p in parts if p.size]
110
134
  if not rows:
@@ -29,9 +29,8 @@ from pathlib import Path
29
29
  from typing import Any, Union
30
30
 
31
31
  import dask.array as da
32
- import zarr
33
32
 
34
- from .._io import load_ome_zarr
33
+ from .._io import from_zarr_any, load_ome_zarr, open_group_any
35
34
  from .ome_zarr import read_ngff_attr
36
35
 
37
36
  logger = logging.getLogger(__name__)
@@ -84,7 +83,7 @@ def _has_multiscales(path: Union[str, Path]) -> bool:
84
83
  bool
85
84
  True if the group has a ``multiscales`` attribute.
86
85
  """
87
- root = zarr.open_group(str(path), mode="r")
86
+ root = open_group_any(path)
88
87
  return read_ngff_attr(root.attrs, "multiscales") is not None
89
88
 
90
89
 
@@ -105,7 +104,7 @@ def _multiscale_levels(
105
104
  list of da.Array
106
105
  One lazy array per resolution level.
107
106
  """
108
- root = zarr.open_group(str(path), mode="r")
107
+ root = open_group_any(path)
109
108
  datasets = read_ngff_attr(root.attrs, "multiscales")[0]["datasets"]
110
109
  return [
111
110
  load_ome_zarr(path, channel=channel, level=i)
@@ -204,7 +203,7 @@ def _label_hint(path: Union[str, Path]) -> dict[str, Any]:
204
203
  ``metadata=``/merge into a bigger dict either way.
205
204
  """
206
205
  try:
207
- attrs = zarr.open_group(str(path), mode="r").attrs
206
+ attrs = open_group_any(path).attrs
208
207
  except Exception:
209
208
  return {}
210
209
  if "n_objects" not in attrs:
@@ -229,7 +228,7 @@ def _inner_label_names(store: Union[str, Path]) -> list[str]:
229
228
  Registered label-image names (empty if there are none).
230
229
  """
231
230
  try:
232
- grp = zarr.open_group(f"{store}/labels", mode="r")
231
+ grp = open_group_any(f"{store}/labels")
233
232
  except Exception:
234
233
  return []
235
234
  return list(read_ngff_attr(grp.attrs, "labels", []) or [])
@@ -256,9 +255,9 @@ def _resolve_labels(
256
255
  if _has_multiscales(source):
257
256
  levels = _multiscale_levels(source, None)
258
257
  return [lvl.astype("int32") for lvl in levels]
259
- arr = da.from_zarr(str(source), component=component)
258
+ arr = from_zarr_any(source, component=component)
260
259
  elif isinstance(source, (str, Path)):
261
- arr = da.from_zarr(str(source))
260
+ arr = from_zarr_any(source)
262
261
  else:
263
262
  arr = da.asarray(source)
264
263
  return arr.astype("int32")
@@ -58,16 +58,17 @@ from itertools import product as _iproduct
58
58
  from pathlib import Path
59
59
  from typing import Union
60
60
 
61
+ import dask
61
62
  import dask.array as da
62
63
  import numpy as np
63
64
  import zarr
64
65
 
65
- from .._chunks import cpu_allocation
66
+ from .._chunks import cpu_allocation, safe_worker_count
66
67
  from .._progress import (
67
68
  PROGRESS_INTERVAL_S as _PROGRESS_INTERVAL_S,
68
69
  )
69
70
  from .._progress import dask_progress, log_progress
70
- from .._io import load_ome_zarr, zarr_compressor_kwargs
71
+ from .._io import load_ome_zarr, open_group_any, zarr_compressor_kwargs
71
72
 
72
73
  logger = logging.getLogger(__name__)
73
74
 
@@ -774,7 +775,26 @@ def _to_zarr_level(
774
775
  shards=sh,
775
776
  dtype=arr.dtype,
776
777
  )
777
- with ctx:
778
+ # Rechunking to the shard size is the one place this module hands work to
779
+ # dask's scheduler, and dask defaults to one thread per *machine* core --
780
+ # on a 128-core node that is 128 tasks each holding a whole shard
781
+ # (~512 MB by default), inside whatever cgroup the job was granted. Bound
782
+ # it here rather than in each caller, so a direct API call is as safe as
783
+ # the workflow's.
784
+ shard_nbytes = int(np.prod(sh)) * arr.dtype.itemsize
785
+ n_workers = max(
786
+ 1,
787
+ min(
788
+ cpu_allocation(),
789
+ safe_worker_count(shard_nbytes, fn_overhead=3),
790
+ ),
791
+ )
792
+ logger.debug(
793
+ "resharding with %d worker(s) for %.0f MB shards",
794
+ n_workers,
795
+ shard_nbytes / 1024**2,
796
+ )
797
+ with ctx, dask.config.set(scheduler="threads", num_workers=n_workers):
778
798
  arr.rechunk(sh).store(z, lock=True, compute=True)
779
799
 
780
800
 
@@ -1124,7 +1144,9 @@ def _read_zarr_calibration(store: Union[str, Path], axes: str) -> PixelSize:
1124
1144
  the store has no multiscales metadata).
1125
1145
  """
1126
1146
  try:
1127
- root = zarr.open_group(str(store), mode="r")
1147
+ # open_group_any, not zarr.open_group: the store may be a .zip
1148
+ # bundle, possibly with a group path after it.
1149
+ root = open_group_any(store)
1128
1150
  ms = read_ngff_attr(root.attrs, "multiscales")[0]
1129
1151
  ax = [a["name"] for a in ms["axes"]]
1130
1152
  scale = ms["datasets"][0]["coordinateTransformations"][0]["scale"]
@@ -104,3 +104,78 @@ def test_label_hint_empty_without_n_objects(tmp_path):
104
104
  def test_label_hint_missing_store_returns_empty():
105
105
  """A path that doesn't exist (or isn't a label group) just yields {}."""
106
106
  assert nplugin._label_hint("/no/such/store.zarr") == {}
107
+
108
+
109
+ def test_a_zip_bundle_opens_exactly_like_the_directory(tmp_path):
110
+ """`pixi run napari` must work on a bundle, not only on a directory.
111
+
112
+ Every reader here builds group paths by string-joining
113
+ (f"{store}/labels/{name}"), which a bundle breaks: the archive is a
114
+ file. Without this, packing a store made it unviewable.
115
+ """
116
+ import zipfile
117
+ from pathlib import Path
118
+
119
+ import zarr
120
+
121
+ from patchworks.plugins import napari as napari_plugin
122
+ from patchworks.plugins.ome_zarr import (
123
+ read_pixel_size,
124
+ register_labels,
125
+ to_ome_zarr,
126
+ )
127
+
128
+ image = np.arange(4 * 64 * 64, dtype="uint16").reshape(4, 64, 64)
129
+ store = to_ome_zarr(
130
+ image,
131
+ tmp_path / "image.zarr",
132
+ axes="zyx",
133
+ n_levels=3,
134
+ chunks=(2, 32, 32),
135
+ pixel_size={"z": 0.24, "y": 0.108, "x": 0.108},
136
+ progress=False,
137
+ )
138
+ labels = np.zeros((4, 64, 64), dtype="uint32")
139
+ labels[1:3, 10:30, 10:30] = 7
140
+ group = zarr.open_group(f"{store}/labels/cilia", mode="a")
141
+ base = group.create_array(
142
+ "0", shape=labels.shape, chunks=(2, 32, 32), dtype="uint32"
143
+ )
144
+ base[:] = labels
145
+ register_labels(store, "cilia", n_levels=3, progress=False, n_objects=1)
146
+
147
+ bundle = tmp_path / "image.zarr.zip"
148
+ with zipfile.ZipFile(bundle, "w", zipfile.ZIP_STORED) as archive:
149
+ for item in sorted(Path(store).rglob("*")):
150
+ if item.is_file():
151
+ archive.write(item, item.relative_to(Path(store).parent))
152
+
153
+ for source in (str(store), str(bundle)):
154
+ label_group = f"{source}/labels/cilia"
155
+ assert napari_plugin._inner_label_names(source) == ["cilia"], source
156
+ assert napari_plugin._has_multiscales(source), source
157
+ assert [
158
+ tuple(level.shape)
159
+ for level in napari_plugin._multiscale_levels(source, None)
160
+ ] == [(4, 64, 64), (4, 32, 32), (4, 16, 16)], source
161
+ assert [
162
+ tuple(level.shape)
163
+ for level in napari_plugin._multiscale_levels(label_group, None)
164
+ ] == [(4, 64, 64), (4, 32, 32), (4, 16, 16)], source
165
+ assert napari_plugin._label_hint(label_group)["n_objects"] == 1
166
+ assert napari_plugin._pyramid_calibration(label_group, 3) == (
167
+ [0.24, 0.108, 0.108],
168
+ ["micrometer"] * 3,
169
+ ), source
170
+ assert read_pixel_size(source) == {
171
+ "z": 0.24,
172
+ "y": 0.108,
173
+ "x": 0.108,
174
+ }, source
175
+
176
+ # And the pixel data itself round-trips.
177
+ from patchworks import load_ome_zarr
178
+
179
+ assert np.array_equal(
180
+ np.asarray(load_ome_zarr(str(bundle), channel=None, level=0)), image
181
+ )