patchworks 2.6.9__tar.gz → 2.6.11__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. {patchworks-2.6.9 → patchworks-2.6.11}/PKG-INFO +1 -1
  2. {patchworks-2.6.9 → patchworks-2.6.11}/docs/api/plugins/cellpose.md +2 -0
  3. patchworks-2.6.11/docs/api/volume_filter.md +9 -0
  4. {patchworks-2.6.9 → patchworks-2.6.11}/docs/examples/cellpose_3d.md +11 -2
  5. {patchworks-2.6.9 → patchworks-2.6.11}/docs/examples/cellpose_3d.py +2 -2
  6. {patchworks-2.6.9 → patchworks-2.6.11}/docs/examples/dog.md +15 -0
  7. {patchworks-2.6.9 → patchworks-2.6.11}/docs/guide/merging.md +53 -0
  8. {patchworks-2.6.9 → patchworks-2.6.11}/docs/guide/snakemake.md +58 -13
  9. {patchworks-2.6.9 → patchworks-2.6.11}/mkdocs.yml +1 -0
  10. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/__init__.py +2 -0
  11. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/_volume_filter.py +68 -17
  12. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/plugins/dog.py +20 -15
  13. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_dog.py +18 -6
  14. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_pw.py +41 -0
  15. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_run_multi.py +165 -0
  16. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_volume_filter.py +58 -0
  17. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/config/config_cilia.yaml +7 -5
  18. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/config/config_cyto.yaml +7 -0
  19. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/scripts/_pw.py +23 -0
  20. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/scripts/merge.py +17 -7
  21. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/scripts/relate.py +39 -0
  22. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/scripts/run_multi.py +125 -30
  23. {patchworks-2.6.9 → patchworks-2.6.11}/.github/workflows/docs.yml +0 -0
  24. {patchworks-2.6.9 → patchworks-2.6.11}/.github/workflows/lint.yml +0 -0
  25. {patchworks-2.6.9 → patchworks-2.6.11}/.github/workflows/release.yml +0 -0
  26. {patchworks-2.6.9 → patchworks-2.6.11}/.gitignore +0 -0
  27. {patchworks-2.6.9 → patchworks-2.6.11}/.markdownlint-cli2.yaml +0 -0
  28. {patchworks-2.6.9 → patchworks-2.6.11}/LICENSE +0 -0
  29. {patchworks-2.6.9 → patchworks-2.6.11}/README.md +0 -0
  30. {patchworks-2.6.9 → patchworks-2.6.11}/cliff.toml +0 -0
  31. {patchworks-2.6.9 → patchworks-2.6.11}/docs/api/chunks.md +0 -0
  32. {patchworks-2.6.9 → patchworks-2.6.11}/docs/api/cluster.md +0 -0
  33. {patchworks-2.6.9 → patchworks-2.6.11}/docs/api/io.md +0 -0
  34. {patchworks-2.6.9 → patchworks-2.6.11}/docs/api/merge_tile_labels.md +0 -0
  35. {patchworks-2.6.9 → patchworks-2.6.11}/docs/api/plugins/dog.md +0 -0
  36. {patchworks-2.6.9 → patchworks-2.6.11}/docs/api/plugins/napari.md +0 -0
  37. {patchworks-2.6.9 → patchworks-2.6.11}/docs/api/plugins/ome_zarr.md +0 -0
  38. {patchworks-2.6.9 → patchworks-2.6.11}/docs/api/postprocess.md +0 -0
  39. {patchworks-2.6.9 → patchworks-2.6.11}/docs/api/relabel.md +0 -0
  40. {patchworks-2.6.9 → patchworks-2.6.11}/docs/api/tile_process.md +0 -0
  41. {patchworks-2.6.9 → patchworks-2.6.11}/docs/assets/logo.png +0 -0
  42. {patchworks-2.6.9 → patchworks-2.6.11}/docs/examples/cellpose_2d.md +0 -0
  43. {patchworks-2.6.9 → patchworks-2.6.11}/docs/examples/cellpose_2d.py +0 -0
  44. {patchworks-2.6.9 → patchworks-2.6.11}/docs/examples/custom.md +0 -0
  45. {patchworks-2.6.9 → patchworks-2.6.11}/docs/examples/custom_method.py +0 -0
  46. {patchworks-2.6.9 → patchworks-2.6.11}/docs/examples/dog.py +0 -0
  47. {patchworks-2.6.9 → patchworks-2.6.11}/docs/examples/standalone_merge.md +0 -0
  48. {patchworks-2.6.9 → patchworks-2.6.11}/docs/examples/stardist.md +0 -0
  49. {patchworks-2.6.9 → patchworks-2.6.11}/docs/examples/stardist_2d.py +0 -0
  50. {patchworks-2.6.9 → patchworks-2.6.11}/docs/getting_started.md +0 -0
  51. {patchworks-2.6.9 → patchworks-2.6.11}/docs/guide/custom_segmentation.md +0 -0
  52. {patchworks-2.6.9 → patchworks-2.6.11}/docs/guide/gpu_distributed.md +0 -0
  53. {patchworks-2.6.9 → patchworks-2.6.11}/docs/guide/label_relations.md +0 -0
  54. {patchworks-2.6.9 → patchworks-2.6.11}/docs/guide/measurements.md +0 -0
  55. {patchworks-2.6.9 → patchworks-2.6.11}/docs/guide/ome_zarr_napari.md +0 -0
  56. {patchworks-2.6.9 → patchworks-2.6.11}/docs/guide/performance.md +0 -0
  57. {patchworks-2.6.9 → patchworks-2.6.11}/docs/guide/pitfalls.md +0 -0
  58. {patchworks-2.6.9 → patchworks-2.6.11}/docs/guide/skip_empty.md +0 -0
  59. {patchworks-2.6.9 → patchworks-2.6.11}/docs/guide/tiling.md +0 -0
  60. {patchworks-2.6.9 → patchworks-2.6.11}/docs/index.md +0 -0
  61. {patchworks-2.6.9 → patchworks-2.6.11}/pyproject.toml +0 -0
  62. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/_chunks.py +0 -0
  63. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/_cluster.py +0 -0
  64. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/_core.py +0 -0
  65. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/_distributed.py +0 -0
  66. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/_gpu.py +0 -0
  67. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/_io.py +0 -0
  68. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/_merge.py +0 -0
  69. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/_notify.py +0 -0
  70. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/_occupancy.py +0 -0
  71. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/_postprocess.py +0 -0
  72. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/_progress.py +0 -0
  73. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/_relabel.py +0 -0
  74. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/_relations.py +0 -0
  75. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/plugins/__init__.py +0 -0
  76. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/plugins/cellpose.py +0 -0
  77. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/plugins/napari.py +0 -0
  78. {patchworks-2.6.9 → patchworks-2.6.11}/src/patchworks/plugins/ome_zarr.py +0 -0
  79. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_allocation.py +0 -0
  80. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_cellpose.py +0 -0
  81. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_core.py +0 -0
  82. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_distributed.py +0 -0
  83. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_gpu.py +0 -0
  84. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_napari.py +0 -0
  85. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_notify.py +0 -0
  86. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_occupancy.py +0 -0
  87. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_ome_zarr.py +0 -0
  88. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_postprocess.py +0 -0
  89. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_progress.py +0 -0
  90. {patchworks-2.6.9 → patchworks-2.6.11}/tests/test_relations.py +0 -0
  91. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/README.md +0 -0
  92. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/Snakefile +0 -0
  93. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/config/common.yaml +0 -0
  94. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/config/config.yaml +0 -0
  95. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/config/config_nuclei.yaml +0 -0
  96. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/config/multi.yaml +0 -0
  97. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/pixi.toml +0 -0
  98. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/profile/slurm/config.yaml +0 -0
  99. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/rules/common.smk +0 -0
  100. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/rules/convert.smk +0 -0
  101. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/rules/merge.smk +0 -0
  102. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/rules/segment.smk +0 -0
  103. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/scripts/build_occupancy.py +0 -0
  104. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/scripts/convert.py +0 -0
  105. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/scripts/fetch_model.py +0 -0
  106. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/scripts/prepare_tiles.py +0 -0
  107. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/scripts/segment_tile.py +0 -0
  108. {patchworks-2.6.9 → patchworks-2.6.11}/workflow/scripts/view.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: patchworks
3
- Version: 2.6.9
3
+ Version: 2.6.11
4
4
  Summary: Tiled processing of arbitrarily large images with globally consistent labels
5
5
  Project-URL: Homepage, https://github.com/imcf/patchworks
6
6
  Project-URL: Issues, https://github.com/imcf/patchworks/issues
@@ -1,3 +1,5 @@
1
1
  # Cellpose plugin
2
2
 
3
3
  ::: patchworks.plugins.cellpose.cellpose_fn
4
+
5
+ ::: patchworks.plugins.cellpose.cellpose_anisotropy
@@ -0,0 +1,9 @@
1
+ # Volume filtering
2
+
3
+ ::: patchworks.filter_labels_by_size
4
+
5
+ ::: patchworks.voxel_volume
6
+
7
+ ::: patchworks.min_voxels_for_volume
8
+
9
+ ::: patchworks.max_voxels_for_volume
@@ -14,19 +14,19 @@ plane orientations and takes a 3-D consensus.
14
14
  from functools import partial
15
15
  from patchworks import auto_tile_shape_cellpose, make_local_cluster, tile_process
16
16
  from patchworks.plugins.cellpose import cellpose_fn
17
+ from patchworks.plugins.ome_zarr import read_pixel_size
17
18
 
18
19
  IMAGE = "image.zarr"
19
20
  OUTPUT = "labels_3d.zarr"
20
21
  CHANNEL = 0
21
22
  DIAMETER = 20 # pixels
22
- ANISOTROPY = 3.0 # z_spacing / xy_spacing
23
23
 
24
24
  fn = cellpose_fn(
25
25
  "cyto3",
26
26
  gpu=True,
27
27
  do_3D=True,
28
28
  diameter=DIAMETER,
29
- anisotropy=ANISOTROPY,
29
+ voxel_size=read_pixel_size(IMAGE), # -> anisotropy = z / lateral
30
30
  )
31
31
 
32
32
  # Tile shape: full z, xy tiled for memory
@@ -59,6 +59,15 @@ finally:
59
59
  cluster.close()
60
60
  ```
61
61
 
62
+ !!! tip "Anisotropy is derived from the calibration, not retyped"
63
+ `do_3D` without `anisotropy` assumes isotropic voxels, which fragments
64
+ objects across z for any real (anisotropic) dataset. Passing
65
+ `voxel_size` derives it as `z / lateral` via
66
+ [`cellpose_anisotropy`](../api/plugins/cellpose.md) instead of keeping a
67
+ second, driftable copy of the calibration in code. An explicit
68
+ `anisotropy=` still wins if you pass one. The Snakemake workflow does
69
+ this automatically — see [Configure the run](../guide/snakemake.md#3-configure-the-run).
70
+
62
71
  ## Memory notes
63
72
 
64
73
  In `do_3D=True` mode, each tile has shape `(z_full, y_tile, x_tile)`.
@@ -12,19 +12,19 @@ from patchworks import (
12
12
  tile_process,
13
13
  )
14
14
  from patchworks.plugins.cellpose import cellpose_fn
15
+ from patchworks.plugins.ome_zarr import read_pixel_size
15
16
 
16
17
  IMAGE = "image.zarr"
17
18
  OUTPUT = "labels_3d.zarr"
18
19
  CHANNEL = 0
19
20
  DIAMETER = 20 # pixels
20
- ANISOTROPY = 3.0 # z-spacing / xy-spacing
21
21
 
22
22
  fn = cellpose_fn(
23
23
  "cyto3",
24
24
  gpu=True,
25
25
  do_3D=True,
26
26
  diameter=DIAMETER,
27
- anisotropy=ANISOTROPY,
27
+ voxel_size=read_pixel_size(IMAGE), # -> anisotropy = z / lateral
28
28
  )
29
29
 
30
30
  tile_fn = partial(
@@ -102,6 +102,21 @@ decon_kwargs=dict(psf=psf, dxpsf=0.05, dzpsf=0.1, wavelength=525, ...)
102
102
  so edge tiles keep enough context (a plain intensity/threshold halo is
103
103
  too thin).
104
104
 
105
+ !!! note "cudaDecon can return a smaller volume than it was given"
106
+ It rounds each axis down to an FFT-efficient length — e.g. a
107
+ `(32, 1084, 1084)` tile comes back `(32, 1080, 1080)`, because
108
+ `1080 = 2³·3³·5` while `1084 = 4·271` — and trims the excess off the
109
+ **high end**, leaving voxel `(0, 0, 0)` where it was. patchworks restores
110
+ the input shape before the DoG step (one label per input voxel is
111
+ required) anchored at that origin, and logs a WARNING with both shapes.
112
+
113
+ Anchoring matters: restoring it *centred* instead moves every voxel by
114
+ `excess // 2` — a 2 px y/x shift for the tile above, identical on every
115
+ tile. That is invisible on a cell tens of voxels wide and obvious on a
116
+ cilium a few voxels wide, which is how it was eventually caught. If the
117
+ logged difference is more than a few voxels, the PSF or the voxel sizes
118
+ are wrong.
119
+
105
120
  ## Growing the labels afterwards
106
121
 
107
122
  DoG spots/threads are often thin — grow each label by a few pixels with
@@ -133,6 +133,59 @@ merged = merge_tile_labels(
133
133
  )
134
134
  ```
135
135
 
136
+ ## Filtering by size after merge
137
+
138
+ Once labels are globally consistent, [`filter_labels_by_size`](../api/volume_filter.md)
139
+ can drop objects outside a voxel-count range, in place — too small, too large,
140
+ or both:
141
+
142
+ ```python
143
+ from patchworks import filter_labels_by_size, merge_tile_labels
144
+
145
+ merged = merge_tile_labels("stage.zarr", write_to="labels.zarr", sequential_labels=True)
146
+ # drop anything under 500 voxels, over 50000, or both -- give either bound alone
147
+ n_kept, n_removed = filter_labels_by_size(
148
+ "labels.zarr", "labels", min_voxels=500, max_voxels=50000
149
+ )
150
+ ```
151
+
152
+ This has to run **after** the merge, not per tile: a tile only sees whatever
153
+ fragment of an object landed inside its own bounds, so a per-tile filter would
154
+ judge (and possibly drop) an object crossing a tile boundary as if it were
155
+ only that fragment's size — including judging it too *large*, for a
156
+ `max_voxels` filter, when several separate objects in one tile would in fact
157
+ merge back into one across the boundary.
158
+
159
+ Like the merge itself, it is a two-pass streaming zarr scan — the array never
160
+ has to fit in RAM. `relabel=True` (the default) folds the size filter into
161
+ the same lookup table that renumbers survivors to a contiguous `1..N` range,
162
+ so dropping out-of-range objects costs no extra pass over the volume beyond
163
+ the scan that already counts them.
164
+
165
+ Physical thresholds (µm³) convert to a voxel count via
166
+ [`min_voxels_for_volume`](../api/volume_filter.md)/[`max_voxels_for_volume`](../api/volume_filter.md),
167
+ using the same `{"z": .., "y": .., "x": ..}` calibration deconvolution and
168
+ Cellpose's `anisotropy` are derived from. The two round in opposite
169
+ directions — `min_voxels_for_volume` rounds up (an object must *reach* the
170
+ threshold), `max_voxels_for_volume` rounds down (an object must not *exceed*
171
+ it):
172
+
173
+ ```python
174
+ from patchworks import max_voxels_for_volume, min_voxels_for_volume
175
+ from patchworks.plugins.ome_zarr import read_pixel_size
176
+
177
+ voxel_size = read_pixel_size("image.zarr")
178
+ min_voxels = min_voxels_for_volume(5.0, voxel_size)
179
+ max_voxels = max_voxels_for_volume(500.0, voxel_size)
180
+ n_kept, n_removed = filter_labels_by_size(
181
+ "labels.zarr", "labels", min_voxels, max_voxels
182
+ )
183
+ ```
184
+
185
+ On the cluster, set `min_volume: 5.0`/`max_volume: 500.0` in the config
186
+ instead — see [Configure the run](snakemake.md#3-configure-the-run). Either
187
+ or both run automatically between `merge` and the pyramid build.
188
+
136
189
  ## Sequential label numbering
137
190
 
138
191
  By default, merged labels are globally unique but may be **gappy** — boundary
@@ -68,12 +68,16 @@ method: "cellpose" # "cellpose" (GPU), "threshold" (no GPU), "custom
68
68
  label_name: "cellpose" # name under image.zarr/labels/
69
69
  dilate: 0 # optional: pixels to grow labels by, any method
70
70
  dilate_gpu: false # dilate via cupy instead of scipy (needs a GPU)
71
+ min_volume: null # optional: drop objects smaller than this many µm³
72
+ max_volume: null # optional: drop objects larger than this many µm³
71
73
  cellpose:
72
74
  model: "cyto3"
73
75
  diameter: 30
74
76
  do_3D: true
75
77
  gpu: true
76
78
  # extra model.eval() kwargs, e.g. flow_threshold: 0.4
79
+ # anisotropy: 2.2 # optional: overrides the value derived automatically
80
+ # # from image.zarr's calibration for do_3D (see tip below)
77
81
 
78
82
  # label pyramid
79
83
  pyramid_levels: 5
@@ -94,6 +98,29 @@ sequential_labels: true # renumber labels to a contiguous 1..N
94
98
  labels afterwards](custom_segmentation.md#growing-labels-afterwards-dilation)
95
99
  for how it works and the equivalent direct-API call.
96
100
 
101
+ !!! tip "Dropping objects by size with `min_volume`/`max_volume`"
102
+ `min_volume: N` drops any object smaller than `N` µm³; `max_volume: N`
103
+ drops any object larger than `N` µm³ (e.g. several objects merged into
104
+ one blob). Set either, both, or neither (`null`, the default, disables
105
+ each). Both run once on the **fully merged** image — not per tile, where
106
+ an object crossing a tile boundary would look smaller or larger than it
107
+ really is. Runs after `merge` and before the pyramid is built, so every
108
+ pyramid level reflects the filtered result, and needs `image.zarr` to
109
+ carry a pixel size (the same calibration deconvolution's voxel sizes and
110
+ Cellpose's `anisotropy` are derived from — see the tip below); an
111
+ uncalibrated store raises rather than silently skipping the filter. See
112
+ [Filtering by size after merge](merging.md#filtering-by-size-after-merge)
113
+ for the equivalent direct-API call.
114
+
115
+ !!! tip "3-D anisotropy is derived automatically"
116
+ Cellpose's `do_3D` assumes isotropic voxels unless told otherwise —
117
+ without an `anisotropy`, a real (anisotropic) dataset gets objects
118
+ fragmented or distorted across z. `segment` now derives it from
119
+ `image.zarr`'s own calibration (`z` voxel size ÷ lateral voxel size)
120
+ whenever `do_3D: true` and `cellpose.anisotropy` isn't set explicitly, so
121
+ there's usually nothing to configure. Set `anisotropy:` yourself in the
122
+ `cellpose:` block to override it.
123
+
97
124
  !!! tip "Tile size vs runtime"
98
125
  `tile_shape: "auto"` sizes each tile to your GPU's VRAM. Smaller tiles =
99
126
  more (faster) jobs; very large 3-D tiles are slow. Keep `do_3D: false` (2-D
@@ -417,10 +444,14 @@ results/image.zarr/labels/cyto_labels/
417
444
 
418
445
  !!! tip "The relate step's log"
419
446
  Unlike `prepare`/`segment`/`merge`, the relate step isn't a Snakemake
420
- rule, so it doesn't get a `log:` directive for free. It writes its own
421
- log to `<work_dir>/logs/relate.log` (override with `relate.py --log`),
422
- the same tee-to-file-and-stdout behaviour as the other steps — check
423
- there instead of scrolling back through `srun`'s live output.
447
+ rule, so it doesn't get a `log:` directive for free. Standalone (or
448
+ under plain `multi`), it writes to `<work_dir>/logs/relate.log`
449
+ (override with `relate.py --log`), the same tee-to-file-and-stdout
450
+ behaviour as the other steps. Under `multi-slurm`, where each pair is
451
+ its own concurrent job, `run_multi.py` points each one at its own file
452
+ instead — `<work_dir>/logs/relate/<a>_to_<b>.log` — so concurrent pairs
453
+ don't interleave into one log; check there instead of scrolling back
454
+ through `srun`'s live output.
424
455
 
425
456
  See [Relating labels across segmentations](label_relations.md) for what
426
457
  `label_relations()` returns and how to save it yourself — the cluster
@@ -471,15 +502,29 @@ abort the others; you get a per-config status and a non-zero exit.
471
502
 
472
503
  !!! tip "The relate step runs on the cluster too, under `multi-slurm`"
473
504
  `label_relations()` streams every chunk of two full-resolution label
474
- volumes — real CPU/IO work, not orchestration. Under `multi-slurm` it is
475
- submitted as its own `srun` job (`scripts/relate.py`) instead of running
476
- in the driver process on the login node, the same fix already applied to
477
- the occupancy map. Tune its allocation with `--relate-partition`,
478
- `--relate-mem`, `--relate-cpus` and `--relate-time` (defaults: `scicore`,
479
- `32G`, `8`, `180` minutes) — these are wide-margin guesses, not measured
480
- numbers, so raise them for a very large or very object-dense pair. Under
481
- plain `multi` (no `--profile`), it still runs locally, in-process, as
482
- before.
505
+ volumes — real CPU/IO work, not orchestration. Under `multi-slurm`,
506
+ **each pair in `relations:` is submitted as its own `srun` job**
507
+ (`scripts/relate.py`) instead of running in the driver process on the
508
+ login node, the same fix already applied to the occupancy map. Tune the
509
+ allocation with `--relate-partition`, `--relate-mem`, `--relate-cpus`
510
+ and `--relate-time` (defaults: `scicore`, `32G`, `8`, `180` minutes,
511
+ **per relation**) — these are wide-margin guesses, not measured numbers,
512
+ so raise them for a very large or very object-dense pair; a pair needing
513
+ a chunk-layout rechunk first (see below) is the usual reason one runs
514
+ long. Set `--relate-qos` if your account's default QOS for the partition
515
+ caps the wall time below `--relate-time` — `srun` fails immediately with
516
+ `QOSMaxWallDurationPerJobLimit` when that happens; `sacctmgr -p show
517
+ assoc user=$USER` and `sacctmgr -p show qos` list what's available and
518
+ each one's `MaxWall`. Under plain `multi` (no `--profile`), relations
519
+ still run locally, in-process, one after another, as before.
520
+
521
+ Because every pair gets its own job, one running long no longer starves
522
+ the others out of a shared time budget, and a pair that gets killed no
523
+ longer takes an already-finished sibling's workbook down with it.
524
+ `relate.py` also skips a pair whose `.xlsx` is already newer than both
525
+ labels' merge marker, so **re-running the exact same `multi-slurm`
526
+ command only recomputes what's still missing or stale** — delete a
527
+ specific `.xlsx` yourself to force just that one to recompute.
483
528
 
484
529
  !!! tip "After a killed run"
485
530
  Snakemake only releases its lock on a clean exit, so a run that was killed
@@ -58,6 +58,7 @@ nav:
58
58
  - tile_process: api/tile_process.md
59
59
  - merge_tile_labels: api/merge_tile_labels.md
60
60
  - dilate_labels: api/postprocess.md
61
+ - Volume filtering: api/volume_filter.md
61
62
  - Tile sizing: api/chunks.md
62
63
  - I/O helpers: api/io.md
63
64
  - Relabelling: api/relabel.md
@@ -57,6 +57,7 @@ from ._relabel import relabel_sequential_array, relabel_sequential_zarr
57
57
  from ._relations import label_relations
58
58
  from ._volume_filter import (
59
59
  filter_labels_by_size,
60
+ max_voxels_for_volume,
60
61
  min_voxels_for_volume,
61
62
  voxel_volume,
62
63
  )
@@ -92,5 +93,6 @@ __all__ = [
92
93
  "dilate_labels",
93
94
  "filter_labels_by_size",
94
95
  "min_voxels_for_volume",
96
+ "max_voxels_for_volume",
95
97
  "voxel_volume",
96
98
  ]
@@ -1,17 +1,18 @@
1
- """Drop label objects below a volume threshold, in place, after merge.
1
+ """Drop label objects outside a size range, in place, after merge.
2
2
 
3
3
  Meant to run once, globally, on the fully merged label array -- not per
4
4
  tile, where an object's true size isn't known yet (a tile only sees
5
5
  whatever fragment of it landed inside that tile's bounds, so a per-tile
6
- filter would clip or drop objects that are only small *within one tile*).
6
+ filter would clip or drop objects that are only small, or only large,
7
+ *within one tile*).
7
8
 
8
9
  Two-pass streaming algorithm, mirroring :func:`patchworks.relabel_sequential_zarr`
9
10
  -- safe for arrays far larger than RAM. Pass 1 does a chunk-wise
10
11
  unique+count to get every label's voxel count (bounded memory: a Python
11
12
  dict keyed by label id, not the voxels themselves). Pass 2 builds a LUT
12
- that zeroes labels under the threshold -- optionally renumbering the
13
- survivors to a contiguous range in the same pass -- and applies it chunk by
14
- chunk, writing back into the same store.
13
+ that zeroes labels outside ``[min_voxels, max_voxels]`` -- optionally
14
+ renumbering the survivors to a contiguous range in the same pass -- and
15
+ applies it chunk by chunk, writing back into the same store.
15
16
  """
16
17
 
17
18
  from __future__ import annotations
@@ -86,6 +87,36 @@ def min_voxels_for_volume(
86
87
  return math.ceil(min_volume / voxel_volume(voxel_size))
87
88
 
88
89
 
90
+ def max_voxels_for_volume(
91
+ max_volume: float, voxel_size: "dict[str, float]"
92
+ ) -> int:
93
+ """Convert a physical volume threshold to a voxel count.
94
+
95
+ Rounds down, the mirror image of :func:`min_voxels_for_volume`: an
96
+ object must not *exceed* *max_volume*, so a voxel count whose volume
97
+ would tip past it must not survive.
98
+
99
+ Parameters
100
+ ----------
101
+ max_volume : float
102
+ Maximum object volume to keep, in the same physical units as
103
+ *voxel_size* (micrometers³ for an NGFF calibration).
104
+ voxel_size : dict
105
+ Per-axis physical size -- see :func:`voxel_volume`.
106
+
107
+ Returns
108
+ -------
109
+ int
110
+ Maximum voxel count for an object to survive filtering.
111
+
112
+ Examples
113
+ --------
114
+ >>> max_voxels_for_volume(5.0, {"z": 0.24, "y": 0.10833, "x": 0.10833})
115
+ 1775
116
+ """
117
+ return math.floor(max_volume / voxel_volume(voxel_size))
118
+
119
+
89
120
  def _chunk_slices(shape, chunks):
90
121
  """Every zarr chunk's index expression, in all dimensions.
91
122
 
@@ -105,11 +136,12 @@ def _chunk_slices(shape, chunks):
105
136
  def filter_labels_by_size(
106
137
  store_path: str,
107
138
  component: str,
108
- min_voxels: int,
139
+ min_voxels: "int | None" = None,
140
+ max_voxels: "int | None" = None,
109
141
  *,
110
142
  relabel: bool = True,
111
143
  ) -> "tuple[int, int]":
112
- """Drop label objects smaller than *min_voxels*, in place.
144
+ """Drop label objects outside ``[min_voxels, max_voxels]``, in place.
113
145
 
114
146
  Two-pass streaming scan (see module docstring) -- the array never has
115
147
  to fit in RAM.
@@ -120,15 +152,21 @@ def filter_labels_by_size(
120
152
  Path to the zarr store containing the label array.
121
153
  component : str
122
154
  Array name inside the store to filter in place.
123
- min_voxels : int
124
- Objects with fewer voxels than this are zeroed (dropped). Use
125
- :func:`min_voxels_for_volume` to derive this from a physical
126
- volume and calibration.
155
+ min_voxels : int, optional
156
+ Objects with fewer voxels than this are zeroed (dropped). ``None``
157
+ (default) sets no lower bound. Use :func:`min_voxels_for_volume` to
158
+ derive this from a physical volume and calibration.
159
+ max_voxels : int, optional
160
+ Objects with more voxels than this are zeroed (dropped) -- e.g. a
161
+ segmentation artifact where several objects merged into one giant
162
+ blob. ``None`` (default) sets no upper bound. Use
163
+ :func:`max_voxels_for_volume` to derive this from a physical volume
164
+ and calibration.
127
165
  relabel : bool, optional
128
166
  Renumber the surviving objects to a contiguous ``1..N`` range in
129
- the same LUT that drops the small ones (default ``True``) --
130
- otherwise the removed ids leave permanent gaps and survivors keep
131
- their original ids.
167
+ the same LUT that drops the out-of-range ones (default ``True``)
168
+ -- otherwise the removed ids leave permanent gaps and survivors
169
+ keep their original ids.
132
170
 
133
171
  Returns
134
172
  -------
@@ -150,6 +188,11 @@ def filter_labels_by_size(
150
188
  >>> filter_labels_by_size("labels.zarr", "labels", min_voxels=2) # doctest: +SKIP
151
189
  (1, 1)
152
190
  """
191
+ if min_voxels is None and max_voxels is None:
192
+ raise ValueError(
193
+ "filter_labels_by_size needs min_voxels, max_voxels, or both"
194
+ )
195
+
153
196
  root = zarr.open_group(store_path, mode="r+")
154
197
  z = root[component]
155
198
  slices = _chunk_slices(z.shape, z.chunks)
@@ -162,7 +205,12 @@ def filter_labels_by_size(
162
205
  continue
163
206
  counts[label_id] = counts.get(label_id, 0) + count
164
207
 
165
- kept = sorted(i for i, c in counts.items() if c >= min_voxels)
208
+ kept = sorted(
209
+ i
210
+ for i, c in counts.items()
211
+ if (min_voxels is None or c >= min_voxels)
212
+ and (max_voxels is None or c <= max_voxels)
213
+ )
166
214
  n_kept = len(kept)
167
215
  n_removed = len(counts) - n_kept
168
216
 
@@ -187,12 +235,15 @@ def filter_labels_by_size(
187
235
  block = np.asarray(z[sl])
188
236
  z[sl] = lut[block].astype(out_dtype)
189
237
 
238
+ bounds = "-".join(
239
+ str(v) if v is not None else "" for v in (min_voxels, max_voxels)
240
+ )
190
241
  logger.info(
191
- "filter_labels_by_size: dropped %d/%d object(s) under %d voxels, "
242
+ "filter_labels_by_size: dropped %d/%d object(s) outside [%s] voxels, "
192
243
  "%d remain",
193
244
  n_removed,
194
245
  len(counts),
195
- min_voxels,
246
+ bounds,
196
247
  n_kept,
197
248
  )
198
249
  return n_kept, n_removed
@@ -239,18 +239,23 @@ def _run(block: np.ndarray, dog_dict: dict[str, Any]) -> np.ndarray:
239
239
 
240
240
 
241
241
  def _restore_shape(arr: np.ndarray, shape: tuple[int, ...]) -> np.ndarray:
242
- """Centre *arr* back into an array of *shape*, cropping or edge-padding.
242
+ """Restore *arr* to *shape*, anchored at the origin, cropping or padding.
243
243
 
244
244
  Deconvolution must not change the field of view: patchworks writes the
245
245
  result into a destination slice derived from the tile's geometry, so one
246
246
  label per input voxel is required.
247
247
 
248
- Centring is the right correction for a symmetric crop, which is what
249
- apodisation produces. The discrepancies observed are small (a voxel in z,
250
- a few in x/y) and land inside the halo, which is discarded anyway -- so
251
- the labels that survive the trim are unaffected. It is logged at WARNING
252
- with the exact shapes so a larger, non-symmetric crop cannot pass
253
- silently.
248
+ The alignment is **origin-anchored**, not centred: cudaDecon rounds each
249
+ axis down to an FFT-efficient length (e.g. 1084 -> 1080, since
250
+ 1080 = 2**3 * 3**3 * 5 while 1084 = 4 * 271) and trims the excess off the
251
+ high end, leaving voxel (0, 0, 0) where it was. Re-centring content that
252
+ was never centred shifts every voxel by ``excess // 2`` -- measured at 2
253
+ px in y and x on a real (32, 1084, 1084) tile, in the same direction on
254
+ every tile. That is invisible on a cell tens of voxels across and glaring
255
+ on a cilium a few voxels across, which is exactly how it was found.
256
+
257
+ A mismatch is still logged at WARNING with the exact shapes, since a
258
+ large one means the PSF or voxel sizes are wrong.
254
259
 
255
260
  Parameters
256
261
  ----------
@@ -271,22 +276,22 @@ def _restore_shape(arr: np.ndarray, shape: tuple[int, ...]) -> np.ndarray:
271
276
  (14, 1024)
272
277
  """
273
278
  logger.warning(
274
- "deconvolution returned %s for a %s input; re-centring to the input "
275
- "shape. patchworks needs one label per input voxel. A large or "
276
- "asymmetric difference here would shift labels -- check the PSF and "
277
- "voxel sizes if this is more than a few voxels.",
279
+ "deconvolution returned %s for a %s input; restoring the input shape "
280
+ "from the origin. patchworks needs one label per input voxel. A large "
281
+ "difference here means the PSF or voxel sizes are wrong -- check them "
282
+ "if this is more than a few voxels.",
278
283
  arr.shape,
279
284
  shape,
280
285
  )
281
286
  # Crop first, so an axis that grew is handled before padding the rest.
287
+ # Both keep voxel 0 where it is: cudaDecon trims off the high end (see the
288
+ # docstring), so the low corner is the one landmark known to be unmoved.
282
289
  crop = tuple(
283
- slice((a - s) // 2, (a - s) // 2 + s) if a > s else slice(None)
284
- for a, s in zip(arr.shape, shape)
290
+ slice(0, s) if a > s else slice(None) for a, s in zip(arr.shape, shape)
285
291
  )
286
292
  arr = arr[crop]
287
293
  pad = tuple(
288
- ((s - a) // 2, s - a - (s - a) // 2) if a < s else (0, 0)
289
- for a, s in zip(arr.shape, shape)
294
+ (0, s - a) if a < s else (0, 0) for a, s in zip(arr.shape, shape)
290
295
  )
291
296
  if any(lo or hi for lo, hi in pad):
292
297
  arr = np.pad(arr, pad, mode="edge")
@@ -109,20 +109,32 @@ def test_explicit_decon_kwargs_win_over_the_calibration(monkeypatch):
109
109
  assert captured["dzpsf"] == 0.2
110
110
 
111
111
 
112
- def test_restore_shape_recentres_a_cropped_decon():
112
+ def test_restore_shape_anchors_a_cropped_decon_at_the_origin():
113
113
  """cudaDecon can hand back a smaller volume than it was given.
114
114
 
115
- Observed on a real edge tile: (14, 1024, 1024) in, (13, 1020, 1020) out.
116
- patchworks needs one label per input voxel, so the field of view has to be
117
- restored before the DoG step.
115
+ Observed on a real tile: (32, 1084, 1084) in, (32, 1080, 1080) out --
116
+ each axis rounded down to an FFT-efficient length, with the excess taken
117
+ off the high end. Restoring it *centred* (what this used to do) moved
118
+ every voxel by excess // 2, measured as a 2 px y/x shift on real data:
119
+ invisible on a cell, glaring on a cilium a few voxels across.
118
120
  """
119
121
  from patchworks.plugins.dog import _restore_shape
120
122
 
121
123
  arr = np.arange(13 * 1020 * 1020, dtype="float32").reshape(13, 1020, 1020)
122
124
  out = _restore_shape(arr, (14, 1024, 1024))
123
125
  assert out.shape == (14, 1024, 1024)
124
- # The original content is preserved, centred, not resampled.
125
- assert np.array_equal(out[0:13, 2:1022, 2:1022], arr)
126
+ # Content keeps its original indices -- voxel 0 stays voxel 0.
127
+ assert np.array_equal(out[0:13, 0:1020, 0:1020], arr)
128
+
129
+
130
+ def test_restore_shape_crops_from_the_high_end():
131
+ """The mirror case: an axis that came back too long keeps its low corner."""
132
+ from patchworks.plugins.dog import _restore_shape
133
+
134
+ arr = np.arange(6 * 12, dtype="float32").reshape(6, 12)
135
+ out = _restore_shape(arr, (4, 8))
136
+ assert out.shape == (4, 8)
137
+ assert np.array_equal(out, arr[0:4, 0:8])
126
138
 
127
139
 
128
140
  def test_restore_shape_handles_growth_and_exact_fit():
@@ -80,3 +80,44 @@ def test_validate_config_rejects_a_non_positive_min_volume():
80
80
  validate_config({"method": "threshold", "min_volume": -1.0})
81
81
  with pytest.raises(ValueError, match="min_volume"):
82
82
  validate_config({"method": "threshold", "min_volume": "5"})
83
+
84
+
85
+ def test_validate_config_accepts_a_positive_max_volume():
86
+ from _pw import validate_config
87
+
88
+ validate_config({"method": "threshold", "max_volume": 500.0})
89
+ validate_config({"method": "threshold", "max_volume": None})
90
+
91
+
92
+ def test_validate_config_rejects_a_non_positive_max_volume():
93
+ import pytest
94
+ from _pw import validate_config
95
+
96
+ with pytest.raises(ValueError, match="max_volume"):
97
+ validate_config({"method": "threshold", "max_volume": 0})
98
+ with pytest.raises(ValueError, match="max_volume"):
99
+ validate_config({"method": "threshold", "max_volume": -1.0})
100
+ with pytest.raises(ValueError, match="max_volume"):
101
+ validate_config({"method": "threshold", "max_volume": "5"})
102
+
103
+
104
+ def test_validate_config_accepts_max_volume_above_min_volume():
105
+ from _pw import validate_config
106
+
107
+ validate_config(
108
+ {"method": "threshold", "min_volume": 5.0, "max_volume": 500.0}
109
+ )
110
+
111
+
112
+ def test_validate_config_rejects_max_volume_at_or_below_min_volume():
113
+ import pytest
114
+ from _pw import validate_config
115
+
116
+ with pytest.raises(ValueError, match="max_volume"):
117
+ validate_config(
118
+ {"method": "threshold", "min_volume": 5.0, "max_volume": 5.0}
119
+ )
120
+ with pytest.raises(ValueError, match="max_volume"):
121
+ validate_config(
122
+ {"method": "threshold", "min_volume": 500.0, "max_volume": 5.0}
123
+ )