dask-setup 2.1.0__tar.gz → 2.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. dask_setup-2.3.0/MANIFEST.in +5 -0
  2. {dask_setup-2.1.0/src/dask_setup.egg-info → dask_setup-2.3.0}/PKG-INFO +91 -12
  3. dask_setup-2.1.0/PKG-INFO → dask_setup-2.3.0/README.md +76 -31
  4. {dask_setup-2.1.0 → dask_setup-2.3.0}/pyproject.toml +31 -3
  5. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/__init__.py +1 -1
  6. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/benchmark.py +195 -40
  7. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/cli.py +9 -4
  8. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/client.py +447 -161
  9. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/cluster.py +89 -14
  10. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/config.py +35 -0
  11. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/config_manager.py +56 -28
  12. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/dashboard.py +37 -1
  13. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/error_handling.py +18 -4
  14. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/exceptions.py +1 -1
  15. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/legacy.py +3 -1
  16. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/logging.py +18 -9
  17. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/multinode.py +134 -49
  18. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/parquet.py +3 -2
  19. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/rechunk.py +99 -53
  20. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/reporting.py +77 -15
  21. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/resources.py +28 -5
  22. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/tempdir.py +31 -2
  23. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/tune.py +84 -30
  24. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/xarray.py +134 -21
  25. dask_setup-2.1.0/README.md → dask_setup-2.3.0/src/dask_setup.egg-info/PKG-INFO +110 -10
  26. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/SOURCES.txt +7 -0
  27. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/requires.txt +5 -1
  28. dask_setup-2.3.0/tests/__init__.py +1 -0
  29. dask_setup-2.3.0/tests/conftest.py +132 -0
  30. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_benchmark.py +205 -0
  31. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_cli.py +14 -3
  32. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_client.py +591 -42
  33. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_cluster.py +102 -0
  34. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_config_manager.py +151 -6
  35. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_dashboard.py +61 -0
  36. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_error_handling.py +66 -4
  37. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_exceptions.py +7 -0
  38. dask_setup-2.3.0/tests/test_logging.py +105 -0
  39. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_multinode.py +388 -3
  40. dask_setup-2.3.0/tests/test_rechunk.py +159 -0
  41. dask_setup-2.3.0/tests/test_reporting.py +180 -0
  42. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_resources.py +101 -11
  43. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_setup_dask_client.py +149 -59
  44. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_tempdir.py +62 -0
  45. dask_setup-2.3.0/tests/test_tune.py +167 -0
  46. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_xarray_chunks.py +322 -0
  47. {dask_setup-2.1.0 → dask_setup-2.3.0}/LICENSE +0 -0
  48. {dask_setup-2.1.0 → dask_setup-2.3.0}/setup.cfg +0 -0
  49. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/callbacks.py +0 -0
  50. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/environment.py +0 -0
  51. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/io_patterns.py +0 -0
  52. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/py.typed +0 -0
  53. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/schema/__init__.py +0 -0
  54. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/schema/profile_schema.json +0 -0
  55. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/topology.py +0 -0
  56. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/types.py +0 -0
  57. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/workload.py +0 -0
  58. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/dependency_links.txt +0 -0
  59. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/entry_points.txt +0 -0
  60. {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/top_level.txt +0 -0
  61. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_compression.py +0 -0
  62. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_config.py +0 -0
  63. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_io_patterns.py +0 -0
  64. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_topology.py +0 -0
  65. {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_types.py +0 -0
@@ -0,0 +1,5 @@
1
+ # setuptools only auto-includes tests/test*.py in the sdist, which leaves out
2
+ # conftest.py (the shared fixtures) and __init__.py -- so the suite could not
3
+ # run from the sdist, as downstream packagers like conda-forge do.
4
+ recursive-include tests *.py
5
+ global-exclude __pycache__ *.py[cod] .DS_Store
@@ -1,9 +1,19 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dask_setup
3
- Version: 2.1.0
3
+ Version: 2.3.0
4
4
  Summary: HPC-tuned Dask helpers for single-node runs on NCI Gadi.
5
5
  Author: Sam Green
6
+ License-Expression: Apache-2.0
6
7
  Project-URL: Homepage, https://github.com/21centuryweather/dask_setup
8
+ Classifier: Development Status :: 5 - Production/Stable
9
+ Classifier: Intended Audience :: Science/Research
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: Programming Language :: Python :: 3.11
12
+ Classifier: Programming Language :: Python :: 3.12
13
+ Classifier: Programming Language :: Python :: 3.13
14
+ Classifier: Programming Language :: Python :: 3.14
15
+ Classifier: Topic :: Scientific/Engineering
16
+ Classifier: Topic :: System :: Distributed Computing
7
17
  Requires-Python: >=3.11
8
18
  Description-Content-Type: text/markdown
9
19
  License-File: LICENSE
@@ -11,21 +21,35 @@ Requires-Dist: dask>=2024.1.0
11
21
  Requires-Dist: distributed>=2024.1.0
12
22
  Requires-Dist: psutil>=5.9
13
23
  Requires-Dist: pyyaml>=6.0
24
+ Provides-Extra: multinode
25
+ Requires-Dist: dask-jobqueue>=0.8; extra == "multinode"
14
26
  Provides-Extra: dev
15
- Requires-Dist: ruff~=0.4; extra == "dev"
27
+ Requires-Dist: ruff==0.16.4; extra == "dev"
16
28
  Requires-Dist: pytest~=8.2; extra == "dev"
17
29
  Requires-Dist: pytest-cov~=5.0; extra == "dev"
18
30
  Requires-Dist: xarray; extra == "dev"
19
31
  Requires-Dist: numpy; extra == "dev"
32
+ Requires-Dist: dask-jobqueue>=0.8; extra == "dev"
20
33
  Dynamic: license-file
21
34
 
22
35
  # dask_setup
23
36
 
24
- [![CI](https://github.com/21centuryweather/dask_setup/workflows/CI/badge.svg)](https://github.com/21centuryweather/dask_setup/actions)
37
+ [![CI](https://github.com/21centuryweather/dask_setup/actions/workflows/ci.yml/badge.svg?branch=main)](https://github.com/21centuryweather/dask_setup/actions/workflows/ci.yml)
38
+ [![Release](https://github.com/21centuryweather/dask_setup/actions/workflows/publish-to-pypi.yml/badge.svg)](https://github.com/21centuryweather/dask_setup/actions/workflows/publish-to-pypi.yml)
39
+ [![PyPI](https://img.shields.io/pypi/v/dask_setup.svg)](https://pypi.org/project/dask_setup/)
40
+ [![conda-forge](https://img.shields.io/conda/vn/conda-forge/dask-setup.svg)](https://anaconda.org/conda-forge/dask-setup)
41
+ [![Python](https://img.shields.io/badge/python-3.11%2B-blue.svg)](https://pypi.org/project/dask_setup/)
42
+ [![License](https://img.shields.io/badge/license-Apache%202.0-blue.svg)](LICENSE)
43
+ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
25
44
 
26
45
  HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask.distributed.LocalCluster` + `Client` with sensible defaults for single-node jobs, and extends to multi-node PBS/SLURM clusters via `dask-jobqueue` in v2.0.
27
46
 
28
- **Python 3.11+** | `pip install dask-setup`
47
+ ```bash
48
+ pip install dask-setup # single-node
49
+ pip install dask-setup[multinode] # + PBS/SLURM via dask-jobqueue
50
+
51
+ conda install -c conda-forge dask-setup # or via conda-forge
52
+ ```
29
53
 
30
54
  ---
31
55
 
@@ -35,9 +59,9 @@ HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask
35
59
  from dask_setup import setup_dask_client
36
60
 
37
61
  # Pick a workload type and go
38
- client, cluster, dask_tmp = setup_dask_client("cpu") # heavy compute
39
- client, cluster, dask_tmp = setup_dask_client("io") # heavy file I/O
40
- client, cluster, dask_tmp = setup_dask_client("mixed") # both
62
+ client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="cpu") # heavy compute
63
+ client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="io") # Zarr / object storage
64
+ client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="mixed") # both
41
65
  ```
42
66
 
43
67
  `dask_tmp` is the path to the spill/temp directory (on `$PBS_JOBFS` if available). Pass it to Rechunker, Zarr, or anywhere else you want fast local I/O.
@@ -81,12 +105,35 @@ client, cluster, shared_tmp = setup_dask_client(
81
105
 
82
106
  | Type | Topology | Best for |
83
107
  |------|----------|----------|
84
- | `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions |
85
- | `"io"` | 1 process, 8–16 threads | Opening many NetCDF/Zarr files concurrently |
108
+ | `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions, **and reading NetCDF/HDF5** |
109
+ | `"io"` | 1 process, 8–16 threads | Zarr, object storage (S3), HTTP — libraries that release the GIL |
86
110
  | `"mixed"` | Processes with 2 threads each | Pipelines that both read and compute |
87
111
  | `"gpu"` | 1 process per GPU, up to 8 threads | CuPy/RAPIDS CUDA workloads |
88
112
  | `"auto"` | Inferred from dataset | Let `dask_setup` decide based on your data |
89
113
 
114
+ ### Reading NetCDF? Use `"cpu"`, not `"io"`
115
+
116
+ The split is not "I/O vs compute" — it is **whether the library releases the GIL
117
+ and is thread-safe**. `"io"` runs one process with many threads, which only wins
118
+ when threads can genuinely overlap.
119
+
120
+ HDF5 (underneath NetCDF4) is usually not built thread-safe, so xarray routes
121
+ *every* NetCDF read through a single process-wide lock. Under `"io"` your threads
122
+ do not become parallel readers — they queue on that one lock, one at a time, and
123
+ zlib decompression is serialised inside it too. Separate processes each get their
124
+ own lock, so `"cpu"` is what actually parallelises NetCDF reads.
125
+
126
+ Zarr is the opposite: it decompresses via numcodecs, which releases the GIL, and
127
+ has no global lock, so threads scale well. Opening, concatenating and writing
128
+ NetCDF is a `"cpu"` job even though it feels like I/O.
129
+
130
+ > **Rule of thumb: NetCDF in → `"cpu"`. Zarr or object storage in → `"io"`.**
131
+ > Pass `ds=` and `dask_setup` will warn if you get this pair the wrong way round.
132
+ >
133
+ > Also avoid `open_mfdataset(..., parallel=True)` together with `"io"`: concurrent
134
+ > metadata reads can kill the worker outright, after which Dask silently
135
+ > recomputes the lost tasks and the job appears to hang rather than fail.
136
+
90
137
  ---
91
138
 
92
139
  ## Key Parameters
@@ -95,14 +142,27 @@ client, cluster, shared_tmp = setup_dask_client(
95
142
  |-----------|---------|-------------|
96
143
  | `workload_type` | `"io"` | Worker topology: `"cpu"`, `"io"`, `"mixed"`, `"gpu"`, `"auto"` |
97
144
  | `max_workers` | all cores | Hard cap on worker count |
98
- | `reserve_mem_gb` | auto (20% RAM) | Memory held back for OS/cache (GiB) |
145
+ | `reserve_mem_gb` | auto | Memory held back for OS/cache (GiB): 20% of RAM, clamped to [4, 50] |
99
146
  | `max_mem_gb` | total RAM | Upper bound on Dask's total memory use |
100
147
  | `dashboard` | `True` | Start dashboard and print SSH tunnel hint |
101
148
  | `profile` | `None` | Named config profile |
102
- | `config` | `None` | Pre-built `DaskSetupConfig` object — mutually exclusive with `profile` |
103
- | `mode` | `"auto"` | `"local"`, `"pbs"`, `"slurm"`, or `"auto"` (v2.0) |
149
+ | `config` | `None` | Pre-built `DaskSetupConfig` object — same layer as `profile`; if both are given, `profile` wins |
150
+ | `mode` | `"interactive"` | `"interactive"` uses the current allocation and never submits jobs (a single node or no job at all is a plain `LocalCluster`); `"auto"` picks `"pbs"`/`"slurm"` in batch jobs, which submit worker jobs; also `"local"`, `"pbs"`, `"slurm"` |
104
151
  | `multi_node_config` | `None` | `MultiNodeConfig` for PBS/SLURM multi-node jobs (v2.0) |
105
152
 
153
+ Settings are layered, lowest to highest:
154
+
155
+ ```
156
+ library defaults < config= or profile= < explicit keyword arguments
157
+ ```
158
+
159
+ A parameter you leave unset inherits from the layer below. Passing a value
160
+ always overrides, even when that value happens to equal the default — so
161
+ `reserve_mem_gb=50.0` overrides a profile that says 40.0.
162
+
163
+ `config=` and `profile=` share a layer rather than stacking: pass one or the
164
+ other, and use explicit keyword arguments for the differences.
165
+
106
166
  ---
107
167
 
108
168
  ## Common Patterns
@@ -177,6 +237,14 @@ Tunnel from your laptop:
177
237
  Then open: http://localhost:8787
178
238
  ```
179
239
 
240
+ The login host is inferred from the compute node's DNS domain, so it is correct
241
+ at other sites too. Override it with `$DASK_SETUP_LOGIN_HOST` if the guess is
242
+ wrong:
243
+
244
+ ```bash
245
+ export DASK_SETUP_LOGIN_HOST=login.mycluster.edu
246
+ ```
247
+
180
248
  ---
181
249
 
182
250
  ## CLI
@@ -214,19 +282,30 @@ Full documentation lives in the [GitHub wiki](https://github.com/21centuryweathe
214
282
  | [Benchmarking](https://github.com/21centuryweather/dask_setup/wiki/Benchmarking) | `benchmark_config`, `scaling_analysis`, `chunk_impact`, `dask-setup benchmark` |
215
283
  | [Internals](https://github.com/21centuryweather/dask_setup/wiki/Internals) | Resource detection, topology decisions, temp/spill routing, module layout |
216
284
  | [Troubleshooting](https://github.com/21centuryweather/dask_setup/wiki/Troubleshooting) | Common errors, OOM, multi-node issues, migration guide |
285
+ | [User Feedback](https://github.com/21centuryweather/dask_setup/wiki/User-Feedback) | Questions from users, the answers, and what changed as a result |
217
286
 
218
287
  ---
219
288
 
220
289
  ## Installation
221
290
 
291
+ From PyPI:
292
+
222
293
  ```bash
223
294
  pip install dask-setup
224
295
  ```
225
296
 
297
+ From conda-forge:
298
+
299
+ ```bash
300
+ conda install -c conda-forge dask-setup
301
+ ```
302
+
226
303
  For multi-node PBS/SLURM support:
227
304
 
228
305
  ```bash
229
306
  pip install dask-setup dask-jobqueue
307
+ # or
308
+ conda install -c conda-forge dask-setup dask-jobqueue
230
309
  ```
231
310
 
232
311
  For GPU workloads (CuPy auto-detection):
@@ -1,31 +1,21 @@
1
- Metadata-Version: 2.4
2
- Name: dask_setup
3
- Version: 2.1.0
4
- Summary: HPC-tuned Dask helpers for single-node runs on NCI Gadi.
5
- Author: Sam Green
6
- Project-URL: Homepage, https://github.com/21centuryweather/dask_setup
7
- Requires-Python: >=3.11
8
- Description-Content-Type: text/markdown
9
- License-File: LICENSE
10
- Requires-Dist: dask>=2024.1.0
11
- Requires-Dist: distributed>=2024.1.0
12
- Requires-Dist: psutil>=5.9
13
- Requires-Dist: pyyaml>=6.0
14
- Provides-Extra: dev
15
- Requires-Dist: ruff~=0.4; extra == "dev"
16
- Requires-Dist: pytest~=8.2; extra == "dev"
17
- Requires-Dist: pytest-cov~=5.0; extra == "dev"
18
- Requires-Dist: xarray; extra == "dev"
19
- Requires-Dist: numpy; extra == "dev"
20
- Dynamic: license-file
21
-
22
1
  # dask_setup
23
2
 
24
- [![CI](https://github.com/21centuryweather/dask_setup/workflows/CI/badge.svg)](https://github.com/21centuryweather/dask_setup/actions)
3
+ [![CI](https://github.com/21centuryweather/dask_setup/actions/workflows/ci.yml/badge.svg?branch=main)](https://github.com/21centuryweather/dask_setup/actions/workflows/ci.yml)
4
+ [![Release](https://github.com/21centuryweather/dask_setup/actions/workflows/publish-to-pypi.yml/badge.svg)](https://github.com/21centuryweather/dask_setup/actions/workflows/publish-to-pypi.yml)
5
+ [![PyPI](https://img.shields.io/pypi/v/dask_setup.svg)](https://pypi.org/project/dask_setup/)
6
+ [![conda-forge](https://img.shields.io/conda/vn/conda-forge/dask-setup.svg)](https://anaconda.org/conda-forge/dask-setup)
7
+ [![Python](https://img.shields.io/badge/python-3.11%2B-blue.svg)](https://pypi.org/project/dask_setup/)
8
+ [![License](https://img.shields.io/badge/license-Apache%202.0-blue.svg)](LICENSE)
9
+ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
25
10
 
26
11
  HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask.distributed.LocalCluster` + `Client` with sensible defaults for single-node jobs, and extends to multi-node PBS/SLURM clusters via `dask-jobqueue` in v2.0.
27
12
 
28
- **Python 3.11+** | `pip install dask-setup`
13
+ ```bash
14
+ pip install dask-setup # single-node
15
+ pip install dask-setup[multinode] # + PBS/SLURM via dask-jobqueue
16
+
17
+ conda install -c conda-forge dask-setup # or via conda-forge
18
+ ```
29
19
 
30
20
  ---
31
21
 
@@ -35,9 +25,9 @@ HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask
35
25
  from dask_setup import setup_dask_client
36
26
 
37
27
  # Pick a workload type and go
38
- client, cluster, dask_tmp = setup_dask_client("cpu") # heavy compute
39
- client, cluster, dask_tmp = setup_dask_client("io") # heavy file I/O
40
- client, cluster, dask_tmp = setup_dask_client("mixed") # both
28
+ client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="cpu") # heavy compute
29
+ client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="io") # Zarr / object storage
30
+ client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="mixed") # both
41
31
  ```
42
32
 
43
33
  `dask_tmp` is the path to the spill/temp directory (on `$PBS_JOBFS` if available). Pass it to Rechunker, Zarr, or anywhere else you want fast local I/O.
@@ -81,12 +71,35 @@ client, cluster, shared_tmp = setup_dask_client(
81
71
 
82
72
  | Type | Topology | Best for |
83
73
  |------|----------|----------|
84
- | `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions |
85
- | `"io"` | 1 process, 8–16 threads | Opening many NetCDF/Zarr files concurrently |
74
+ | `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions, **and reading NetCDF/HDF5** |
75
+ | `"io"` | 1 process, 8–16 threads | Zarr, object storage (S3), HTTP — libraries that release the GIL |
86
76
  | `"mixed"` | Processes with 2 threads each | Pipelines that both read and compute |
87
77
  | `"gpu"` | 1 process per GPU, up to 8 threads | CuPy/RAPIDS CUDA workloads |
88
78
  | `"auto"` | Inferred from dataset | Let `dask_setup` decide based on your data |
89
79
 
80
+ ### Reading NetCDF? Use `"cpu"`, not `"io"`
81
+
82
+ The split is not "I/O vs compute" — it is **whether the library releases the GIL
83
+ and is thread-safe**. `"io"` runs one process with many threads, which only wins
84
+ when threads can genuinely overlap.
85
+
86
+ HDF5 (underneath NetCDF4) is usually not built thread-safe, so xarray routes
87
+ *every* NetCDF read through a single process-wide lock. Under `"io"` your threads
88
+ do not become parallel readers — they queue on that one lock, one at a time, and
89
+ zlib decompression is serialised inside it too. Separate processes each get their
90
+ own lock, so `"cpu"` is what actually parallelises NetCDF reads.
91
+
92
+ Zarr is the opposite: it decompresses via numcodecs, which releases the GIL, and
93
+ has no global lock, so threads scale well. Opening, concatenating and writing
94
+ NetCDF is a `"cpu"` job even though it feels like I/O.
95
+
96
+ > **Rule of thumb: NetCDF in → `"cpu"`. Zarr or object storage in → `"io"`.**
97
+ > Pass `ds=` and `dask_setup` will warn if you get this pair the wrong way round.
98
+ >
99
+ > Also avoid `open_mfdataset(..., parallel=True)` together with `"io"`: concurrent
100
+ > metadata reads can kill the worker outright, after which Dask silently
101
+ > recomputes the lost tasks and the job appears to hang rather than fail.
102
+
90
103
  ---
91
104
 
92
105
  ## Key Parameters
@@ -95,14 +108,27 @@ client, cluster, shared_tmp = setup_dask_client(
95
108
  |-----------|---------|-------------|
96
109
  | `workload_type` | `"io"` | Worker topology: `"cpu"`, `"io"`, `"mixed"`, `"gpu"`, `"auto"` |
97
110
  | `max_workers` | all cores | Hard cap on worker count |
98
- | `reserve_mem_gb` | auto (20% RAM) | Memory held back for OS/cache (GiB) |
111
+ | `reserve_mem_gb` | auto | Memory held back for OS/cache (GiB): 20% of RAM, clamped to [4, 50] |
99
112
  | `max_mem_gb` | total RAM | Upper bound on Dask's total memory use |
100
113
  | `dashboard` | `True` | Start dashboard and print SSH tunnel hint |
101
114
  | `profile` | `None` | Named config profile |
102
- | `config` | `None` | Pre-built `DaskSetupConfig` object — mutually exclusive with `profile` |
103
- | `mode` | `"auto"` | `"local"`, `"pbs"`, `"slurm"`, or `"auto"` (v2.0) |
115
+ | `config` | `None` | Pre-built `DaskSetupConfig` object — same layer as `profile`; if both are given, `profile` wins |
116
+ | `mode` | `"interactive"` | `"interactive"` uses the current allocation and never submits jobs (a single node or no job at all is a plain `LocalCluster`); `"auto"` picks `"pbs"`/`"slurm"` in batch jobs, which submit worker jobs; also `"local"`, `"pbs"`, `"slurm"` |
104
117
  | `multi_node_config` | `None` | `MultiNodeConfig` for PBS/SLURM multi-node jobs (v2.0) |
105
118
 
119
+ Settings are layered, lowest to highest:
120
+
121
+ ```
122
+ library defaults < config= or profile= < explicit keyword arguments
123
+ ```
124
+
125
+ A parameter you leave unset inherits from the layer below. Passing a value
126
+ always overrides, even when that value happens to equal the default — so
127
+ `reserve_mem_gb=50.0` overrides a profile that says 40.0.
128
+
129
+ `config=` and `profile=` share a layer rather than stacking: pass one or the
130
+ other, and use explicit keyword arguments for the differences.
131
+
106
132
  ---
107
133
 
108
134
  ## Common Patterns
@@ -177,6 +203,14 @@ Tunnel from your laptop:
177
203
  Then open: http://localhost:8787
178
204
  ```
179
205
 
206
+ The login host is inferred from the compute node's DNS domain, so it is correct
207
+ at other sites too. Override it with `$DASK_SETUP_LOGIN_HOST` if the guess is
208
+ wrong:
209
+
210
+ ```bash
211
+ export DASK_SETUP_LOGIN_HOST=login.mycluster.edu
212
+ ```
213
+
180
214
  ---
181
215
 
182
216
  ## CLI
@@ -214,19 +248,30 @@ Full documentation lives in the [GitHub wiki](https://github.com/21centuryweathe
214
248
  | [Benchmarking](https://github.com/21centuryweather/dask_setup/wiki/Benchmarking) | `benchmark_config`, `scaling_analysis`, `chunk_impact`, `dask-setup benchmark` |
215
249
  | [Internals](https://github.com/21centuryweather/dask_setup/wiki/Internals) | Resource detection, topology decisions, temp/spill routing, module layout |
216
250
  | [Troubleshooting](https://github.com/21centuryweather/dask_setup/wiki/Troubleshooting) | Common errors, OOM, multi-node issues, migration guide |
251
+ | [User Feedback](https://github.com/21centuryweather/dask_setup/wiki/User-Feedback) | Questions from users, the answers, and what changed as a result |
217
252
 
218
253
  ---
219
254
 
220
255
  ## Installation
221
256
 
257
+ From PyPI:
258
+
222
259
  ```bash
223
260
  pip install dask-setup
224
261
  ```
225
262
 
263
+ From conda-forge:
264
+
265
+ ```bash
266
+ conda install -c conda-forge dask-setup
267
+ ```
268
+
226
269
  For multi-node PBS/SLURM support:
227
270
 
228
271
  ```bash
229
272
  pip install dask-setup dask-jobqueue
273
+ # or
274
+ conda install -c conda-forge dask-setup dask-jobqueue
230
275
  ```
231
276
 
232
277
  For GPU workloads (CuPy auto-detection):
@@ -1,14 +1,27 @@
1
1
  [build-system]
2
- requires = ["setuptools>=64", "wheel"]
2
+ requires = ["setuptools>=77", "wheel"]
3
3
  build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "dask_setup"
7
- version = "2.1.0"
7
+ version = "2.3.0"
8
8
  description = "HPC-tuned Dask helpers for single-node runs on NCI Gadi."
9
9
  authors = [{name="Sam Green"}]
10
10
  readme = "README.md"
11
11
  requires-python = ">=3.11"
12
+ license = "Apache-2.0"
13
+ license-files = ["LICENSE"]
14
+ classifiers = [
15
+ "Development Status :: 5 - Production/Stable",
16
+ "Intended Audience :: Science/Research",
17
+ "Programming Language :: Python :: 3",
18
+ "Programming Language :: Python :: 3.11",
19
+ "Programming Language :: Python :: 3.12",
20
+ "Programming Language :: Python :: 3.13",
21
+ "Programming Language :: Python :: 3.14",
22
+ "Topic :: Scientific/Engineering",
23
+ "Topic :: System :: Distributed Computing",
24
+ ]
12
25
  dependencies = [
13
26
  "dask>=2024.1.0",
14
27
  "distributed>=2024.1.0",
@@ -23,13 +36,28 @@ Homepage = "https://github.com/21centuryweather/dask_setup"
23
36
  dask-setup = "dask_setup.cli:main"
24
37
 
25
38
  [project.optional-dependencies]
39
+ # Multi-node PBS/SLURM cluster support. Optional: the single-node path never
40
+ # imports it, and setup_pbs_cluster()/setup_slurm_cluster() raise a helpful
41
+ # ImportError if it is missing.
42
+ multinode = [
43
+ "dask-jobqueue>=0.8",
44
+ ]
45
+
26
46
  dev = [
27
- "ruff ~= 0.4",
47
+ # Pinned exactly, not a range. `ruff ~= 0.4` means >=0.4,<1.0, so CI picked
48
+ # up whatever ruff had released that morning -- and ruff's formatter changes
49
+ # between minor versions. That turns `ruff format --check` into a job that
50
+ # can fail on a commit that touched nothing. Bump this deliberately.
51
+ "ruff == 0.16.4",
28
52
  "pytest ~= 8.2",
29
53
  "pytest-cov ~= 5.0",
30
54
  # Xarray integration testing
31
55
  "xarray",
32
56
  "numpy",
57
+ # Multi-node tests exercise a real PBSJob. Without this, CI silently skipped
58
+ # the multi-node path -- including the regression test for the 4x worker
59
+ # under-provisioning fixed in 2.2.0.
60
+ "dask-jobqueue>=0.8",
33
61
  ]
34
62
 
35
63
  [tool.setuptools.package-data]
@@ -281,7 +281,7 @@ except ImportError:
281
281
  )
282
282
 
283
283
 
284
- __version__ = "2.1.0"
284
+ __version__ = "2.3.0"
285
285
 
286
286
  __all__ = [
287
287
  # Core API — always available