dask-setup 2.1.0__tar.gz → 2.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dask_setup-2.3.0/MANIFEST.in +5 -0
- {dask_setup-2.1.0/src/dask_setup.egg-info → dask_setup-2.3.0}/PKG-INFO +91 -12
- dask_setup-2.1.0/PKG-INFO → dask_setup-2.3.0/README.md +76 -31
- {dask_setup-2.1.0 → dask_setup-2.3.0}/pyproject.toml +31 -3
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/__init__.py +1 -1
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/benchmark.py +195 -40
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/cli.py +9 -4
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/client.py +447 -161
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/cluster.py +89 -14
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/config.py +35 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/config_manager.py +56 -28
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/dashboard.py +37 -1
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/error_handling.py +18 -4
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/exceptions.py +1 -1
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/legacy.py +3 -1
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/logging.py +18 -9
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/multinode.py +134 -49
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/parquet.py +3 -2
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/rechunk.py +99 -53
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/reporting.py +77 -15
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/resources.py +28 -5
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/tempdir.py +31 -2
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/tune.py +84 -30
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/xarray.py +134 -21
- dask_setup-2.1.0/README.md → dask_setup-2.3.0/src/dask_setup.egg-info/PKG-INFO +110 -10
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/SOURCES.txt +7 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/requires.txt +5 -1
- dask_setup-2.3.0/tests/__init__.py +1 -0
- dask_setup-2.3.0/tests/conftest.py +132 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_benchmark.py +205 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_cli.py +14 -3
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_client.py +591 -42
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_cluster.py +102 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_config_manager.py +151 -6
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_dashboard.py +61 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_error_handling.py +66 -4
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_exceptions.py +7 -0
- dask_setup-2.3.0/tests/test_logging.py +105 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_multinode.py +388 -3
- dask_setup-2.3.0/tests/test_rechunk.py +159 -0
- dask_setup-2.3.0/tests/test_reporting.py +180 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_resources.py +101 -11
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_setup_dask_client.py +149 -59
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_tempdir.py +62 -0
- dask_setup-2.3.0/tests/test_tune.py +167 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_xarray_chunks.py +322 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/LICENSE +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/setup.cfg +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/callbacks.py +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/environment.py +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/io_patterns.py +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/py.typed +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/schema/__init__.py +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/schema/profile_schema.json +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/topology.py +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/types.py +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup/workload.py +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/dependency_links.txt +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/entry_points.txt +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/top_level.txt +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_compression.py +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_config.py +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_io_patterns.py +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_topology.py +0 -0
- {dask_setup-2.1.0 → dask_setup-2.3.0}/tests/test_types.py +0 -0
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
# setuptools only auto-includes tests/test*.py in the sdist, which leaves out
|
|
2
|
+
# conftest.py (the shared fixtures) and __init__.py -- so the suite could not
|
|
3
|
+
# run from the sdist, as downstream packagers like conda-forge do.
|
|
4
|
+
recursive-include tests *.py
|
|
5
|
+
global-exclude __pycache__ *.py[cod] .DS_Store
|
|
@@ -1,9 +1,19 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dask_setup
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.3.0
|
|
4
4
|
Summary: HPC-tuned Dask helpers for single-node runs on NCI Gadi.
|
|
5
5
|
Author: Sam Green
|
|
6
|
+
License-Expression: Apache-2.0
|
|
6
7
|
Project-URL: Homepage, https://github.com/21centuryweather/dask_setup
|
|
8
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
9
|
+
Classifier: Intended Audience :: Science/Research
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering
|
|
16
|
+
Classifier: Topic :: System :: Distributed Computing
|
|
7
17
|
Requires-Python: >=3.11
|
|
8
18
|
Description-Content-Type: text/markdown
|
|
9
19
|
License-File: LICENSE
|
|
@@ -11,21 +21,35 @@ Requires-Dist: dask>=2024.1.0
|
|
|
11
21
|
Requires-Dist: distributed>=2024.1.0
|
|
12
22
|
Requires-Dist: psutil>=5.9
|
|
13
23
|
Requires-Dist: pyyaml>=6.0
|
|
24
|
+
Provides-Extra: multinode
|
|
25
|
+
Requires-Dist: dask-jobqueue>=0.8; extra == "multinode"
|
|
14
26
|
Provides-Extra: dev
|
|
15
|
-
Requires-Dist: ruff
|
|
27
|
+
Requires-Dist: ruff==0.16.4; extra == "dev"
|
|
16
28
|
Requires-Dist: pytest~=8.2; extra == "dev"
|
|
17
29
|
Requires-Dist: pytest-cov~=5.0; extra == "dev"
|
|
18
30
|
Requires-Dist: xarray; extra == "dev"
|
|
19
31
|
Requires-Dist: numpy; extra == "dev"
|
|
32
|
+
Requires-Dist: dask-jobqueue>=0.8; extra == "dev"
|
|
20
33
|
Dynamic: license-file
|
|
21
34
|
|
|
22
35
|
# dask_setup
|
|
23
36
|
|
|
24
|
-
[](https://github.com/21centuryweather/dask_setup/actions/workflows/ci.yml)
|
|
38
|
+
[](https://github.com/21centuryweather/dask_setup/actions/workflows/publish-to-pypi.yml)
|
|
39
|
+
[](https://pypi.org/project/dask_setup/)
|
|
40
|
+
[](https://anaconda.org/conda-forge/dask-setup)
|
|
41
|
+
[](https://pypi.org/project/dask_setup/)
|
|
42
|
+
[](LICENSE)
|
|
43
|
+
[](https://github.com/astral-sh/ruff)
|
|
25
44
|
|
|
26
45
|
HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask.distributed.LocalCluster` + `Client` with sensible defaults for single-node jobs, and extends to multi-node PBS/SLURM clusters via `dask-jobqueue` in v2.0.
|
|
27
46
|
|
|
28
|
-
|
|
47
|
+
```bash
|
|
48
|
+
pip install dask-setup # single-node
|
|
49
|
+
pip install dask-setup[multinode] # + PBS/SLURM via dask-jobqueue
|
|
50
|
+
|
|
51
|
+
conda install -c conda-forge dask-setup # or via conda-forge
|
|
52
|
+
```
|
|
29
53
|
|
|
30
54
|
---
|
|
31
55
|
|
|
@@ -35,9 +59,9 @@ HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask
|
|
|
35
59
|
from dask_setup import setup_dask_client
|
|
36
60
|
|
|
37
61
|
# Pick a workload type and go
|
|
38
|
-
client, cluster, dask_tmp = setup_dask_client("cpu") # heavy compute
|
|
39
|
-
client, cluster, dask_tmp = setup_dask_client("io") #
|
|
40
|
-
client, cluster, dask_tmp = setup_dask_client("mixed") # both
|
|
62
|
+
client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="cpu") # heavy compute
|
|
63
|
+
client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="io") # Zarr / object storage
|
|
64
|
+
client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="mixed") # both
|
|
41
65
|
```
|
|
42
66
|
|
|
43
67
|
`dask_tmp` is the path to the spill/temp directory (on `$PBS_JOBFS` if available). Pass it to Rechunker, Zarr, or anywhere else you want fast local I/O.
|
|
@@ -81,12 +105,35 @@ client, cluster, shared_tmp = setup_dask_client(
|
|
|
81
105
|
|
|
82
106
|
| Type | Topology | Best for |
|
|
83
107
|
|------|----------|----------|
|
|
84
|
-
| `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions |
|
|
85
|
-
| `"io"` | 1 process, 8–16 threads |
|
|
108
|
+
| `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions, **and reading NetCDF/HDF5** |
|
|
109
|
+
| `"io"` | 1 process, 8–16 threads | Zarr, object storage (S3), HTTP — libraries that release the GIL |
|
|
86
110
|
| `"mixed"` | Processes with 2 threads each | Pipelines that both read and compute |
|
|
87
111
|
| `"gpu"` | 1 process per GPU, up to 8 threads | CuPy/RAPIDS CUDA workloads |
|
|
88
112
|
| `"auto"` | Inferred from dataset | Let `dask_setup` decide based on your data |
|
|
89
113
|
|
|
114
|
+
### Reading NetCDF? Use `"cpu"`, not `"io"`
|
|
115
|
+
|
|
116
|
+
The split is not "I/O vs compute" — it is **whether the library releases the GIL
|
|
117
|
+
and is thread-safe**. `"io"` runs one process with many threads, which only wins
|
|
118
|
+
when threads can genuinely overlap.
|
|
119
|
+
|
|
120
|
+
HDF5 (underneath NetCDF4) is usually not built thread-safe, so xarray routes
|
|
121
|
+
*every* NetCDF read through a single process-wide lock. Under `"io"` your threads
|
|
122
|
+
do not become parallel readers — they queue on that one lock, one at a time, and
|
|
123
|
+
zlib decompression is serialised inside it too. Separate processes each get their
|
|
124
|
+
own lock, so `"cpu"` is what actually parallelises NetCDF reads.
|
|
125
|
+
|
|
126
|
+
Zarr is the opposite: it decompresses via numcodecs, which releases the GIL, and
|
|
127
|
+
has no global lock, so threads scale well. Opening, concatenating and writing
|
|
128
|
+
NetCDF is a `"cpu"` job even though it feels like I/O.
|
|
129
|
+
|
|
130
|
+
> **Rule of thumb: NetCDF in → `"cpu"`. Zarr or object storage in → `"io"`.**
|
|
131
|
+
> Pass `ds=` and `dask_setup` will warn if you get this pair the wrong way round.
|
|
132
|
+
>
|
|
133
|
+
> Also avoid `open_mfdataset(..., parallel=True)` together with `"io"`: concurrent
|
|
134
|
+
> metadata reads can kill the worker outright, after which Dask silently
|
|
135
|
+
> recomputes the lost tasks and the job appears to hang rather than fail.
|
|
136
|
+
|
|
90
137
|
---
|
|
91
138
|
|
|
92
139
|
## Key Parameters
|
|
@@ -95,14 +142,27 @@ client, cluster, shared_tmp = setup_dask_client(
|
|
|
95
142
|
|-----------|---------|-------------|
|
|
96
143
|
| `workload_type` | `"io"` | Worker topology: `"cpu"`, `"io"`, `"mixed"`, `"gpu"`, `"auto"` |
|
|
97
144
|
| `max_workers` | all cores | Hard cap on worker count |
|
|
98
|
-
| `reserve_mem_gb` | auto
|
|
145
|
+
| `reserve_mem_gb` | auto | Memory held back for OS/cache (GiB): 20% of RAM, clamped to [4, 50] |
|
|
99
146
|
| `max_mem_gb` | total RAM | Upper bound on Dask's total memory use |
|
|
100
147
|
| `dashboard` | `True` | Start dashboard and print SSH tunnel hint |
|
|
101
148
|
| `profile` | `None` | Named config profile |
|
|
102
|
-
| `config` | `None` | Pre-built `DaskSetupConfig` object —
|
|
103
|
-
| `mode` | `"
|
|
149
|
+
| `config` | `None` | Pre-built `DaskSetupConfig` object — same layer as `profile`; if both are given, `profile` wins |
|
|
150
|
+
| `mode` | `"interactive"` | `"interactive"` uses the current allocation and never submits jobs (a single node or no job at all is a plain `LocalCluster`); `"auto"` picks `"pbs"`/`"slurm"` in batch jobs, which submit worker jobs; also `"local"`, `"pbs"`, `"slurm"` |
|
|
104
151
|
| `multi_node_config` | `None` | `MultiNodeConfig` for PBS/SLURM multi-node jobs (v2.0) |
|
|
105
152
|
|
|
153
|
+
Settings are layered, lowest to highest:
|
|
154
|
+
|
|
155
|
+
```
|
|
156
|
+
library defaults < config= or profile= < explicit keyword arguments
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
A parameter you leave unset inherits from the layer below. Passing a value
|
|
160
|
+
always overrides, even when that value happens to equal the default — so
|
|
161
|
+
`reserve_mem_gb=50.0` overrides a profile that says 40.0.
|
|
162
|
+
|
|
163
|
+
`config=` and `profile=` share a layer rather than stacking: pass one or the
|
|
164
|
+
other, and use explicit keyword arguments for the differences.
|
|
165
|
+
|
|
106
166
|
---
|
|
107
167
|
|
|
108
168
|
## Common Patterns
|
|
@@ -177,6 +237,14 @@ Tunnel from your laptop:
|
|
|
177
237
|
Then open: http://localhost:8787
|
|
178
238
|
```
|
|
179
239
|
|
|
240
|
+
The login host is inferred from the compute node's DNS domain, so it is correct
|
|
241
|
+
at other sites too. Override it with `$DASK_SETUP_LOGIN_HOST` if the guess is
|
|
242
|
+
wrong:
|
|
243
|
+
|
|
244
|
+
```bash
|
|
245
|
+
export DASK_SETUP_LOGIN_HOST=login.mycluster.edu
|
|
246
|
+
```
|
|
247
|
+
|
|
180
248
|
---
|
|
181
249
|
|
|
182
250
|
## CLI
|
|
@@ -214,19 +282,30 @@ Full documentation lives in the [GitHub wiki](https://github.com/21centuryweathe
|
|
|
214
282
|
| [Benchmarking](https://github.com/21centuryweather/dask_setup/wiki/Benchmarking) | `benchmark_config`, `scaling_analysis`, `chunk_impact`, `dask-setup benchmark` |
|
|
215
283
|
| [Internals](https://github.com/21centuryweather/dask_setup/wiki/Internals) | Resource detection, topology decisions, temp/spill routing, module layout |
|
|
216
284
|
| [Troubleshooting](https://github.com/21centuryweather/dask_setup/wiki/Troubleshooting) | Common errors, OOM, multi-node issues, migration guide |
|
|
285
|
+
| [User Feedback](https://github.com/21centuryweather/dask_setup/wiki/User-Feedback) | Questions from users, the answers, and what changed as a result |
|
|
217
286
|
|
|
218
287
|
---
|
|
219
288
|
|
|
220
289
|
## Installation
|
|
221
290
|
|
|
291
|
+
From PyPI:
|
|
292
|
+
|
|
222
293
|
```bash
|
|
223
294
|
pip install dask-setup
|
|
224
295
|
```
|
|
225
296
|
|
|
297
|
+
From conda-forge:
|
|
298
|
+
|
|
299
|
+
```bash
|
|
300
|
+
conda install -c conda-forge dask-setup
|
|
301
|
+
```
|
|
302
|
+
|
|
226
303
|
For multi-node PBS/SLURM support:
|
|
227
304
|
|
|
228
305
|
```bash
|
|
229
306
|
pip install dask-setup dask-jobqueue
|
|
307
|
+
# or
|
|
308
|
+
conda install -c conda-forge dask-setup dask-jobqueue
|
|
230
309
|
```
|
|
231
310
|
|
|
232
311
|
For GPU workloads (CuPy auto-detection):
|
|
@@ -1,31 +1,21 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: dask_setup
|
|
3
|
-
Version: 2.1.0
|
|
4
|
-
Summary: HPC-tuned Dask helpers for single-node runs on NCI Gadi.
|
|
5
|
-
Author: Sam Green
|
|
6
|
-
Project-URL: Homepage, https://github.com/21centuryweather/dask_setup
|
|
7
|
-
Requires-Python: >=3.11
|
|
8
|
-
Description-Content-Type: text/markdown
|
|
9
|
-
License-File: LICENSE
|
|
10
|
-
Requires-Dist: dask>=2024.1.0
|
|
11
|
-
Requires-Dist: distributed>=2024.1.0
|
|
12
|
-
Requires-Dist: psutil>=5.9
|
|
13
|
-
Requires-Dist: pyyaml>=6.0
|
|
14
|
-
Provides-Extra: dev
|
|
15
|
-
Requires-Dist: ruff~=0.4; extra == "dev"
|
|
16
|
-
Requires-Dist: pytest~=8.2; extra == "dev"
|
|
17
|
-
Requires-Dist: pytest-cov~=5.0; extra == "dev"
|
|
18
|
-
Requires-Dist: xarray; extra == "dev"
|
|
19
|
-
Requires-Dist: numpy; extra == "dev"
|
|
20
|
-
Dynamic: license-file
|
|
21
|
-
|
|
22
1
|
# dask_setup
|
|
23
2
|
|
|
24
|
-
[](https://github.com/21centuryweather/dask_setup/actions/workflows/ci.yml)
|
|
4
|
+
[](https://github.com/21centuryweather/dask_setup/actions/workflows/publish-to-pypi.yml)
|
|
5
|
+
[](https://pypi.org/project/dask_setup/)
|
|
6
|
+
[](https://anaconda.org/conda-forge/dask-setup)
|
|
7
|
+
[](https://pypi.org/project/dask_setup/)
|
|
8
|
+
[](LICENSE)
|
|
9
|
+
[](https://github.com/astral-sh/ruff)
|
|
25
10
|
|
|
26
11
|
HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask.distributed.LocalCluster` + `Client` with sensible defaults for single-node jobs, and extends to multi-node PBS/SLURM clusters via `dask-jobqueue` in v2.0.
|
|
27
12
|
|
|
28
|
-
|
|
13
|
+
```bash
|
|
14
|
+
pip install dask-setup # single-node
|
|
15
|
+
pip install dask-setup[multinode] # + PBS/SLURM via dask-jobqueue
|
|
16
|
+
|
|
17
|
+
conda install -c conda-forge dask-setup # or via conda-forge
|
|
18
|
+
```
|
|
29
19
|
|
|
30
20
|
---
|
|
31
21
|
|
|
@@ -35,9 +25,9 @@ HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask
|
|
|
35
25
|
from dask_setup import setup_dask_client
|
|
36
26
|
|
|
37
27
|
# Pick a workload type and go
|
|
38
|
-
client, cluster, dask_tmp = setup_dask_client("cpu") # heavy compute
|
|
39
|
-
client, cluster, dask_tmp = setup_dask_client("io") #
|
|
40
|
-
client, cluster, dask_tmp = setup_dask_client("mixed") # both
|
|
28
|
+
client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="cpu") # heavy compute
|
|
29
|
+
client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="io") # Zarr / object storage
|
|
30
|
+
client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="mixed") # both
|
|
41
31
|
```
|
|
42
32
|
|
|
43
33
|
`dask_tmp` is the path to the spill/temp directory (on `$PBS_JOBFS` if available). Pass it to Rechunker, Zarr, or anywhere else you want fast local I/O.
|
|
@@ -81,12 +71,35 @@ client, cluster, shared_tmp = setup_dask_client(
|
|
|
81
71
|
|
|
82
72
|
| Type | Topology | Best for |
|
|
83
73
|
|------|----------|----------|
|
|
84
|
-
| `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions |
|
|
85
|
-
| `"io"` | 1 process, 8–16 threads |
|
|
74
|
+
| `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions, **and reading NetCDF/HDF5** |
|
|
75
|
+
| `"io"` | 1 process, 8–16 threads | Zarr, object storage (S3), HTTP — libraries that release the GIL |
|
|
86
76
|
| `"mixed"` | Processes with 2 threads each | Pipelines that both read and compute |
|
|
87
77
|
| `"gpu"` | 1 process per GPU, up to 8 threads | CuPy/RAPIDS CUDA workloads |
|
|
88
78
|
| `"auto"` | Inferred from dataset | Let `dask_setup` decide based on your data |
|
|
89
79
|
|
|
80
|
+
### Reading NetCDF? Use `"cpu"`, not `"io"`
|
|
81
|
+
|
|
82
|
+
The split is not "I/O vs compute" — it is **whether the library releases the GIL
|
|
83
|
+
and is thread-safe**. `"io"` runs one process with many threads, which only wins
|
|
84
|
+
when threads can genuinely overlap.
|
|
85
|
+
|
|
86
|
+
HDF5 (underneath NetCDF4) is usually not built thread-safe, so xarray routes
|
|
87
|
+
*every* NetCDF read through a single process-wide lock. Under `"io"` your threads
|
|
88
|
+
do not become parallel readers — they queue on that one lock, one at a time, and
|
|
89
|
+
zlib decompression is serialised inside it too. Separate processes each get their
|
|
90
|
+
own lock, so `"cpu"` is what actually parallelises NetCDF reads.
|
|
91
|
+
|
|
92
|
+
Zarr is the opposite: it decompresses via numcodecs, which releases the GIL, and
|
|
93
|
+
has no global lock, so threads scale well. Opening, concatenating and writing
|
|
94
|
+
NetCDF is a `"cpu"` job even though it feels like I/O.
|
|
95
|
+
|
|
96
|
+
> **Rule of thumb: NetCDF in → `"cpu"`. Zarr or object storage in → `"io"`.**
|
|
97
|
+
> Pass `ds=` and `dask_setup` will warn if you get this pair the wrong way round.
|
|
98
|
+
>
|
|
99
|
+
> Also avoid `open_mfdataset(..., parallel=True)` together with `"io"`: concurrent
|
|
100
|
+
> metadata reads can kill the worker outright, after which Dask silently
|
|
101
|
+
> recomputes the lost tasks and the job appears to hang rather than fail.
|
|
102
|
+
|
|
90
103
|
---
|
|
91
104
|
|
|
92
105
|
## Key Parameters
|
|
@@ -95,14 +108,27 @@ client, cluster, shared_tmp = setup_dask_client(
|
|
|
95
108
|
|-----------|---------|-------------|
|
|
96
109
|
| `workload_type` | `"io"` | Worker topology: `"cpu"`, `"io"`, `"mixed"`, `"gpu"`, `"auto"` |
|
|
97
110
|
| `max_workers` | all cores | Hard cap on worker count |
|
|
98
|
-
| `reserve_mem_gb` | auto
|
|
111
|
+
| `reserve_mem_gb` | auto | Memory held back for OS/cache (GiB): 20% of RAM, clamped to [4, 50] |
|
|
99
112
|
| `max_mem_gb` | total RAM | Upper bound on Dask's total memory use |
|
|
100
113
|
| `dashboard` | `True` | Start dashboard and print SSH tunnel hint |
|
|
101
114
|
| `profile` | `None` | Named config profile |
|
|
102
|
-
| `config` | `None` | Pre-built `DaskSetupConfig` object —
|
|
103
|
-
| `mode` | `"
|
|
115
|
+
| `config` | `None` | Pre-built `DaskSetupConfig` object — same layer as `profile`; if both are given, `profile` wins |
|
|
116
|
+
| `mode` | `"interactive"` | `"interactive"` uses the current allocation and never submits jobs (a single node or no job at all is a plain `LocalCluster`); `"auto"` picks `"pbs"`/`"slurm"` in batch jobs, which submit worker jobs; also `"local"`, `"pbs"`, `"slurm"` |
|
|
104
117
|
| `multi_node_config` | `None` | `MultiNodeConfig` for PBS/SLURM multi-node jobs (v2.0) |
|
|
105
118
|
|
|
119
|
+
Settings are layered, lowest to highest:
|
|
120
|
+
|
|
121
|
+
```
|
|
122
|
+
library defaults < config= or profile= < explicit keyword arguments
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
A parameter you leave unset inherits from the layer below. Passing a value
|
|
126
|
+
always overrides, even when that value happens to equal the default — so
|
|
127
|
+
`reserve_mem_gb=50.0` overrides a profile that says 40.0.
|
|
128
|
+
|
|
129
|
+
`config=` and `profile=` share a layer rather than stacking: pass one or the
|
|
130
|
+
other, and use explicit keyword arguments for the differences.
|
|
131
|
+
|
|
106
132
|
---
|
|
107
133
|
|
|
108
134
|
## Common Patterns
|
|
@@ -177,6 +203,14 @@ Tunnel from your laptop:
|
|
|
177
203
|
Then open: http://localhost:8787
|
|
178
204
|
```
|
|
179
205
|
|
|
206
|
+
The login host is inferred from the compute node's DNS domain, so it is correct
|
|
207
|
+
at other sites too. Override it with `$DASK_SETUP_LOGIN_HOST` if the guess is
|
|
208
|
+
wrong:
|
|
209
|
+
|
|
210
|
+
```bash
|
|
211
|
+
export DASK_SETUP_LOGIN_HOST=login.mycluster.edu
|
|
212
|
+
```
|
|
213
|
+
|
|
180
214
|
---
|
|
181
215
|
|
|
182
216
|
## CLI
|
|
@@ -214,19 +248,30 @@ Full documentation lives in the [GitHub wiki](https://github.com/21centuryweathe
|
|
|
214
248
|
| [Benchmarking](https://github.com/21centuryweather/dask_setup/wiki/Benchmarking) | `benchmark_config`, `scaling_analysis`, `chunk_impact`, `dask-setup benchmark` |
|
|
215
249
|
| [Internals](https://github.com/21centuryweather/dask_setup/wiki/Internals) | Resource detection, topology decisions, temp/spill routing, module layout |
|
|
216
250
|
| [Troubleshooting](https://github.com/21centuryweather/dask_setup/wiki/Troubleshooting) | Common errors, OOM, multi-node issues, migration guide |
|
|
251
|
+
| [User Feedback](https://github.com/21centuryweather/dask_setup/wiki/User-Feedback) | Questions from users, the answers, and what changed as a result |
|
|
217
252
|
|
|
218
253
|
---
|
|
219
254
|
|
|
220
255
|
## Installation
|
|
221
256
|
|
|
257
|
+
From PyPI:
|
|
258
|
+
|
|
222
259
|
```bash
|
|
223
260
|
pip install dask-setup
|
|
224
261
|
```
|
|
225
262
|
|
|
263
|
+
From conda-forge:
|
|
264
|
+
|
|
265
|
+
```bash
|
|
266
|
+
conda install -c conda-forge dask-setup
|
|
267
|
+
```
|
|
268
|
+
|
|
226
269
|
For multi-node PBS/SLURM support:
|
|
227
270
|
|
|
228
271
|
```bash
|
|
229
272
|
pip install dask-setup dask-jobqueue
|
|
273
|
+
# or
|
|
274
|
+
conda install -c conda-forge dask-setup dask-jobqueue
|
|
230
275
|
```
|
|
231
276
|
|
|
232
277
|
For GPU workloads (CuPy auto-detection):
|
|
@@ -1,14 +1,27 @@
|
|
|
1
1
|
[build-system]
|
|
2
|
-
requires = ["setuptools>=
|
|
2
|
+
requires = ["setuptools>=77", "wheel"]
|
|
3
3
|
build-backend = "setuptools.build_meta"
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "dask_setup"
|
|
7
|
-
version = "2.
|
|
7
|
+
version = "2.3.0"
|
|
8
8
|
description = "HPC-tuned Dask helpers for single-node runs on NCI Gadi."
|
|
9
9
|
authors = [{name="Sam Green"}]
|
|
10
10
|
readme = "README.md"
|
|
11
11
|
requires-python = ">=3.11"
|
|
12
|
+
license = "Apache-2.0"
|
|
13
|
+
license-files = ["LICENSE"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 5 - Production/Stable",
|
|
16
|
+
"Intended Audience :: Science/Research",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Programming Language :: Python :: 3.11",
|
|
19
|
+
"Programming Language :: Python :: 3.12",
|
|
20
|
+
"Programming Language :: Python :: 3.13",
|
|
21
|
+
"Programming Language :: Python :: 3.14",
|
|
22
|
+
"Topic :: Scientific/Engineering",
|
|
23
|
+
"Topic :: System :: Distributed Computing",
|
|
24
|
+
]
|
|
12
25
|
dependencies = [
|
|
13
26
|
"dask>=2024.1.0",
|
|
14
27
|
"distributed>=2024.1.0",
|
|
@@ -23,13 +36,28 @@ Homepage = "https://github.com/21centuryweather/dask_setup"
|
|
|
23
36
|
dask-setup = "dask_setup.cli:main"
|
|
24
37
|
|
|
25
38
|
[project.optional-dependencies]
|
|
39
|
+
# Multi-node PBS/SLURM cluster support. Optional: the single-node path never
|
|
40
|
+
# imports it, and setup_pbs_cluster()/setup_slurm_cluster() raise a helpful
|
|
41
|
+
# ImportError if it is missing.
|
|
42
|
+
multinode = [
|
|
43
|
+
"dask-jobqueue>=0.8",
|
|
44
|
+
]
|
|
45
|
+
|
|
26
46
|
dev = [
|
|
27
|
-
|
|
47
|
+
# Pinned exactly, not a range. `ruff ~= 0.4` means >=0.4,<1.0, so CI picked
|
|
48
|
+
# up whatever ruff had released that morning -- and ruff's formatter changes
|
|
49
|
+
# between minor versions. That turns `ruff format --check` into a job that
|
|
50
|
+
# can fail on a commit that touched nothing. Bump this deliberately.
|
|
51
|
+
"ruff == 0.16.4",
|
|
28
52
|
"pytest ~= 8.2",
|
|
29
53
|
"pytest-cov ~= 5.0",
|
|
30
54
|
# Xarray integration testing
|
|
31
55
|
"xarray",
|
|
32
56
|
"numpy",
|
|
57
|
+
# Multi-node tests exercise a real PBSJob. Without this, CI silently skipped
|
|
58
|
+
# the multi-node path -- including the regression test for the 4x worker
|
|
59
|
+
# under-provisioning fixed in 2.2.0.
|
|
60
|
+
"dask-jobqueue>=0.8",
|
|
33
61
|
]
|
|
34
62
|
|
|
35
63
|
[tool.setuptools.package-data]
|