dask-setup 1.0.0__tar.gz → 2.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dask_setup-2.0.0/PKG-INFO +246 -0
- dask_setup-2.0.0/README.md +225 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/pyproject.toml +4 -1
- dask_setup-2.0.0/src/dask_setup/__init__.py +351 -0
- dask_setup-2.0.0/src/dask_setup/benchmark.py +1176 -0
- dask_setup-2.0.0/src/dask_setup/callbacks.py +162 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/cli.py +333 -0
- dask_setup-2.0.0/src/dask_setup/client.py +721 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/cluster.py +25 -2
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/config.py +59 -3
- dask_setup-2.0.0/src/dask_setup/config_manager.py +644 -0
- dask_setup-2.0.0/src/dask_setup/dashboard.py +123 -0
- dask_setup-2.0.0/src/dask_setup/environment.py +82 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/error_handling.py +24 -1
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/io_patterns.py +376 -8
- dask_setup-2.0.0/src/dask_setup/multinode.py +768 -0
- dask_setup-2.0.0/src/dask_setup/parquet.py +335 -0
- dask_setup-2.0.0/src/dask_setup/py.typed +0 -0
- dask_setup-2.0.0/src/dask_setup/rechunk.py +196 -0
- dask_setup-2.0.0/src/dask_setup/reporting.py +237 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/resources.py +63 -5
- dask_setup-2.0.0/src/dask_setup/schema/__init__.py +35 -0
- dask_setup-2.0.0/src/dask_setup/schema/profile_schema.json +262 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/topology.py +82 -8
- dask_setup-2.0.0/src/dask_setup/tune.py +283 -0
- dask_setup-2.0.0/src/dask_setup/workload.py +270 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/xarray.py +284 -35
- dask_setup-2.0.0/src/dask_setup.egg-info/PKG-INFO +246 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup.egg-info/SOURCES.txt +14 -0
- dask_setup-2.0.0/tests/test_benchmark.py +517 -0
- dask_setup-2.0.0/tests/test_multinode.py +890 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_xarray_chunks.py +330 -0
- dask_setup-1.0.0/PKG-INFO +0 -817
- dask_setup-1.0.0/README.md +0 -796
- dask_setup-1.0.0/src/dask_setup/__init__.py +0 -101
- dask_setup-1.0.0/src/dask_setup/client.py +0 -280
- dask_setup-1.0.0/src/dask_setup/config_manager.py +0 -343
- dask_setup-1.0.0/src/dask_setup/dashboard.py +0 -71
- dask_setup-1.0.0/src/dask_setup.egg-info/PKG-INFO +0 -817
- {dask_setup-1.0.0 → dask_setup-2.0.0}/LICENSE +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/setup.cfg +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/exceptions.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/legacy.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/logging.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/tempdir.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/types.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup.egg-info/dependency_links.txt +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup.egg-info/entry_points.txt +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup.egg-info/requires.txt +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup.egg-info/top_level.txt +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_cli.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_client.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_cluster.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_compression.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_config.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_config_manager.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_dashboard.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_error_handling.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_exceptions.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_io_patterns.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_resources.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_setup_dask_client.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_tempdir.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_topology.py +0 -0
- {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_types.py +0 -0
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: dask_setup
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: HPC-tuned Dask helpers for single-node runs on NCI Gadi.
|
|
5
|
+
Author: Sam Green
|
|
6
|
+
Project-URL: Homepage, https://github.com/21centuryweather/dask_setup
|
|
7
|
+
Requires-Python: >=3.11
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Dist: dask>=2024.1.0
|
|
11
|
+
Requires-Dist: distributed>=2024.1.0
|
|
12
|
+
Requires-Dist: psutil>=5.9
|
|
13
|
+
Requires-Dist: pyyaml>=6.0
|
|
14
|
+
Provides-Extra: dev
|
|
15
|
+
Requires-Dist: ruff~=0.4; extra == "dev"
|
|
16
|
+
Requires-Dist: pytest~=8.2; extra == "dev"
|
|
17
|
+
Requires-Dist: pytest-cov~=5.0; extra == "dev"
|
|
18
|
+
Requires-Dist: xarray; extra == "dev"
|
|
19
|
+
Requires-Dist: numpy; extra == "dev"
|
|
20
|
+
Dynamic: license-file
|
|
21
|
+
|
|
22
|
+
# dask_setup
|
|
23
|
+
|
|
24
|
+
[](https://github.com/21centuryweather/dask_setup/actions)
|
|
25
|
+
|
|
26
|
+
HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask.distributed.LocalCluster` + `Client` with sensible defaults for single-node jobs, and extends to multi-node PBS/SLURM clusters via `dask-jobqueue` in v2.0.
|
|
27
|
+
|
|
28
|
+
**Python 3.11+** | `pip install dask-setup`
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
|
|
32
|
+
## Quick Start
|
|
33
|
+
|
|
34
|
+
```python
|
|
35
|
+
from dask_setup import setup_dask_client
|
|
36
|
+
|
|
37
|
+
# Pick a workload type and go
|
|
38
|
+
client, cluster, dask_tmp = setup_dask_client("cpu") # heavy compute
|
|
39
|
+
client, cluster, dask_tmp = setup_dask_client("io") # heavy file I/O
|
|
40
|
+
client, cluster, dask_tmp = setup_dask_client("mixed") # both
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
`dask_tmp` is the path to the spill/temp directory (on `$PBS_JOBFS` if available). Pass it to Rechunker, Zarr, or anywhere else you want fast local I/O.
|
|
44
|
+
|
|
45
|
+
For more control, pass a `DaskSetupConfig` object or a named profile:
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
from dask_setup import setup_dask_client, DaskSetupConfig
|
|
49
|
+
|
|
50
|
+
config = DaskSetupConfig(
|
|
51
|
+
workload_type="cpu",
|
|
52
|
+
max_workers=8,
|
|
53
|
+
reserve_mem_gb=32.0,
|
|
54
|
+
spill_compression="lz4",
|
|
55
|
+
)
|
|
56
|
+
client, cluster, dask_tmp = setup_dask_client(config=config)
|
|
57
|
+
|
|
58
|
+
# Or use a built-in profile
|
|
59
|
+
client, cluster, dask_tmp = setup_dask_client(profile="climate_analysis")
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
**Multi-node (v2.0):**
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
from dask_setup import setup_dask_client, MultiNodeConfig
|
|
66
|
+
|
|
67
|
+
client, cluster, shared_tmp = setup_dask_client(
|
|
68
|
+
mode="auto", # detects PBS_JOBID / SLURM_JOB_ID, falls back to local
|
|
69
|
+
multi_node_config=MultiNodeConfig(
|
|
70
|
+
workers_per_node=4,
|
|
71
|
+
cores_per_worker=12,
|
|
72
|
+
mem_per_worker_gb=32.0,
|
|
73
|
+
walltime="04:00:00",
|
|
74
|
+
),
|
|
75
|
+
)
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
---
|
|
79
|
+
|
|
80
|
+
## Workload Types
|
|
81
|
+
|
|
82
|
+
| Type | Topology | Best for |
|
|
83
|
+
|------|----------|----------|
|
|
84
|
+
| `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions |
|
|
85
|
+
| `"io"` | 1 process, 8–16 threads | Opening many NetCDF/Zarr files concurrently |
|
|
86
|
+
| `"mixed"` | Processes with 2 threads each | Pipelines that both read and compute |
|
|
87
|
+
| `"gpu"` | 1 process per GPU, up to 8 threads | CuPy/RAPIDS CUDA workloads |
|
|
88
|
+
| `"auto"` | Inferred from dataset | Let `dask_setup` decide based on your data |
|
|
89
|
+
|
|
90
|
+
---
|
|
91
|
+
|
|
92
|
+
## Key Parameters
|
|
93
|
+
|
|
94
|
+
| Parameter | Default | Description |
|
|
95
|
+
|-----------|---------|-------------|
|
|
96
|
+
| `workload_type` | `"io"` | Worker topology: `"cpu"`, `"io"`, `"mixed"`, `"gpu"`, `"auto"` |
|
|
97
|
+
| `max_workers` | all cores | Hard cap on worker count |
|
|
98
|
+
| `reserve_mem_gb` | auto (20% RAM) | Memory held back for OS/cache (GiB) |
|
|
99
|
+
| `max_mem_gb` | total RAM | Upper bound on Dask's total memory use |
|
|
100
|
+
| `dashboard` | `True` | Start dashboard and print SSH tunnel hint |
|
|
101
|
+
| `profile` | `None` | Named config profile |
|
|
102
|
+
| `config` | `None` | Pre-built `DaskSetupConfig` object — mutually exclusive with `profile` |
|
|
103
|
+
| `mode` | `"auto"` | `"local"`, `"pbs"`, `"slurm"`, or `"auto"` (v2.0) |
|
|
104
|
+
| `multi_node_config` | `None` | `MultiNodeConfig` for PBS/SLURM multi-node jobs (v2.0) |
|
|
105
|
+
|
|
106
|
+
---
|
|
107
|
+
|
|
108
|
+
## Common Patterns
|
|
109
|
+
|
|
110
|
+
**Big xarray reductions (CPU-bound)**
|
|
111
|
+
```python
|
|
112
|
+
client, cluster, dask_tmp = setup_dask_client("cpu", reserve_mem_gb=60)
|
|
113
|
+
ds = ds.chunk({"time": 240, "y": 512, "x": 512})
|
|
114
|
+
out = ds.mean(("y", "x")).compute()
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
**Opening many NetCDF/Zarr files (I/O-bound)**
|
|
118
|
+
```python
|
|
119
|
+
client, cluster, dask_tmp = setup_dask_client("io", reserve_mem_gb=40)
|
|
120
|
+
ds = xr.open_mfdataset(files, engine="netcdf4", chunks={}, parallel=True)
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
**Rechunking with Rechunker (spill stays on fast local storage)**
|
|
124
|
+
```python
|
|
125
|
+
client, cluster, dask_tmp = setup_dask_client("cpu", reserve_mem_gb=60)
|
|
126
|
+
plan = rechunker.rechunk(
|
|
127
|
+
ds.to_array().data,
|
|
128
|
+
target_chunks={"time": 240, "y": 512, "x": 512},
|
|
129
|
+
max_mem="6GB",
|
|
130
|
+
target_store="out.zarr",
|
|
131
|
+
temp_store=f"{dask_tmp}/tmp_rechunk.zarr",
|
|
132
|
+
)
|
|
133
|
+
plan.execute()
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
**Writing to Zarr in time windows**
|
|
137
|
+
```python
|
|
138
|
+
client, cluster, dask_tmp = setup_dask_client("cpu", max_workers=1, reserve_mem_gb=50)
|
|
139
|
+
ds.isel(time=slice(0, 240)).to_zarr("out.zarr", mode="w", consolidated=True)
|
|
140
|
+
for start in range(240, ds.sizes["time"], 240):
|
|
141
|
+
stop = min(start + 240, ds.sizes["time"])
|
|
142
|
+
ds.isel(time=slice(start, stop)).to_zarr(
|
|
143
|
+
"out.zarr", mode="a", region={"time": slice(start, stop)}
|
|
144
|
+
)
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
---
|
|
148
|
+
|
|
149
|
+
## Built-in Profiles
|
|
150
|
+
|
|
151
|
+
Use `profile=` to load a named configuration:
|
|
152
|
+
|
|
153
|
+
```python
|
|
154
|
+
client, cluster, dask_tmp = setup_dask_client(profile="climate_analysis")
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
| Profile | Workload | Reserve | Notes |
|
|
158
|
+
|---------|----------|---------|-------|
|
|
159
|
+
| `climate_analysis` | cpu | 60 GB | Heavy compute with large arrays |
|
|
160
|
+
| `zarr_io_heavy` | io | 40 GB | Many Zarr files |
|
|
161
|
+
| `development` | mixed | 8 GB | Local testing, 2 workers max |
|
|
162
|
+
| `production` | mixed | 80 GB | Adaptive, dashboard off |
|
|
163
|
+
| `interactive` | mixed | 20 GB | Jupyter notebooks, 4 workers |
|
|
164
|
+
|
|
165
|
+
See [Configuration](https://github.com/21centuryweather/dask_setup/wiki/Configuration) for how to create and save your own profiles.
|
|
166
|
+
|
|
167
|
+
---
|
|
168
|
+
|
|
169
|
+
## Dashboard
|
|
170
|
+
|
|
171
|
+
With `dashboard=True` (default), the cluster prints an SSH tunnel command:
|
|
172
|
+
|
|
173
|
+
```
|
|
174
|
+
Dask dashboard: http://127.0.0.1:<PORT>/status
|
|
175
|
+
Tunnel from your laptop:
|
|
176
|
+
ssh -N -L 8787:<COMPUTE_HOST>:<PORT> gadi.nci.org.au
|
|
177
|
+
Then open: http://localhost:8787
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
---
|
|
181
|
+
|
|
182
|
+
## CLI
|
|
183
|
+
|
|
184
|
+
```bash
|
|
185
|
+
# Profile management
|
|
186
|
+
dask-setup list # list all profiles
|
|
187
|
+
dask-setup show climate_analysis # show profile details
|
|
188
|
+
dask-setup create my_profile --from-profile zarr_io_heavy
|
|
189
|
+
dask-setup validate my_profile
|
|
190
|
+
dask-setup export climate_analysis -o profile.yaml
|
|
191
|
+
dask-setup import https://example.com/team.yaml
|
|
192
|
+
dask-setup delete my_profile
|
|
193
|
+
|
|
194
|
+
# Synthetic benchmark
|
|
195
|
+
dask-setup benchmark --profile development --size small --operation mean
|
|
196
|
+
|
|
197
|
+
# Generate a PBS/SLURM job script (v2.0)
|
|
198
|
+
dask-setup submit my_analysis.py --scheduler pbs \
|
|
199
|
+
--workers-per-node 4 --cores-per-worker 12 --mem-per-worker 32 \
|
|
200
|
+
--walltime 04:00:00 --queue normal --project ab01
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
---
|
|
204
|
+
|
|
205
|
+
## Documentation
|
|
206
|
+
|
|
207
|
+
Full documentation lives in the [GitHub wiki](https://github.com/21centuryweather/dask_setup/wiki):
|
|
208
|
+
|
|
209
|
+
| Page | What's covered |
|
|
210
|
+
|------|----------------|
|
|
211
|
+
| [Configuration](https://github.com/21centuryweather/dask_setup/wiki/Configuration) | `DaskSetupConfig`, profiles, site-wide profiles, profile inheritance, CLI, JSON Schema |
|
|
212
|
+
| [Multi-Node](https://github.com/21centuryweather/dask_setup/wiki/Multi-Node) | `MultiNodeConfig`, PBS/SLURM cluster setup, GPU topology, shared temp dirs, `dask-setup submit` |
|
|
213
|
+
| [IO-Optimization](https://github.com/21centuryweather/dask_setup/wiki/IO-Optimization) | `recommend_chunks`, `recommend_io_chunks`, Zarr v3, Kerchunk, Parquet/Arrow, storage-aware chunking |
|
|
214
|
+
| [Benchmarking](https://github.com/21centuryweather/dask_setup/wiki/Benchmarking) | `benchmark_config`, `scaling_analysis`, `chunk_impact`, `dask-setup benchmark` |
|
|
215
|
+
| [Internals](https://github.com/21centuryweather/dask_setup/wiki/Internals) | Resource detection, topology decisions, temp/spill routing, module layout |
|
|
216
|
+
| [Troubleshooting](https://github.com/21centuryweather/dask_setup/wiki/Troubleshooting) | Common errors, OOM, multi-node issues, migration guide |
|
|
217
|
+
|
|
218
|
+
---
|
|
219
|
+
|
|
220
|
+
## Installation
|
|
221
|
+
|
|
222
|
+
```bash
|
|
223
|
+
pip install dask-setup
|
|
224
|
+
```
|
|
225
|
+
|
|
226
|
+
For multi-node PBS/SLURM support:
|
|
227
|
+
|
|
228
|
+
```bash
|
|
229
|
+
pip install dask-setup dask-jobqueue
|
|
230
|
+
```
|
|
231
|
+
|
|
232
|
+
For GPU workloads (CuPy auto-detection):
|
|
233
|
+
|
|
234
|
+
```bash
|
|
235
|
+
pip install dask-setup cupy-cuda12x # match your CUDA version
|
|
236
|
+
```
|
|
237
|
+
|
|
238
|
+
---
|
|
239
|
+
|
|
240
|
+
## Contributing
|
|
241
|
+
|
|
242
|
+
Bug reports, feature requests, and pull requests are welcome. Please include tests and, for performance changes, benchmarks.
|
|
243
|
+
|
|
244
|
+
## License
|
|
245
|
+
|
|
246
|
+
Apache-2.0 — see [LICENSE](LICENSE) for details.
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
# dask_setup
|
|
2
|
+
|
|
3
|
+
[](https://github.com/21centuryweather/dask_setup/actions)
|
|
4
|
+
|
|
5
|
+
HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask.distributed.LocalCluster` + `Client` with sensible defaults for single-node jobs, and extends to multi-node PBS/SLURM clusters via `dask-jobqueue` in v2.0.
|
|
6
|
+
|
|
7
|
+
**Python 3.11+** | `pip install dask-setup`
|
|
8
|
+
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
## Quick Start
|
|
12
|
+
|
|
13
|
+
```python
|
|
14
|
+
from dask_setup import setup_dask_client
|
|
15
|
+
|
|
16
|
+
# Pick a workload type and go
|
|
17
|
+
client, cluster, dask_tmp = setup_dask_client("cpu") # heavy compute
|
|
18
|
+
client, cluster, dask_tmp = setup_dask_client("io") # heavy file I/O
|
|
19
|
+
client, cluster, dask_tmp = setup_dask_client("mixed") # both
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
`dask_tmp` is the path to the spill/temp directory (on `$PBS_JOBFS` if available). Pass it to Rechunker, Zarr, or anywhere else you want fast local I/O.
|
|
23
|
+
|
|
24
|
+
For more control, pass a `DaskSetupConfig` object or a named profile:
|
|
25
|
+
|
|
26
|
+
```python
|
|
27
|
+
from dask_setup import setup_dask_client, DaskSetupConfig
|
|
28
|
+
|
|
29
|
+
config = DaskSetupConfig(
|
|
30
|
+
workload_type="cpu",
|
|
31
|
+
max_workers=8,
|
|
32
|
+
reserve_mem_gb=32.0,
|
|
33
|
+
spill_compression="lz4",
|
|
34
|
+
)
|
|
35
|
+
client, cluster, dask_tmp = setup_dask_client(config=config)
|
|
36
|
+
|
|
37
|
+
# Or use a built-in profile
|
|
38
|
+
client, cluster, dask_tmp = setup_dask_client(profile="climate_analysis")
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
**Multi-node (v2.0):**
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from dask_setup import setup_dask_client, MultiNodeConfig
|
|
45
|
+
|
|
46
|
+
client, cluster, shared_tmp = setup_dask_client(
|
|
47
|
+
mode="auto", # detects PBS_JOBID / SLURM_JOB_ID, falls back to local
|
|
48
|
+
multi_node_config=MultiNodeConfig(
|
|
49
|
+
workers_per_node=4,
|
|
50
|
+
cores_per_worker=12,
|
|
51
|
+
mem_per_worker_gb=32.0,
|
|
52
|
+
walltime="04:00:00",
|
|
53
|
+
),
|
|
54
|
+
)
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
---
|
|
58
|
+
|
|
59
|
+
## Workload Types
|
|
60
|
+
|
|
61
|
+
| Type | Topology | Best for |
|
|
62
|
+
|------|----------|----------|
|
|
63
|
+
| `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions |
|
|
64
|
+
| `"io"` | 1 process, 8–16 threads | Opening many NetCDF/Zarr files concurrently |
|
|
65
|
+
| `"mixed"` | Processes with 2 threads each | Pipelines that both read and compute |
|
|
66
|
+
| `"gpu"` | 1 process per GPU, up to 8 threads | CuPy/RAPIDS CUDA workloads |
|
|
67
|
+
| `"auto"` | Inferred from dataset | Let `dask_setup` decide based on your data |
|
|
68
|
+
|
|
69
|
+
---
|
|
70
|
+
|
|
71
|
+
## Key Parameters
|
|
72
|
+
|
|
73
|
+
| Parameter | Default | Description |
|
|
74
|
+
|-----------|---------|-------------|
|
|
75
|
+
| `workload_type` | `"io"` | Worker topology: `"cpu"`, `"io"`, `"mixed"`, `"gpu"`, `"auto"` |
|
|
76
|
+
| `max_workers` | all cores | Hard cap on worker count |
|
|
77
|
+
| `reserve_mem_gb` | auto (20% RAM) | Memory held back for OS/cache (GiB) |
|
|
78
|
+
| `max_mem_gb` | total RAM | Upper bound on Dask's total memory use |
|
|
79
|
+
| `dashboard` | `True` | Start dashboard and print SSH tunnel hint |
|
|
80
|
+
| `profile` | `None` | Named config profile |
|
|
81
|
+
| `config` | `None` | Pre-built `DaskSetupConfig` object — mutually exclusive with `profile` |
|
|
82
|
+
| `mode` | `"auto"` | `"local"`, `"pbs"`, `"slurm"`, or `"auto"` (v2.0) |
|
|
83
|
+
| `multi_node_config` | `None` | `MultiNodeConfig` for PBS/SLURM multi-node jobs (v2.0) |
|
|
84
|
+
|
|
85
|
+
---
|
|
86
|
+
|
|
87
|
+
## Common Patterns
|
|
88
|
+
|
|
89
|
+
**Big xarray reductions (CPU-bound)**
|
|
90
|
+
```python
|
|
91
|
+
client, cluster, dask_tmp = setup_dask_client("cpu", reserve_mem_gb=60)
|
|
92
|
+
ds = ds.chunk({"time": 240, "y": 512, "x": 512})
|
|
93
|
+
out = ds.mean(("y", "x")).compute()
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
**Opening many NetCDF/Zarr files (I/O-bound)**
|
|
97
|
+
```python
|
|
98
|
+
client, cluster, dask_tmp = setup_dask_client("io", reserve_mem_gb=40)
|
|
99
|
+
ds = xr.open_mfdataset(files, engine="netcdf4", chunks={}, parallel=True)
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
**Rechunking with Rechunker (spill stays on fast local storage)**
|
|
103
|
+
```python
|
|
104
|
+
client, cluster, dask_tmp = setup_dask_client("cpu", reserve_mem_gb=60)
|
|
105
|
+
plan = rechunker.rechunk(
|
|
106
|
+
ds.to_array().data,
|
|
107
|
+
target_chunks={"time": 240, "y": 512, "x": 512},
|
|
108
|
+
max_mem="6GB",
|
|
109
|
+
target_store="out.zarr",
|
|
110
|
+
temp_store=f"{dask_tmp}/tmp_rechunk.zarr",
|
|
111
|
+
)
|
|
112
|
+
plan.execute()
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
**Writing to Zarr in time windows**
|
|
116
|
+
```python
|
|
117
|
+
client, cluster, dask_tmp = setup_dask_client("cpu", max_workers=1, reserve_mem_gb=50)
|
|
118
|
+
ds.isel(time=slice(0, 240)).to_zarr("out.zarr", mode="w", consolidated=True)
|
|
119
|
+
for start in range(240, ds.sizes["time"], 240):
|
|
120
|
+
stop = min(start + 240, ds.sizes["time"])
|
|
121
|
+
ds.isel(time=slice(start, stop)).to_zarr(
|
|
122
|
+
"out.zarr", mode="a", region={"time": slice(start, stop)}
|
|
123
|
+
)
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
---
|
|
127
|
+
|
|
128
|
+
## Built-in Profiles
|
|
129
|
+
|
|
130
|
+
Use `profile=` to load a named configuration:
|
|
131
|
+
|
|
132
|
+
```python
|
|
133
|
+
client, cluster, dask_tmp = setup_dask_client(profile="climate_analysis")
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
| Profile | Workload | Reserve | Notes |
|
|
137
|
+
|---------|----------|---------|-------|
|
|
138
|
+
| `climate_analysis` | cpu | 60 GB | Heavy compute with large arrays |
|
|
139
|
+
| `zarr_io_heavy` | io | 40 GB | Many Zarr files |
|
|
140
|
+
| `development` | mixed | 8 GB | Local testing, 2 workers max |
|
|
141
|
+
| `production` | mixed | 80 GB | Adaptive, dashboard off |
|
|
142
|
+
| `interactive` | mixed | 20 GB | Jupyter notebooks, 4 workers |
|
|
143
|
+
|
|
144
|
+
See [Configuration](https://github.com/21centuryweather/dask_setup/wiki/Configuration) for how to create and save your own profiles.
|
|
145
|
+
|
|
146
|
+
---
|
|
147
|
+
|
|
148
|
+
## Dashboard
|
|
149
|
+
|
|
150
|
+
With `dashboard=True` (default), the cluster prints an SSH tunnel command:
|
|
151
|
+
|
|
152
|
+
```
|
|
153
|
+
Dask dashboard: http://127.0.0.1:<PORT>/status
|
|
154
|
+
Tunnel from your laptop:
|
|
155
|
+
ssh -N -L 8787:<COMPUTE_HOST>:<PORT> gadi.nci.org.au
|
|
156
|
+
Then open: http://localhost:8787
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
---
|
|
160
|
+
|
|
161
|
+
## CLI
|
|
162
|
+
|
|
163
|
+
```bash
|
|
164
|
+
# Profile management
|
|
165
|
+
dask-setup list # list all profiles
|
|
166
|
+
dask-setup show climate_analysis # show profile details
|
|
167
|
+
dask-setup create my_profile --from-profile zarr_io_heavy
|
|
168
|
+
dask-setup validate my_profile
|
|
169
|
+
dask-setup export climate_analysis -o profile.yaml
|
|
170
|
+
dask-setup import https://example.com/team.yaml
|
|
171
|
+
dask-setup delete my_profile
|
|
172
|
+
|
|
173
|
+
# Synthetic benchmark
|
|
174
|
+
dask-setup benchmark --profile development --size small --operation mean
|
|
175
|
+
|
|
176
|
+
# Generate a PBS/SLURM job script (v2.0)
|
|
177
|
+
dask-setup submit my_analysis.py --scheduler pbs \
|
|
178
|
+
--workers-per-node 4 --cores-per-worker 12 --mem-per-worker 32 \
|
|
179
|
+
--walltime 04:00:00 --queue normal --project ab01
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
---
|
|
183
|
+
|
|
184
|
+
## Documentation
|
|
185
|
+
|
|
186
|
+
Full documentation lives in the [GitHub wiki](https://github.com/21centuryweather/dask_setup/wiki):
|
|
187
|
+
|
|
188
|
+
| Page | What's covered |
|
|
189
|
+
|------|----------------|
|
|
190
|
+
| [Configuration](https://github.com/21centuryweather/dask_setup/wiki/Configuration) | `DaskSetupConfig`, profiles, site-wide profiles, profile inheritance, CLI, JSON Schema |
|
|
191
|
+
| [Multi-Node](https://github.com/21centuryweather/dask_setup/wiki/Multi-Node) | `MultiNodeConfig`, PBS/SLURM cluster setup, GPU topology, shared temp dirs, `dask-setup submit` |
|
|
192
|
+
| [IO-Optimization](https://github.com/21centuryweather/dask_setup/wiki/IO-Optimization) | `recommend_chunks`, `recommend_io_chunks`, Zarr v3, Kerchunk, Parquet/Arrow, storage-aware chunking |
|
|
193
|
+
| [Benchmarking](https://github.com/21centuryweather/dask_setup/wiki/Benchmarking) | `benchmark_config`, `scaling_analysis`, `chunk_impact`, `dask-setup benchmark` |
|
|
194
|
+
| [Internals](https://github.com/21centuryweather/dask_setup/wiki/Internals) | Resource detection, topology decisions, temp/spill routing, module layout |
|
|
195
|
+
| [Troubleshooting](https://github.com/21centuryweather/dask_setup/wiki/Troubleshooting) | Common errors, OOM, multi-node issues, migration guide |
|
|
196
|
+
|
|
197
|
+
---
|
|
198
|
+
|
|
199
|
+
## Installation
|
|
200
|
+
|
|
201
|
+
```bash
|
|
202
|
+
pip install dask-setup
|
|
203
|
+
```
|
|
204
|
+
|
|
205
|
+
For multi-node PBS/SLURM support:
|
|
206
|
+
|
|
207
|
+
```bash
|
|
208
|
+
pip install dask-setup dask-jobqueue
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
For GPU workloads (CuPy auto-detection):
|
|
212
|
+
|
|
213
|
+
```bash
|
|
214
|
+
pip install dask-setup cupy-cuda12x # match your CUDA version
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
---
|
|
218
|
+
|
|
219
|
+
## Contributing
|
|
220
|
+
|
|
221
|
+
Bug reports, feature requests, and pull requests are welcome. Please include tests and, for performance changes, benchmarks.
|
|
222
|
+
|
|
223
|
+
## License
|
|
224
|
+
|
|
225
|
+
Apache-2.0 — see [LICENSE](LICENSE) for details.
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "dask_setup"
|
|
7
|
-
version = "
|
|
7
|
+
version = "2.0.0"
|
|
8
8
|
description = "HPC-tuned Dask helpers for single-node runs on NCI Gadi."
|
|
9
9
|
authors = [{name="Sam Green"}]
|
|
10
10
|
readme = "README.md"
|
|
@@ -32,6 +32,9 @@ dev = [
|
|
|
32
32
|
"numpy",
|
|
33
33
|
]
|
|
34
34
|
|
|
35
|
+
[tool.setuptools.package-data]
|
|
36
|
+
dask_setup = ["py.typed", "schema/profile_schema.json"]
|
|
37
|
+
|
|
35
38
|
[tool.ruff]
|
|
36
39
|
line-length = 100
|
|
37
40
|
target-version = "py311"
|