dask-setup 1.0.0__tar.gz → 2.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. dask_setup-2.0.0/PKG-INFO +246 -0
  2. dask_setup-2.0.0/README.md +225 -0
  3. {dask_setup-1.0.0 → dask_setup-2.0.0}/pyproject.toml +4 -1
  4. dask_setup-2.0.0/src/dask_setup/__init__.py +351 -0
  5. dask_setup-2.0.0/src/dask_setup/benchmark.py +1176 -0
  6. dask_setup-2.0.0/src/dask_setup/callbacks.py +162 -0
  7. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/cli.py +333 -0
  8. dask_setup-2.0.0/src/dask_setup/client.py +721 -0
  9. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/cluster.py +25 -2
  10. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/config.py +59 -3
  11. dask_setup-2.0.0/src/dask_setup/config_manager.py +644 -0
  12. dask_setup-2.0.0/src/dask_setup/dashboard.py +123 -0
  13. dask_setup-2.0.0/src/dask_setup/environment.py +82 -0
  14. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/error_handling.py +24 -1
  15. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/io_patterns.py +376 -8
  16. dask_setup-2.0.0/src/dask_setup/multinode.py +768 -0
  17. dask_setup-2.0.0/src/dask_setup/parquet.py +335 -0
  18. dask_setup-2.0.0/src/dask_setup/py.typed +0 -0
  19. dask_setup-2.0.0/src/dask_setup/rechunk.py +196 -0
  20. dask_setup-2.0.0/src/dask_setup/reporting.py +237 -0
  21. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/resources.py +63 -5
  22. dask_setup-2.0.0/src/dask_setup/schema/__init__.py +35 -0
  23. dask_setup-2.0.0/src/dask_setup/schema/profile_schema.json +262 -0
  24. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/topology.py +82 -8
  25. dask_setup-2.0.0/src/dask_setup/tune.py +283 -0
  26. dask_setup-2.0.0/src/dask_setup/workload.py +270 -0
  27. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/xarray.py +284 -35
  28. dask_setup-2.0.0/src/dask_setup.egg-info/PKG-INFO +246 -0
  29. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup.egg-info/SOURCES.txt +14 -0
  30. dask_setup-2.0.0/tests/test_benchmark.py +517 -0
  31. dask_setup-2.0.0/tests/test_multinode.py +890 -0
  32. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_xarray_chunks.py +330 -0
  33. dask_setup-1.0.0/PKG-INFO +0 -817
  34. dask_setup-1.0.0/README.md +0 -796
  35. dask_setup-1.0.0/src/dask_setup/__init__.py +0 -101
  36. dask_setup-1.0.0/src/dask_setup/client.py +0 -280
  37. dask_setup-1.0.0/src/dask_setup/config_manager.py +0 -343
  38. dask_setup-1.0.0/src/dask_setup/dashboard.py +0 -71
  39. dask_setup-1.0.0/src/dask_setup.egg-info/PKG-INFO +0 -817
  40. {dask_setup-1.0.0 → dask_setup-2.0.0}/LICENSE +0 -0
  41. {dask_setup-1.0.0 → dask_setup-2.0.0}/setup.cfg +0 -0
  42. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/exceptions.py +0 -0
  43. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/legacy.py +0 -0
  44. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/logging.py +0 -0
  45. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/tempdir.py +0 -0
  46. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup/types.py +0 -0
  47. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup.egg-info/dependency_links.txt +0 -0
  48. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup.egg-info/entry_points.txt +0 -0
  49. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup.egg-info/requires.txt +0 -0
  50. {dask_setup-1.0.0 → dask_setup-2.0.0}/src/dask_setup.egg-info/top_level.txt +0 -0
  51. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_cli.py +0 -0
  52. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_client.py +0 -0
  53. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_cluster.py +0 -0
  54. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_compression.py +0 -0
  55. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_config.py +0 -0
  56. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_config_manager.py +0 -0
  57. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_dashboard.py +0 -0
  58. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_error_handling.py +0 -0
  59. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_exceptions.py +0 -0
  60. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_io_patterns.py +0 -0
  61. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_resources.py +0 -0
  62. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_setup_dask_client.py +0 -0
  63. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_tempdir.py +0 -0
  64. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_topology.py +0 -0
  65. {dask_setup-1.0.0 → dask_setup-2.0.0}/tests/test_types.py +0 -0
@@ -0,0 +1,246 @@
1
+ Metadata-Version: 2.4
2
+ Name: dask_setup
3
+ Version: 2.0.0
4
+ Summary: HPC-tuned Dask helpers for single-node runs on NCI Gadi.
5
+ Author: Sam Green
6
+ Project-URL: Homepage, https://github.com/21centuryweather/dask_setup
7
+ Requires-Python: >=3.11
8
+ Description-Content-Type: text/markdown
9
+ License-File: LICENSE
10
+ Requires-Dist: dask>=2024.1.0
11
+ Requires-Dist: distributed>=2024.1.0
12
+ Requires-Dist: psutil>=5.9
13
+ Requires-Dist: pyyaml>=6.0
14
+ Provides-Extra: dev
15
+ Requires-Dist: ruff~=0.4; extra == "dev"
16
+ Requires-Dist: pytest~=8.2; extra == "dev"
17
+ Requires-Dist: pytest-cov~=5.0; extra == "dev"
18
+ Requires-Dist: xarray; extra == "dev"
19
+ Requires-Dist: numpy; extra == "dev"
20
+ Dynamic: license-file
21
+
22
+ # dask_setup
23
+
24
+ [![CI](https://github.com/21centuryweather/dask_setup/workflows/CI/badge.svg)](https://github.com/21centuryweather/dask_setup/actions)
25
+
26
+ HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask.distributed.LocalCluster` + `Client` with sensible defaults for single-node jobs, and extends to multi-node PBS/SLURM clusters via `dask-jobqueue` in v2.0.
27
+
28
+ **Python 3.11+** | `pip install dask-setup`
29
+
30
+ ---
31
+
32
+ ## Quick Start
33
+
34
+ ```python
35
+ from dask_setup import setup_dask_client
36
+
37
+ # Pick a workload type and go
38
+ client, cluster, dask_tmp = setup_dask_client("cpu") # heavy compute
39
+ client, cluster, dask_tmp = setup_dask_client("io") # heavy file I/O
40
+ client, cluster, dask_tmp = setup_dask_client("mixed") # both
41
+ ```
42
+
43
+ `dask_tmp` is the path to the spill/temp directory (on `$PBS_JOBFS` if available). Pass it to Rechunker, Zarr, or anywhere else you want fast local I/O.
44
+
45
+ For more control, pass a `DaskSetupConfig` object or a named profile:
46
+
47
+ ```python
48
+ from dask_setup import setup_dask_client, DaskSetupConfig
49
+
50
+ config = DaskSetupConfig(
51
+ workload_type="cpu",
52
+ max_workers=8,
53
+ reserve_mem_gb=32.0,
54
+ spill_compression="lz4",
55
+ )
56
+ client, cluster, dask_tmp = setup_dask_client(config=config)
57
+
58
+ # Or use a built-in profile
59
+ client, cluster, dask_tmp = setup_dask_client(profile="climate_analysis")
60
+ ```
61
+
62
+ **Multi-node (v2.0):**
63
+
64
+ ```python
65
+ from dask_setup import setup_dask_client, MultiNodeConfig
66
+
67
+ client, cluster, shared_tmp = setup_dask_client(
68
+ mode="auto", # detects PBS_JOBID / SLURM_JOB_ID, falls back to local
69
+ multi_node_config=MultiNodeConfig(
70
+ workers_per_node=4,
71
+ cores_per_worker=12,
72
+ mem_per_worker_gb=32.0,
73
+ walltime="04:00:00",
74
+ ),
75
+ )
76
+ ```
77
+
78
+ ---
79
+
80
+ ## Workload Types
81
+
82
+ | Type | Topology | Best for |
83
+ |------|----------|----------|
84
+ | `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions |
85
+ | `"io"` | 1 process, 8–16 threads | Opening many NetCDF/Zarr files concurrently |
86
+ | `"mixed"` | Processes with 2 threads each | Pipelines that both read and compute |
87
+ | `"gpu"` | 1 process per GPU, up to 8 threads | CuPy/RAPIDS CUDA workloads |
88
+ | `"auto"` | Inferred from dataset | Let `dask_setup` decide based on your data |
89
+
90
+ ---
91
+
92
+ ## Key Parameters
93
+
94
+ | Parameter | Default | Description |
95
+ |-----------|---------|-------------|
96
+ | `workload_type` | `"io"` | Worker topology: `"cpu"`, `"io"`, `"mixed"`, `"gpu"`, `"auto"` |
97
+ | `max_workers` | all cores | Hard cap on worker count |
98
+ | `reserve_mem_gb` | auto (20% RAM) | Memory held back for OS/cache (GiB) |
99
+ | `max_mem_gb` | total RAM | Upper bound on Dask's total memory use |
100
+ | `dashboard` | `True` | Start dashboard and print SSH tunnel hint |
101
+ | `profile` | `None` | Named config profile |
102
+ | `config` | `None` | Pre-built `DaskSetupConfig` object — mutually exclusive with `profile` |
103
+ | `mode` | `"auto"` | `"local"`, `"pbs"`, `"slurm"`, or `"auto"` (v2.0) |
104
+ | `multi_node_config` | `None` | `MultiNodeConfig` for PBS/SLURM multi-node jobs (v2.0) |
105
+
106
+ ---
107
+
108
+ ## Common Patterns
109
+
110
+ **Big xarray reductions (CPU-bound)**
111
+ ```python
112
+ client, cluster, dask_tmp = setup_dask_client("cpu", reserve_mem_gb=60)
113
+ ds = ds.chunk({"time": 240, "y": 512, "x": 512})
114
+ out = ds.mean(("y", "x")).compute()
115
+ ```
116
+
117
+ **Opening many NetCDF/Zarr files (I/O-bound)**
118
+ ```python
119
+ client, cluster, dask_tmp = setup_dask_client("io", reserve_mem_gb=40)
120
+ ds = xr.open_mfdataset(files, engine="netcdf4", chunks={}, parallel=True)
121
+ ```
122
+
123
+ **Rechunking with Rechunker (spill stays on fast local storage)**
124
+ ```python
125
+ client, cluster, dask_tmp = setup_dask_client("cpu", reserve_mem_gb=60)
126
+ plan = rechunker.rechunk(
127
+ ds.to_array().data,
128
+ target_chunks={"time": 240, "y": 512, "x": 512},
129
+ max_mem="6GB",
130
+ target_store="out.zarr",
131
+ temp_store=f"{dask_tmp}/tmp_rechunk.zarr",
132
+ )
133
+ plan.execute()
134
+ ```
135
+
136
+ **Writing to Zarr in time windows**
137
+ ```python
138
+ client, cluster, dask_tmp = setup_dask_client("cpu", max_workers=1, reserve_mem_gb=50)
139
+ ds.isel(time=slice(0, 240)).to_zarr("out.zarr", mode="w", consolidated=True)
140
+ for start in range(240, ds.sizes["time"], 240):
141
+ stop = min(start + 240, ds.sizes["time"])
142
+ ds.isel(time=slice(start, stop)).to_zarr(
143
+ "out.zarr", mode="a", region={"time": slice(start, stop)}
144
+ )
145
+ ```
146
+
147
+ ---
148
+
149
+ ## Built-in Profiles
150
+
151
+ Use `profile=` to load a named configuration:
152
+
153
+ ```python
154
+ client, cluster, dask_tmp = setup_dask_client(profile="climate_analysis")
155
+ ```
156
+
157
+ | Profile | Workload | Reserve | Notes |
158
+ |---------|----------|---------|-------|
159
+ | `climate_analysis` | cpu | 60 GB | Heavy compute with large arrays |
160
+ | `zarr_io_heavy` | io | 40 GB | Many Zarr files |
161
+ | `development` | mixed | 8 GB | Local testing, 2 workers max |
162
+ | `production` | mixed | 80 GB | Adaptive, dashboard off |
163
+ | `interactive` | mixed | 20 GB | Jupyter notebooks, 4 workers |
164
+
165
+ See [Configuration](https://github.com/21centuryweather/dask_setup/wiki/Configuration) for how to create and save your own profiles.
166
+
167
+ ---
168
+
169
+ ## Dashboard
170
+
171
+ With `dashboard=True` (default), the cluster prints an SSH tunnel command:
172
+
173
+ ```
174
+ Dask dashboard: http://127.0.0.1:<PORT>/status
175
+ Tunnel from your laptop:
176
+ ssh -N -L 8787:<COMPUTE_HOST>:<PORT> gadi.nci.org.au
177
+ Then open: http://localhost:8787
178
+ ```
179
+
180
+ ---
181
+
182
+ ## CLI
183
+
184
+ ```bash
185
+ # Profile management
186
+ dask-setup list # list all profiles
187
+ dask-setup show climate_analysis # show profile details
188
+ dask-setup create my_profile --from-profile zarr_io_heavy
189
+ dask-setup validate my_profile
190
+ dask-setup export climate_analysis -o profile.yaml
191
+ dask-setup import https://example.com/team.yaml
192
+ dask-setup delete my_profile
193
+
194
+ # Synthetic benchmark
195
+ dask-setup benchmark --profile development --size small --operation mean
196
+
197
+ # Generate a PBS/SLURM job script (v2.0)
198
+ dask-setup submit my_analysis.py --scheduler pbs \
199
+ --workers-per-node 4 --cores-per-worker 12 --mem-per-worker 32 \
200
+ --walltime 04:00:00 --queue normal --project ab01
201
+ ```
202
+
203
+ ---
204
+
205
+ ## Documentation
206
+
207
+ Full documentation lives in the [GitHub wiki](https://github.com/21centuryweather/dask_setup/wiki):
208
+
209
+ | Page | What's covered |
210
+ |------|----------------|
211
+ | [Configuration](https://github.com/21centuryweather/dask_setup/wiki/Configuration) | `DaskSetupConfig`, profiles, site-wide profiles, profile inheritance, CLI, JSON Schema |
212
+ | [Multi-Node](https://github.com/21centuryweather/dask_setup/wiki/Multi-Node) | `MultiNodeConfig`, PBS/SLURM cluster setup, GPU topology, shared temp dirs, `dask-setup submit` |
213
+ | [IO-Optimization](https://github.com/21centuryweather/dask_setup/wiki/IO-Optimization) | `recommend_chunks`, `recommend_io_chunks`, Zarr v3, Kerchunk, Parquet/Arrow, storage-aware chunking |
214
+ | [Benchmarking](https://github.com/21centuryweather/dask_setup/wiki/Benchmarking) | `benchmark_config`, `scaling_analysis`, `chunk_impact`, `dask-setup benchmark` |
215
+ | [Internals](https://github.com/21centuryweather/dask_setup/wiki/Internals) | Resource detection, topology decisions, temp/spill routing, module layout |
216
+ | [Troubleshooting](https://github.com/21centuryweather/dask_setup/wiki/Troubleshooting) | Common errors, OOM, multi-node issues, migration guide |
217
+
218
+ ---
219
+
220
+ ## Installation
221
+
222
+ ```bash
223
+ pip install dask-setup
224
+ ```
225
+
226
+ For multi-node PBS/SLURM support:
227
+
228
+ ```bash
229
+ pip install dask-setup dask-jobqueue
230
+ ```
231
+
232
+ For GPU workloads (CuPy auto-detection):
233
+
234
+ ```bash
235
+ pip install dask-setup cupy-cuda12x # match your CUDA version
236
+ ```
237
+
238
+ ---
239
+
240
+ ## Contributing
241
+
242
+ Bug reports, feature requests, and pull requests are welcome. Please include tests and, for performance changes, benchmarks.
243
+
244
+ ## License
245
+
246
+ Apache-2.0 — see [LICENSE](LICENSE) for details.
@@ -0,0 +1,225 @@
1
+ # dask_setup
2
+
3
+ [![CI](https://github.com/21centuryweather/dask_setup/workflows/CI/badge.svg)](https://github.com/21centuryweather/dask_setup/actions)
4
+
5
+ HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask.distributed.LocalCluster` + `Client` with sensible defaults for single-node jobs, and extends to multi-node PBS/SLURM clusters via `dask-jobqueue` in v2.0.
6
+
7
+ **Python 3.11+** | `pip install dask-setup`
8
+
9
+ ---
10
+
11
+ ## Quick Start
12
+
13
+ ```python
14
+ from dask_setup import setup_dask_client
15
+
16
+ # Pick a workload type and go
17
+ client, cluster, dask_tmp = setup_dask_client("cpu") # heavy compute
18
+ client, cluster, dask_tmp = setup_dask_client("io") # heavy file I/O
19
+ client, cluster, dask_tmp = setup_dask_client("mixed") # both
20
+ ```
21
+
22
+ `dask_tmp` is the path to the spill/temp directory (on `$PBS_JOBFS` if available). Pass it to Rechunker, Zarr, or anywhere else you want fast local I/O.
23
+
24
+ For more control, pass a `DaskSetupConfig` object or a named profile:
25
+
26
+ ```python
27
+ from dask_setup import setup_dask_client, DaskSetupConfig
28
+
29
+ config = DaskSetupConfig(
30
+ workload_type="cpu",
31
+ max_workers=8,
32
+ reserve_mem_gb=32.0,
33
+ spill_compression="lz4",
34
+ )
35
+ client, cluster, dask_tmp = setup_dask_client(config=config)
36
+
37
+ # Or use a built-in profile
38
+ client, cluster, dask_tmp = setup_dask_client(profile="climate_analysis")
39
+ ```
40
+
41
+ **Multi-node (v2.0):**
42
+
43
+ ```python
44
+ from dask_setup import setup_dask_client, MultiNodeConfig
45
+
46
+ client, cluster, shared_tmp = setup_dask_client(
47
+ mode="auto", # detects PBS_JOBID / SLURM_JOB_ID, falls back to local
48
+ multi_node_config=MultiNodeConfig(
49
+ workers_per_node=4,
50
+ cores_per_worker=12,
51
+ mem_per_worker_gb=32.0,
52
+ walltime="04:00:00",
53
+ ),
54
+ )
55
+ ```
56
+
57
+ ---
58
+
59
+ ## Workload Types
60
+
61
+ | Type | Topology | Best for |
62
+ |------|----------|----------|
63
+ | `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions |
64
+ | `"io"` | 1 process, 8–16 threads | Opening many NetCDF/Zarr files concurrently |
65
+ | `"mixed"` | Processes with 2 threads each | Pipelines that both read and compute |
66
+ | `"gpu"` | 1 process per GPU, up to 8 threads | CuPy/RAPIDS CUDA workloads |
67
+ | `"auto"` | Inferred from dataset | Let `dask_setup` decide based on your data |
68
+
69
+ ---
70
+
71
+ ## Key Parameters
72
+
73
+ | Parameter | Default | Description |
74
+ |-----------|---------|-------------|
75
+ | `workload_type` | `"io"` | Worker topology: `"cpu"`, `"io"`, `"mixed"`, `"gpu"`, `"auto"` |
76
+ | `max_workers` | all cores | Hard cap on worker count |
77
+ | `reserve_mem_gb` | auto (20% RAM) | Memory held back for OS/cache (GiB) |
78
+ | `max_mem_gb` | total RAM | Upper bound on Dask's total memory use |
79
+ | `dashboard` | `True` | Start dashboard and print SSH tunnel hint |
80
+ | `profile` | `None` | Named config profile |
81
+ | `config` | `None` | Pre-built `DaskSetupConfig` object — mutually exclusive with `profile` |
82
+ | `mode` | `"auto"` | `"local"`, `"pbs"`, `"slurm"`, or `"auto"` (v2.0) |
83
+ | `multi_node_config` | `None` | `MultiNodeConfig` for PBS/SLURM multi-node jobs (v2.0) |
84
+
85
+ ---
86
+
87
+ ## Common Patterns
88
+
89
+ **Big xarray reductions (CPU-bound)**
90
+ ```python
91
+ client, cluster, dask_tmp = setup_dask_client("cpu", reserve_mem_gb=60)
92
+ ds = ds.chunk({"time": 240, "y": 512, "x": 512})
93
+ out = ds.mean(("y", "x")).compute()
94
+ ```
95
+
96
+ **Opening many NetCDF/Zarr files (I/O-bound)**
97
+ ```python
98
+ client, cluster, dask_tmp = setup_dask_client("io", reserve_mem_gb=40)
99
+ ds = xr.open_mfdataset(files, engine="netcdf4", chunks={}, parallel=True)
100
+ ```
101
+
102
+ **Rechunking with Rechunker (spill stays on fast local storage)**
103
+ ```python
104
+ client, cluster, dask_tmp = setup_dask_client("cpu", reserve_mem_gb=60)
105
+ plan = rechunker.rechunk(
106
+ ds.to_array().data,
107
+ target_chunks={"time": 240, "y": 512, "x": 512},
108
+ max_mem="6GB",
109
+ target_store="out.zarr",
110
+ temp_store=f"{dask_tmp}/tmp_rechunk.zarr",
111
+ )
112
+ plan.execute()
113
+ ```
114
+
115
+ **Writing to Zarr in time windows**
116
+ ```python
117
+ client, cluster, dask_tmp = setup_dask_client("cpu", max_workers=1, reserve_mem_gb=50)
118
+ ds.isel(time=slice(0, 240)).to_zarr("out.zarr", mode="w", consolidated=True)
119
+ for start in range(240, ds.sizes["time"], 240):
120
+ stop = min(start + 240, ds.sizes["time"])
121
+ ds.isel(time=slice(start, stop)).to_zarr(
122
+ "out.zarr", mode="a", region={"time": slice(start, stop)}
123
+ )
124
+ ```
125
+
126
+ ---
127
+
128
+ ## Built-in Profiles
129
+
130
+ Use `profile=` to load a named configuration:
131
+
132
+ ```python
133
+ client, cluster, dask_tmp = setup_dask_client(profile="climate_analysis")
134
+ ```
135
+
136
+ | Profile | Workload | Reserve | Notes |
137
+ |---------|----------|---------|-------|
138
+ | `climate_analysis` | cpu | 60 GB | Heavy compute with large arrays |
139
+ | `zarr_io_heavy` | io | 40 GB | Many Zarr files |
140
+ | `development` | mixed | 8 GB | Local testing, 2 workers max |
141
+ | `production` | mixed | 80 GB | Adaptive, dashboard off |
142
+ | `interactive` | mixed | 20 GB | Jupyter notebooks, 4 workers |
143
+
144
+ See [Configuration](https://github.com/21centuryweather/dask_setup/wiki/Configuration) for how to create and save your own profiles.
145
+
146
+ ---
147
+
148
+ ## Dashboard
149
+
150
+ With `dashboard=True` (default), the cluster prints an SSH tunnel command:
151
+
152
+ ```
153
+ Dask dashboard: http://127.0.0.1:<PORT>/status
154
+ Tunnel from your laptop:
155
+ ssh -N -L 8787:<COMPUTE_HOST>:<PORT> gadi.nci.org.au
156
+ Then open: http://localhost:8787
157
+ ```
158
+
159
+ ---
160
+
161
+ ## CLI
162
+
163
+ ```bash
164
+ # Profile management
165
+ dask-setup list # list all profiles
166
+ dask-setup show climate_analysis # show profile details
167
+ dask-setup create my_profile --from-profile zarr_io_heavy
168
+ dask-setup validate my_profile
169
+ dask-setup export climate_analysis -o profile.yaml
170
+ dask-setup import https://example.com/team.yaml
171
+ dask-setup delete my_profile
172
+
173
+ # Synthetic benchmark
174
+ dask-setup benchmark --profile development --size small --operation mean
175
+
176
+ # Generate a PBS/SLURM job script (v2.0)
177
+ dask-setup submit my_analysis.py --scheduler pbs \
178
+ --workers-per-node 4 --cores-per-worker 12 --mem-per-worker 32 \
179
+ --walltime 04:00:00 --queue normal --project ab01
180
+ ```
181
+
182
+ ---
183
+
184
+ ## Documentation
185
+
186
+ Full documentation lives in the [GitHub wiki](https://github.com/21centuryweather/dask_setup/wiki):
187
+
188
+ | Page | What's covered |
189
+ |------|----------------|
190
+ | [Configuration](https://github.com/21centuryweather/dask_setup/wiki/Configuration) | `DaskSetupConfig`, profiles, site-wide profiles, profile inheritance, CLI, JSON Schema |
191
+ | [Multi-Node](https://github.com/21centuryweather/dask_setup/wiki/Multi-Node) | `MultiNodeConfig`, PBS/SLURM cluster setup, GPU topology, shared temp dirs, `dask-setup submit` |
192
+ | [IO-Optimization](https://github.com/21centuryweather/dask_setup/wiki/IO-Optimization) | `recommend_chunks`, `recommend_io_chunks`, Zarr v3, Kerchunk, Parquet/Arrow, storage-aware chunking |
193
+ | [Benchmarking](https://github.com/21centuryweather/dask_setup/wiki/Benchmarking) | `benchmark_config`, `scaling_analysis`, `chunk_impact`, `dask-setup benchmark` |
194
+ | [Internals](https://github.com/21centuryweather/dask_setup/wiki/Internals) | Resource detection, topology decisions, temp/spill routing, module layout |
195
+ | [Troubleshooting](https://github.com/21centuryweather/dask_setup/wiki/Troubleshooting) | Common errors, OOM, multi-node issues, migration guide |
196
+
197
+ ---
198
+
199
+ ## Installation
200
+
201
+ ```bash
202
+ pip install dask-setup
203
+ ```
204
+
205
+ For multi-node PBS/SLURM support:
206
+
207
+ ```bash
208
+ pip install dask-setup dask-jobqueue
209
+ ```
210
+
211
+ For GPU workloads (CuPy auto-detection):
212
+
213
+ ```bash
214
+ pip install dask-setup cupy-cuda12x # match your CUDA version
215
+ ```
216
+
217
+ ---
218
+
219
+ ## Contributing
220
+
221
+ Bug reports, feature requests, and pull requests are welcome. Please include tests and, for performance changes, benchmarks.
222
+
223
+ ## License
224
+
225
+ Apache-2.0 — see [LICENSE](LICENSE) for details.
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "dask_setup"
7
- version = "1.0.0"
7
+ version = "2.0.0"
8
8
  description = "HPC-tuned Dask helpers for single-node runs on NCI Gadi."
9
9
  authors = [{name="Sam Green"}]
10
10
  readme = "README.md"
@@ -32,6 +32,9 @@ dev = [
32
32
  "numpy",
33
33
  ]
34
34
 
35
+ [tool.setuptools.package-data]
36
+ dask_setup = ["py.typed", "schema/profile_schema.json"]
37
+
35
38
  [tool.ruff]
36
39
  line-length = 100
37
40
  target-version = "py311"