dask-setup 2.2.0__tar.gz → 2.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. dask_setup-2.3.0/MANIFEST.in +5 -0
  2. {dask_setup-2.2.0/src/dask_setup.egg-info → dask_setup-2.3.0}/PKG-INFO +62 -7
  3. dask_setup-2.2.0/PKG-INFO → dask_setup-2.3.0/README.md +51 -30
  4. {dask_setup-2.2.0 → dask_setup-2.3.0}/pyproject.toml +15 -2
  5. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/__init__.py +1 -1
  6. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/benchmark.py +3 -2
  7. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/client.py +80 -31
  8. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/dashboard.py +6 -1
  9. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/exceptions.py +1 -1
  10. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/multinode.py +31 -20
  11. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/parquet.py +3 -2
  12. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/rechunk.py +58 -51
  13. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/reporting.py +46 -8
  14. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/tune.py +2 -3
  15. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/xarray.py +102 -5
  16. dask_setup-2.2.0/README.md → dask_setup-2.3.0/src/dask_setup.egg-info/PKG-INFO +85 -6
  17. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/SOURCES.txt +3 -0
  18. dask_setup-2.3.0/tests/__init__.py +1 -0
  19. dask_setup-2.3.0/tests/conftest.py +132 -0
  20. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_client.py +127 -4
  21. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_dashboard.py +14 -0
  22. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_error_handling.py +5 -4
  23. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_exceptions.py +7 -0
  24. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_multinode.py +61 -0
  25. dask_setup-2.3.0/tests/test_rechunk.py +159 -0
  26. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_reporting.py +77 -0
  27. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_setup_dask_client.py +75 -0
  28. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_xarray_chunks.py +194 -0
  29. dask_setup-2.2.0/tests/test_rechunk.py +0 -70
  30. {dask_setup-2.2.0 → dask_setup-2.3.0}/LICENSE +0 -0
  31. {dask_setup-2.2.0 → dask_setup-2.3.0}/setup.cfg +0 -0
  32. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/callbacks.py +0 -0
  33. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/cli.py +0 -0
  34. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/cluster.py +0 -0
  35. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/config.py +0 -0
  36. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/config_manager.py +0 -0
  37. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/environment.py +0 -0
  38. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/error_handling.py +0 -0
  39. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/io_patterns.py +0 -0
  40. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/legacy.py +0 -0
  41. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/logging.py +0 -0
  42. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/py.typed +0 -0
  43. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/resources.py +0 -0
  44. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/schema/__init__.py +0 -0
  45. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/schema/profile_schema.json +0 -0
  46. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/tempdir.py +0 -0
  47. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/topology.py +0 -0
  48. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/types.py +0 -0
  49. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/workload.py +0 -0
  50. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/dependency_links.txt +0 -0
  51. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/entry_points.txt +0 -0
  52. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/requires.txt +0 -0
  53. {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/top_level.txt +0 -0
  54. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_benchmark.py +0 -0
  55. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_cli.py +0 -0
  56. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_cluster.py +0 -0
  57. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_compression.py +0 -0
  58. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_config.py +0 -0
  59. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_config_manager.py +0 -0
  60. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_io_patterns.py +0 -0
  61. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_logging.py +0 -0
  62. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_resources.py +0 -0
  63. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_tempdir.py +0 -0
  64. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_topology.py +0 -0
  65. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_tune.py +0 -0
  66. {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_types.py +0 -0
@@ -0,0 +1,5 @@
1
+ # setuptools only auto-includes tests/test*.py in the sdist, which leaves out
2
+ # conftest.py (the shared fixtures) and __init__.py -- so the suite could not
3
+ # run from the sdist, as downstream packagers like conda-forge do.
4
+ recursive-include tests *.py
5
+ global-exclude __pycache__ *.py[cod] .DS_Store
@@ -1,9 +1,19 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dask_setup
3
- Version: 2.2.0
3
+ Version: 2.3.0
4
4
  Summary: HPC-tuned Dask helpers for single-node runs on NCI Gadi.
5
5
  Author: Sam Green
6
+ License-Expression: Apache-2.0
6
7
  Project-URL: Homepage, https://github.com/21centuryweather/dask_setup
8
+ Classifier: Development Status :: 5 - Production/Stable
9
+ Classifier: Intended Audience :: Science/Research
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: Programming Language :: Python :: 3.11
12
+ Classifier: Programming Language :: Python :: 3.12
13
+ Classifier: Programming Language :: Python :: 3.13
14
+ Classifier: Programming Language :: Python :: 3.14
15
+ Classifier: Topic :: Scientific/Engineering
16
+ Classifier: Topic :: System :: Distributed Computing
7
17
  Requires-Python: >=3.11
8
18
  Description-Content-Type: text/markdown
9
19
  License-File: LICENSE
@@ -24,11 +34,22 @@ Dynamic: license-file
24
34
 
25
35
  # dask_setup
26
36
 
27
- [![CI](https://github.com/21centuryweather/dask_setup/workflows/CI/badge.svg)](https://github.com/21centuryweather/dask_setup/actions)
37
+ [![CI](https://github.com/21centuryweather/dask_setup/actions/workflows/ci.yml/badge.svg?branch=main)](https://github.com/21centuryweather/dask_setup/actions/workflows/ci.yml)
38
+ [![Release](https://github.com/21centuryweather/dask_setup/actions/workflows/publish-to-pypi.yml/badge.svg)](https://github.com/21centuryweather/dask_setup/actions/workflows/publish-to-pypi.yml)
39
+ [![PyPI](https://img.shields.io/pypi/v/dask_setup.svg)](https://pypi.org/project/dask_setup/)
40
+ [![conda-forge](https://img.shields.io/conda/vn/conda-forge/dask-setup.svg)](https://anaconda.org/conda-forge/dask-setup)
41
+ [![Python](https://img.shields.io/badge/python-3.11%2B-blue.svg)](https://pypi.org/project/dask_setup/)
42
+ [![License](https://img.shields.io/badge/license-Apache%202.0-blue.svg)](LICENSE)
43
+ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
28
44
 
29
45
  HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask.distributed.LocalCluster` + `Client` with sensible defaults for single-node jobs, and extends to multi-node PBS/SLURM clusters via `dask-jobqueue` in v2.0.
30
46
 
31
- **Python 3.11+** | `pip install dask-setup`
47
+ ```bash
48
+ pip install dask-setup # single-node
49
+ pip install dask-setup[multinode] # + PBS/SLURM via dask-jobqueue
50
+
51
+ conda install -c conda-forge dask-setup # or via conda-forge
52
+ ```
32
53
 
33
54
  ---
34
55
 
@@ -39,7 +60,7 @@ from dask_setup import setup_dask_client
39
60
 
40
61
  # Pick a workload type and go
41
62
  client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="cpu") # heavy compute
42
- client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="io") # heavy file I/O
63
+ client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="io") # Zarr / object storage
43
64
  client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="mixed") # both
44
65
  ```
45
66
 
@@ -84,12 +105,35 @@ client, cluster, shared_tmp = setup_dask_client(
84
105
 
85
106
  | Type | Topology | Best for |
86
107
  |------|----------|----------|
87
- | `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions |
88
- | `"io"` | 1 process, 8–16 threads | Opening many NetCDF/Zarr files concurrently |
108
+ | `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions, **and reading NetCDF/HDF5** |
109
+ | `"io"` | 1 process, 8–16 threads | Zarr, object storage (S3), HTTP — libraries that release the GIL |
89
110
  | `"mixed"` | Processes with 2 threads each | Pipelines that both read and compute |
90
111
  | `"gpu"` | 1 process per GPU, up to 8 threads | CuPy/RAPIDS CUDA workloads |
91
112
  | `"auto"` | Inferred from dataset | Let `dask_setup` decide based on your data |
92
113
 
114
+ ### Reading NetCDF? Use `"cpu"`, not `"io"`
115
+
116
+ The split is not "I/O vs compute" — it is **whether the library releases the GIL
117
+ and is thread-safe**. `"io"` runs one process with many threads, which only wins
118
+ when threads can genuinely overlap.
119
+
120
+ HDF5 (underneath NetCDF4) is usually not built thread-safe, so xarray routes
121
+ *every* NetCDF read through a single process-wide lock. Under `"io"` your threads
122
+ do not become parallel readers — they queue on that one lock, one at a time, and
123
+ zlib decompression is serialised inside it too. Separate processes each get their
124
+ own lock, so `"cpu"` is what actually parallelises NetCDF reads.
125
+
126
+ Zarr is the opposite: it decompresses via numcodecs, which releases the GIL, and
127
+ has no global lock, so threads scale well. Opening, concatenating and writing
128
+ NetCDF is a `"cpu"` job even though it feels like I/O.
129
+
130
+ > **Rule of thumb: NetCDF in → `"cpu"`. Zarr or object storage in → `"io"`.**
131
+ > Pass `ds=` and `dask_setup` will warn if you get this pair the wrong way round.
132
+ >
133
+ > Also avoid `open_mfdataset(..., parallel=True)` together with `"io"`: concurrent
134
+ > metadata reads can kill the worker outright, after which Dask silently
135
+ > recomputes the lost tasks and the job appears to hang rather than fail.
136
+
93
137
  ---
94
138
 
95
139
  ## Key Parameters
@@ -103,7 +147,7 @@ client, cluster, shared_tmp = setup_dask_client(
103
147
  | `dashboard` | `True` | Start dashboard and print SSH tunnel hint |
104
148
  | `profile` | `None` | Named config profile |
105
149
  | `config` | `None` | Pre-built `DaskSetupConfig` object — same layer as `profile`; if both are given, `profile` wins |
106
- | `mode` | `"auto"` | `"local"`, `"pbs"`, `"slurm"`, or `"auto"` (v2.0) |
150
+ | `mode` | `"interactive"` | `"interactive"` uses the current allocation and never submits jobs (a single node or no job at all is a plain `LocalCluster`); `"auto"` picks `"pbs"`/`"slurm"` in batch jobs, which submit worker jobs; also `"local"`, `"pbs"`, `"slurm"` |
107
151
  | `multi_node_config` | `None` | `MultiNodeConfig` for PBS/SLURM multi-node jobs (v2.0) |
108
152
 
109
153
  Settings are layered, lowest to highest:
@@ -238,19 +282,30 @@ Full documentation lives in the [GitHub wiki](https://github.com/21centuryweathe
238
282
  | [Benchmarking](https://github.com/21centuryweather/dask_setup/wiki/Benchmarking) | `benchmark_config`, `scaling_analysis`, `chunk_impact`, `dask-setup benchmark` |
239
283
  | [Internals](https://github.com/21centuryweather/dask_setup/wiki/Internals) | Resource detection, topology decisions, temp/spill routing, module layout |
240
284
  | [Troubleshooting](https://github.com/21centuryweather/dask_setup/wiki/Troubleshooting) | Common errors, OOM, multi-node issues, migration guide |
285
+ | [User Feedback](https://github.com/21centuryweather/dask_setup/wiki/User-Feedback) | Questions from users, the answers, and what changed as a result |
241
286
 
242
287
  ---
243
288
 
244
289
  ## Installation
245
290
 
291
+ From PyPI:
292
+
246
293
  ```bash
247
294
  pip install dask-setup
248
295
  ```
249
296
 
297
+ From conda-forge:
298
+
299
+ ```bash
300
+ conda install -c conda-forge dask-setup
301
+ ```
302
+
250
303
  For multi-node PBS/SLURM support:
251
304
 
252
305
  ```bash
253
306
  pip install dask-setup dask-jobqueue
307
+ # or
308
+ conda install -c conda-forge dask-setup dask-jobqueue
254
309
  ```
255
310
 
256
311
  For GPU workloads (CuPy auto-detection):
@@ -1,34 +1,21 @@
1
- Metadata-Version: 2.4
2
- Name: dask_setup
3
- Version: 2.2.0
4
- Summary: HPC-tuned Dask helpers for single-node runs on NCI Gadi.
5
- Author: Sam Green
6
- Project-URL: Homepage, https://github.com/21centuryweather/dask_setup
7
- Requires-Python: >=3.11
8
- Description-Content-Type: text/markdown
9
- License-File: LICENSE
10
- Requires-Dist: dask>=2024.1.0
11
- Requires-Dist: distributed>=2024.1.0
12
- Requires-Dist: psutil>=5.9
13
- Requires-Dist: pyyaml>=6.0
14
- Provides-Extra: multinode
15
- Requires-Dist: dask-jobqueue>=0.8; extra == "multinode"
16
- Provides-Extra: dev
17
- Requires-Dist: ruff==0.16.4; extra == "dev"
18
- Requires-Dist: pytest~=8.2; extra == "dev"
19
- Requires-Dist: pytest-cov~=5.0; extra == "dev"
20
- Requires-Dist: xarray; extra == "dev"
21
- Requires-Dist: numpy; extra == "dev"
22
- Requires-Dist: dask-jobqueue>=0.8; extra == "dev"
23
- Dynamic: license-file
24
-
25
1
  # dask_setup
26
2
 
27
- [![CI](https://github.com/21centuryweather/dask_setup/workflows/CI/badge.svg)](https://github.com/21centuryweather/dask_setup/actions)
3
+ [![CI](https://github.com/21centuryweather/dask_setup/actions/workflows/ci.yml/badge.svg?branch=main)](https://github.com/21centuryweather/dask_setup/actions/workflows/ci.yml)
4
+ [![Release](https://github.com/21centuryweather/dask_setup/actions/workflows/publish-to-pypi.yml/badge.svg)](https://github.com/21centuryweather/dask_setup/actions/workflows/publish-to-pypi.yml)
5
+ [![PyPI](https://img.shields.io/pypi/v/dask_setup.svg)](https://pypi.org/project/dask_setup/)
6
+ [![conda-forge](https://img.shields.io/conda/vn/conda-forge/dask-setup.svg)](https://anaconda.org/conda-forge/dask-setup)
7
+ [![Python](https://img.shields.io/badge/python-3.11%2B-blue.svg)](https://pypi.org/project/dask_setup/)
8
+ [![License](https://img.shields.io/badge/license-Apache%202.0-blue.svg)](LICENSE)
9
+ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
28
10
 
29
11
  HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask.distributed.LocalCluster` + `Client` with sensible defaults for single-node jobs, and extends to multi-node PBS/SLURM clusters via `dask-jobqueue` in v2.0.
30
12
 
31
- **Python 3.11+** | `pip install dask-setup`
13
+ ```bash
14
+ pip install dask-setup # single-node
15
+ pip install dask-setup[multinode] # + PBS/SLURM via dask-jobqueue
16
+
17
+ conda install -c conda-forge dask-setup # or via conda-forge
18
+ ```
32
19
 
33
20
  ---
34
21
 
@@ -39,7 +26,7 @@ from dask_setup import setup_dask_client
39
26
 
40
27
  # Pick a workload type and go
41
28
  client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="cpu") # heavy compute
42
- client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="io") # heavy file I/O
29
+ client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="io") # Zarr / object storage
43
30
  client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="mixed") # both
44
31
  ```
45
32
 
@@ -84,12 +71,35 @@ client, cluster, shared_tmp = setup_dask_client(
84
71
 
85
72
  | Type | Topology | Best for |
86
73
  |------|----------|----------|
87
- | `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions |
88
- | `"io"` | 1 process, 8–16 threads | Opening many NetCDF/Zarr files concurrently |
74
+ | `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions, **and reading NetCDF/HDF5** |
75
+ | `"io"` | 1 process, 8–16 threads | Zarr, object storage (S3), HTTP — libraries that release the GIL |
89
76
  | `"mixed"` | Processes with 2 threads each | Pipelines that both read and compute |
90
77
  | `"gpu"` | 1 process per GPU, up to 8 threads | CuPy/RAPIDS CUDA workloads |
91
78
  | `"auto"` | Inferred from dataset | Let `dask_setup` decide based on your data |
92
79
 
80
+ ### Reading NetCDF? Use `"cpu"`, not `"io"`
81
+
82
+ The split is not "I/O vs compute" — it is **whether the library releases the GIL
83
+ and is thread-safe**. `"io"` runs one process with many threads, which only wins
84
+ when threads can genuinely overlap.
85
+
86
+ HDF5 (underneath NetCDF4) is usually not built thread-safe, so xarray routes
87
+ *every* NetCDF read through a single process-wide lock. Under `"io"` your threads
88
+ do not become parallel readers — they queue on that one lock, one at a time, and
89
+ zlib decompression is serialised inside it too. Separate processes each get their
90
+ own lock, so `"cpu"` is what actually parallelises NetCDF reads.
91
+
92
+ Zarr is the opposite: it decompresses via numcodecs, which releases the GIL, and
93
+ has no global lock, so threads scale well. Opening, concatenating and writing
94
+ NetCDF is a `"cpu"` job even though it feels like I/O.
95
+
96
+ > **Rule of thumb: NetCDF in → `"cpu"`. Zarr or object storage in → `"io"`.**
97
+ > Pass `ds=` and `dask_setup` will warn if you get this pair the wrong way round.
98
+ >
99
+ > Also avoid `open_mfdataset(..., parallel=True)` together with `"io"`: concurrent
100
+ > metadata reads can kill the worker outright, after which Dask silently
101
+ > recomputes the lost tasks and the job appears to hang rather than fail.
102
+
93
103
  ---
94
104
 
95
105
  ## Key Parameters
@@ -103,7 +113,7 @@ client, cluster, shared_tmp = setup_dask_client(
103
113
  | `dashboard` | `True` | Start dashboard and print SSH tunnel hint |
104
114
  | `profile` | `None` | Named config profile |
105
115
  | `config` | `None` | Pre-built `DaskSetupConfig` object — same layer as `profile`; if both are given, `profile` wins |
106
- | `mode` | `"auto"` | `"local"`, `"pbs"`, `"slurm"`, or `"auto"` (v2.0) |
116
+ | `mode` | `"interactive"` | `"interactive"` uses the current allocation and never submits jobs (a single node or no job at all is a plain `LocalCluster`); `"auto"` picks `"pbs"`/`"slurm"` in batch jobs, which submit worker jobs; also `"local"`, `"pbs"`, `"slurm"` |
107
117
  | `multi_node_config` | `None` | `MultiNodeConfig` for PBS/SLURM multi-node jobs (v2.0) |
108
118
 
109
119
  Settings are layered, lowest to highest:
@@ -238,19 +248,30 @@ Full documentation lives in the [GitHub wiki](https://github.com/21centuryweathe
238
248
  | [Benchmarking](https://github.com/21centuryweather/dask_setup/wiki/Benchmarking) | `benchmark_config`, `scaling_analysis`, `chunk_impact`, `dask-setup benchmark` |
239
249
  | [Internals](https://github.com/21centuryweather/dask_setup/wiki/Internals) | Resource detection, topology decisions, temp/spill routing, module layout |
240
250
  | [Troubleshooting](https://github.com/21centuryweather/dask_setup/wiki/Troubleshooting) | Common errors, OOM, multi-node issues, migration guide |
251
+ | [User Feedback](https://github.com/21centuryweather/dask_setup/wiki/User-Feedback) | Questions from users, the answers, and what changed as a result |
241
252
 
242
253
  ---
243
254
 
244
255
  ## Installation
245
256
 
257
+ From PyPI:
258
+
246
259
  ```bash
247
260
  pip install dask-setup
248
261
  ```
249
262
 
263
+ From conda-forge:
264
+
265
+ ```bash
266
+ conda install -c conda-forge dask-setup
267
+ ```
268
+
250
269
  For multi-node PBS/SLURM support:
251
270
 
252
271
  ```bash
253
272
  pip install dask-setup dask-jobqueue
273
+ # or
274
+ conda install -c conda-forge dask-setup dask-jobqueue
254
275
  ```
255
276
 
256
277
  For GPU workloads (CuPy auto-detection):
@@ -1,14 +1,27 @@
1
1
  [build-system]
2
- requires = ["setuptools>=64", "wheel"]
2
+ requires = ["setuptools>=77", "wheel"]
3
3
  build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "dask_setup"
7
- version = "2.2.0"
7
+ version = "2.3.0"
8
8
  description = "HPC-tuned Dask helpers for single-node runs on NCI Gadi."
9
9
  authors = [{name="Sam Green"}]
10
10
  readme = "README.md"
11
11
  requires-python = ">=3.11"
12
+ license = "Apache-2.0"
13
+ license-files = ["LICENSE"]
14
+ classifiers = [
15
+ "Development Status :: 5 - Production/Stable",
16
+ "Intended Audience :: Science/Research",
17
+ "Programming Language :: Python :: 3",
18
+ "Programming Language :: Python :: 3.11",
19
+ "Programming Language :: Python :: 3.12",
20
+ "Programming Language :: Python :: 3.13",
21
+ "Programming Language :: Python :: 3.14",
22
+ "Topic :: Scientific/Engineering",
23
+ "Topic :: System :: Distributed Computing",
24
+ ]
12
25
  dependencies = [
13
26
  "dask>=2024.1.0",
14
27
  "distributed>=2024.1.0",
@@ -281,7 +281,7 @@ except ImportError:
281
281
  )
282
282
 
283
283
 
284
- __version__ = "2.2.0"
284
+ __version__ = "2.3.0"
285
285
 
286
286
  __all__ = [
287
287
  # Core API — always available
@@ -47,6 +47,7 @@ from dataclasses import dataclass, field
47
47
  from typing import TYPE_CHECKING, Any
48
48
 
49
49
  from .logging import get_logger
50
+ from .reporting import scheduler_workers
50
51
 
51
52
  if TYPE_CHECKING:
52
53
  from dask.distributed import Client
@@ -561,7 +562,7 @@ def _measure_one(
561
562
 
562
563
  # Worker count
563
564
  with contextlib.suppress(Exception):
564
- n_workers = len(client.scheduler_info().get("workers", {}))
565
+ n_workers = len(scheduler_workers(client))
565
566
 
566
567
  # Optional warmup (not sampled -- it is not part of the measured run)
567
568
  if warmup:
@@ -1286,7 +1287,7 @@ def run_synthetic_benchmark(
1286
1287
  dashboard=False,
1287
1288
  )
1288
1289
 
1289
- n_workers = len(client.scheduler_info().get("workers", {}))
1290
+ n_workers = len(scheduler_workers(client))
1290
1291
 
1291
1292
  if verbose:
1292
1293
  print(f"Cluster ready: {n_workers} workers. Running {operation!r} × {repeats} …")
@@ -25,6 +25,7 @@ from .logging import get_logger
25
25
  from .multinode import (
26
26
  MultiNodeConfig,
27
27
  detect_cluster_mode,
28
+ discover_allocated_nodes,
28
29
  setup_interactive_cluster,
29
30
  setup_pbs_cluster,
30
31
  setup_slurm_cluster,
@@ -82,6 +83,43 @@ def _compute_smart_reserve_default() -> float:
82
83
  return 50.0 # Safe HPC fallback if psutil is unexpectedly unavailable
83
84
 
84
85
 
86
+ def _warn_if_threaded_topology_reads_netcdf(ds: Any, workload_type: str) -> bool:
87
+ """Warn when ``workload_type="io"`` is paired with NetCDF/HDF5 input.
88
+
89
+ ``"io"`` runs one process with many threads, which only pays off when the
90
+ reading library releases the GIL. HDF5 -- underneath NetCDF4 -- is usually
91
+ not built thread-safe, so xarray funnels every read through a single
92
+ process-wide lock: the threads queue on it instead of reading in parallel,
93
+ and decompression is serialised inside it too. ``"cpu"`` gives each process
94
+ its own lock and is typically far faster on the same data.
95
+
96
+ Returns ``True`` if a warning was emitted, so callers and tests can tell.
97
+ """
98
+ if workload_type != "io":
99
+ return False
100
+
101
+ try:
102
+ from .xarray import detect_storage_format
103
+
104
+ storage = detect_storage_format(ds)
105
+ except Exception as e: # pragma: no cover - never block setup on a hint
106
+ logger.debug("Could not determine dataset storage format", error=str(e))
107
+ return False
108
+
109
+ if storage != "netcdf":
110
+ return False
111
+
112
+ logger.warning(
113
+ "workload_type='io' is usually the wrong choice for NetCDF/HDF5 input",
114
+ reason="HDF5 is not thread-safe, so xarray serialises every read through "
115
+ "one process-wide lock; the threads queue rather than read in parallel",
116
+ suggestion="use workload_type='cpu' (one lock per process) for NetCDF",
117
+ note="also avoid open_mfdataset(..., parallel=True) with 'io' -- concurrent "
118
+ "metadata reads can kill the worker and Dask's silent retries look like a hang",
119
+ )
120
+ return True
121
+
122
+
85
123
  def _chunk_recommendations_for(
86
124
  ds: Any,
87
125
  client: Client,
@@ -458,7 +496,7 @@ def setup_dask_client(
458
496
  ds: Any = None, # xr.Dataset | xr.DataArray | None
459
497
  fallback_on_detection_failure: bool = False,
460
498
  adaptive_memory: bool = False,
461
- mode: str = "auto",
499
+ mode: str = "interactive",
462
500
  multi_node_config: MultiNodeConfig | None = None,
463
501
  ) -> tuple[Client, LocalCluster, str] | tuple[Client, LocalCluster, str, dict[str, int]]:
464
502
  """Create a single-node Dask LocalCluster tuned for HPC login/compute nodes.
@@ -542,21 +580,22 @@ def setup_dask_client(
542
580
  and tightens the worker ``memory.target`` / ``memory.spill`` thresholds
543
581
  slightly, giving workers more head-room from the start. Default
544
582
  ``False``.
545
- mode : {"auto", "local", "pbs", "slurm", "interactive"}
583
+ mode : {"interactive", "auto", "local", "pbs", "slurm"}
546
584
  Backend selection.
547
585
 
586
+ - ``"interactive"`` (default) — use resources already allocated in
587
+ the current PBS or SLURM job, interactive (``qsub -I``,
588
+ ``salloc``) or batch. Never submits new jobs. Multi-node
589
+ allocations create an ``SSHCluster`` across all nodes in
590
+ ``PBS_NODEFILE`` / ``SLURM_NODELIST``; a single node, or no
591
+ allocation at all, gets the same ``LocalCluster`` as ``"local"``.
548
592
  - ``"local"`` — always use a single-node ``LocalCluster`` (default
549
593
  behaviour prior to v2.0).
550
594
  - ``"pbs"`` — launch via ``dask-jobqueue.PBSCluster`` (submits new
551
595
  batch jobs). Requires ``pip install dask-jobqueue``.
552
596
  - ``"slurm"`` — launch via ``dask-jobqueue.SLURMCluster`` (submits
553
597
  new batch jobs).
554
- - ``"interactive"`` — use resources already allocated in the current
555
- interactive PBS (``qsub -I``) or SLURM (``salloc``) session.
556
- Single-node allocations create a ``LocalCluster``; multi-node
557
- allocations create an ``SSHCluster`` across all nodes in
558
- ``PBS_NODEFILE`` / ``SLURM_NODELIST``.
559
- - ``"auto"`` (default) — inspect the environment and choose
598
+ - ``"auto"`` — inspect the environment and choose
560
599
  ``"interactive"`` when inside a PBS interactive job
561
600
  (``PBS_ENVIRONMENT=PBS_INTERACTIVE``) or a SLURM interactive
562
601
  allocation (``SLURM_BATCH_FLAG`` not set to ``"1"``);
@@ -610,7 +649,7 @@ def setup_dask_client(
610
649
  client, cluster, tmp, chunks = setup_dask_client(ds=ds, suggest_chunks=True)
611
650
  ds_opt = ds.chunk(chunks)
612
651
 
613
- # Multi-node PBS — auto-detects from environment
652
+ # Multi-node PBS — submits worker jobs via dask-jobqueue
614
653
  client, cluster, tmp = setup_dask_client(
615
654
  mode="pbs",
616
655
  multi_node_config=MultiNodeConfig(
@@ -634,6 +673,14 @@ def setup_dask_client(
634
673
  resolved_mode = detect_cluster_mode()
635
674
  logger.debug("Mode auto-resolved", mode=resolved_mode)
636
675
 
676
+ # Interactive on one node (or none — a laptop, a login node) is just a
677
+ # LocalCluster, so take the local path directly. Going through
678
+ # setup_interactive_cluster() would drop ds-based workload inference,
679
+ # fallback_on_detection_failure and adaptive_memory.
680
+ if resolved_mode == "interactive" and len(discover_allocated_nodes()) <= 1:
681
+ resolved_mode = "local"
682
+ logger.debug("Interactive mode with at most one allocated node — using local path")
683
+
637
684
  # --- Profile auto-selection -----------------------------------------
638
685
  # Must happen before _resolve_configuration so the selected profile name
639
686
  # can be passed in. Requires a preliminary resource detection pass.
@@ -697,18 +744,13 @@ def setup_dask_client(
697
744
  # Chunk recommendations need a live client, which we now have — so
698
745
  # the ds= contract (4-tuple) holds on these paths too.
699
746
  if ds is not None:
747
+ _warn_if_threaded_topology_reads_netcdf(ds, mn_resolved.workload_type)
700
748
  chunks = _chunk_recommendations_for(
701
749
  ds, client, mn_resolved.workload_type, mn_resolved.suggest_chunks
702
750
  )
703
751
  return client, cluster, tmp_path, chunks # type: ignore[return-value]
704
752
  return client, cluster, tmp_path # type: ignore[return-value]
705
753
 
706
- logger.info(
707
- "Starting Dask client setup",
708
- workload_type=workload_type or _DEFAULT_WORKLOAD_TYPE,
709
- environment=env_type,
710
- )
711
-
712
754
  # Load and merge configuration.
713
755
  # Save the caller-supplied config object before the local variable is
714
756
  # rebound by _resolve_configuration so it can be used as the base layer.
@@ -733,11 +775,14 @@ def setup_dask_client(
733
775
  config.fallback_on_detection_failure = fallback_on_detection_failure
734
776
  config.adaptive_memory = adaptive_memory
735
777
 
736
- logger.debug(
737
- "Configuration resolved",
778
+ # Logged after resolution: before it, an unset workload_type reads as the
779
+ # library default, so profile= and config= users saw "io" whatever they chose.
780
+ logger.info(
781
+ "Starting Dask client setup",
738
782
  workload_type=config.workload_type,
739
- reserve_mem_gb=config.reserve_mem_gb,
783
+ environment=env_type,
740
784
  )
785
+ logger.debug("Configuration resolved", reserve_mem_gb=config.reserve_mem_gb)
741
786
 
742
787
  # Detect system resources
743
788
  resources = detect_resources(fallback=config.fallback_on_detection_failure)
@@ -826,26 +871,29 @@ def setup_dask_client(
826
871
  max_mem_gb=config.max_mem_gb,
827
872
  )
828
873
  except ValueError as e:
829
- # Extract memory values for better error reporting
874
+ # The only failure here is "nothing left after the reserve": the worker
875
+ # count was already fitted to memory above, so fewer workers can't help.
876
+ # Measure against the same budget compute_usable_mem_gb() used --
877
+ # max_mem_gb caps the total *before* the reserve comes off, and
878
+ # ignoring it reported memory the caller had explicitly excluded.
830
879
  total_gib = resources.total_mem_bytes / (1024**3)
831
- available_gb = total_gib - config.reserve_mem_gb
832
- required_gb = topology.n_workers * 1.0 # Rough estimate: 1 GB per worker minimum
880
+ budget_gb = min(config.max_mem_gb or total_gib, total_gib)
881
+ available_gb = max(0.0, budget_gb - config.reserve_mem_gb)
882
+ required_gb = MIN_MEM_PER_WORKER_GB # enough for a single worker
833
883
 
834
- # Generate suggested actions based on the configuration
835
884
  suggestions = []
836
- if config.reserve_mem_gb > available_gb / 2: # Reserve more than half of available
885
+ if config.max_mem_gb is not None and config.max_mem_gb < total_gib:
837
886
  suggestions.append(
838
- f"Reduce reserve_mem_gb from {config.reserve_mem_gb:.1f} GB to {available_gb * 0.3:.1f} GB"
887
+ f"Raise max_mem_gb above {config.reserve_mem_gb + required_gb:.1f} GB: it caps "
888
+ f"total memory and reserve_mem_gb ({config.reserve_mem_gb:.1f} GB) comes out of it"
839
889
  )
840
- if topology.n_workers > 1:
890
+ max_reserve_gb = budget_gb - required_gb
891
+ if max_reserve_gb > 0:
841
892
  suggestions.append(
842
- f"Limit max_workers to 1 or 2 workers instead of {topology.n_workers}"
893
+ f"Reduce reserve_mem_gb from {config.reserve_mem_gb:.1f} GB to at most "
894
+ f"{max_reserve_gb:.1f} GB"
843
895
  )
844
- if not suggestions: # Fallback suggestions
845
- suggestions = [
846
- "Close other applications to free up memory",
847
- "Request a larger memory allocation for your job",
848
- ]
896
+ suggestions.append("Request a larger memory allocation for your job")
849
897
 
850
898
  raise InsufficientResourcesError(
851
899
  required_mem=required_gb, available_mem=available_gb, suggested_actions=suggestions
@@ -931,6 +979,7 @@ def setup_dask_client(
931
979
  chunk_recommendations: dict[str, int] | None = None
932
980
 
933
981
  if ds is not None:
982
+ _warn_if_threaded_topology_reads_netcdf(ds, config.workload_type)
934
983
  chunk_recommendations = _chunk_recommendations_for(
935
984
  ds, client, config.workload_type, config.suggest_chunks
936
985
  )
@@ -25,13 +25,18 @@ def get_login_host() -> str:
25
25
  unconditionally, including to SLURM users who have never heard of it.
26
26
  3. ``"<login-node>"`` as an obvious placeholder when there is no domain
27
27
  to work from (a laptop, a container with no DNS suffix).
28
+
29
+ ``getfqdn()`` falls back to a reverse-DNS pointer name when the host has no
30
+ forward entry -- on macOS typically ``1.0.0.[...].ip6.arpa`` -- and splitting
31
+ that gave a 60-character "login host". Pointer names never name a host you
32
+ can SSH to, so they get the placeholder too.
28
33
  """
29
34
  override = os.environ.get(LOGIN_HOST_ENV)
30
35
  if override:
31
36
  return override
32
37
 
33
38
  fqdn = socket.getfqdn()
34
- if "." in fqdn:
39
+ if "." in fqdn and not fqdn.lower().rstrip(".").endswith(".arpa"):
35
40
  domain = fqdn.split(".", 1)[1]
36
41
  if domain and domain not in {"local", "localdomain"}:
37
42
  return domain
@@ -23,7 +23,7 @@ class InsufficientResourcesError(DaskSetupError):
23
23
  f"❌ Insufficient memory for configuration:\n"
24
24
  f" - Required: {required_mem:.1f} GB\n"
25
25
  f" - Available: {available_mem:.1f} GB\n"
26
- f" - Shortfall: {required_mem - available_mem:.1f} GB"
26
+ f" - Shortfall: {max(0.0, required_mem - available_mem):.1f} GB"
27
27
  )
28
28
 
29
29
  if self.suggested_actions:
@@ -480,7 +480,9 @@ def _wait_for_workers(
480
480
  f"worker job logs (usually in ~/dask-worker-space/ or the PBS log dir)\n"
481
481
  ) from exc
482
482
 
483
- n_connected = len(client.scheduler_info().get("workers", {}))
483
+ from .reporting import scheduler_workers
484
+
485
+ n_connected = len(scheduler_workers(client))
484
486
  logger.debug("Workers connected", n_connected=n_connected, expected_total=expected_total)
485
487
 
486
488
 
@@ -756,6 +758,33 @@ def _parse_slurm_nodelist(nodelist: str, cpus_per_node: int = 1) -> dict[str, in
756
758
  return dict.fromkeys(hosts, cpus_per_node)
757
759
 
758
760
 
761
+ def discover_allocated_nodes() -> dict[str, int]:
762
+ """Return ``{hostname: cores}`` for the nodes allocated to the current job.
763
+
764
+ Reads ``PBS_NODEFILE`` first, then ``SLURM_NODELIST`` /
765
+ ``SLURM_JOB_NODELIST``. Returns an empty dict outside a PBS or SLURM
766
+ allocation (a laptop, a login node).
767
+ """
768
+ nodefile = os.getenv("PBS_NODEFILE", "")
769
+ slurm_nodelist = os.getenv("SLURM_NODELIST") or os.getenv("SLURM_JOB_NODELIST", "")
770
+
771
+ if nodefile and Path(nodefile).exists():
772
+ node_cores = _parse_pbs_nodefile(nodefile)
773
+ logger.debug("PBS_NODEFILE parsed", nodes=list(node_cores.keys()))
774
+ return node_cores
775
+ if slurm_nodelist:
776
+ # SLURM_CPUS_ON_NODE can be a comma-separated list (one per node)
777
+ cpus_str = os.getenv("SLURM_CPUS_ON_NODE", "1")
778
+ try:
779
+ cpus_per_node = int(cpus_str.split(",")[0])
780
+ except ValueError:
781
+ cpus_per_node = 1
782
+ node_cores = _parse_slurm_nodelist(slurm_nodelist, cpus_per_node)
783
+ logger.debug("SLURM nodelist parsed", nodes=list(node_cores.keys()))
784
+ return node_cores
785
+ return {}
786
+
787
+
759
788
  def setup_interactive_cluster(
760
789
  workload_type: str = "cpu",
761
790
  workers_per_node: int | None = None,
@@ -811,25 +840,7 @@ def setup_interactive_cluster(
811
840
  """
812
841
  from dask.distributed import Client
813
842
 
814
- # --- Discover allocated nodes ------------------------------------------
815
- node_cores: dict[str, int] = {}
816
-
817
- nodefile = os.getenv("PBS_NODEFILE", "")
818
- slurm_nodelist = os.getenv("SLURM_NODELIST") or os.getenv("SLURM_JOB_NODELIST", "")
819
-
820
- if nodefile and Path(nodefile).exists():
821
- node_cores = _parse_pbs_nodefile(nodefile)
822
- logger.debug("PBS_NODEFILE parsed", nodes=list(node_cores.keys()))
823
- elif slurm_nodelist:
824
- # SLURM_CPUS_ON_NODE can be a comma-separated list (one per node)
825
- cpus_str = os.getenv("SLURM_CPUS_ON_NODE", "1")
826
- try:
827
- cpus_per_node = int(cpus_str.split(",")[0])
828
- except ValueError:
829
- cpus_per_node = 1
830
- node_cores = _parse_slurm_nodelist(slurm_nodelist, cpus_per_node)
831
- logger.debug("SLURM nodelist parsed", nodes=list(node_cores.keys()))
832
-
843
+ node_cores = discover_allocated_nodes()
833
844
  unique_nodes = list(node_cores.keys())
834
845
 
835
846
  # --- Single-node path (LocalCluster) ------------------------------------
@@ -177,8 +177,9 @@ def recommend_parquet_chunks(
177
177
  max_partition_mb = target_partition_mb[1]
178
178
  if client is not None:
179
179
  try:
180
- info = client.scheduler_info()
181
- workers = info.get("workers", {})
180
+ from .reporting import scheduler_workers
181
+
182
+ workers = scheduler_workers(client)
182
183
  if workers:
183
184
  min_worker_mem_bytes = min(
184
185
  w.get("memory_limit", float("inf")) for w in workers.values()