dask-setup 2.2.0__tar.gz → 2.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dask_setup-2.3.0/MANIFEST.in +5 -0
- {dask_setup-2.2.0/src/dask_setup.egg-info → dask_setup-2.3.0}/PKG-INFO +62 -7
- dask_setup-2.2.0/PKG-INFO → dask_setup-2.3.0/README.md +51 -30
- {dask_setup-2.2.0 → dask_setup-2.3.0}/pyproject.toml +15 -2
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/__init__.py +1 -1
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/benchmark.py +3 -2
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/client.py +80 -31
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/dashboard.py +6 -1
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/exceptions.py +1 -1
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/multinode.py +31 -20
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/parquet.py +3 -2
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/rechunk.py +58 -51
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/reporting.py +46 -8
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/tune.py +2 -3
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/xarray.py +102 -5
- dask_setup-2.2.0/README.md → dask_setup-2.3.0/src/dask_setup.egg-info/PKG-INFO +85 -6
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/SOURCES.txt +3 -0
- dask_setup-2.3.0/tests/__init__.py +1 -0
- dask_setup-2.3.0/tests/conftest.py +132 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_client.py +127 -4
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_dashboard.py +14 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_error_handling.py +5 -4
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_exceptions.py +7 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_multinode.py +61 -0
- dask_setup-2.3.0/tests/test_rechunk.py +159 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_reporting.py +77 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_setup_dask_client.py +75 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_xarray_chunks.py +194 -0
- dask_setup-2.2.0/tests/test_rechunk.py +0 -70
- {dask_setup-2.2.0 → dask_setup-2.3.0}/LICENSE +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/setup.cfg +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/callbacks.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/cli.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/cluster.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/config.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/config_manager.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/environment.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/error_handling.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/io_patterns.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/legacy.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/logging.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/py.typed +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/resources.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/schema/__init__.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/schema/profile_schema.json +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/tempdir.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/topology.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/types.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup/workload.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/dependency_links.txt +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/entry_points.txt +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/requires.txt +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/src/dask_setup.egg-info/top_level.txt +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_benchmark.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_cli.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_cluster.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_compression.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_config.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_config_manager.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_io_patterns.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_logging.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_resources.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_tempdir.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_topology.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_tune.py +0 -0
- {dask_setup-2.2.0 → dask_setup-2.3.0}/tests/test_types.py +0 -0
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
# setuptools only auto-includes tests/test*.py in the sdist, which leaves out
|
|
2
|
+
# conftest.py (the shared fixtures) and __init__.py -- so the suite could not
|
|
3
|
+
# run from the sdist, as downstream packagers like conda-forge do.
|
|
4
|
+
recursive-include tests *.py
|
|
5
|
+
global-exclude __pycache__ *.py[cod] .DS_Store
|
|
@@ -1,9 +1,19 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dask_setup
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.3.0
|
|
4
4
|
Summary: HPC-tuned Dask helpers for single-node runs on NCI Gadi.
|
|
5
5
|
Author: Sam Green
|
|
6
|
+
License-Expression: Apache-2.0
|
|
6
7
|
Project-URL: Homepage, https://github.com/21centuryweather/dask_setup
|
|
8
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
9
|
+
Classifier: Intended Audience :: Science/Research
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering
|
|
16
|
+
Classifier: Topic :: System :: Distributed Computing
|
|
7
17
|
Requires-Python: >=3.11
|
|
8
18
|
Description-Content-Type: text/markdown
|
|
9
19
|
License-File: LICENSE
|
|
@@ -24,11 +34,22 @@ Dynamic: license-file
|
|
|
24
34
|
|
|
25
35
|
# dask_setup
|
|
26
36
|
|
|
27
|
-
[](https://github.com/21centuryweather/dask_setup/actions/workflows/ci.yml)
|
|
38
|
+
[](https://github.com/21centuryweather/dask_setup/actions/workflows/publish-to-pypi.yml)
|
|
39
|
+
[](https://pypi.org/project/dask_setup/)
|
|
40
|
+
[](https://anaconda.org/conda-forge/dask-setup)
|
|
41
|
+
[](https://pypi.org/project/dask_setup/)
|
|
42
|
+
[](LICENSE)
|
|
43
|
+
[](https://github.com/astral-sh/ruff)
|
|
28
44
|
|
|
29
45
|
HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask.distributed.LocalCluster` + `Client` with sensible defaults for single-node jobs, and extends to multi-node PBS/SLURM clusters via `dask-jobqueue` in v2.0.
|
|
30
46
|
|
|
31
|
-
|
|
47
|
+
```bash
|
|
48
|
+
pip install dask-setup # single-node
|
|
49
|
+
pip install dask-setup[multinode] # + PBS/SLURM via dask-jobqueue
|
|
50
|
+
|
|
51
|
+
conda install -c conda-forge dask-setup # or via conda-forge
|
|
52
|
+
```
|
|
32
53
|
|
|
33
54
|
---
|
|
34
55
|
|
|
@@ -39,7 +60,7 @@ from dask_setup import setup_dask_client
|
|
|
39
60
|
|
|
40
61
|
# Pick a workload type and go
|
|
41
62
|
client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="cpu") # heavy compute
|
|
42
|
-
client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="io") #
|
|
63
|
+
client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="io") # Zarr / object storage
|
|
43
64
|
client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="mixed") # both
|
|
44
65
|
```
|
|
45
66
|
|
|
@@ -84,12 +105,35 @@ client, cluster, shared_tmp = setup_dask_client(
|
|
|
84
105
|
|
|
85
106
|
| Type | Topology | Best for |
|
|
86
107
|
|------|----------|----------|
|
|
87
|
-
| `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions |
|
|
88
|
-
| `"io"` | 1 process, 8–16 threads |
|
|
108
|
+
| `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions, **and reading NetCDF/HDF5** |
|
|
109
|
+
| `"io"` | 1 process, 8–16 threads | Zarr, object storage (S3), HTTP — libraries that release the GIL |
|
|
89
110
|
| `"mixed"` | Processes with 2 threads each | Pipelines that both read and compute |
|
|
90
111
|
| `"gpu"` | 1 process per GPU, up to 8 threads | CuPy/RAPIDS CUDA workloads |
|
|
91
112
|
| `"auto"` | Inferred from dataset | Let `dask_setup` decide based on your data |
|
|
92
113
|
|
|
114
|
+
### Reading NetCDF? Use `"cpu"`, not `"io"`
|
|
115
|
+
|
|
116
|
+
The split is not "I/O vs compute" — it is **whether the library releases the GIL
|
|
117
|
+
and is thread-safe**. `"io"` runs one process with many threads, which only wins
|
|
118
|
+
when threads can genuinely overlap.
|
|
119
|
+
|
|
120
|
+
HDF5 (underneath NetCDF4) is usually not built thread-safe, so xarray routes
|
|
121
|
+
*every* NetCDF read through a single process-wide lock. Under `"io"` your threads
|
|
122
|
+
do not become parallel readers — they queue on that one lock, one at a time, and
|
|
123
|
+
zlib decompression is serialised inside it too. Separate processes each get their
|
|
124
|
+
own lock, so `"cpu"` is what actually parallelises NetCDF reads.
|
|
125
|
+
|
|
126
|
+
Zarr is the opposite: it decompresses via numcodecs, which releases the GIL, and
|
|
127
|
+
has no global lock, so threads scale well. Opening, concatenating and writing
|
|
128
|
+
NetCDF is a `"cpu"` job even though it feels like I/O.
|
|
129
|
+
|
|
130
|
+
> **Rule of thumb: NetCDF in → `"cpu"`. Zarr or object storage in → `"io"`.**
|
|
131
|
+
> Pass `ds=` and `dask_setup` will warn if you get this pair the wrong way round.
|
|
132
|
+
>
|
|
133
|
+
> Also avoid `open_mfdataset(..., parallel=True)` together with `"io"`: concurrent
|
|
134
|
+
> metadata reads can kill the worker outright, after which Dask silently
|
|
135
|
+
> recomputes the lost tasks and the job appears to hang rather than fail.
|
|
136
|
+
|
|
93
137
|
---
|
|
94
138
|
|
|
95
139
|
## Key Parameters
|
|
@@ -103,7 +147,7 @@ client, cluster, shared_tmp = setup_dask_client(
|
|
|
103
147
|
| `dashboard` | `True` | Start dashboard and print SSH tunnel hint |
|
|
104
148
|
| `profile` | `None` | Named config profile |
|
|
105
149
|
| `config` | `None` | Pre-built `DaskSetupConfig` object — same layer as `profile`; if both are given, `profile` wins |
|
|
106
|
-
| `mode` | `"
|
|
150
|
+
| `mode` | `"interactive"` | `"interactive"` uses the current allocation and never submits jobs (a single node or no job at all is a plain `LocalCluster`); `"auto"` picks `"pbs"`/`"slurm"` in batch jobs, which submit worker jobs; also `"local"`, `"pbs"`, `"slurm"` |
|
|
107
151
|
| `multi_node_config` | `None` | `MultiNodeConfig` for PBS/SLURM multi-node jobs (v2.0) |
|
|
108
152
|
|
|
109
153
|
Settings are layered, lowest to highest:
|
|
@@ -238,19 +282,30 @@ Full documentation lives in the [GitHub wiki](https://github.com/21centuryweathe
|
|
|
238
282
|
| [Benchmarking](https://github.com/21centuryweather/dask_setup/wiki/Benchmarking) | `benchmark_config`, `scaling_analysis`, `chunk_impact`, `dask-setup benchmark` |
|
|
239
283
|
| [Internals](https://github.com/21centuryweather/dask_setup/wiki/Internals) | Resource detection, topology decisions, temp/spill routing, module layout |
|
|
240
284
|
| [Troubleshooting](https://github.com/21centuryweather/dask_setup/wiki/Troubleshooting) | Common errors, OOM, multi-node issues, migration guide |
|
|
285
|
+
| [User Feedback](https://github.com/21centuryweather/dask_setup/wiki/User-Feedback) | Questions from users, the answers, and what changed as a result |
|
|
241
286
|
|
|
242
287
|
---
|
|
243
288
|
|
|
244
289
|
## Installation
|
|
245
290
|
|
|
291
|
+
From PyPI:
|
|
292
|
+
|
|
246
293
|
```bash
|
|
247
294
|
pip install dask-setup
|
|
248
295
|
```
|
|
249
296
|
|
|
297
|
+
From conda-forge:
|
|
298
|
+
|
|
299
|
+
```bash
|
|
300
|
+
conda install -c conda-forge dask-setup
|
|
301
|
+
```
|
|
302
|
+
|
|
250
303
|
For multi-node PBS/SLURM support:
|
|
251
304
|
|
|
252
305
|
```bash
|
|
253
306
|
pip install dask-setup dask-jobqueue
|
|
307
|
+
# or
|
|
308
|
+
conda install -c conda-forge dask-setup dask-jobqueue
|
|
254
309
|
```
|
|
255
310
|
|
|
256
311
|
For GPU workloads (CuPy auto-detection):
|
|
@@ -1,34 +1,21 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: dask_setup
|
|
3
|
-
Version: 2.2.0
|
|
4
|
-
Summary: HPC-tuned Dask helpers for single-node runs on NCI Gadi.
|
|
5
|
-
Author: Sam Green
|
|
6
|
-
Project-URL: Homepage, https://github.com/21centuryweather/dask_setup
|
|
7
|
-
Requires-Python: >=3.11
|
|
8
|
-
Description-Content-Type: text/markdown
|
|
9
|
-
License-File: LICENSE
|
|
10
|
-
Requires-Dist: dask>=2024.1.0
|
|
11
|
-
Requires-Dist: distributed>=2024.1.0
|
|
12
|
-
Requires-Dist: psutil>=5.9
|
|
13
|
-
Requires-Dist: pyyaml>=6.0
|
|
14
|
-
Provides-Extra: multinode
|
|
15
|
-
Requires-Dist: dask-jobqueue>=0.8; extra == "multinode"
|
|
16
|
-
Provides-Extra: dev
|
|
17
|
-
Requires-Dist: ruff==0.16.4; extra == "dev"
|
|
18
|
-
Requires-Dist: pytest~=8.2; extra == "dev"
|
|
19
|
-
Requires-Dist: pytest-cov~=5.0; extra == "dev"
|
|
20
|
-
Requires-Dist: xarray; extra == "dev"
|
|
21
|
-
Requires-Dist: numpy; extra == "dev"
|
|
22
|
-
Requires-Dist: dask-jobqueue>=0.8; extra == "dev"
|
|
23
|
-
Dynamic: license-file
|
|
24
|
-
|
|
25
1
|
# dask_setup
|
|
26
2
|
|
|
27
|
-
[](https://github.com/21centuryweather/dask_setup/actions/workflows/ci.yml)
|
|
4
|
+
[](https://github.com/21centuryweather/dask_setup/actions/workflows/publish-to-pypi.yml)
|
|
5
|
+
[](https://pypi.org/project/dask_setup/)
|
|
6
|
+
[](https://anaconda.org/conda-forge/dask-setup)
|
|
7
|
+
[](https://pypi.org/project/dask_setup/)
|
|
8
|
+
[](LICENSE)
|
|
9
|
+
[](https://github.com/astral-sh/ruff)
|
|
28
10
|
|
|
29
11
|
HPC-tuned Dask helpers for **NCI Gadi** and other PBS/SLURM systems. Wraps `dask.distributed.LocalCluster` + `Client` with sensible defaults for single-node jobs, and extends to multi-node PBS/SLURM clusters via `dask-jobqueue` in v2.0.
|
|
30
12
|
|
|
31
|
-
|
|
13
|
+
```bash
|
|
14
|
+
pip install dask-setup # single-node
|
|
15
|
+
pip install dask-setup[multinode] # + PBS/SLURM via dask-jobqueue
|
|
16
|
+
|
|
17
|
+
conda install -c conda-forge dask-setup # or via conda-forge
|
|
18
|
+
```
|
|
32
19
|
|
|
33
20
|
---
|
|
34
21
|
|
|
@@ -39,7 +26,7 @@ from dask_setup import setup_dask_client
|
|
|
39
26
|
|
|
40
27
|
# Pick a workload type and go
|
|
41
28
|
client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="cpu") # heavy compute
|
|
42
|
-
client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="io") #
|
|
29
|
+
client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="io") # Zarr / object storage
|
|
43
30
|
client, cluster, dask_tmp = setup_dask_client(mode="interactive", workload_type="mixed") # both
|
|
44
31
|
```
|
|
45
32
|
|
|
@@ -84,12 +71,35 @@ client, cluster, shared_tmp = setup_dask_client(
|
|
|
84
71
|
|
|
85
72
|
| Type | Topology | Best for |
|
|
86
73
|
|------|----------|----------|
|
|
87
|
-
| `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions |
|
|
88
|
-
| `"io"` | 1 process, 8–16 threads |
|
|
74
|
+
| `"cpu"` | Many processes, 1 thread each | NumPy/Numba math, xarray reductions, **and reading NetCDF/HDF5** |
|
|
75
|
+
| `"io"` | 1 process, 8–16 threads | Zarr, object storage (S3), HTTP — libraries that release the GIL |
|
|
89
76
|
| `"mixed"` | Processes with 2 threads each | Pipelines that both read and compute |
|
|
90
77
|
| `"gpu"` | 1 process per GPU, up to 8 threads | CuPy/RAPIDS CUDA workloads |
|
|
91
78
|
| `"auto"` | Inferred from dataset | Let `dask_setup` decide based on your data |
|
|
92
79
|
|
|
80
|
+
### Reading NetCDF? Use `"cpu"`, not `"io"`
|
|
81
|
+
|
|
82
|
+
The split is not "I/O vs compute" — it is **whether the library releases the GIL
|
|
83
|
+
and is thread-safe**. `"io"` runs one process with many threads, which only wins
|
|
84
|
+
when threads can genuinely overlap.
|
|
85
|
+
|
|
86
|
+
HDF5 (underneath NetCDF4) is usually not built thread-safe, so xarray routes
|
|
87
|
+
*every* NetCDF read through a single process-wide lock. Under `"io"` your threads
|
|
88
|
+
do not become parallel readers — they queue on that one lock, one at a time, and
|
|
89
|
+
zlib decompression is serialised inside it too. Separate processes each get their
|
|
90
|
+
own lock, so `"cpu"` is what actually parallelises NetCDF reads.
|
|
91
|
+
|
|
92
|
+
Zarr is the opposite: it decompresses via numcodecs, which releases the GIL, and
|
|
93
|
+
has no global lock, so threads scale well. Opening, concatenating and writing
|
|
94
|
+
NetCDF is a `"cpu"` job even though it feels like I/O.
|
|
95
|
+
|
|
96
|
+
> **Rule of thumb: NetCDF in → `"cpu"`. Zarr or object storage in → `"io"`.**
|
|
97
|
+
> Pass `ds=` and `dask_setup` will warn if you get this pair the wrong way round.
|
|
98
|
+
>
|
|
99
|
+
> Also avoid `open_mfdataset(..., parallel=True)` together with `"io"`: concurrent
|
|
100
|
+
> metadata reads can kill the worker outright, after which Dask silently
|
|
101
|
+
> recomputes the lost tasks and the job appears to hang rather than fail.
|
|
102
|
+
|
|
93
103
|
---
|
|
94
104
|
|
|
95
105
|
## Key Parameters
|
|
@@ -103,7 +113,7 @@ client, cluster, shared_tmp = setup_dask_client(
|
|
|
103
113
|
| `dashboard` | `True` | Start dashboard and print SSH tunnel hint |
|
|
104
114
|
| `profile` | `None` | Named config profile |
|
|
105
115
|
| `config` | `None` | Pre-built `DaskSetupConfig` object — same layer as `profile`; if both are given, `profile` wins |
|
|
106
|
-
| `mode` | `"
|
|
116
|
+
| `mode` | `"interactive"` | `"interactive"` uses the current allocation and never submits jobs (a single node or no job at all is a plain `LocalCluster`); `"auto"` picks `"pbs"`/`"slurm"` in batch jobs, which submit worker jobs; also `"local"`, `"pbs"`, `"slurm"` |
|
|
107
117
|
| `multi_node_config` | `None` | `MultiNodeConfig` for PBS/SLURM multi-node jobs (v2.0) |
|
|
108
118
|
|
|
109
119
|
Settings are layered, lowest to highest:
|
|
@@ -238,19 +248,30 @@ Full documentation lives in the [GitHub wiki](https://github.com/21centuryweathe
|
|
|
238
248
|
| [Benchmarking](https://github.com/21centuryweather/dask_setup/wiki/Benchmarking) | `benchmark_config`, `scaling_analysis`, `chunk_impact`, `dask-setup benchmark` |
|
|
239
249
|
| [Internals](https://github.com/21centuryweather/dask_setup/wiki/Internals) | Resource detection, topology decisions, temp/spill routing, module layout |
|
|
240
250
|
| [Troubleshooting](https://github.com/21centuryweather/dask_setup/wiki/Troubleshooting) | Common errors, OOM, multi-node issues, migration guide |
|
|
251
|
+
| [User Feedback](https://github.com/21centuryweather/dask_setup/wiki/User-Feedback) | Questions from users, the answers, and what changed as a result |
|
|
241
252
|
|
|
242
253
|
---
|
|
243
254
|
|
|
244
255
|
## Installation
|
|
245
256
|
|
|
257
|
+
From PyPI:
|
|
258
|
+
|
|
246
259
|
```bash
|
|
247
260
|
pip install dask-setup
|
|
248
261
|
```
|
|
249
262
|
|
|
263
|
+
From conda-forge:
|
|
264
|
+
|
|
265
|
+
```bash
|
|
266
|
+
conda install -c conda-forge dask-setup
|
|
267
|
+
```
|
|
268
|
+
|
|
250
269
|
For multi-node PBS/SLURM support:
|
|
251
270
|
|
|
252
271
|
```bash
|
|
253
272
|
pip install dask-setup dask-jobqueue
|
|
273
|
+
# or
|
|
274
|
+
conda install -c conda-forge dask-setup dask-jobqueue
|
|
254
275
|
```
|
|
255
276
|
|
|
256
277
|
For GPU workloads (CuPy auto-detection):
|
|
@@ -1,14 +1,27 @@
|
|
|
1
1
|
[build-system]
|
|
2
|
-
requires = ["setuptools>=
|
|
2
|
+
requires = ["setuptools>=77", "wheel"]
|
|
3
3
|
build-backend = "setuptools.build_meta"
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "dask_setup"
|
|
7
|
-
version = "2.
|
|
7
|
+
version = "2.3.0"
|
|
8
8
|
description = "HPC-tuned Dask helpers for single-node runs on NCI Gadi."
|
|
9
9
|
authors = [{name="Sam Green"}]
|
|
10
10
|
readme = "README.md"
|
|
11
11
|
requires-python = ">=3.11"
|
|
12
|
+
license = "Apache-2.0"
|
|
13
|
+
license-files = ["LICENSE"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 5 - Production/Stable",
|
|
16
|
+
"Intended Audience :: Science/Research",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Programming Language :: Python :: 3.11",
|
|
19
|
+
"Programming Language :: Python :: 3.12",
|
|
20
|
+
"Programming Language :: Python :: 3.13",
|
|
21
|
+
"Programming Language :: Python :: 3.14",
|
|
22
|
+
"Topic :: Scientific/Engineering",
|
|
23
|
+
"Topic :: System :: Distributed Computing",
|
|
24
|
+
]
|
|
12
25
|
dependencies = [
|
|
13
26
|
"dask>=2024.1.0",
|
|
14
27
|
"distributed>=2024.1.0",
|
|
@@ -47,6 +47,7 @@ from dataclasses import dataclass, field
|
|
|
47
47
|
from typing import TYPE_CHECKING, Any
|
|
48
48
|
|
|
49
49
|
from .logging import get_logger
|
|
50
|
+
from .reporting import scheduler_workers
|
|
50
51
|
|
|
51
52
|
if TYPE_CHECKING:
|
|
52
53
|
from dask.distributed import Client
|
|
@@ -561,7 +562,7 @@ def _measure_one(
|
|
|
561
562
|
|
|
562
563
|
# Worker count
|
|
563
564
|
with contextlib.suppress(Exception):
|
|
564
|
-
n_workers = len(client
|
|
565
|
+
n_workers = len(scheduler_workers(client))
|
|
565
566
|
|
|
566
567
|
# Optional warmup (not sampled -- it is not part of the measured run)
|
|
567
568
|
if warmup:
|
|
@@ -1286,7 +1287,7 @@ def run_synthetic_benchmark(
|
|
|
1286
1287
|
dashboard=False,
|
|
1287
1288
|
)
|
|
1288
1289
|
|
|
1289
|
-
n_workers = len(client
|
|
1290
|
+
n_workers = len(scheduler_workers(client))
|
|
1290
1291
|
|
|
1291
1292
|
if verbose:
|
|
1292
1293
|
print(f"Cluster ready: {n_workers} workers. Running {operation!r} × {repeats} …")
|
|
@@ -25,6 +25,7 @@ from .logging import get_logger
|
|
|
25
25
|
from .multinode import (
|
|
26
26
|
MultiNodeConfig,
|
|
27
27
|
detect_cluster_mode,
|
|
28
|
+
discover_allocated_nodes,
|
|
28
29
|
setup_interactive_cluster,
|
|
29
30
|
setup_pbs_cluster,
|
|
30
31
|
setup_slurm_cluster,
|
|
@@ -82,6 +83,43 @@ def _compute_smart_reserve_default() -> float:
|
|
|
82
83
|
return 50.0 # Safe HPC fallback if psutil is unexpectedly unavailable
|
|
83
84
|
|
|
84
85
|
|
|
86
|
+
def _warn_if_threaded_topology_reads_netcdf(ds: Any, workload_type: str) -> bool:
|
|
87
|
+
"""Warn when ``workload_type="io"`` is paired with NetCDF/HDF5 input.
|
|
88
|
+
|
|
89
|
+
``"io"`` runs one process with many threads, which only pays off when the
|
|
90
|
+
reading library releases the GIL. HDF5 -- underneath NetCDF4 -- is usually
|
|
91
|
+
not built thread-safe, so xarray funnels every read through a single
|
|
92
|
+
process-wide lock: the threads queue on it instead of reading in parallel,
|
|
93
|
+
and decompression is serialised inside it too. ``"cpu"`` gives each process
|
|
94
|
+
its own lock and is typically far faster on the same data.
|
|
95
|
+
|
|
96
|
+
Returns ``True`` if a warning was emitted, so callers and tests can tell.
|
|
97
|
+
"""
|
|
98
|
+
if workload_type != "io":
|
|
99
|
+
return False
|
|
100
|
+
|
|
101
|
+
try:
|
|
102
|
+
from .xarray import detect_storage_format
|
|
103
|
+
|
|
104
|
+
storage = detect_storage_format(ds)
|
|
105
|
+
except Exception as e: # pragma: no cover - never block setup on a hint
|
|
106
|
+
logger.debug("Could not determine dataset storage format", error=str(e))
|
|
107
|
+
return False
|
|
108
|
+
|
|
109
|
+
if storage != "netcdf":
|
|
110
|
+
return False
|
|
111
|
+
|
|
112
|
+
logger.warning(
|
|
113
|
+
"workload_type='io' is usually the wrong choice for NetCDF/HDF5 input",
|
|
114
|
+
reason="HDF5 is not thread-safe, so xarray serialises every read through "
|
|
115
|
+
"one process-wide lock; the threads queue rather than read in parallel",
|
|
116
|
+
suggestion="use workload_type='cpu' (one lock per process) for NetCDF",
|
|
117
|
+
note="also avoid open_mfdataset(..., parallel=True) with 'io' -- concurrent "
|
|
118
|
+
"metadata reads can kill the worker and Dask's silent retries look like a hang",
|
|
119
|
+
)
|
|
120
|
+
return True
|
|
121
|
+
|
|
122
|
+
|
|
85
123
|
def _chunk_recommendations_for(
|
|
86
124
|
ds: Any,
|
|
87
125
|
client: Client,
|
|
@@ -458,7 +496,7 @@ def setup_dask_client(
|
|
|
458
496
|
ds: Any = None, # xr.Dataset | xr.DataArray | None
|
|
459
497
|
fallback_on_detection_failure: bool = False,
|
|
460
498
|
adaptive_memory: bool = False,
|
|
461
|
-
mode: str = "
|
|
499
|
+
mode: str = "interactive",
|
|
462
500
|
multi_node_config: MultiNodeConfig | None = None,
|
|
463
501
|
) -> tuple[Client, LocalCluster, str] | tuple[Client, LocalCluster, str, dict[str, int]]:
|
|
464
502
|
"""Create a single-node Dask LocalCluster tuned for HPC login/compute nodes.
|
|
@@ -542,21 +580,22 @@ def setup_dask_client(
|
|
|
542
580
|
and tightens the worker ``memory.target`` / ``memory.spill`` thresholds
|
|
543
581
|
slightly, giving workers more head-room from the start. Default
|
|
544
582
|
``False``.
|
|
545
|
-
mode : {"auto", "local", "pbs", "slurm"
|
|
583
|
+
mode : {"interactive", "auto", "local", "pbs", "slurm"}
|
|
546
584
|
Backend selection.
|
|
547
585
|
|
|
586
|
+
- ``"interactive"`` (default) — use resources already allocated in
|
|
587
|
+
the current PBS or SLURM job, interactive (``qsub -I``,
|
|
588
|
+
``salloc``) or batch. Never submits new jobs. Multi-node
|
|
589
|
+
allocations create an ``SSHCluster`` across all nodes in
|
|
590
|
+
``PBS_NODEFILE`` / ``SLURM_NODELIST``; a single node, or no
|
|
591
|
+
allocation at all, gets the same ``LocalCluster`` as ``"local"``.
|
|
548
592
|
- ``"local"`` — always use a single-node ``LocalCluster`` (default
|
|
549
593
|
behaviour prior to v2.0).
|
|
550
594
|
- ``"pbs"`` — launch via ``dask-jobqueue.PBSCluster`` (submits new
|
|
551
595
|
batch jobs). Requires ``pip install dask-jobqueue``.
|
|
552
596
|
- ``"slurm"`` — launch via ``dask-jobqueue.SLURMCluster`` (submits
|
|
553
597
|
new batch jobs).
|
|
554
|
-
- ``"
|
|
555
|
-
interactive PBS (``qsub -I``) or SLURM (``salloc``) session.
|
|
556
|
-
Single-node allocations create a ``LocalCluster``; multi-node
|
|
557
|
-
allocations create an ``SSHCluster`` across all nodes in
|
|
558
|
-
``PBS_NODEFILE`` / ``SLURM_NODELIST``.
|
|
559
|
-
- ``"auto"`` (default) — inspect the environment and choose
|
|
598
|
+
- ``"auto"`` — inspect the environment and choose
|
|
560
599
|
``"interactive"`` when inside a PBS interactive job
|
|
561
600
|
(``PBS_ENVIRONMENT=PBS_INTERACTIVE``) or a SLURM interactive
|
|
562
601
|
allocation (``SLURM_BATCH_FLAG`` not set to ``"1"``);
|
|
@@ -610,7 +649,7 @@ def setup_dask_client(
|
|
|
610
649
|
client, cluster, tmp, chunks = setup_dask_client(ds=ds, suggest_chunks=True)
|
|
611
650
|
ds_opt = ds.chunk(chunks)
|
|
612
651
|
|
|
613
|
-
# Multi-node PBS —
|
|
652
|
+
# Multi-node PBS — submits worker jobs via dask-jobqueue
|
|
614
653
|
client, cluster, tmp = setup_dask_client(
|
|
615
654
|
mode="pbs",
|
|
616
655
|
multi_node_config=MultiNodeConfig(
|
|
@@ -634,6 +673,14 @@ def setup_dask_client(
|
|
|
634
673
|
resolved_mode = detect_cluster_mode()
|
|
635
674
|
logger.debug("Mode auto-resolved", mode=resolved_mode)
|
|
636
675
|
|
|
676
|
+
# Interactive on one node (or none — a laptop, a login node) is just a
|
|
677
|
+
# LocalCluster, so take the local path directly. Going through
|
|
678
|
+
# setup_interactive_cluster() would drop ds-based workload inference,
|
|
679
|
+
# fallback_on_detection_failure and adaptive_memory.
|
|
680
|
+
if resolved_mode == "interactive" and len(discover_allocated_nodes()) <= 1:
|
|
681
|
+
resolved_mode = "local"
|
|
682
|
+
logger.debug("Interactive mode with at most one allocated node — using local path")
|
|
683
|
+
|
|
637
684
|
# --- Profile auto-selection -----------------------------------------
|
|
638
685
|
# Must happen before _resolve_configuration so the selected profile name
|
|
639
686
|
# can be passed in. Requires a preliminary resource detection pass.
|
|
@@ -697,18 +744,13 @@ def setup_dask_client(
|
|
|
697
744
|
# Chunk recommendations need a live client, which we now have — so
|
|
698
745
|
# the ds= contract (4-tuple) holds on these paths too.
|
|
699
746
|
if ds is not None:
|
|
747
|
+
_warn_if_threaded_topology_reads_netcdf(ds, mn_resolved.workload_type)
|
|
700
748
|
chunks = _chunk_recommendations_for(
|
|
701
749
|
ds, client, mn_resolved.workload_type, mn_resolved.suggest_chunks
|
|
702
750
|
)
|
|
703
751
|
return client, cluster, tmp_path, chunks # type: ignore[return-value]
|
|
704
752
|
return client, cluster, tmp_path # type: ignore[return-value]
|
|
705
753
|
|
|
706
|
-
logger.info(
|
|
707
|
-
"Starting Dask client setup",
|
|
708
|
-
workload_type=workload_type or _DEFAULT_WORKLOAD_TYPE,
|
|
709
|
-
environment=env_type,
|
|
710
|
-
)
|
|
711
|
-
|
|
712
754
|
# Load and merge configuration.
|
|
713
755
|
# Save the caller-supplied config object before the local variable is
|
|
714
756
|
# rebound by _resolve_configuration so it can be used as the base layer.
|
|
@@ -733,11 +775,14 @@ def setup_dask_client(
|
|
|
733
775
|
config.fallback_on_detection_failure = fallback_on_detection_failure
|
|
734
776
|
config.adaptive_memory = adaptive_memory
|
|
735
777
|
|
|
736
|
-
|
|
737
|
-
|
|
778
|
+
# Logged after resolution: before it, an unset workload_type reads as the
|
|
779
|
+
# library default, so profile= and config= users saw "io" whatever they chose.
|
|
780
|
+
logger.info(
|
|
781
|
+
"Starting Dask client setup",
|
|
738
782
|
workload_type=config.workload_type,
|
|
739
|
-
|
|
783
|
+
environment=env_type,
|
|
740
784
|
)
|
|
785
|
+
logger.debug("Configuration resolved", reserve_mem_gb=config.reserve_mem_gb)
|
|
741
786
|
|
|
742
787
|
# Detect system resources
|
|
743
788
|
resources = detect_resources(fallback=config.fallback_on_detection_failure)
|
|
@@ -826,26 +871,29 @@ def setup_dask_client(
|
|
|
826
871
|
max_mem_gb=config.max_mem_gb,
|
|
827
872
|
)
|
|
828
873
|
except ValueError as e:
|
|
829
|
-
#
|
|
874
|
+
# The only failure here is "nothing left after the reserve": the worker
|
|
875
|
+
# count was already fitted to memory above, so fewer workers can't help.
|
|
876
|
+
# Measure against the same budget compute_usable_mem_gb() used --
|
|
877
|
+
# max_mem_gb caps the total *before* the reserve comes off, and
|
|
878
|
+
# ignoring it reported memory the caller had explicitly excluded.
|
|
830
879
|
total_gib = resources.total_mem_bytes / (1024**3)
|
|
831
|
-
|
|
832
|
-
|
|
880
|
+
budget_gb = min(config.max_mem_gb or total_gib, total_gib)
|
|
881
|
+
available_gb = max(0.0, budget_gb - config.reserve_mem_gb)
|
|
882
|
+
required_gb = MIN_MEM_PER_WORKER_GB # enough for a single worker
|
|
833
883
|
|
|
834
|
-
# Generate suggested actions based on the configuration
|
|
835
884
|
suggestions = []
|
|
836
|
-
if config.
|
|
885
|
+
if config.max_mem_gb is not None and config.max_mem_gb < total_gib:
|
|
837
886
|
suggestions.append(
|
|
838
|
-
f"
|
|
887
|
+
f"Raise max_mem_gb above {config.reserve_mem_gb + required_gb:.1f} GB: it caps "
|
|
888
|
+
f"total memory and reserve_mem_gb ({config.reserve_mem_gb:.1f} GB) comes out of it"
|
|
839
889
|
)
|
|
840
|
-
|
|
890
|
+
max_reserve_gb = budget_gb - required_gb
|
|
891
|
+
if max_reserve_gb > 0:
|
|
841
892
|
suggestions.append(
|
|
842
|
-
f"
|
|
893
|
+
f"Reduce reserve_mem_gb from {config.reserve_mem_gb:.1f} GB to at most "
|
|
894
|
+
f"{max_reserve_gb:.1f} GB"
|
|
843
895
|
)
|
|
844
|
-
|
|
845
|
-
suggestions = [
|
|
846
|
-
"Close other applications to free up memory",
|
|
847
|
-
"Request a larger memory allocation for your job",
|
|
848
|
-
]
|
|
896
|
+
suggestions.append("Request a larger memory allocation for your job")
|
|
849
897
|
|
|
850
898
|
raise InsufficientResourcesError(
|
|
851
899
|
required_mem=required_gb, available_mem=available_gb, suggested_actions=suggestions
|
|
@@ -931,6 +979,7 @@ def setup_dask_client(
|
|
|
931
979
|
chunk_recommendations: dict[str, int] | None = None
|
|
932
980
|
|
|
933
981
|
if ds is not None:
|
|
982
|
+
_warn_if_threaded_topology_reads_netcdf(ds, config.workload_type)
|
|
934
983
|
chunk_recommendations = _chunk_recommendations_for(
|
|
935
984
|
ds, client, config.workload_type, config.suggest_chunks
|
|
936
985
|
)
|
|
@@ -25,13 +25,18 @@ def get_login_host() -> str:
|
|
|
25
25
|
unconditionally, including to SLURM users who have never heard of it.
|
|
26
26
|
3. ``"<login-node>"`` as an obvious placeholder when there is no domain
|
|
27
27
|
to work from (a laptop, a container with no DNS suffix).
|
|
28
|
+
|
|
29
|
+
``getfqdn()`` falls back to a reverse-DNS pointer name when the host has no
|
|
30
|
+
forward entry -- on macOS typically ``1.0.0.[...].ip6.arpa`` -- and splitting
|
|
31
|
+
that gave a 60-character "login host". Pointer names never name a host you
|
|
32
|
+
can SSH to, so they get the placeholder too.
|
|
28
33
|
"""
|
|
29
34
|
override = os.environ.get(LOGIN_HOST_ENV)
|
|
30
35
|
if override:
|
|
31
36
|
return override
|
|
32
37
|
|
|
33
38
|
fqdn = socket.getfqdn()
|
|
34
|
-
if "." in fqdn:
|
|
39
|
+
if "." in fqdn and not fqdn.lower().rstrip(".").endswith(".arpa"):
|
|
35
40
|
domain = fqdn.split(".", 1)[1]
|
|
36
41
|
if domain and domain not in {"local", "localdomain"}:
|
|
37
42
|
return domain
|
|
@@ -23,7 +23,7 @@ class InsufficientResourcesError(DaskSetupError):
|
|
|
23
23
|
f"❌ Insufficient memory for configuration:\n"
|
|
24
24
|
f" - Required: {required_mem:.1f} GB\n"
|
|
25
25
|
f" - Available: {available_mem:.1f} GB\n"
|
|
26
|
-
f" - Shortfall: {required_mem - available_mem:.1f} GB"
|
|
26
|
+
f" - Shortfall: {max(0.0, required_mem - available_mem):.1f} GB"
|
|
27
27
|
)
|
|
28
28
|
|
|
29
29
|
if self.suggested_actions:
|
|
@@ -480,7 +480,9 @@ def _wait_for_workers(
|
|
|
480
480
|
f"worker job logs (usually in ~/dask-worker-space/ or the PBS log dir)\n"
|
|
481
481
|
) from exc
|
|
482
482
|
|
|
483
|
-
|
|
483
|
+
from .reporting import scheduler_workers
|
|
484
|
+
|
|
485
|
+
n_connected = len(scheduler_workers(client))
|
|
484
486
|
logger.debug("Workers connected", n_connected=n_connected, expected_total=expected_total)
|
|
485
487
|
|
|
486
488
|
|
|
@@ -756,6 +758,33 @@ def _parse_slurm_nodelist(nodelist: str, cpus_per_node: int = 1) -> dict[str, in
|
|
|
756
758
|
return dict.fromkeys(hosts, cpus_per_node)
|
|
757
759
|
|
|
758
760
|
|
|
761
|
+
def discover_allocated_nodes() -> dict[str, int]:
|
|
762
|
+
"""Return ``{hostname: cores}`` for the nodes allocated to the current job.
|
|
763
|
+
|
|
764
|
+
Reads ``PBS_NODEFILE`` first, then ``SLURM_NODELIST`` /
|
|
765
|
+
``SLURM_JOB_NODELIST``. Returns an empty dict outside a PBS or SLURM
|
|
766
|
+
allocation (a laptop, a login node).
|
|
767
|
+
"""
|
|
768
|
+
nodefile = os.getenv("PBS_NODEFILE", "")
|
|
769
|
+
slurm_nodelist = os.getenv("SLURM_NODELIST") or os.getenv("SLURM_JOB_NODELIST", "")
|
|
770
|
+
|
|
771
|
+
if nodefile and Path(nodefile).exists():
|
|
772
|
+
node_cores = _parse_pbs_nodefile(nodefile)
|
|
773
|
+
logger.debug("PBS_NODEFILE parsed", nodes=list(node_cores.keys()))
|
|
774
|
+
return node_cores
|
|
775
|
+
if slurm_nodelist:
|
|
776
|
+
# SLURM_CPUS_ON_NODE can be a comma-separated list (one per node)
|
|
777
|
+
cpus_str = os.getenv("SLURM_CPUS_ON_NODE", "1")
|
|
778
|
+
try:
|
|
779
|
+
cpus_per_node = int(cpus_str.split(",")[0])
|
|
780
|
+
except ValueError:
|
|
781
|
+
cpus_per_node = 1
|
|
782
|
+
node_cores = _parse_slurm_nodelist(slurm_nodelist, cpus_per_node)
|
|
783
|
+
logger.debug("SLURM nodelist parsed", nodes=list(node_cores.keys()))
|
|
784
|
+
return node_cores
|
|
785
|
+
return {}
|
|
786
|
+
|
|
787
|
+
|
|
759
788
|
def setup_interactive_cluster(
|
|
760
789
|
workload_type: str = "cpu",
|
|
761
790
|
workers_per_node: int | None = None,
|
|
@@ -811,25 +840,7 @@ def setup_interactive_cluster(
|
|
|
811
840
|
"""
|
|
812
841
|
from dask.distributed import Client
|
|
813
842
|
|
|
814
|
-
|
|
815
|
-
node_cores: dict[str, int] = {}
|
|
816
|
-
|
|
817
|
-
nodefile = os.getenv("PBS_NODEFILE", "")
|
|
818
|
-
slurm_nodelist = os.getenv("SLURM_NODELIST") or os.getenv("SLURM_JOB_NODELIST", "")
|
|
819
|
-
|
|
820
|
-
if nodefile and Path(nodefile).exists():
|
|
821
|
-
node_cores = _parse_pbs_nodefile(nodefile)
|
|
822
|
-
logger.debug("PBS_NODEFILE parsed", nodes=list(node_cores.keys()))
|
|
823
|
-
elif slurm_nodelist:
|
|
824
|
-
# SLURM_CPUS_ON_NODE can be a comma-separated list (one per node)
|
|
825
|
-
cpus_str = os.getenv("SLURM_CPUS_ON_NODE", "1")
|
|
826
|
-
try:
|
|
827
|
-
cpus_per_node = int(cpus_str.split(",")[0])
|
|
828
|
-
except ValueError:
|
|
829
|
-
cpus_per_node = 1
|
|
830
|
-
node_cores = _parse_slurm_nodelist(slurm_nodelist, cpus_per_node)
|
|
831
|
-
logger.debug("SLURM nodelist parsed", nodes=list(node_cores.keys()))
|
|
832
|
-
|
|
843
|
+
node_cores = discover_allocated_nodes()
|
|
833
844
|
unique_nodes = list(node_cores.keys())
|
|
834
845
|
|
|
835
846
|
# --- Single-node path (LocalCluster) ------------------------------------
|
|
@@ -177,8 +177,9 @@ def recommend_parquet_chunks(
|
|
|
177
177
|
max_partition_mb = target_partition_mb[1]
|
|
178
178
|
if client is not None:
|
|
179
179
|
try:
|
|
180
|
-
|
|
181
|
-
|
|
180
|
+
from .reporting import scheduler_workers
|
|
181
|
+
|
|
182
|
+
workers = scheduler_workers(client)
|
|
182
183
|
if workers:
|
|
183
184
|
min_worker_mem_bytes = min(
|
|
184
185
|
w.get("memory_limit", float("inf")) for w in workers.values()
|