datalad-slurm 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- datalad_slurm-0.2.3/CONTRIBUTORS +4 -0
- datalad_slurm-0.2.3/LICENSE +31 -0
- datalad_slurm-0.2.3/MANIFEST.in +8 -0
- datalad_slurm-0.2.3/PKG-INFO +199 -0
- datalad_slurm-0.2.3/README.md +156 -0
- datalad_slurm-0.2.3/_datalad_buildsupport/__init__.py +13 -0
- datalad_slurm-0.2.3/_datalad_buildsupport/formatters.py +314 -0
- datalad_slurm-0.2.3/_datalad_buildsupport/setup.py +221 -0
- datalad_slurm-0.2.3/docs/README.md +35 -0
- datalad_slurm-0.2.3/docs/source/_static/datalad_logo.png +0 -0
- datalad_slurm-0.2.3/docs/source/_templates/autosummary/module.rst +23 -0
- datalad_slurm-0.2.3/docs/source/cli_reference.rst +9 -0
- datalad_slurm-0.2.3/docs/source/index.rst +89 -0
- datalad_slurm-0.2.3/docs/source/python_reference.rst +10 -0
- datalad_slurm-0.2.3/pyproject.toml +72 -0
- datalad_slurm-0.2.3/requirements.txt +3 -0
- datalad_slurm-0.2.3/setup.cfg +4 -0
- datalad_slurm-0.2.3/setup.py +7 -0
- datalad_slurm-0.2.3/src/datalad_slurm/__init__.py +51 -0
- datalad_slurm-0.2.3/src/datalad_slurm/common.py +102 -0
- datalad_slurm-0.2.3/src/datalad_slurm/conftest.py +1 -0
- datalad_slurm-0.2.3/src/datalad_slurm/finish.py +585 -0
- datalad_slurm-0.2.3/src/datalad_slurm/reschedule.py +796 -0
- datalad_slurm-0.2.3/src/datalad_slurm/schedule.py +1212 -0
- datalad_slurm-0.2.3/src/datalad_slurm/tests/__init__.py +0 -0
- datalad_slurm-0.2.3/src/datalad_slurm/tests/test_register.py +21 -0
- datalad_slurm-0.2.3/src/datalad_slurm.egg-info/PKG-INFO +199 -0
- datalad_slurm-0.2.3/src/datalad_slurm.egg-info/SOURCES.txt +31 -0
- datalad_slurm-0.2.3/src/datalad_slurm.egg-info/dependency_links.txt +1 -0
- datalad_slurm-0.2.3/src/datalad_slurm.egg-info/entry_points.txt +2 -0
- datalad_slurm-0.2.3/src/datalad_slurm.egg-info/requires.txt +24 -0
- datalad_slurm-0.2.3/src/datalad_slurm.egg-info/top_level.txt +1 -0
- datalad_slurm-0.2.3/versioneer.py +2277 -0
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# Main Copyright/License
|
|
2
|
+
|
|
3
|
+
DataLad-slurm, including all examples, code snippets and attached
|
|
4
|
+
documentation is covered by the MIT license.
|
|
5
|
+
|
|
6
|
+
The MIT License
|
|
7
|
+
|
|
8
|
+
Copyright (c) 2024- Andreas Knüpfer and Timothy Callow
|
|
9
|
+
Center for Advanced Systems Understanding
|
|
10
|
+
Helmholtz-Zentrum Dresden-Rossendorf
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
14
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
15
|
+
in the Software without restriction, including without limitation the rights
|
|
16
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
17
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
18
|
+
furnished to do so, subject to the following conditions:
|
|
19
|
+
|
|
20
|
+
The above copyright notice and this permission notice shall be included in
|
|
21
|
+
all copies or substantial portions of the Software.
|
|
22
|
+
|
|
23
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
24
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
25
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
26
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
27
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
28
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
|
29
|
+
THE SOFTWARE.
|
|
30
|
+
|
|
31
|
+
See CONTRIBUTORS file for a full list of contributors.
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
include CONTRIBUTORS LICENSE versioneer.py README.md
|
|
2
|
+
recursive-include src *.py
|
|
3
|
+
recursive-exclude * __pycache__
|
|
4
|
+
recursive-exclude * *.py[cod]
|
|
5
|
+
recursive-include _datalad_buildsupport *.py
|
|
6
|
+
recursive-include tests *.py
|
|
7
|
+
recursive-include docs *.rst *.png *.md
|
|
8
|
+
prune docs/build
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: datalad-slurm
|
|
3
|
+
Version: 0.2.3
|
|
4
|
+
Summary: A DataLad extension for managing research data workflows on SLURM-based HPC systems
|
|
5
|
+
Author-email: Andreas Knüpfer <a.knuepfer@hzdr.de>, Timothy Callow <t.callow@hzdr.de>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/knuedd/datalad-slurm
|
|
8
|
+
Project-URL: Repository, https://github.com/knuedd/datalad-slurm.git
|
|
9
|
+
Project-URL: Issues, https://github.com/knuedd/datalad-slurm/issues
|
|
10
|
+
Keywords: datalad,slurm,hpc,high-performance-computing,workflow,scientific-computing
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Requires-Python: >=3.8
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Requires-Dist: datalad>=0.18.0
|
|
23
|
+
Requires-Dist: sqlalchemy>=1.4.0
|
|
24
|
+
Requires-Dist: tqdm>=4.0.0
|
|
25
|
+
Provides-Extra: test
|
|
26
|
+
Requires-Dist: pytest>=6.0; extra == "test"
|
|
27
|
+
Requires-Dist: pytest-cov>=2.0; extra == "test"
|
|
28
|
+
Provides-Extra: dev
|
|
29
|
+
Requires-Dist: black>=23.0; extra == "dev"
|
|
30
|
+
Requires-Dist: flake8>=5.0; extra == "dev"
|
|
31
|
+
Requires-Dist: isort>=5.0; extra == "dev"
|
|
32
|
+
Requires-Dist: codespell>=2.0; extra == "dev"
|
|
33
|
+
Provides-Extra: devel
|
|
34
|
+
Requires-Dist: pytest; extra == "devel"
|
|
35
|
+
Requires-Dist: pytest-cov; extra == "devel"
|
|
36
|
+
Requires-Dist: coverage; extra == "devel"
|
|
37
|
+
Requires-Dist: sphinx; extra == "devel"
|
|
38
|
+
Requires-Dist: sphinx_rtd_theme; extra == "devel"
|
|
39
|
+
Provides-Extra: devel-utils
|
|
40
|
+
Requires-Dist: pytest-xdist; extra == "devel-utils"
|
|
41
|
+
Requires-Dist: scriv; extra == "devel-utils"
|
|
42
|
+
Dynamic: license-file
|
|
43
|
+
|
|
44
|
+
# datalad-slurm: A DataLad extension for HPC (slurm) systems
|
|
45
|
+
|
|
46
|
+
[](https://pypi.org/project/datalad-slurm/) [](https://github.com/datalad/datalad-slurm/actions) [](https://codecov.io/github/datalad/datalad-slurm?branch=main) [](https://datalad-slurm.readthedocs.io/en/latest/?badge=latest)
|
|
47
|
+
|
|
48
|
+
[](https://pypi.org/project/datalad-slurm/) [](https://pypi.org/project/datalad-slurm/) [](https://doi.org/10.5281/zenodo.12345678)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
`datalad-slurm` is an extension to the [DataLad](http://datalad.org) package for high-performance computing (HPC), specifically slurm systems.
|
|
52
|
+
|
|
53
|
+
DataLad is a package which facilitates adherence to the [FAIR](https://www.nature.com/articles/sdata201618) research data management principles.
|
|
54
|
+
|
|
55
|
+
`datalad-slurm` sits on top of the main DataLad package, and it is designed to improve the DataLad workflow on HPC systems. The package is aimed at slurm systems due to the prominence of slurm in HPC settings, but in the future it may be extended to HPC systems more generally.
|
|
56
|
+
|
|
57
|
+
`datalad-slurm` makes it easier for users to manage their research data on HPC systems with DataLad, and also solves the following conflicts of DataLad usage in HPC systems:
|
|
58
|
+
|
|
59
|
+
- **Inefficient** sequential sections in highly parallel HPC jobs
|
|
60
|
+
- **Critical** race conditions between git commands in concurrent jobs
|
|
61
|
+
|
|
62
|
+
## Installation
|
|
63
|
+
|
|
64
|
+
First, install the main [DataLad](http://datalad.org) package and its dependencies.
|
|
65
|
+
|
|
66
|
+
Then, clone this repository and install the extension with:
|
|
67
|
+
|
|
68
|
+
pip install -e .
|
|
69
|
+
|
|
70
|
+
## Example usage
|
|
71
|
+
|
|
72
|
+
To **schedule** a slurm script:
|
|
73
|
+
|
|
74
|
+
datalad slurm-schedule --output=<output_files_or_dir> <slurm_submission_command>
|
|
75
|
+
|
|
76
|
+
where `<output_files_or_dir>` are the expected outputs from the job, and `<slurm_submission_command>` is for example `sbatch submit_script`. Further optional command line arguments can be found in the documentation.
|
|
77
|
+
|
|
78
|
+
Multiple jobs (including array jobs) can be scheduled sequentially. They are tracked in an SQLite database. Note that any open jobs must not have conflicting outputs with previously scheduled jobs. This is so that the outputs of each slurm run can be tracked to the slurm job which generated them.
|
|
79
|
+
|
|
80
|
+
To **finish** (i.e. post-process) these jobs (once they are complete), simply run:
|
|
81
|
+
|
|
82
|
+
datalad slurm-finish
|
|
83
|
+
|
|
84
|
+
Alternatively, to finish a particular scheduled job, run:
|
|
85
|
+
|
|
86
|
+
datalad slurm-finish <slurm_job_id>
|
|
87
|
+
|
|
88
|
+
This will create a `[DATALAD SLURM RUN]` entry in the git log, analogous to a `datalad run` command.
|
|
89
|
+
|
|
90
|
+
`datalad-slurm` will flag an error for any jobs which could not be post-processed, either because they are still running, or the job failed. These are not automatically cleared from the SQLite database. The output files should first be removed or manually added in git, before running
|
|
91
|
+
|
|
92
|
+
datalad slurm-finish --close-failed-jobs
|
|
93
|
+
|
|
94
|
+
To clear the SQLite database. To inspect the current status of all open jobs (without saving anything in git), run:
|
|
95
|
+
|
|
96
|
+
datalad slurm-finish --list-open-jobs
|
|
97
|
+
|
|
98
|
+
To **reschedule** a previously scheduled job:
|
|
99
|
+
|
|
100
|
+
datalad slurm-reschedule <schedule_commit_hash>
|
|
101
|
+
|
|
102
|
+
where `<schedule_commit_hash>` is the commit hash of the previously scheduled job. There must also be a corresponding `datalad slurm-finish` command to the original `datalad slurm-schedule`, otherwise `datalad slurm-reschedule` will throw an error.
|
|
103
|
+
|
|
104
|
+
In the lingo of the original DataLad package, the combination of `datalad slurm-schedule + datalad slurm-finish` is similar to `datalad run`, and `datalad slurm-reschedule + datalad slurm-finish` is similar to `datalad rerun`.
|
|
105
|
+
|
|
106
|
+
An example workflow could look like this (constructed deliberately to have some failed jobs):
|
|
107
|
+
|
|
108
|
+
datalad slurm-schedule -o models/abrupt/gold/ sbatch submit_gold.slurm
|
|
109
|
+
datalad slurm-schedule -o models/abrupt/silver/ sbatch submit_silver.slurm
|
|
110
|
+
datalad slurm-schedule -o models/abrupt/bronze/ sbatch submit_bronze.slurm
|
|
111
|
+
datalad slurm-schedule -o models/abrupt/platinum/ sbatch submit_array_platinum.slurm
|
|
112
|
+
|
|
113
|
+
Checking the job statuses at some point while they are running:
|
|
114
|
+
|
|
115
|
+
datalad slurm-finish --list-open-jobs
|
|
116
|
+
|
|
117
|
+
The following jobs are open:
|
|
118
|
+
|
|
119
|
+
slurm-job-id slurm-job-status
|
|
120
|
+
10524442 COMPLETED
|
|
121
|
+
10524535 RUNNING
|
|
122
|
+
10524556 FAILED
|
|
123
|
+
10524620 PENDING
|
|
124
|
+
|
|
125
|
+
Later, once all the jobs have finished running:
|
|
126
|
+
|
|
127
|
+
datalad slurm-finish
|
|
128
|
+
|
|
129
|
+
add(ok): models/abrupt/gold/05_02/slurm-10524442.out (file)
|
|
130
|
+
add(ok): models/abrupt/gold/05_02/slurm-job-10524442.env.json (file)
|
|
131
|
+
add(ok): models/abrupt/gold/05_02/model_0.model.gz (file)
|
|
132
|
+
save(ok): . (dataset)
|
|
133
|
+
add(ok): models/abrupt/silver/05_02/slurm-10524535.out (file)
|
|
134
|
+
add(ok): models/abrupt/silver/05_02/slurm-job-10524535.env.json (file)
|
|
135
|
+
add(ok): models/abrupt/silver/05_02/model_0.model.gz (file)
|
|
136
|
+
add(ok): models/abrupt/silver/05_02/model.scaler.gz (file)
|
|
137
|
+
save(ok): . (dataset)
|
|
138
|
+
finish(impossible): [Slurm job(s) for job 10524556 are not complete.Statuses: 10524556: FAILED]
|
|
139
|
+
finish(impossible): [Slurm job(s) for job 10524620 are not complete.Statuses: 10524620_0: COMPLETED, 10524620_1: COMPLETED, 10524620_2: TIMEOUT]
|
|
140
|
+
action summary:
|
|
141
|
+
add (ok: 7)
|
|
142
|
+
finish (impossible: 2)
|
|
143
|
+
save (ok: 2)
|
|
144
|
+
|
|
145
|
+
To close the failed jobs:
|
|
146
|
+
|
|
147
|
+
datalad slurm-finish --close-failed-jobs
|
|
148
|
+
|
|
149
|
+
finish(ok): [Closing failed / cancelled jobs. Statuses: 10524556: FAILED]
|
|
150
|
+
finish(ok): [Closing failed / cancelled jobs. Statuses: 10524620_0: COMPLETED, 10524620_1: COMPLETED, 10524620_2: TIMEOUT]
|
|
151
|
+
action summary:
|
|
152
|
+
finish (ok: 2)
|
|
153
|
+
|
|
154
|
+
Note that if any sub-job of an array job fails, that whole job is treated as a failed job. The user always has the option to manually commit the successful outputs if desired.
|
|
155
|
+
|
|
156
|
+
The git history would then appear like so:
|
|
157
|
+
|
|
158
|
+
git log --oneline
|
|
159
|
+
|
|
160
|
+
a8e4aa6 (HEAD -> master) [DATALAD SLURM RUN] Slurm job 10524535: Completed
|
|
161
|
+
25067fe [DATALAD SLURM RUN] Slurm job 10524442: Completed
|
|
162
|
+
|
|
163
|
+
With one particular entry looking like:
|
|
164
|
+
|
|
165
|
+
commit a8e4aa62519db3b5f63243cc925ee918984bf506 (HEAD -> master)
|
|
166
|
+
Author: Tim Callow <tim@notmyrealemail.com>
|
|
167
|
+
Date: Tue Feb 18 09:31:47 2025 +0100
|
|
168
|
+
|
|
169
|
+
[DATALAD SLURM RUN] Slurm job 10524535: Completed
|
|
170
|
+
|
|
171
|
+
=== Do not change lines below ===
|
|
172
|
+
{
|
|
173
|
+
"chain": [],
|
|
174
|
+
"cmd": "sbatch submit_silver.slurm",
|
|
175
|
+
"commit_id": null,
|
|
176
|
+
"dsid": "61576cad-ea4f-4425-8f35-16b9955c9926",
|
|
177
|
+
"extra_inputs": [],
|
|
178
|
+
"inputs": [],
|
|
179
|
+
"outputs": [
|
|
180
|
+
"models/abrupt/silver",
|
|
181
|
+
"models/abrupt/silver/05_02/slurm-10524535.out",
|
|
182
|
+
"models/abrupt/silver/05_02/slurm-job-10524535.env.json"
|
|
183
|
+
],
|
|
184
|
+
"pwd": ".",
|
|
185
|
+
"slurm_job_id": 10524535,
|
|
186
|
+
"slurm_outputs": [
|
|
187
|
+
"models/abrupt/silver/05_02/slurm-10524535.out",
|
|
188
|
+
"models/abrupt/silver/05_02/slurm-job-10524535.env.json"
|
|
189
|
+
]
|
|
190
|
+
}
|
|
191
|
+
^^^ Do not change lines above ^^^
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
## Contributing
|
|
195
|
+
|
|
196
|
+
The `datalad-slurm` extension is still in the very early stages of development. We welcome contributors and testers of the package. Please document any issues on GitHub and we will try to resolve them.
|
|
197
|
+
|
|
198
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md) if you are interested in internals or
|
|
199
|
+
contributing to the project.
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
# datalad-slurm: A DataLad extension for HPC (slurm) systems
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/datalad-slurm/) [](https://github.com/datalad/datalad-slurm/actions) [](https://codecov.io/github/datalad/datalad-slurm?branch=main) [](https://datalad-slurm.readthedocs.io/en/latest/?badge=latest)
|
|
4
|
+
|
|
5
|
+
[](https://pypi.org/project/datalad-slurm/) [](https://pypi.org/project/datalad-slurm/) [](https://doi.org/10.5281/zenodo.12345678)
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
`datalad-slurm` is an extension to the [DataLad](http://datalad.org) package for high-performance computing (HPC), specifically slurm systems.
|
|
9
|
+
|
|
10
|
+
DataLad is a package which facilitates adherence to the [FAIR](https://www.nature.com/articles/sdata201618) research data management principles.
|
|
11
|
+
|
|
12
|
+
`datalad-slurm` sits on top of the main DataLad package, and it is designed to improve the DataLad workflow on HPC systems. The package is aimed at slurm systems due to the prominence of slurm in HPC settings, but in the future it may be extended to HPC systems more generally.
|
|
13
|
+
|
|
14
|
+
`datalad-slurm` makes it easier for users to manage their research data on HPC systems with DataLad, and also solves the following conflicts of DataLad usage in HPC systems:
|
|
15
|
+
|
|
16
|
+
- **Inefficient** sequential sections in highly parallel HPC jobs
|
|
17
|
+
- **Critical** race conditions between git commands in concurrent jobs
|
|
18
|
+
|
|
19
|
+
## Installation
|
|
20
|
+
|
|
21
|
+
First, install the main [DataLad](http://datalad.org) package and its dependencies.
|
|
22
|
+
|
|
23
|
+
Then, clone this repository and install the extension with:
|
|
24
|
+
|
|
25
|
+
pip install -e .
|
|
26
|
+
|
|
27
|
+
## Example usage
|
|
28
|
+
|
|
29
|
+
To **schedule** a slurm script:
|
|
30
|
+
|
|
31
|
+
datalad slurm-schedule --output=<output_files_or_dir> <slurm_submission_command>
|
|
32
|
+
|
|
33
|
+
where `<output_files_or_dir>` are the expected outputs from the job, and `<slurm_submission_command>` is for example `sbatch submit_script`. Further optional command line arguments can be found in the documentation.
|
|
34
|
+
|
|
35
|
+
Multiple jobs (including array jobs) can be scheduled sequentially. They are tracked in an SQLite database. Note that any open jobs must not have conflicting outputs with previously scheduled jobs. This is so that the outputs of each slurm run can be tracked to the slurm job which generated them.
|
|
36
|
+
|
|
37
|
+
To **finish** (i.e. post-process) these jobs (once they are complete), simply run:
|
|
38
|
+
|
|
39
|
+
datalad slurm-finish
|
|
40
|
+
|
|
41
|
+
Alternatively, to finish a particular scheduled job, run:
|
|
42
|
+
|
|
43
|
+
datalad slurm-finish <slurm_job_id>
|
|
44
|
+
|
|
45
|
+
This will create a `[DATALAD SLURM RUN]` entry in the git log, analogous to a `datalad run` command.
|
|
46
|
+
|
|
47
|
+
`datalad-slurm` will flag an error for any jobs which could not be post-processed, either because they are still running, or the job failed. These are not automatically cleared from the SQLite database. The output files should first be removed or manually added in git, before running
|
|
48
|
+
|
|
49
|
+
datalad slurm-finish --close-failed-jobs
|
|
50
|
+
|
|
51
|
+
To clear the SQLite database. To inspect the current status of all open jobs (without saving anything in git), run:
|
|
52
|
+
|
|
53
|
+
datalad slurm-finish --list-open-jobs
|
|
54
|
+
|
|
55
|
+
To **reschedule** a previously scheduled job:
|
|
56
|
+
|
|
57
|
+
datalad slurm-reschedule <schedule_commit_hash>
|
|
58
|
+
|
|
59
|
+
where `<schedule_commit_hash>` is the commit hash of the previously scheduled job. There must also be a corresponding `datalad slurm-finish` command to the original `datalad slurm-schedule`, otherwise `datalad slurm-reschedule` will throw an error.
|
|
60
|
+
|
|
61
|
+
In the lingo of the original DataLad package, the combination of `datalad slurm-schedule + datalad slurm-finish` is similar to `datalad run`, and `datalad slurm-reschedule + datalad slurm-finish` is similar to `datalad rerun`.
|
|
62
|
+
|
|
63
|
+
An example workflow could look like this (constructed deliberately to have some failed jobs):
|
|
64
|
+
|
|
65
|
+
datalad slurm-schedule -o models/abrupt/gold/ sbatch submit_gold.slurm
|
|
66
|
+
datalad slurm-schedule -o models/abrupt/silver/ sbatch submit_silver.slurm
|
|
67
|
+
datalad slurm-schedule -o models/abrupt/bronze/ sbatch submit_bronze.slurm
|
|
68
|
+
datalad slurm-schedule -o models/abrupt/platinum/ sbatch submit_array_platinum.slurm
|
|
69
|
+
|
|
70
|
+
Checking the job statuses at some point while they are running:
|
|
71
|
+
|
|
72
|
+
datalad slurm-finish --list-open-jobs
|
|
73
|
+
|
|
74
|
+
The following jobs are open:
|
|
75
|
+
|
|
76
|
+
slurm-job-id slurm-job-status
|
|
77
|
+
10524442 COMPLETED
|
|
78
|
+
10524535 RUNNING
|
|
79
|
+
10524556 FAILED
|
|
80
|
+
10524620 PENDING
|
|
81
|
+
|
|
82
|
+
Later, once all the jobs have finished running:
|
|
83
|
+
|
|
84
|
+
datalad slurm-finish
|
|
85
|
+
|
|
86
|
+
add(ok): models/abrupt/gold/05_02/slurm-10524442.out (file)
|
|
87
|
+
add(ok): models/abrupt/gold/05_02/slurm-job-10524442.env.json (file)
|
|
88
|
+
add(ok): models/abrupt/gold/05_02/model_0.model.gz (file)
|
|
89
|
+
save(ok): . (dataset)
|
|
90
|
+
add(ok): models/abrupt/silver/05_02/slurm-10524535.out (file)
|
|
91
|
+
add(ok): models/abrupt/silver/05_02/slurm-job-10524535.env.json (file)
|
|
92
|
+
add(ok): models/abrupt/silver/05_02/model_0.model.gz (file)
|
|
93
|
+
add(ok): models/abrupt/silver/05_02/model.scaler.gz (file)
|
|
94
|
+
save(ok): . (dataset)
|
|
95
|
+
finish(impossible): [Slurm job(s) for job 10524556 are not complete.Statuses: 10524556: FAILED]
|
|
96
|
+
finish(impossible): [Slurm job(s) for job 10524620 are not complete.Statuses: 10524620_0: COMPLETED, 10524620_1: COMPLETED, 10524620_2: TIMEOUT]
|
|
97
|
+
action summary:
|
|
98
|
+
add (ok: 7)
|
|
99
|
+
finish (impossible: 2)
|
|
100
|
+
save (ok: 2)
|
|
101
|
+
|
|
102
|
+
To close the failed jobs:
|
|
103
|
+
|
|
104
|
+
datalad slurm-finish --close-failed-jobs
|
|
105
|
+
|
|
106
|
+
finish(ok): [Closing failed / cancelled jobs. Statuses: 10524556: FAILED]
|
|
107
|
+
finish(ok): [Closing failed / cancelled jobs. Statuses: 10524620_0: COMPLETED, 10524620_1: COMPLETED, 10524620_2: TIMEOUT]
|
|
108
|
+
action summary:
|
|
109
|
+
finish (ok: 2)
|
|
110
|
+
|
|
111
|
+
Note that if any sub-job of an array job fails, that whole job is treated as a failed job. The user always has the option to manually commit the successful outputs if desired.
|
|
112
|
+
|
|
113
|
+
The git history would then appear like so:
|
|
114
|
+
|
|
115
|
+
git log --oneline
|
|
116
|
+
|
|
117
|
+
a8e4aa6 (HEAD -> master) [DATALAD SLURM RUN] Slurm job 10524535: Completed
|
|
118
|
+
25067fe [DATALAD SLURM RUN] Slurm job 10524442: Completed
|
|
119
|
+
|
|
120
|
+
With one particular entry looking like:
|
|
121
|
+
|
|
122
|
+
commit a8e4aa62519db3b5f63243cc925ee918984bf506 (HEAD -> master)
|
|
123
|
+
Author: Tim Callow <tim@notmyrealemail.com>
|
|
124
|
+
Date: Tue Feb 18 09:31:47 2025 +0100
|
|
125
|
+
|
|
126
|
+
[DATALAD SLURM RUN] Slurm job 10524535: Completed
|
|
127
|
+
|
|
128
|
+
=== Do not change lines below ===
|
|
129
|
+
{
|
|
130
|
+
"chain": [],
|
|
131
|
+
"cmd": "sbatch submit_silver.slurm",
|
|
132
|
+
"commit_id": null,
|
|
133
|
+
"dsid": "61576cad-ea4f-4425-8f35-16b9955c9926",
|
|
134
|
+
"extra_inputs": [],
|
|
135
|
+
"inputs": [],
|
|
136
|
+
"outputs": [
|
|
137
|
+
"models/abrupt/silver",
|
|
138
|
+
"models/abrupt/silver/05_02/slurm-10524535.out",
|
|
139
|
+
"models/abrupt/silver/05_02/slurm-job-10524535.env.json"
|
|
140
|
+
],
|
|
141
|
+
"pwd": ".",
|
|
142
|
+
"slurm_job_id": 10524535,
|
|
143
|
+
"slurm_outputs": [
|
|
144
|
+
"models/abrupt/silver/05_02/slurm-10524535.out",
|
|
145
|
+
"models/abrupt/silver/05_02/slurm-job-10524535.env.json"
|
|
146
|
+
]
|
|
147
|
+
}
|
|
148
|
+
^^^ Do not change lines above ^^^
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
## Contributing
|
|
152
|
+
|
|
153
|
+
The `datalad-slurm` extension is still in the very early stages of development. We welcome contributors and testers of the package. Please document any issues on GitHub and we will try to resolve them.
|
|
154
|
+
|
|
155
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md) if you are interested in internals or
|
|
156
|
+
contributing to the project.
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# ## ### ### ### ### ### ### ### ### ### ### ### ### ### ### ### ### ### ### ##
|
|
2
|
+
#
|
|
3
|
+
# See COPYING file distributed along with the DataLad package for the
|
|
4
|
+
# copyright and license terms.
|
|
5
|
+
#
|
|
6
|
+
# ## ### ### ### ### ### ### ### ### ### ### ### ### ### ### ### ### ### ### ##
|
|
7
|
+
"""Python package for functionality needed at package 'build' time by DataLad and its extensions
|
|
8
|
+
|
|
9
|
+
__init__ here should be really minimalistic, not import submodules by default
|
|
10
|
+
and submodules should also not require heavy dependencies.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
__version__ = '0.1'
|