access-profiling 0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- access/profiling/__init__.py +32 -0
- access/profiling/access_models.py +84 -0
- access/profiling/cice5_parser.py +80 -0
- access/profiling/cylc_manager.py +209 -0
- access/profiling/cylc_parser.py +146 -0
- access/profiling/esmf_parser.py +183 -0
- access/profiling/experiment.py +263 -0
- access/profiling/fms_parser.py +80 -0
- access/profiling/manager.py +559 -0
- access/profiling/metrics.py +85 -0
- access/profiling/parser.py +263 -0
- access/profiling/payu_manager.py +338 -0
- access/profiling/payujson_parser.py +75 -0
- access/profiling/plotting_utils.py +116 -0
- access/profiling/scaling.py +128 -0
- access/profiling/um_parser.py +242 -0
- access_profiling-0.1.dist-info/METADATA +100 -0
- access_profiling-0.1.dist-info/RECORD +21 -0
- access_profiling-0.1.dist-info/WHEEL +5 -0
- access_profiling-0.1.dist-info/licenses/LICENSE +201 -0
- access_profiling-0.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
# Copyright 2025 ACCESS-NRI and contributors. See the top-level COPYRIGHT file for details.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
|
|
4
|
+
"""Classes and utilities for reading and transforming profiling data.
|
|
5
|
+
|
|
6
|
+
Data formats
|
|
7
|
+
------------
|
|
8
|
+
Parsers return a plain dict. Three shapes are supported:
|
|
9
|
+
|
|
10
|
+
Flat (standard)
|
|
11
|
+
One list per metric, all the same length as 'region':
|
|
12
|
+
|
|
13
|
+
{'region': [...], metric_a: [...], metric_b: [...]}
|
|
14
|
+
|
|
15
|
+
Hierarchical (nested dict)
|
|
16
|
+
Used when regions form a call-stack tree (e.g. ESMF). String keys are child
|
|
17
|
+
region names; ProfilingMetric keys are metric values. The two key types never
|
|
18
|
+
collide, so no separator is needed:
|
|
19
|
+
|
|
20
|
+
{'[ESMF]': {tavg: 2558.6, '[ICE] RunPhase1': {tavg: 155.8, ...}}}
|
|
21
|
+
|
|
22
|
+
Use flatten_hierarchical() to convert to the flat format.
|
|
23
|
+
|
|
24
|
+
Per-PE
|
|
25
|
+
Each region has one measurement per processing element (MPI process). Metric
|
|
26
|
+
values are 2D lists of shape (n_regions, n_pes); a 'pe' key holds the PE IDs:
|
|
27
|
+
|
|
28
|
+
{'region': [...], 'pe': [0, 1, ..., N-1], metric_a: [[...], [...]]}
|
|
29
|
+
|
|
30
|
+
Use aggregate_pe_data() on the resulting xarray Dataset to reduce over PEs
|
|
31
|
+
and compute load-imbalance statistics (see metrics.py for the naming convention).
|
|
32
|
+
|
|
33
|
+
Utilities
|
|
34
|
+
---------
|
|
35
|
+
flatten_hierarchical(data, metrics)
|
|
36
|
+
Converts a hierarchical nested dict to the standard flat dict (DFS pre-order).
|
|
37
|
+
aggregate_pe_data(ds)
|
|
38
|
+
Reduces a per-PE Dataset along 'pe', producing variables named
|
|
39
|
+
{var}_{stat}_pe for each input variable and statistic.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
import os
|
|
43
|
+
from abc import ABC, abstractmethod
|
|
44
|
+
from pathlib import Path
|
|
45
|
+
from typing import Any
|
|
46
|
+
|
|
47
|
+
# Next import is required to register pint with xarray
|
|
48
|
+
import pint_xarray # noqa: F401
|
|
49
|
+
import xarray as xr
|
|
50
|
+
|
|
51
|
+
from access.profiling.metrics import ProfilingMetric
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class ProfilingParser(ABC):
|
|
55
|
+
"""Abstract parser of profiling data.
|
|
56
|
+
|
|
57
|
+
The main purpose of a parser is to read profiling data from a file and return it
|
|
58
|
+
as a dict. Three output shapes are supported (see module docstring for full details):
|
|
59
|
+
|
|
60
|
+
Flat (standard)::
|
|
61
|
+
|
|
62
|
+
{
|
|
63
|
+
'region': ['region1', 'region2', ...],
|
|
64
|
+
metric_a: [val1a, val2a, ...],
|
|
65
|
+
metric_b: [val1b, val2b, ...],
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
Hierarchical (nested dict, no 'region' key)::
|
|
69
|
+
|
|
70
|
+
{
|
|
71
|
+
'root_region': {
|
|
72
|
+
metric_a: val,
|
|
73
|
+
'child_region': {metric_a: val, ...},
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
Per-PE (flat with an additional 'pe' key; metric values are 2D lists)::
|
|
78
|
+
|
|
79
|
+
{
|
|
80
|
+
'region': ['region1', 'region2', ...],
|
|
81
|
+
'pe': [0, 1, ..., N-1],
|
|
82
|
+
metric_a: [[pe0_val1, pe1_val1, ...], [pe0_val2, pe1_val2, ...]],
|
|
83
|
+
}
|
|
84
|
+
"""
|
|
85
|
+
|
|
86
|
+
_metrics: list[ProfilingMetric]
|
|
87
|
+
|
|
88
|
+
@property
|
|
89
|
+
def metrics(self) -> list[ProfilingMetric]:
|
|
90
|
+
"""list: Metrics available when using this parser."""
|
|
91
|
+
return self._metrics
|
|
92
|
+
|
|
93
|
+
@abstractmethod
|
|
94
|
+
def parse(self, file_path: str | Path | os.PathLike) -> dict:
|
|
95
|
+
"""Parse the given file.
|
|
96
|
+
|
|
97
|
+
Args:
|
|
98
|
+
file_path (str | Path | os.PathLike): file to parse.
|
|
99
|
+
|
|
100
|
+
Returns:
|
|
101
|
+
dict: profiling data in one of the three formats described in the class
|
|
102
|
+
docstring (flat, hierarchical, or per-PE).
|
|
103
|
+
|
|
104
|
+
Raises:
|
|
105
|
+
ValueError: If no parsable text is found in file_path.
|
|
106
|
+
TypeError: If file_path cannot be converted to a valid Path object.
|
|
107
|
+
FileNotFoundError: If file_path doesn't exist or isn't a file.
|
|
108
|
+
"""
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def flatten_hierarchical(data: dict, metrics: list[ProfilingMetric]) -> dict:
|
|
112
|
+
"""Converts a hierarchical (nested dict) parser output into the standard flat format.
|
|
113
|
+
|
|
114
|
+
Traverses the nested dict depth-first (pre-order: parent before children).
|
|
115
|
+
At each node, string keys are treated as child region names and ProfilingMetric
|
|
116
|
+
keys as metric values — these types never collide so no separator is needed.
|
|
117
|
+
|
|
118
|
+
Args:
|
|
119
|
+
data (dict): Nested dict as returned by a hierarchical parser. At each level,
|
|
120
|
+
string keys are child region names and ProfilingMetric keys are metric values.
|
|
121
|
+
metrics (list[ProfilingMetric]): Metrics to extract from each node.
|
|
122
|
+
|
|
123
|
+
Returns:
|
|
124
|
+
dict: Standard flat profiling dict with a 'region' key and one list per metric.
|
|
125
|
+
"""
|
|
126
|
+
result: dict = {"region": []}
|
|
127
|
+
for m in metrics:
|
|
128
|
+
result[m] = []
|
|
129
|
+
|
|
130
|
+
def _visit(node: dict, name: str) -> None:
|
|
131
|
+
result["region"].append(name)
|
|
132
|
+
for m in metrics:
|
|
133
|
+
result[m].append(node.get(m))
|
|
134
|
+
for key, value in node.items():
|
|
135
|
+
if isinstance(key, str): # child region — ProfilingMetric keys are not str
|
|
136
|
+
_visit(value, key)
|
|
137
|
+
|
|
138
|
+
for region_name, region_data in data.items():
|
|
139
|
+
_visit(region_data, region_name)
|
|
140
|
+
|
|
141
|
+
return result
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def aggregate_pe_data(ds: xr.Dataset) -> xr.Dataset:
|
|
145
|
+
"""Aggregates a per-PE profiling Dataset into summary statistics over the pe dimension.
|
|
146
|
+
|
|
147
|
+
For each data variable, computes the following reductions over the 'pe' dimension
|
|
148
|
+
and returns them as new string-named variables following the {var}_{stat}_pe
|
|
149
|
+
convention defined in metrics.py:
|
|
150
|
+
|
|
151
|
+
======================== =============================================
|
|
152
|
+
Variable Description
|
|
153
|
+
======================== =============================================
|
|
154
|
+
``{var}_min_pe`` minimum across PEs
|
|
155
|
+
``{var}_max_pe`` maximum across PEs
|
|
156
|
+
``{var}_mean_pe`` mean across PEs
|
|
157
|
+
``{var}_median_pe`` median across PEs
|
|
158
|
+
``{var}_std_pe`` standard deviation across PEs
|
|
159
|
+
``{var}_total_pe`` sum across PEs (total work done by all PEs)
|
|
160
|
+
``{var}_argmin_pe`` PE index of the minimum value (dimensionless)
|
|
161
|
+
``{var}_argmax_pe`` PE index of the maximum value (dimensionless)
|
|
162
|
+
``{var}_imbalance_pe`` (max - min) / mean; 0 = perfectly balanced
|
|
163
|
+
======================== =============================================
|
|
164
|
+
|
|
165
|
+
Args:
|
|
166
|
+
ds (xr.Dataset): Dataset with a 'pe' dimension, as returned by
|
|
167
|
+
ProfilingLog.parse() for per-PE parser output.
|
|
168
|
+
|
|
169
|
+
Returns:
|
|
170
|
+
xr.Dataset: New Dataset with 'pe' reduced, containing the derived variables
|
|
171
|
+
described above. All non-pe coordinates are preserved.
|
|
172
|
+
|
|
173
|
+
Raises:
|
|
174
|
+
ValueError: If ds does not have a 'pe' dimension.
|
|
175
|
+
"""
|
|
176
|
+
if "pe" not in ds.dims:
|
|
177
|
+
raise ValueError("Dataset does not have a 'pe' dimension.")
|
|
178
|
+
|
|
179
|
+
result_vars = {}
|
|
180
|
+
for var in ds.data_vars:
|
|
181
|
+
da = ds[var]
|
|
182
|
+
base = str(var).replace(" ", "_")
|
|
183
|
+
vmin = da.min("pe")
|
|
184
|
+
vmax = da.max("pe")
|
|
185
|
+
vmean = da.mean("pe")
|
|
186
|
+
result_vars[f"{base}_min_pe"] = vmin
|
|
187
|
+
result_vars[f"{base}_max_pe"] = vmax
|
|
188
|
+
result_vars[f"{base}_mean_pe"] = vmean
|
|
189
|
+
result_vars[f"{base}_median_pe"] = da.median("pe")
|
|
190
|
+
result_vars[f"{base}_std_pe"] = da.std("pe")
|
|
191
|
+
result_vars[f"{base}_total_pe"] = da.sum("pe")
|
|
192
|
+
result_vars[f"{base}_argmin_pe"] = da.argmin("pe")
|
|
193
|
+
result_vars[f"{base}_argmax_pe"] = da.argmax("pe")
|
|
194
|
+
result_vars[f"{base}_imbalance_pe"] = xr.where(vmean != 0, (vmax - vmin) / vmean, 0)
|
|
195
|
+
|
|
196
|
+
coords = {k: v for k, v in ds.coords.items() if k != "pe"}
|
|
197
|
+
return xr.Dataset(result_vars, coords=coords)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _convert_from_string(value: str) -> Any:
|
|
201
|
+
"""Tries to convert a string to the most appropriate numeric type. Leaves it unchanged if conversion does not
|
|
202
|
+
succeed.
|
|
203
|
+
|
|
204
|
+
Args:
|
|
205
|
+
value (str): string to convert.
|
|
206
|
+
|
|
207
|
+
Returns:
|
|
208
|
+
Any: the converted string or the original string.
|
|
209
|
+
"""
|
|
210
|
+
for type_conversion in (int, float):
|
|
211
|
+
try:
|
|
212
|
+
return type_conversion(value)
|
|
213
|
+
except Exception:
|
|
214
|
+
continue
|
|
215
|
+
return value
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _test_file(file_path: str | Path | os.PathLike) -> Path:
|
|
219
|
+
"""Checks whether file_path is a valid path.
|
|
220
|
+
|
|
221
|
+
Args:
|
|
222
|
+
file_path (str | Path | os.PathLike): the path to check.
|
|
223
|
+
|
|
224
|
+
Returns
|
|
225
|
+
Path: file_path as a Path object.
|
|
226
|
+
|
|
227
|
+
Raises:
|
|
228
|
+
TypeError: if file_path cannot be turned into a Path object.
|
|
229
|
+
FileNotFoundError: if file_path can be converted into a Path, but the file dosen't exist.
|
|
230
|
+
"""
|
|
231
|
+
|
|
232
|
+
try:
|
|
233
|
+
path = Path(file_path)
|
|
234
|
+
except TypeError as e:
|
|
235
|
+
raise TypeError(f"{file_path} is not a valid path.") from e
|
|
236
|
+
|
|
237
|
+
if not path.is_file():
|
|
238
|
+
raise FileNotFoundError(f"{file_path} is not a file or doesn't exist.")
|
|
239
|
+
|
|
240
|
+
return path
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def _read_text_file(file_path: str | Path | os.PathLike) -> str:
|
|
244
|
+
"""Checks whether file_path is a valid path to a text file and tries to read it.
|
|
245
|
+
|
|
246
|
+
Args:
|
|
247
|
+
file_path (str | Path | os.PathLike): the path to check/read
|
|
248
|
+
|
|
249
|
+
Returns:
|
|
250
|
+
str: The text within the file.
|
|
251
|
+
|
|
252
|
+
Raises:
|
|
253
|
+
TypeError: if file_path is not a valid path
|
|
254
|
+
FileNotFoundError: if file_path is a path, but is not a file or doesn't exist.
|
|
255
|
+
ValueError: if file_path is a file, but cannot be read as a text file.
|
|
256
|
+
"""
|
|
257
|
+
|
|
258
|
+
path = _test_file(file_path)
|
|
259
|
+
|
|
260
|
+
try:
|
|
261
|
+
return path.read_text()
|
|
262
|
+
except UnicodeDecodeError as e:
|
|
263
|
+
raise ValueError(f"{file_path} is not a text file.") from e
|
|
@@ -0,0 +1,338 @@
|
|
|
1
|
+
# Copyright 2025 ACCESS-NRI and contributors. See the top-level COPYRIGHT file for details.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
|
|
4
|
+
import logging
|
|
5
|
+
from abc import ABC, abstractmethod
|
|
6
|
+
from collections.abc import Callable
|
|
7
|
+
from datetime import timedelta
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from access.config import YAMLParser
|
|
11
|
+
from access.config.esm1p6_layout_input import LayoutSearchConfig
|
|
12
|
+
from access.config.layout_config import LayoutTuple
|
|
13
|
+
from experiment_generator.experiment_generator import ExperimentGenerator
|
|
14
|
+
from experiment_runner.experiment_runner import ExperimentRunner
|
|
15
|
+
|
|
16
|
+
from access.profiling.experiment import ProfilingLog
|
|
17
|
+
from access.profiling.manager import ProfilingExperiment, ProfilingExperimentStatus, ProfilingManager
|
|
18
|
+
from access.profiling.payujson_parser import PayuJSONProfilingParser
|
|
19
|
+
|
|
20
|
+
logger = logging.getLogger(__name__)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class PayuManager(ProfilingManager, ABC):
|
|
24
|
+
"""Abstract base class to handle profiling of Payu configurations."""
|
|
25
|
+
|
|
26
|
+
_repository_directory: str = "config" # Repository directory name needed by the experiment generator and runner.
|
|
27
|
+
_nruns: int = 1 # Number of repetitions for the Payu experiments.
|
|
28
|
+
_startfrom_restart: str = "cold" # Restart option for the Payu experiments.
|
|
29
|
+
|
|
30
|
+
@abstractmethod
|
|
31
|
+
def get_component_logs(self, path: Path) -> dict[str, ProfilingLog]:
|
|
32
|
+
"""Returns available profiling logs for the components in the configuration.
|
|
33
|
+
|
|
34
|
+
Args:
|
|
35
|
+
path (Path): Path to the output directory.
|
|
36
|
+
Returns:
|
|
37
|
+
dict[str, ProfilingLog]: Dictionary mapping component names to their ProfilingLog instances.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
@abstractmethod
|
|
42
|
+
def model_type(self) -> str:
|
|
43
|
+
"""Returns the model type identifier, as defined in Payu."""
|
|
44
|
+
|
|
45
|
+
@abstractmethod
|
|
46
|
+
def generate_core_layouts_from_node_count(
|
|
47
|
+
self,
|
|
48
|
+
num_nodes: float,
|
|
49
|
+
cores_per_node: int,
|
|
50
|
+
layout_search_config: LayoutSearchConfig | None = None,
|
|
51
|
+
) -> list:
|
|
52
|
+
"""Generates core layouts from the given number of nodes.
|
|
53
|
+
|
|
54
|
+
Args:
|
|
55
|
+
num_nodes (float): Number of nodes.
|
|
56
|
+
cores_per_node (int): Number of cores per node.
|
|
57
|
+
layout_search_config (LayoutSearchConfig | None): Configuration for layout search.
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
@abstractmethod
|
|
61
|
+
def generate_perturbation_block(self, layout: LayoutTuple, branch_name_prefix: str) -> dict:
|
|
62
|
+
"""Generates a perturbation block for the given layout to be passed to the experiment generator.
|
|
63
|
+
|
|
64
|
+
Args:
|
|
65
|
+
layout (LayoutTuple): Core layout tuple.
|
|
66
|
+
branch_name_prefix (str): Branch name prefix.
|
|
67
|
+
Returns:
|
|
68
|
+
dict: Perturbation block configuration.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
@property
|
|
72
|
+
def nruns(self) -> int:
|
|
73
|
+
"""Returns the number of repetitions for the Payu experiments.
|
|
74
|
+
|
|
75
|
+
Returns:
|
|
76
|
+
int: Number of repetitions.
|
|
77
|
+
"""
|
|
78
|
+
return self._nruns
|
|
79
|
+
|
|
80
|
+
@nruns.setter
|
|
81
|
+
def nruns(self, value: int) -> None:
|
|
82
|
+
"""Sets the number of repetitions for the Payu experiments.
|
|
83
|
+
|
|
84
|
+
Args:
|
|
85
|
+
value (int): Number of repetitions.
|
|
86
|
+
"""
|
|
87
|
+
if value < 0:
|
|
88
|
+
raise ValueError("Number of runs must be at least 0.")
|
|
89
|
+
self._nruns = value
|
|
90
|
+
|
|
91
|
+
@property
|
|
92
|
+
def startfrom_restart(self) -> str:
|
|
93
|
+
"""Returns the restart option for the Payu experiments.
|
|
94
|
+
|
|
95
|
+
Returns:
|
|
96
|
+
str: Restart option.
|
|
97
|
+
"""
|
|
98
|
+
return self._startfrom_restart
|
|
99
|
+
|
|
100
|
+
@startfrom_restart.setter
|
|
101
|
+
def startfrom_restart(self, value: str) -> None:
|
|
102
|
+
"""Sets the restart option for the Payu experiments.
|
|
103
|
+
|
|
104
|
+
Args:
|
|
105
|
+
value (str): Restart option.
|
|
106
|
+
"""
|
|
107
|
+
self._startfrom_restart = value
|
|
108
|
+
|
|
109
|
+
def set_control(self, repository, commit) -> None:
|
|
110
|
+
"""Sets the control experiment from an existing Payu configuration.
|
|
111
|
+
|
|
112
|
+
Args:
|
|
113
|
+
repository: Git repository URL or path.
|
|
114
|
+
commit: Git commit hash or identifier.
|
|
115
|
+
"""
|
|
116
|
+
self._repository = repository
|
|
117
|
+
self._control_commit = commit
|
|
118
|
+
|
|
119
|
+
def generate_scaling_experiments(
|
|
120
|
+
self,
|
|
121
|
+
num_nodes_list: list[float],
|
|
122
|
+
control_options: dict,
|
|
123
|
+
cores_per_node: int,
|
|
124
|
+
tol_around_ctrl_ratio: float,
|
|
125
|
+
max_wasted_ncores_frac: float | Callable[[float], float],
|
|
126
|
+
walltime: float | Callable[[float], float],
|
|
127
|
+
) -> None:
|
|
128
|
+
"""Generates scaling experiments using the ExperimentGenerator.
|
|
129
|
+
|
|
130
|
+
Args:
|
|
131
|
+
num_nodes_list (list[int]): List of number of nodes to generate experiments for.
|
|
132
|
+
control_options (dict): Options for the control experiment.
|
|
133
|
+
cores_per_node (int): Number of cores per node.
|
|
134
|
+
tol_around_ctrl_ratio (float): Tolerance around control core ratio for layout generation.
|
|
135
|
+
max_wasted_ncores_frac (float | Callable[[float], float]): Maximum fraction of wasted cores allowed.
|
|
136
|
+
walltime (float | Callable[[float], float]): Walltime in hours for each experiment.
|
|
137
|
+
"""
|
|
138
|
+
|
|
139
|
+
generator_config = {
|
|
140
|
+
"model_type": self.model_type,
|
|
141
|
+
"repository_url": self._repository,
|
|
142
|
+
"start_point": self._control_commit,
|
|
143
|
+
"test_path": str(self.work_dir),
|
|
144
|
+
"repository_directory": self._repository_directory,
|
|
145
|
+
"control_branch_name": "ctrl",
|
|
146
|
+
"Control_Experiment": control_options,
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
seen_layouts = set()
|
|
150
|
+
seqnum = 1
|
|
151
|
+
generator_config["Perturbation_Experiment"] = {}
|
|
152
|
+
for num_nodes in num_nodes_list:
|
|
153
|
+
mwf = max_wasted_ncores_frac(num_nodes) if callable(max_wasted_ncores_frac) else max_wasted_ncores_frac
|
|
154
|
+
layout_config = LayoutSearchConfig(tol_around_ctrl_ratio=tol_around_ctrl_ratio, max_wasted_ncores_frac=mwf)
|
|
155
|
+
layouts = self.generate_core_layouts_from_node_count(
|
|
156
|
+
num_nodes,
|
|
157
|
+
cores_per_node=cores_per_node,
|
|
158
|
+
layout_search_config=layout_config,
|
|
159
|
+
)
|
|
160
|
+
if not layouts:
|
|
161
|
+
logger.warning(f"No layouts found for {num_nodes} nodes")
|
|
162
|
+
continue
|
|
163
|
+
|
|
164
|
+
layouts = [x for x in layouts if x not in seen_layouts]
|
|
165
|
+
seen_layouts.update(layouts)
|
|
166
|
+
logger.info(f"Generated {len(layouts)} layouts for {num_nodes} nodes. Layouts: {layouts}")
|
|
167
|
+
|
|
168
|
+
# TODO: the branch name needs to be simpler and model agnostic
|
|
169
|
+
branch_name = f"layout-unused-cores-to-cice-{layout_config.allocate_unused_cores_to_ice}"
|
|
170
|
+
walltime_hrs = walltime(num_nodes) if callable(walltime) else walltime
|
|
171
|
+
|
|
172
|
+
for layout in layouts:
|
|
173
|
+
pert_config = self.generate_perturbation_block(layout=layout, branch_name_prefix=branch_name)
|
|
174
|
+
branch = pert_config["branches"][0]
|
|
175
|
+
pert_config["config.yaml"]["walltime"] = str(timedelta(hours=walltime_hrs))
|
|
176
|
+
|
|
177
|
+
generator_config["Perturbation_Experiment"][f"Experiment_{seqnum}"] = pert_config
|
|
178
|
+
self.experiments[branch] = ProfilingExperiment(path=self.work_dir / branch / self._repository_directory)
|
|
179
|
+
|
|
180
|
+
seqnum += 1
|
|
181
|
+
|
|
182
|
+
ExperimentGenerator(generator_config).run()
|
|
183
|
+
|
|
184
|
+
def run_experiments(self) -> None:
|
|
185
|
+
"""Runs Payu experiments for profiling data generation."""
|
|
186
|
+
|
|
187
|
+
runner_config = {
|
|
188
|
+
"test_path": self.work_dir,
|
|
189
|
+
"repository_directory": self._repository_directory,
|
|
190
|
+
"running_branches": [],
|
|
191
|
+
"keep_uuid": True,
|
|
192
|
+
"nruns": [],
|
|
193
|
+
"startfrom_restart": [],
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
for path, exp in self.experiments.items():
|
|
197
|
+
if exp.status == ProfilingExperimentStatus.NEW:
|
|
198
|
+
runner_config["running_branches"].append(path)
|
|
199
|
+
runner_config["nruns"].append(self.nruns)
|
|
200
|
+
runner_config["startfrom_restart"].append(self.startfrom_restart)
|
|
201
|
+
exp.status = ProfilingExperimentStatus.RUNNING
|
|
202
|
+
|
|
203
|
+
# Run the experiment runner
|
|
204
|
+
if runner_config["running_branches"]:
|
|
205
|
+
ExperimentRunner(runner_config).run()
|
|
206
|
+
else:
|
|
207
|
+
logger.info("No new experiments to run. Will skip execution.")
|
|
208
|
+
|
|
209
|
+
# We are marking all running experiments as done here, but later this should be implemented properly
|
|
210
|
+
# so that an actual check is performed, probably somewhere else.
|
|
211
|
+
for exp in self.experiments.values():
|
|
212
|
+
if exp.status == ProfilingExperimentStatus.RUNNING:
|
|
213
|
+
exp.status = ProfilingExperimentStatus.DONE
|
|
214
|
+
|
|
215
|
+
def delete_experiments(
|
|
216
|
+
self,
|
|
217
|
+
experiments: list[str] | None = None,
|
|
218
|
+
all_experiments: bool = False,
|
|
219
|
+
dry_run: bool = False,
|
|
220
|
+
remove_repo_dir: bool = False,
|
|
221
|
+
) -> None:
|
|
222
|
+
"""Deletes Payu experiments from the work directory and remove them from the manager.
|
|
223
|
+
|
|
224
|
+
Args:
|
|
225
|
+
experiments (list[str] | None): List of experiments (branches) to delete.
|
|
226
|
+
all_experiments (bool): If True, deletes all experiments managed by this instance.
|
|
227
|
+
dry_run (bool): If True, performs a dry run without deleting files. Defaults to False.
|
|
228
|
+
remove_repo_dir (bool): If True, removes the base repository directory if no branches are using it.
|
|
229
|
+
"""
|
|
230
|
+
# remove_repo_dir would already be forwarded to _delete_experiment via the base class **kwargs, but this
|
|
231
|
+
# override declares it explicitly so it stays a documented, discoverable and typo-checked argument of the
|
|
232
|
+
# public Payu API rather than a hidden keyword convention.
|
|
233
|
+
super().delete_experiments(
|
|
234
|
+
experiments=experiments,
|
|
235
|
+
all_experiments=all_experiments,
|
|
236
|
+
dry_run=dry_run,
|
|
237
|
+
remove_repo_dir=remove_repo_dir,
|
|
238
|
+
)
|
|
239
|
+
|
|
240
|
+
def _delete_experiment(self, name: str, dry_run: bool, remove_repo_dir: bool = False) -> None:
|
|
241
|
+
"""Deletes a single Payu experiment (branch) via the experiment runner.
|
|
242
|
+
|
|
243
|
+
Args:
|
|
244
|
+
name (str): Name of the experiment (branch) to delete.
|
|
245
|
+
dry_run (bool): If True, performs a dry run without deleting files.
|
|
246
|
+
remove_repo_dir (bool): If True, removes the base repository directory if no branches are using it.
|
|
247
|
+
"""
|
|
248
|
+
runner_config = {
|
|
249
|
+
"test_path": self.work_dir,
|
|
250
|
+
"repository_directory": self._repository_directory,
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
runner = ExperimentRunner(runner_config)
|
|
254
|
+
|
|
255
|
+
runner.delete_experiments(
|
|
256
|
+
branches=[name],
|
|
257
|
+
hard=True,
|
|
258
|
+
dry_run=dry_run,
|
|
259
|
+
remove_repo_dir=remove_repo_dir,
|
|
260
|
+
)
|
|
261
|
+
|
|
262
|
+
def archive_experiments(
|
|
263
|
+
self,
|
|
264
|
+
exclude_dirs: list[str] | None = None,
|
|
265
|
+
exclude_files: list[str] | None = None,
|
|
266
|
+
follow_symlinks: bool = True,
|
|
267
|
+
overwrite: bool = False,
|
|
268
|
+
) -> None:
|
|
269
|
+
"""Archives completed experiments to the specified archive path.
|
|
270
|
+
|
|
271
|
+
Args:
|
|
272
|
+
exclude_dirs (list[str] | None): Directory patterns to exclude when archiving experiments. Defaults to
|
|
273
|
+
[".git", "restart*"] if not provided.
|
|
274
|
+
exclude_files (list[str] | None): File patterns to exclude when archiving experiments. Defaults to
|
|
275
|
+
["*.nc"] if not provided.
|
|
276
|
+
follow_symlinks (bool): Whether to follow symlinks when archiving experiments. Defaults to True.
|
|
277
|
+
overwrite (bool): Whether to overwrite existing archives. Defaults to False.
|
|
278
|
+
"""
|
|
279
|
+
if exclude_dirs is None:
|
|
280
|
+
exclude_dirs = [".git", "restart*"]
|
|
281
|
+
if exclude_files is None:
|
|
282
|
+
exclude_files = ["*.nc"]
|
|
283
|
+
super().archive_experiments(
|
|
284
|
+
exclude_dirs=exclude_dirs, exclude_files=exclude_files, follow_symlinks=follow_symlinks, overwrite=overwrite
|
|
285
|
+
)
|
|
286
|
+
|
|
287
|
+
def parse_ncpus(self, path: Path, run_path: Path | None = None) -> int:
|
|
288
|
+
"""Parses the number of CPUs used in a given Payu experiment.
|
|
289
|
+
|
|
290
|
+
Args:
|
|
291
|
+
path (Path): Path to the Payu experiment directory. Must contain a config.yaml file.
|
|
292
|
+
run_path (Path | None): Optional path to a separate runs directory. Unused for Payu experiments.
|
|
293
|
+
Returns:
|
|
294
|
+
int: Number of CPUs used in the experiment. If multiple submodels are defined, returns the sum of their
|
|
295
|
+
ncpus.
|
|
296
|
+
"""
|
|
297
|
+
config_path = path / "config.yaml"
|
|
298
|
+
payu_config = YAMLParser().parse(config_path.read_text())
|
|
299
|
+
if "submodels" in payu_config:
|
|
300
|
+
return sum(submodel["ncpus"] for submodel in payu_config["submodels"])
|
|
301
|
+
else:
|
|
302
|
+
return payu_config["ncpus"]
|
|
303
|
+
|
|
304
|
+
def profiling_logs(self, path: Path, run_path: Path | None = None) -> dict[str, dict[int, ProfilingLog]]:
|
|
305
|
+
"""Returns all profiling logs from the specified path.
|
|
306
|
+
|
|
307
|
+
Payu can be asked to submit the same experiment several times, in which case each run produces its own
|
|
308
|
+
output directory and its own telemetry log. Payu numbers both after the same run counter, so the logs of
|
|
309
|
+
every run are returned, keyed by that number.
|
|
310
|
+
|
|
311
|
+
Args:
|
|
312
|
+
path (Path): Path to the experiment directory.
|
|
313
|
+
run_path (Path | None): Optional path to a separate runs directory. Unused for Payu experiments.
|
|
314
|
+
Returns:
|
|
315
|
+
dict[str, dict[int, ProfilingLog]]: Dictionary mapping log names to their logs, keyed by run number.
|
|
316
|
+
"""
|
|
317
|
+
logs: dict[str, dict[int, ProfilingLog]] = {}
|
|
318
|
+
|
|
319
|
+
# Check archive directory exists
|
|
320
|
+
archive = path / "archive"
|
|
321
|
+
if not archive.is_dir():
|
|
322
|
+
raise FileNotFoundError(f"Directory {archive} does not exist!")
|
|
323
|
+
|
|
324
|
+
# Parse payu json profiling data if available. Payu names the directory holding each log after the run number.
|
|
325
|
+
for json_path in archive.glob("payu_jobs/*/run/*.json"):
|
|
326
|
+
logs.setdefault("payu", {})[int(json_path.parts[-3])] = ProfilingLog(json_path, PayuJSONProfilingParser())
|
|
327
|
+
|
|
328
|
+
# Get the logs of each component of every output directory. Payu names these outputNNN, NNN being the run
|
|
329
|
+
# number, so output003 holds the same run as payu_jobs/3.
|
|
330
|
+
output_dirs = sorted(archive.glob("output*"))
|
|
331
|
+
if not output_dirs:
|
|
332
|
+
raise FileNotFoundError(f"No output files found in {path}!")
|
|
333
|
+
for output_dir in output_dirs:
|
|
334
|
+
run = int(output_dir.name.removeprefix("output"))
|
|
335
|
+
for name, log in self.get_component_logs(output_dir).items():
|
|
336
|
+
logs.setdefault(name, {})[run] = log
|
|
337
|
+
|
|
338
|
+
return logs
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
# Copyright 2025 ACCESS-NRI and contributors. See the top-level COPYRIGHT file for details.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
|
|
4
|
+
"""Parser for payu JSON walltime data generated by payu.
|
|
5
|
+
The data to be parsed is written in the following form:
|
|
6
|
+
|
|
7
|
+
{
|
|
8
|
+
"scheduler_job_id": "149764665.gadi-pbs",
|
|
9
|
+
"scheduler_type": "pbs",
|
|
10
|
+
# ... many more fields ...
|
|
11
|
+
"timings": {
|
|
12
|
+
"payu_start_time": "2025-09-16T08:52:50.748807",
|
|
13
|
+
"payu_setup_duration_seconds": 47.73822930175811,
|
|
14
|
+
"payu_model_run_duration_seconds": 6776.044810215011,
|
|
15
|
+
"payu_run_duration_seconds": 6779.385873348918,
|
|
16
|
+
"payu_archive_duration_seconds": 8.063649574294686,
|
|
17
|
+
"payu_finish_time": "2025-09-16T10:46:48.974451",
|
|
18
|
+
"payu_total_duration_seconds": 6838.225644
|
|
19
|
+
},
|
|
20
|
+
# ... more fields
|
|
21
|
+
}
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
import json
|
|
25
|
+
import os
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
|
|
28
|
+
from access.profiling.metrics import tmax
|
|
29
|
+
from access.profiling.parser import ProfilingParser, _read_text_file
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class PayuJSONProfilingParser(ProfilingParser):
|
|
33
|
+
"""Payu JSON job output profiling parser."""
|
|
34
|
+
|
|
35
|
+
_metrics = [tmax]
|
|
36
|
+
|
|
37
|
+
def parse(self, file_path: str | Path | os.PathLike) -> dict:
|
|
38
|
+
"""Implements "parse" abstract method to parse a JSON file generated by Payu.
|
|
39
|
+
|
|
40
|
+
Args:
|
|
41
|
+
file_path (str | Path | os.PathLike): file to parse.
|
|
42
|
+
|
|
43
|
+
Returns:
|
|
44
|
+
dict: Parsed timing information.
|
|
45
|
+
|
|
46
|
+
Raises:
|
|
47
|
+
ValueError: when input stream is not valid Payu JSON output.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
stream = _read_text_file(file_path)
|
|
51
|
+
|
|
52
|
+
errmsg = "No Payu profiling data found"
|
|
53
|
+
|
|
54
|
+
try:
|
|
55
|
+
timings = json.loads(stream)["timings"]
|
|
56
|
+
except Exception as e:
|
|
57
|
+
raise ValueError(errmsg) from e
|
|
58
|
+
|
|
59
|
+
# remove known keys not relevant to profiling
|
|
60
|
+
for unwanted_key in ("payu_start_time", "payu_finish_time"):
|
|
61
|
+
if unwanted_key in timings:
|
|
62
|
+
del timings[unwanted_key]
|
|
63
|
+
|
|
64
|
+
# error if no relevant keys in timings
|
|
65
|
+
if not timings:
|
|
66
|
+
raise ValueError(errmsg)
|
|
67
|
+
|
|
68
|
+
result = {"region": [], tmax: []}
|
|
69
|
+
|
|
70
|
+
# transpose dict to be consistent with other profiling parsers.
|
|
71
|
+
for k, v in timings.items():
|
|
72
|
+
result["region"].append(k)
|
|
73
|
+
result[tmax].append(v)
|
|
74
|
+
|
|
75
|
+
return result
|