access-profiling 0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- access/profiling/__init__.py +32 -0
- access/profiling/access_models.py +84 -0
- access/profiling/cice5_parser.py +80 -0
- access/profiling/cylc_manager.py +209 -0
- access/profiling/cylc_parser.py +146 -0
- access/profiling/esmf_parser.py +183 -0
- access/profiling/experiment.py +263 -0
- access/profiling/fms_parser.py +80 -0
- access/profiling/manager.py +559 -0
- access/profiling/metrics.py +85 -0
- access/profiling/parser.py +263 -0
- access/profiling/payu_manager.py +338 -0
- access/profiling/payujson_parser.py +75 -0
- access/profiling/plotting_utils.py +116 -0
- access/profiling/scaling.py +128 -0
- access/profiling/um_parser.py +242 -0
- access_profiling-0.1.dist-info/METADATA +100 -0
- access_profiling-0.1.dist-info/RECORD +21 -0
- access_profiling-0.1.dist-info/WHEEL +5 -0
- access_profiling-0.1.dist-info/licenses/LICENSE +201 -0
- access_profiling-0.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,559 @@
|
|
|
1
|
+
# Copyright 2025 ACCESS-NRI and contributors. See the top-level COPYRIGHT file for details.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
|
|
4
|
+
import logging
|
|
5
|
+
import textwrap
|
|
6
|
+
from abc import ABC, abstractmethod
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
import xarray as xr
|
|
10
|
+
from matplotlib.figure import Figure
|
|
11
|
+
|
|
12
|
+
from access.profiling.experiment import ProfilingExperiment, ProfilingExperimentStatus, ProfilingLog
|
|
13
|
+
from access.profiling.metrics import ProfilingMetric
|
|
14
|
+
from access.profiling.plotting_utils import plot_bar_metrics
|
|
15
|
+
from access.profiling.scaling import plot_scaling_metrics
|
|
16
|
+
|
|
17
|
+
logger = logging.getLogger(__name__)
|
|
18
|
+
|
|
19
|
+
_RUN_DIM_ERROR = (
|
|
20
|
+
"Profiling data still has a 'run' dimension. Use select_best_run() to keep the best run of each experiment, "
|
|
21
|
+
"or aggregate_runs() to reduce over runs."
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class ProfilingManager(ABC):
|
|
26
|
+
"""Abstract base class to handle profiling data and workflows.
|
|
27
|
+
|
|
28
|
+
This high-level class defines methods to parse different types of profiling data. Currently,
|
|
29
|
+
it supports parsing and plotting scaling data, including selecting the best performing experiment
|
|
30
|
+
for each number of CPUs.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
work_dir (Path): Working directory where profiling experiments will be generated and run.
|
|
34
|
+
archive_dir (Path): Directory where completed experiments will be archived.
|
|
35
|
+
archive_exclude_patterns (list[str] | None): File patterns to exclude when archiving experiments.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
work_dir: Path # Working directory where profiling experiments will be generated and run.
|
|
39
|
+
archive_dir: Path # Directory where completed experiments will be archived.
|
|
40
|
+
experiments: dict[str, ProfilingExperiment] # Dictionary storing ProfilingExperiment instances.
|
|
41
|
+
data: dict[
|
|
42
|
+
str, dict[str, xr.Dataset]
|
|
43
|
+
] # Dictionary mapping experiments to component names and their profiling datasets.
|
|
44
|
+
_ncpus_cache: dict[str, int] # Number of CPUs of each experiment, parsed on demand.
|
|
45
|
+
|
|
46
|
+
def __init__(self, work_dir: Path, archive_dir: Path):
|
|
47
|
+
super().__init__()
|
|
48
|
+
self.work_dir = work_dir
|
|
49
|
+
self.archive_dir = archive_dir
|
|
50
|
+
self.experiments = {}
|
|
51
|
+
self.data = {}
|
|
52
|
+
self._ncpus_cache = {}
|
|
53
|
+
|
|
54
|
+
# Discover experiments in the archive directory
|
|
55
|
+
if self.archive_dir.is_dir():
|
|
56
|
+
for branch_path in self.archive_dir.glob("*.tar.gz"):
|
|
57
|
+
if branch_path.is_file():
|
|
58
|
+
branch_name = branch_path.name[: -len(".tar.gz")]
|
|
59
|
+
logger.info(f"Found archived experiment: {branch_name}")
|
|
60
|
+
self.experiments[branch_name] = ProfilingExperiment(path=branch_path)
|
|
61
|
+
|
|
62
|
+
def __repr__(self) -> str:
|
|
63
|
+
"""Returns a string representation of the ProfilingManager."""
|
|
64
|
+
|
|
65
|
+
indent = " "
|
|
66
|
+
summary = f"<{type(self).__name__}>\n"
|
|
67
|
+
summary += indent + f"Working directory: {self.work_dir!r}\n"
|
|
68
|
+
summary += indent + f"Archive directory: {self.archive_dir!r}\n"
|
|
69
|
+
summary += indent + "Experiments:\n"
|
|
70
|
+
for name, exp in self.experiments.items():
|
|
71
|
+
summary += indent * 2 + f"'{name}': {exp!r}\n"
|
|
72
|
+
summary += indent + "Data:\n"
|
|
73
|
+
if self.data == {}:
|
|
74
|
+
summary += indent * 2 + "No parsed data.\n"
|
|
75
|
+
else:
|
|
76
|
+
for name, exp_data in self.data.items():
|
|
77
|
+
summary += indent * 2 + f"'{name}':\n"
|
|
78
|
+
for comp_name, ds in exp_data.items():
|
|
79
|
+
summary += indent * 3 + f"'{comp_name}':\n"
|
|
80
|
+
summary += textwrap.indent(f"{ds}\n", indent * 4)
|
|
81
|
+
return summary
|
|
82
|
+
|
|
83
|
+
@abstractmethod
|
|
84
|
+
def profiling_logs(self, path: Path, run_path: Path | None = None) -> dict[str, dict[int, ProfilingLog]]:
|
|
85
|
+
"""Returns all profiling logs from the specified path.
|
|
86
|
+
|
|
87
|
+
Args:
|
|
88
|
+
path (Path): Path to the experiment directory.
|
|
89
|
+
run_path (Path | None): Optional path to a separate runs directory.
|
|
90
|
+
|
|
91
|
+
Returns:
|
|
92
|
+
dict[str, dict[int, ProfilingLog]]: Dictionary mapping log names to their logs, keyed by run number.
|
|
93
|
+
Configurations with no concept of repeated runs should return a single run, numbered 0.
|
|
94
|
+
"""
|
|
95
|
+
|
|
96
|
+
@abstractmethod
|
|
97
|
+
def parse_ncpus(self, path: Path, run_path: Path | None = None) -> int:
|
|
98
|
+
"""Parses the number of CPUs used in a given experiment in the specified path.
|
|
99
|
+
|
|
100
|
+
Args:
|
|
101
|
+
path (Path): Path to the experiment directory.
|
|
102
|
+
run_path (Path | None): Optional path to a separate runs directory.
|
|
103
|
+
|
|
104
|
+
Returns:
|
|
105
|
+
int: Number of CPUs used in the experiment.
|
|
106
|
+
"""
|
|
107
|
+
|
|
108
|
+
def archive_experiments(
|
|
109
|
+
self,
|
|
110
|
+
exclude_dirs: list[str] | None = None,
|
|
111
|
+
exclude_files: list[str] | None = None,
|
|
112
|
+
follow_symlinks: bool = False,
|
|
113
|
+
overwrite: bool = False,
|
|
114
|
+
) -> None:
|
|
115
|
+
"""Archives completed experiments to the specified archive path.
|
|
116
|
+
|
|
117
|
+
This method will create a tar.gz archive containing relevant data from an experiment. No data will be deleted
|
|
118
|
+
once an experiment is archived, but data will be parsed directly from the archive instead of the original
|
|
119
|
+
experiment directory.
|
|
120
|
+
|
|
121
|
+
Args:
|
|
122
|
+
exclude_dirs (list[str] | None): Directory patterns to exclude when archiving experiments.
|
|
123
|
+
exclude_files (list[str] | None): File patterns to exclude when archiving experiments.
|
|
124
|
+
follow_symlinks (bool): Whether to follow symlinks when archiving experiments. Defaults to False.
|
|
125
|
+
overwrite (bool): Whether to overwrite existing archives. Defaults to False.
|
|
126
|
+
"""
|
|
127
|
+
self.archive_dir.mkdir(parents=True, exist_ok=True)
|
|
128
|
+
for branch, exp in self.experiments.items():
|
|
129
|
+
exp.archive(
|
|
130
|
+
self.archive_dir / branch,
|
|
131
|
+
exclude_dirs=exclude_dirs,
|
|
132
|
+
exclude_files=exclude_files,
|
|
133
|
+
follow_symlinks=follow_symlinks,
|
|
134
|
+
overwrite=overwrite,
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
def add_experiment_from_directory(self, name: str, path: Path) -> None:
|
|
138
|
+
"""Adds an existing experiment from the specified directory.
|
|
139
|
+
|
|
140
|
+
Note that the directory must already exist on disk and be inside the working directory. Also, the experiment
|
|
141
|
+
will be marked as DONE, so any runs associated with the experiment must already be completed.
|
|
142
|
+
|
|
143
|
+
Args:
|
|
144
|
+
name (str): Name of the experiment.
|
|
145
|
+
path (Path): Path to the experiment directory.
|
|
146
|
+
Raises:
|
|
147
|
+
ValueError: If the specified path does not exist, is not a directory, or is not inside the working
|
|
148
|
+
directory.
|
|
149
|
+
"""
|
|
150
|
+
if not path.is_absolute():
|
|
151
|
+
path = self.work_dir / path
|
|
152
|
+
if not path.is_dir():
|
|
153
|
+
raise ValueError(f"Experiment path '{path}' does not exist or is not a directory.")
|
|
154
|
+
if not path.resolve().is_relative_to(self.work_dir.resolve()):
|
|
155
|
+
raise ValueError(f"Experiment path '{path}' is not inside the working directory '{self.work_dir}'.")
|
|
156
|
+
self.experiments[name] = ProfilingExperiment(path=path)
|
|
157
|
+
self.experiments[name].status = ProfilingExperimentStatus.DONE
|
|
158
|
+
|
|
159
|
+
def delete_experiment(self, name: str) -> None:
|
|
160
|
+
"""Deletes the specified experiment.
|
|
161
|
+
|
|
162
|
+
Note that this only removes the experiment from the manager's tracking; it does not delete any files on disk.
|
|
163
|
+
|
|
164
|
+
Args:
|
|
165
|
+
name (str): Name of the experiment to delete.
|
|
166
|
+
"""
|
|
167
|
+
if name in self.experiments:
|
|
168
|
+
del self.experiments[name]
|
|
169
|
+
else:
|
|
170
|
+
logger.warning(f"Experiment '{name}' not found; cannot delete.")
|
|
171
|
+
|
|
172
|
+
@abstractmethod
|
|
173
|
+
def _delete_experiment(self, name: str, dry_run: bool, **kwargs) -> None:
|
|
174
|
+
"""Deletes the on-disk artifacts of a single experiment.
|
|
175
|
+
|
|
176
|
+
This is the configuration-specific counterpart to delete_experiments, which handles selection, validation and
|
|
177
|
+
manager-state bookkeeping. Implementations should only remove files and, when dry_run is True, log what would
|
|
178
|
+
be removed without making any changes.
|
|
179
|
+
|
|
180
|
+
Args:
|
|
181
|
+
name (str): Name of the experiment to delete. Guaranteed to be managed by this instance.
|
|
182
|
+
dry_run (bool): If True, log what would be deleted without making any changes.
|
|
183
|
+
**kwargs: Configuration-specific options forwarded verbatim from delete_experiments.
|
|
184
|
+
"""
|
|
185
|
+
|
|
186
|
+
def delete_experiments(
|
|
187
|
+
self,
|
|
188
|
+
experiments: list[str] | None = None,
|
|
189
|
+
all_experiments: bool = False,
|
|
190
|
+
dry_run: bool = False,
|
|
191
|
+
**kwargs,
|
|
192
|
+
) -> None:
|
|
193
|
+
"""Deletes experiments and removes them from the manager.
|
|
194
|
+
|
|
195
|
+
The selection, validation and manager-state bookkeeping are handled here, while the actual on-disk deletion is
|
|
196
|
+
delegated to the configuration-specific _delete_experiment method.
|
|
197
|
+
|
|
198
|
+
Args:
|
|
199
|
+
experiments (list[str] | None): List of experiment names to delete.
|
|
200
|
+
all_experiments (bool): If True, deletes all experiments managed by this instance.
|
|
201
|
+
dry_run (bool): If True, logs what would be deleted without making any changes. Defaults to False.
|
|
202
|
+
**kwargs: Configuration-specific options forwarded to _delete_experiment.
|
|
203
|
+
|
|
204
|
+
Raises:
|
|
205
|
+
ValueError: If both experiments and all_experiments are specified, or neither is.
|
|
206
|
+
KeyError: If any experiment name is not managed by this instance.
|
|
207
|
+
"""
|
|
208
|
+
if all_experiments and experiments is not None:
|
|
209
|
+
raise ValueError("Pass either experiments=[...] or all_experiments=True, not both.")
|
|
210
|
+
if not all_experiments and not experiments:
|
|
211
|
+
raise ValueError("No experiments specified. Pass either experiments=[...] or all_experiments=True.")
|
|
212
|
+
existing = set(self.experiments.keys())
|
|
213
|
+
names_to_delete = existing if all_experiments else set(experiments)
|
|
214
|
+
unmanaged = names_to_delete - existing
|
|
215
|
+
if unmanaged:
|
|
216
|
+
raise KeyError(
|
|
217
|
+
f"Experiments {unmanaged} are not managed by this manager "
|
|
218
|
+
f"(existing: {existing}). Please check the names and try again."
|
|
219
|
+
)
|
|
220
|
+
|
|
221
|
+
for name in names_to_delete:
|
|
222
|
+
self._delete_experiment(name, dry_run=dry_run, **kwargs)
|
|
223
|
+
|
|
224
|
+
if dry_run:
|
|
225
|
+
return
|
|
226
|
+
|
|
227
|
+
for name in names_to_delete:
|
|
228
|
+
del self.experiments[name]
|
|
229
|
+
|
|
230
|
+
def parse_profiling_data(self):
|
|
231
|
+
"""Parses profiling data from the experiments.
|
|
232
|
+
|
|
233
|
+
Configurations that can be run several times produce one set of profiling logs per run. In that case the
|
|
234
|
+
parsed datasets are concatenated along a 'run' dimension, coordinated by the run number reported by the
|
|
235
|
+
configuration. Experiments with a single run are stored without a 'run' dimension, so that they can be used
|
|
236
|
+
directly. Use select_best_run() or aggregate_runs() to reduce the 'run' dimension before plotting. Note that
|
|
237
|
+
if all but one run fail to produce a log, the result has no 'run' dimension.
|
|
238
|
+
"""
|
|
239
|
+
self.data = {}
|
|
240
|
+
for exp_name, exp in self.experiments.items():
|
|
241
|
+
if exp.status == ProfilingExperimentStatus.DONE or exp.status == ProfilingExperimentStatus.ARCHIVED:
|
|
242
|
+
logger.info(f"Parsing profiling data for experiment '{exp_name}'.")
|
|
243
|
+
self.data[exp_name] = {}
|
|
244
|
+
with exp.directory() as (exp_path, run_path):
|
|
245
|
+
# Parse all logs
|
|
246
|
+
logs = self.profiling_logs(exp_path, run_path)
|
|
247
|
+
for log_name, run_logs in logs.items():
|
|
248
|
+
datasets = {}
|
|
249
|
+
for run, log in run_logs.items():
|
|
250
|
+
logger.info(f"Parsing {log_name} profiling log for run {run}: {log.filepath}. ")
|
|
251
|
+
if log.optional:
|
|
252
|
+
try:
|
|
253
|
+
datasets[run] = log.parse()
|
|
254
|
+
except FileNotFoundError:
|
|
255
|
+
logger.info(f"Optional profiling log '{log.filepath}' not found. Skipping.")
|
|
256
|
+
continue
|
|
257
|
+
except Exception as e:
|
|
258
|
+
# might be useful to make this a warning instead of info to help catch parse
|
|
259
|
+
# failures for logs that should've succeeded.
|
|
260
|
+
logger.info(
|
|
261
|
+
f"Failed to parse optional profiling log '{log.filepath}' with exception:\n"
|
|
262
|
+
f" {e}\nSkipping."
|
|
263
|
+
)
|
|
264
|
+
continue
|
|
265
|
+
else:
|
|
266
|
+
datasets[run] = log.parse()
|
|
267
|
+
logger.info(" Done.")
|
|
268
|
+
# A single run is stored as is; several runs are concatenated along a new 'run' dimension.
|
|
269
|
+
if len(datasets) == 1:
|
|
270
|
+
self.data[exp_name][log_name] = next(iter(datasets.values()))
|
|
271
|
+
elif datasets:
|
|
272
|
+
self.data[exp_name][log_name] = xr.concat(
|
|
273
|
+
[ds.expand_dims({"run": [run]}) for run, ds in sorted(datasets.items())],
|
|
274
|
+
dim="run",
|
|
275
|
+
join="outer",
|
|
276
|
+
)
|
|
277
|
+
else:
|
|
278
|
+
logger.warning(
|
|
279
|
+
f"Experiment '{exp_name}' is not completed (status: {exp.status.name}). Skipping parsing profiling "
|
|
280
|
+
"data."
|
|
281
|
+
)
|
|
282
|
+
|
|
283
|
+
def _ncpus(self, exp_name: str) -> int:
|
|
284
|
+
"""Returns the number of CPUs used by an experiment, parsing it at most once.
|
|
285
|
+
|
|
286
|
+
Args:
|
|
287
|
+
exp_name (str): Name of the experiment.
|
|
288
|
+
|
|
289
|
+
Returns:
|
|
290
|
+
int: Number of CPUs used by the experiment.
|
|
291
|
+
"""
|
|
292
|
+
if exp_name not in self._ncpus_cache:
|
|
293
|
+
with self.experiments[exp_name].directory() as (exp_path, run_path):
|
|
294
|
+
self._ncpus_cache[exp_name] = self.parse_ncpus(exp_path, run_path)
|
|
295
|
+
return self._ncpus_cache[exp_name]
|
|
296
|
+
|
|
297
|
+
def select_best_experiments(
|
|
298
|
+
self,
|
|
299
|
+
component: str,
|
|
300
|
+
region: str,
|
|
301
|
+
metric: ProfilingMetric,
|
|
302
|
+
experiments: list[str] | None = None,
|
|
303
|
+
) -> list[str]:
|
|
304
|
+
"""Selects the best performing experiment for each number of CPUs.
|
|
305
|
+
|
|
306
|
+
Scaling studies often contain several experiments that use the same number of CPUs, for instance different
|
|
307
|
+
domain decomposition layouts of the same total core count. Plotting all of them produces duplicated ncpus
|
|
308
|
+
coordinates and meaningless speedup and efficiency curves. This method keeps a single experiment per CPU
|
|
309
|
+
count: the one with the smallest value of the given metric, measured on the given region of the given
|
|
310
|
+
component. Smaller is always better.
|
|
311
|
+
|
|
312
|
+
The returned list is meant to be passed to the experiments argument of the plotting methods. If two
|
|
313
|
+
experiments with the same number of CPUs have exactly the same value, the first one is kept and a warning
|
|
314
|
+
is logged.
|
|
315
|
+
|
|
316
|
+
Args:
|
|
317
|
+
component (str): Name of the component holding the region used to rank experiments.
|
|
318
|
+
region (str): Name of the region used to rank experiments.
|
|
319
|
+
metric (ProfilingMetric): Metric used to rank experiments. The smallest value wins.
|
|
320
|
+
experiments (list[str] | None): Optional list of experiment names to select from. If None, all
|
|
321
|
+
experiments with parsed profiling data are considered.
|
|
322
|
+
|
|
323
|
+
Returns:
|
|
324
|
+
list[str]: Names of the selected experiments, one per distinct number of CPUs, ordered by increasing
|
|
325
|
+
number of CPUs.
|
|
326
|
+
|
|
327
|
+
Raises:
|
|
328
|
+
KeyError: If an experiment has no parsed profiling data, or if the component, region or metric is not
|
|
329
|
+
available in one of them.
|
|
330
|
+
ValueError: If the profiling data still has a 'run' dimension.
|
|
331
|
+
"""
|
|
332
|
+
exp_names = experiments if experiments is not None else list(self.data.keys())
|
|
333
|
+
|
|
334
|
+
best: dict[int, tuple[str, float]] = {}
|
|
335
|
+
for exp_name in exp_names:
|
|
336
|
+
ds = self.data[exp_name][component]
|
|
337
|
+
if "run" in ds.dims:
|
|
338
|
+
raise ValueError(_RUN_DIM_ERROR)
|
|
339
|
+
value = float(ds[metric].sel(region=region).pint.dequantify().values)
|
|
340
|
+
ncpus = self._ncpus(exp_name)
|
|
341
|
+
incumbent = best.get(ncpus)
|
|
342
|
+
if incumbent is None or value < incumbent[1]:
|
|
343
|
+
best[ncpus] = (exp_name, value)
|
|
344
|
+
elif value == incumbent[1]:
|
|
345
|
+
logger.warning(
|
|
346
|
+
f"Experiments '{incumbent[0]}' and '{exp_name}' have the same {metric} ({value} "
|
|
347
|
+
f"{metric.units}) for region '{region}' of component '{component}' at {ncpus} CPUs. "
|
|
348
|
+
f"Keeping '{incumbent[0]}'."
|
|
349
|
+
)
|
|
350
|
+
|
|
351
|
+
return [name for _, (name, _) in sorted(best.items())]
|
|
352
|
+
|
|
353
|
+
def select_best_run(self, component: str, region: str, metric: ProfilingMetric) -> None:
|
|
354
|
+
"""Keeps only the best performing run of each experiment, discarding the others.
|
|
355
|
+
|
|
356
|
+
Configurations that can be run several times produce profiling data with a 'run' dimension. This method
|
|
357
|
+
keeps a single run per experiment: the one with the smallest value of the given metric, measured on the
|
|
358
|
+
given region of the given component. Smaller is always better.
|
|
359
|
+
|
|
360
|
+
The chosen run is selected in *every* component of the experiment, so the result always describes a single
|
|
361
|
+
run that actually took place, and never mixes measurements taken during different runs. Use
|
|
362
|
+
aggregate_runs() instead to compute statistics over the runs.
|
|
363
|
+
|
|
364
|
+
Datasets without a 'run' dimension are left untouched, so this is safe to call on any manager. The reduction
|
|
365
|
+
is destructive: call parse_profiling_data() again to recover the individual runs.
|
|
366
|
+
|
|
367
|
+
Args:
|
|
368
|
+
component (str): Name of the component holding the region used to rank runs.
|
|
369
|
+
region (str): Name of the region used to rank runs.
|
|
370
|
+
metric (ProfilingMetric): Metric used to rank runs. The smallest value wins.
|
|
371
|
+
|
|
372
|
+
Raises:
|
|
373
|
+
KeyError: If the component, region or metric is not available in one of the experiments, or if the
|
|
374
|
+
chosen run is missing from one of the components of that experiment.
|
|
375
|
+
"""
|
|
376
|
+
for exp_name, components in self.data.items():
|
|
377
|
+
ranking = components[component]
|
|
378
|
+
if "run" not in ranking.dims:
|
|
379
|
+
continue
|
|
380
|
+
values = ranking[metric].sel(region=region).pint.dequantify()
|
|
381
|
+
run = int(values.run.values[int(values.argmin("run"))])
|
|
382
|
+
logger.info(f"Keeping run {run} of experiment '{exp_name}'.")
|
|
383
|
+
self.data[exp_name] = {
|
|
384
|
+
name: ds.sel(run=run, drop=True) if "run" in ds.dims else ds for name, ds in components.items()
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
def aggregate_runs(self, how: str = "min") -> None:
|
|
388
|
+
"""Reduces the 'run' dimension of every experiment with the given statistic.
|
|
389
|
+
|
|
390
|
+
Configurations that can be run several times produce profiling data with a 'run' dimension. This method
|
|
391
|
+
collapses it, computing the requested statistic over the runs.
|
|
392
|
+
|
|
393
|
+
Note that each region and each metric is reduced independently, so unlike select_best_run() the result does
|
|
394
|
+
not correspond to any run that actually took place: with how="min", different regions can come from
|
|
395
|
+
different runs. Use "min" to estimate the best achievable timings and "mean" or "median" to describe the
|
|
396
|
+
typical ones. Integer metrics, such as call counts, become floats.
|
|
397
|
+
|
|
398
|
+
Datasets without a 'run' dimension are left untouched, so this is safe to call on any manager. The reduction
|
|
399
|
+
is destructive: call parse_profiling_data() again to recover the individual runs.
|
|
400
|
+
|
|
401
|
+
Args:
|
|
402
|
+
how (str): Statistic to compute over the runs. One of "min", "mean" or "median". Defaults to "min".
|
|
403
|
+
|
|
404
|
+
Raises:
|
|
405
|
+
ValueError: If how is not one of the supported statistics.
|
|
406
|
+
"""
|
|
407
|
+
if how not in ("min", "mean", "median"):
|
|
408
|
+
raise ValueError(f"Unknown reduction '{how}'. Use 'min', 'mean' or 'median'.")
|
|
409
|
+
for exp_name, components in self.data.items():
|
|
410
|
+
self.data[exp_name] = {
|
|
411
|
+
name: getattr(ds, how)("run") if "run" in ds.dims else ds for name, ds in components.items()
|
|
412
|
+
}
|
|
413
|
+
|
|
414
|
+
def plot_scaling_data(
|
|
415
|
+
self,
|
|
416
|
+
components: list[str],
|
|
417
|
+
regions: list[list[str]],
|
|
418
|
+
metric: ProfilingMetric,
|
|
419
|
+
region_relabel_map: dict | None = None,
|
|
420
|
+
experiments: list[str] | None = None,
|
|
421
|
+
) -> Figure:
|
|
422
|
+
"""Plots scaling data for the specified components, regions and metric.
|
|
423
|
+
|
|
424
|
+
Args:
|
|
425
|
+
components (list[str]): List of component names to plot.
|
|
426
|
+
regions (list[list[str]]): List of regions to plot for each component.
|
|
427
|
+
metric (ProfilingMetric): Metric to use for the scaling plots.
|
|
428
|
+
region_relabel_map (dict | None): Optional mapping to relabel regions in the plots.
|
|
429
|
+
experiments (list[str] | None): Optional list of experiment names to include. If None, all experiments
|
|
430
|
+
with parsed profiling data are included.
|
|
431
|
+
|
|
432
|
+
Returns:
|
|
433
|
+
Figure: The Matplotlib figure containing the scaling plots.
|
|
434
|
+
|
|
435
|
+
Raises:
|
|
436
|
+
ValueError: If no experiments are selected, if a selected experiment has no parsed profiling data, if no
|
|
437
|
+
profiling data is found for a specified component, if a requested region is missing, if the
|
|
438
|
+
profiling data still has a 'run' dimension, or if several of the selected experiments use the same
|
|
439
|
+
number of CPUs.
|
|
440
|
+
"""
|
|
441
|
+
|
|
442
|
+
exp_names = experiments if experiments is not None else list(self.data.keys())
|
|
443
|
+
if not exp_names:
|
|
444
|
+
raise ValueError("No experiments selected for scaling plot.")
|
|
445
|
+
|
|
446
|
+
missing_experiments = [exp_name for exp_name in exp_names if exp_name not in self.data]
|
|
447
|
+
if missing_experiments:
|
|
448
|
+
raise ValueError(
|
|
449
|
+
f"No parsed profiling data found for experiment(s): {missing_experiments}. "
|
|
450
|
+
f"Available experiments: {list(self.data.keys())}."
|
|
451
|
+
)
|
|
452
|
+
|
|
453
|
+
if any("run" in ds.dims for exp_name in exp_names for ds in self.data[exp_name].values()):
|
|
454
|
+
raise ValueError(_RUN_DIM_ERROR)
|
|
455
|
+
|
|
456
|
+
# Find number of cpus used for each experiment
|
|
457
|
+
ncpus = {exp_name: self._ncpus(exp_name) for exp_name in exp_names}
|
|
458
|
+
|
|
459
|
+
# Speedup and efficiency are ill-defined if several experiments share the same number of cpus
|
|
460
|
+
cpu_counts = list(ncpus.values())
|
|
461
|
+
duplicated_ncpus = sorted({n for n in cpu_counts if cpu_counts.count(n) > 1})
|
|
462
|
+
if duplicated_ncpus:
|
|
463
|
+
raise ValueError(
|
|
464
|
+
f"Several selected experiments use the same number of CPUs {duplicated_ncpus}, which makes speedup "
|
|
465
|
+
"and efficiency ill-defined. Use select_best_experiments() to keep only the best performing "
|
|
466
|
+
"experiment for each number of CPUs, or restrict the selection with experiments=[...]."
|
|
467
|
+
)
|
|
468
|
+
|
|
469
|
+
# Gather scaling data for each component
|
|
470
|
+
scaling_data = []
|
|
471
|
+
for component, component_regions in zip(components, regions, strict=True):
|
|
472
|
+
component_data = []
|
|
473
|
+
for exp_name in exp_names:
|
|
474
|
+
ds = self.data[exp_name].get(component)
|
|
475
|
+
if ds is None:
|
|
476
|
+
raise ValueError(f"No profiling data found for component '{component}' in experiment '{exp_name}'.")
|
|
477
|
+
|
|
478
|
+
available_regions = ds.coords["region"].values.tolist()
|
|
479
|
+
missing_regions = [region for region in component_regions if region not in available_regions]
|
|
480
|
+
if missing_regions:
|
|
481
|
+
raise ValueError(
|
|
482
|
+
f"Requested region(s) {missing_regions} not found for component '{component}' "
|
|
483
|
+
f"in experiment '{exp_name}'. Available regions: {available_regions}."
|
|
484
|
+
)
|
|
485
|
+
|
|
486
|
+
# Select only the desired regions
|
|
487
|
+
ds = ds.sel(region=component_regions)
|
|
488
|
+
|
|
489
|
+
# Relabel regions if a relabel map is provided
|
|
490
|
+
if region_relabel_map is not None:
|
|
491
|
+
ds = ds.assign_coords(region=[region_relabel_map.get(n, n) for n in ds.region.values])
|
|
492
|
+
|
|
493
|
+
# Add ncpus dimension
|
|
494
|
+
component_data.append(ds.expand_dims({"ncpus": [ncpus[exp_name]]}))
|
|
495
|
+
|
|
496
|
+
# Concatenate data along ncpus dimension
|
|
497
|
+
scaling_data.append(xr.concat(component_data, dim="ncpus", join="outer").sortby("ncpus"))
|
|
498
|
+
|
|
499
|
+
return plot_scaling_metrics(scaling_data, metric)
|
|
500
|
+
|
|
501
|
+
def plot_bar_chart(
|
|
502
|
+
self,
|
|
503
|
+
components: list[str],
|
|
504
|
+
regions: list[list[str]],
|
|
505
|
+
metric: ProfilingMetric,
|
|
506
|
+
region_relabel_map: dict | None = None,
|
|
507
|
+
experiment_relabel_map: dict | None = None,
|
|
508
|
+
experiments: list[str] | None = None,
|
|
509
|
+
show: bool = True,
|
|
510
|
+
) -> Figure:
|
|
511
|
+
"""Plots a bar chart of a profiling metric over regions, grouped by experiment.
|
|
512
|
+
|
|
513
|
+
Regions are placed along the x-axis. Within each region group, there is one bar per
|
|
514
|
+
experiment, coloured by experiment name.
|
|
515
|
+
|
|
516
|
+
Args:
|
|
517
|
+
components (list[str]): List of component names to include.
|
|
518
|
+
regions (list[list[str]]): List of regions to include for each component.
|
|
519
|
+
metric (ProfilingMetric): Metric to plot.
|
|
520
|
+
region_relabel_map (dict | None): Optional mapping to relabel regions in the plot.
|
|
521
|
+
experiment_relabel_map (dict | None): Optional mapping to relabel experiments in the plot.
|
|
522
|
+
experiments (list[str] | None): Optional list of experiment names to include. If None, all experiments
|
|
523
|
+
are included.
|
|
524
|
+
show (bool): Whether to show the generated plot. Default: True.
|
|
525
|
+
|
|
526
|
+
Returns:
|
|
527
|
+
Figure: The Matplotlib figure containing the bar chart.
|
|
528
|
+
|
|
529
|
+
Raises:
|
|
530
|
+
ValueError: If no profiling data is found for a specified component in any experiment, or if the
|
|
531
|
+
profiling data still has a 'run' dimension.
|
|
532
|
+
"""
|
|
533
|
+
exp_names = experiments if experiments is not None else list(self.data.keys())
|
|
534
|
+
relabel = region_relabel_map or {}
|
|
535
|
+
|
|
536
|
+
# Build a lookup from display label to (component, original_region) and preserve input order.
|
|
537
|
+
region_info: list[tuple[str, str, str]] = [] # (component, original_region, display_label)
|
|
538
|
+
for component, component_regions in zip(components, regions, strict=True):
|
|
539
|
+
for region in component_regions:
|
|
540
|
+
region_info.append((component, region, relabel.get(region, region)))
|
|
541
|
+
region_labels = [label for _, _, label in region_info]
|
|
542
|
+
|
|
543
|
+
# Extract metric values per experiment, reading directly from the datasets
|
|
544
|
+
bar_data: dict[str, list[float]] = {}
|
|
545
|
+
for exp_name in exp_names:
|
|
546
|
+
values = []
|
|
547
|
+
for component, region, _ in region_info:
|
|
548
|
+
ds = self.data[exp_name].get(component)
|
|
549
|
+
if ds is None:
|
|
550
|
+
raise ValueError(f"No profiling data found for component '{component}' in experiment '{exp_name}'.")
|
|
551
|
+
if "run" in ds.dims:
|
|
552
|
+
raise ValueError(_RUN_DIM_ERROR)
|
|
553
|
+
values.append(float(ds[metric].sel(region=region).pint.dequantify().values))
|
|
554
|
+
bar_data[exp_name] = values
|
|
555
|
+
|
|
556
|
+
exp_relabel = experiment_relabel_map or {}
|
|
557
|
+
relabelled_bar_data = {exp_relabel.get(k, k): v for k, v in bar_data.items()}
|
|
558
|
+
|
|
559
|
+
return plot_bar_metrics(relabelled_bar_data, region_labels, metric, show=show)
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
# Copyright 2025 ACCESS-NRI and contributors. See the top-level COPYRIGHT file for details.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
|
|
4
|
+
"""Classes and utilities to define profiling metrics.
|
|
5
|
+
|
|
6
|
+
Metrics are classified by which dimension(s) they aggregate over.
|
|
7
|
+
|
|
8
|
+
Aggregation dimensions
|
|
9
|
+
----------------------
|
|
10
|
+
call
|
|
11
|
+
A single invocation of a profiling region. Statistics over this dimension
|
|
12
|
+
describe how timing varies across repeated calls to the same region within one run.
|
|
13
|
+
pe
|
|
14
|
+
A processing element (MPI process). Statistics over this dimension describe how
|
|
15
|
+
timing varies across parallel processes for a given call (or a call-aggregated value).
|
|
16
|
+
call_pe
|
|
17
|
+
Both dimensions simultaneously — global statistics across all calls and PEs.
|
|
18
|
+
|
|
19
|
+
Metric naming convention
|
|
20
|
+
------------------------
|
|
21
|
+
New metric names encode aggregation dimension(s) as underscore-separated suffixes:
|
|
22
|
+
|
|
23
|
+
<base>_<stat>_<dimension>
|
|
24
|
+
|
|
25
|
+
Examples:
|
|
26
|
+
tmin minimum time over calls (implicit _call suffix; legacy name)
|
|
27
|
+
tavg_max_pe maximum across PEs of the per-call average time
|
|
28
|
+
t_min_call_pe global minimum across both calls and PEs
|
|
29
|
+
|
|
30
|
+
The pre-defined constants below (tmin, tmax, tavg, …) carry an implicit _call
|
|
31
|
+
dimension for backward compatibility. New metrics should always make the dimension
|
|
32
|
+
suffix explicit.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
from pint import Unit
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class ProfilingMetric:
|
|
39
|
+
def __init__(self, name: str, units: Unit, description: str):
|
|
40
|
+
"""Class representing a profiling metric.
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
name (str): Name of the metric.
|
|
44
|
+
units (pint.Unit): Units of the metric.
|
|
45
|
+
description (str): Description of the metric.
|
|
46
|
+
Raises:
|
|
47
|
+
ValueError: If name, units or description are empty or whitespace-only strings.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
if not name.strip():
|
|
51
|
+
raise ValueError("Metric name cannot be empty!")
|
|
52
|
+
|
|
53
|
+
if not description.strip():
|
|
54
|
+
raise ValueError("Metric description cannot be empty!")
|
|
55
|
+
|
|
56
|
+
self._name = name
|
|
57
|
+
self._units = units
|
|
58
|
+
self._description = description
|
|
59
|
+
|
|
60
|
+
@property
|
|
61
|
+
def name(self) -> str:
|
|
62
|
+
return self._name
|
|
63
|
+
|
|
64
|
+
@property
|
|
65
|
+
def units(self) -> Unit:
|
|
66
|
+
return self._units
|
|
67
|
+
|
|
68
|
+
@property
|
|
69
|
+
def description(self) -> str:
|
|
70
|
+
return self._description
|
|
71
|
+
|
|
72
|
+
def __str__(self) -> str:
|
|
73
|
+
return self._name
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
# Per-call statistics (reduced over repeated invocations of the same region)
|
|
77
|
+
count = ProfilingMetric("count", Unit("dimensionless"), "Number of calls to region")
|
|
78
|
+
tmin = ProfilingMetric("minimum time", Unit("second"), "Minimum time over calls to region")
|
|
79
|
+
tmax = ProfilingMetric("maximum time", Unit("second"), "Maximum time over calls to region")
|
|
80
|
+
pemin = ProfilingMetric("minimum PE", Unit("dimensionless"), "Processing element where minimum call time was recorded")
|
|
81
|
+
pemax = ProfilingMetric("maximum PE", Unit("dimensionless"), "Processing element where maximum call time was recorded")
|
|
82
|
+
tavg = ProfilingMetric("average time", Unit("second"), "Mean time over calls to region")
|
|
83
|
+
tmed = ProfilingMetric("median time", Unit("second"), "Median time over calls to region")
|
|
84
|
+
tstd = ProfilingMetric("time std", Unit("second"), "Standard deviation of time over calls to region")
|
|
85
|
+
tfrac = ProfilingMetric("time fraction", Unit("%"), "Fraction of total time over calls to region")
|