access-profiling 0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- access/profiling/__init__.py +32 -0
- access/profiling/access_models.py +84 -0
- access/profiling/cice5_parser.py +80 -0
- access/profiling/cylc_manager.py +209 -0
- access/profiling/cylc_parser.py +146 -0
- access/profiling/esmf_parser.py +183 -0
- access/profiling/experiment.py +263 -0
- access/profiling/fms_parser.py +80 -0
- access/profiling/manager.py +559 -0
- access/profiling/metrics.py +85 -0
- access/profiling/parser.py +263 -0
- access/profiling/payu_manager.py +338 -0
- access/profiling/payujson_parser.py +75 -0
- access/profiling/plotting_utils.py +116 -0
- access/profiling/scaling.py +128 -0
- access/profiling/um_parser.py +242 -0
- access_profiling-0.1.dist-info/METADATA +100 -0
- access_profiling-0.1.dist-info/RECORD +21 -0
- access_profiling-0.1.dist-info/WHEEL +5 -0
- access_profiling-0.1.dist-info/licenses/LICENSE +201 -0
- access_profiling-0.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
# Copyright 2025 ACCESS-NRI and contributors. See the top-level COPYRIGHT file for details.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
|
|
4
|
+
"""Parser for ESMF profiling text summary data, such as output by nuopc.
|
|
5
|
+
The data to be parsed is written in the following form:
|
|
6
|
+
|
|
7
|
+
Region PETs PEs Count Mean (s) Min (s) Min PET Max (s) Max PET
|
|
8
|
+
[ESMF] 1664 1664 1 2558.5684 2555.1450 279 2559.5801 817
|
|
9
|
+
[ensemble] RunPhase1 1664 1664 1 1879.7292 1872.5078 376 1905.4939 1
|
|
10
|
+
[ESM0001] RunPhase1 1664 1664 1 1879.7286 1872.5059 858 1905.4937 1
|
|
11
|
+
[OCN] RunPhase1 1300 1300 960 1850.4170 1848.3905 1023 1858.6404 364
|
|
12
|
+
[OCN-TO-MED] RunPhase1 1664 1664 960 365.9532 0.1688 405 1673.2255 18
|
|
13
|
+
[ICE] RunPhase1 364 364 960 155.8202 154.7637 94 160.2443 0
|
|
14
|
+
cice_run_total 364 364 960 155.4648 154.3687 94 159.8980 0
|
|
15
|
+
cice_run_import 364 364 960 3.7565 3.5426 218 8.3892 1
|
|
16
|
+
cice_imp_halo 364 364 1920 2.1386 1.5405 111 6.9782 0
|
|
17
|
+
cice_imp_t2u 364 364 960 0.9242 0.5578 355 1.5026 111
|
|
18
|
+
cice_imp_atm 364 364 960 0.0579 0.0399 331 0.0834 18
|
|
19
|
+
cice_imp_ocn 364 364 960 0.0499 0.0419 50 0.0632 12
|
|
20
|
+
cice_run_export 364 364 960 1.1015 0.8846 361 1.3607 194
|
|
21
|
+
[MED-TO-OCN] RunPhase1 1664 1664 960 16.7498 0.4588 256 23.2875 1023
|
|
22
|
+
[MED] med_phases_restart_write 364 364 960 31.8234 31.8123 203 33.0513 363
|
|
23
|
+
MED:(med_phases_restart_write) 364 364 960 31.7651 31.7559 203 32.9919 363
|
|
24
|
+
...
|
|
25
|
+
|
|
26
|
+
Where indentation depth indicates depth in the call-stack. Usually there is a header like
|
|
27
|
+
********
|
|
28
|
+
A warning about identifying load-imbalance.
|
|
29
|
+
********
|
|
30
|
+
Note that the profiling summary stats may contain identical region names
|
|
31
|
+
e.g. [ATM-TO-MED] RunPhase1 is found twice in OM3.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
import os
|
|
35
|
+
from pathlib import Path
|
|
36
|
+
|
|
37
|
+
from pint import Unit
|
|
38
|
+
|
|
39
|
+
from access.profiling.metrics import (
|
|
40
|
+
ProfilingMetric,
|
|
41
|
+
count,
|
|
42
|
+
pemax,
|
|
43
|
+
pemin,
|
|
44
|
+
tavg,
|
|
45
|
+
tmax,
|
|
46
|
+
tmin,
|
|
47
|
+
)
|
|
48
|
+
from access.profiling.parser import ProfilingParser, _read_text_file
|
|
49
|
+
|
|
50
|
+
pets = ProfilingMetric("PETs", Unit("dimensionless"), "ESMF Virtual Machine Persistent Execution Threads")
|
|
51
|
+
pes = ProfilingMetric("PEs", Unit("dimensionless"), "Processing Elements")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class ESMFSummaryProfilingParser(ProfilingParser):
|
|
55
|
+
"""ESMF text summary profiling output parser."""
|
|
56
|
+
|
|
57
|
+
hierarchical: bool # whether the call-stack hierarchy is parsed.
|
|
58
|
+
|
|
59
|
+
def __init__(self, hierarchical: bool = False):
|
|
60
|
+
"""Instantiate ESMF profiling parser.
|
|
61
|
+
|
|
62
|
+
Args:
|
|
63
|
+
hierarchical (bool): Whether call-stack hierarchy is parsed.
|
|
64
|
+
"""
|
|
65
|
+
super().__init__()
|
|
66
|
+
|
|
67
|
+
self.hierarchical = hierarchical
|
|
68
|
+
|
|
69
|
+
# ESMF provides the following metrics
|
|
70
|
+
self._metrics = [pets, pes, count, tavg, tmin, pemin, tmax, pemax]
|
|
71
|
+
|
|
72
|
+
def parse(self, file_path: str | Path | os.PathLike) -> dict:
|
|
73
|
+
stream = _read_text_file(file_path)
|
|
74
|
+
|
|
75
|
+
lines = stream.strip().split("\n")
|
|
76
|
+
|
|
77
|
+
if self.hierarchical:
|
|
78
|
+
result = {}
|
|
79
|
+
stack = [(result, -1)] # (current_dict, indent_level)
|
|
80
|
+
else:
|
|
81
|
+
result = {m: [] for m in self._metrics}
|
|
82
|
+
result["region"] = []
|
|
83
|
+
|
|
84
|
+
for line in lines:
|
|
85
|
+
# Split the line into region name and statistics
|
|
86
|
+
parts = line.split()
|
|
87
|
+
if len(parts) < (1 + len(self._metrics)): # Need at least 1 region + the number of stat columns
|
|
88
|
+
continue
|
|
89
|
+
|
|
90
|
+
# Extract region name and statistics
|
|
91
|
+
region = " ".join(parts[:-8])
|
|
92
|
+
stats = parts[-8:]
|
|
93
|
+
|
|
94
|
+
# Validate that all statistics can be parsed correctly
|
|
95
|
+
try:
|
|
96
|
+
stats_dict = {
|
|
97
|
+
pets: int(stats[0]),
|
|
98
|
+
pes: int(stats[1]),
|
|
99
|
+
count: int(stats[2]),
|
|
100
|
+
tavg: float(stats[3]),
|
|
101
|
+
tmin: float(stats[4]),
|
|
102
|
+
pemin: int(stats[5]),
|
|
103
|
+
tmax: float(stats[6]),
|
|
104
|
+
pemax: int(stats[7]),
|
|
105
|
+
}
|
|
106
|
+
except (ValueError, IndexError):
|
|
107
|
+
# Skip lines that don't match the expected format
|
|
108
|
+
continue
|
|
109
|
+
|
|
110
|
+
if self.hierarchical:
|
|
111
|
+
# Calculate indentation level (each level is 2 spaces)
|
|
112
|
+
indent = len(line) - len(line.lstrip())
|
|
113
|
+
indent_level = indent // 2
|
|
114
|
+
|
|
115
|
+
# Pop stack until we find the parent level
|
|
116
|
+
while stack and stack[-1][1] >= indent_level:
|
|
117
|
+
stack.pop()
|
|
118
|
+
|
|
119
|
+
# Get parent dictionary
|
|
120
|
+
parent_dict = stack[-1][0]
|
|
121
|
+
|
|
122
|
+
# Create entry for this region
|
|
123
|
+
if region not in parent_dict:
|
|
124
|
+
parent_dict[region] = {}
|
|
125
|
+
|
|
126
|
+
# Add statistics to this region
|
|
127
|
+
parent_dict[region].update(stats_dict)
|
|
128
|
+
|
|
129
|
+
# Push this level onto stack for potential children
|
|
130
|
+
stack.append((parent_dict[region], indent_level))
|
|
131
|
+
else:
|
|
132
|
+
_update_flat_result(result, stats_dict, region)
|
|
133
|
+
|
|
134
|
+
# fewer if statements to pass ruff checks
|
|
135
|
+
if (self.hierarchical and not result) or (not self.hierarchical and len(result["region"]) == 0):
|
|
136
|
+
raise ValueError("No ESMF summary profiling data found")
|
|
137
|
+
|
|
138
|
+
return result
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _update_flat_result(result: dict, stats_dict: dict, region: str):
|
|
142
|
+
"""Helper function to update flat result.
|
|
143
|
+
|
|
144
|
+
Besides appending results, this function also checks whether the region already exists
|
|
145
|
+
and aggregates the metric values in result if possible.
|
|
146
|
+
|
|
147
|
+
Args:
|
|
148
|
+
result (dict): The flat result dictionary to update.
|
|
149
|
+
stats_dict (dict): The stats to update the result with.
|
|
150
|
+
region (str): The region to append the results to.
|
|
151
|
+
|
|
152
|
+
Raises:
|
|
153
|
+
NotImplementedError: If a stats_dict["region"] is already in result["region"],
|
|
154
|
+
but the PETs or PEs value aren't the same.
|
|
155
|
+
"""
|
|
156
|
+
# Flat structure: just use region name as key
|
|
157
|
+
try:
|
|
158
|
+
idx = result["region"].index(region)
|
|
159
|
+
# only update existing region if PETs and PEs are same
|
|
160
|
+
if (
|
|
161
|
+
result[pets][idx] == stats_dict[pets]
|
|
162
|
+
and result[pes][idx] == stats_dict[pes]
|
|
163
|
+
and stats_dict[pets] == stats_dict[pes]
|
|
164
|
+
):
|
|
165
|
+
# new avg is weighted average using count as the weight
|
|
166
|
+
result[tavg][idx] = (result[tavg][idx] * result[count][idx] + stats_dict[tavg] * stats_dict[count]) / (
|
|
167
|
+
result[count][idx] + stats_dict[count]
|
|
168
|
+
)
|
|
169
|
+
result[count][idx] += stats_dict[count]
|
|
170
|
+
if stats_dict[tmin] < result[tmin][idx]:
|
|
171
|
+
result[tmin][idx] = stats_dict[tmin]
|
|
172
|
+
result[pemin][idx] = stats_dict[pemin]
|
|
173
|
+
if stats_dict[tmax] > result[tmax][idx]:
|
|
174
|
+
result[tmax][idx] = stats_dict[tmax]
|
|
175
|
+
result[pemax][idx] = stats_dict[pemax]
|
|
176
|
+
else:
|
|
177
|
+
raise NotImplementedError(
|
|
178
|
+
"I don't know what to do with multiple regions with same name, but different PETs/PEs."
|
|
179
|
+
)
|
|
180
|
+
except ValueError:
|
|
181
|
+
result["region"].append(region)
|
|
182
|
+
for k, v in stats_dict.items():
|
|
183
|
+
result[k].append(v)
|
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
# Copyright 2025 ACCESS-NRI and contributors. See the top-level COPYRIGHT file for details.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
|
|
4
|
+
import logging
|
|
5
|
+
import tarfile
|
|
6
|
+
import tempfile
|
|
7
|
+
from contextlib import contextmanager
|
|
8
|
+
from enum import Enum
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
import xarray as xr
|
|
12
|
+
|
|
13
|
+
from access.profiling.parser import ProfilingParser, flatten_hierarchical
|
|
14
|
+
|
|
15
|
+
logger = logging.getLogger(__name__)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _make_unique_region_names(regions: list[object]) -> list[object]:
|
|
19
|
+
"""Return region names with deterministic suffixes for duplicates."""
|
|
20
|
+
|
|
21
|
+
counts: dict[object, int] = {}
|
|
22
|
+
unique_regions: list[object] = []
|
|
23
|
+
for region in regions:
|
|
24
|
+
count = counts.get(region, 0) + 1
|
|
25
|
+
counts[region] = count
|
|
26
|
+
unique_regions.append(region if count == 1 else f"{region}_{count}")
|
|
27
|
+
return unique_regions
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class ProfilingLog:
|
|
31
|
+
"""Represents a profiling log file.
|
|
32
|
+
|
|
33
|
+
Args:
|
|
34
|
+
filepath (Path): Path to the log file.
|
|
35
|
+
parser (ProfilingParser): Parser to use for this log file.
|
|
36
|
+
optional (bool): Whether this log might be missing or does not contain parsable data. If True, no error should
|
|
37
|
+
be raised if the log is missing or unparsable. Defaults to False.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
filepath: Path # Path to the log file
|
|
41
|
+
parser: ProfilingParser # Parser to use for this log file
|
|
42
|
+
_optional: bool = False # Whether this log might not be present
|
|
43
|
+
|
|
44
|
+
def __init__(self, filepath: Path, parser: ProfilingParser, optional: bool = False) -> None:
|
|
45
|
+
self.filepath = filepath
|
|
46
|
+
self.parser = parser
|
|
47
|
+
self._optional = optional
|
|
48
|
+
|
|
49
|
+
@property
|
|
50
|
+
def optional(self) -> bool:
|
|
51
|
+
"""bool: Whether this log might not be present."""
|
|
52
|
+
return self._optional
|
|
53
|
+
|
|
54
|
+
def parse(self) -> xr.Dataset:
|
|
55
|
+
"""Parses the log file and returns the profiling data as an xarray Dataset.
|
|
56
|
+
|
|
57
|
+
Accepts all three parser output formats (see parser.py module docstring):
|
|
58
|
+
|
|
59
|
+
- **Flat**: standard 1D Dataset over the ``region`` dimension.
|
|
60
|
+
- **Hierarchical nested dict**: automatically flattened via
|
|
61
|
+
:func:`flatten_hierarchical` before building the Dataset.
|
|
62
|
+
- **Per-PE**: produces a 2D Dataset with both ``region`` and ``pe`` dimensions.
|
|
63
|
+
Use :func:`aggregate_pe_data` on the result to compute summary statistics.
|
|
64
|
+
|
|
65
|
+
Returns:
|
|
66
|
+
xr.Dataset: Parsed profiling data.
|
|
67
|
+
"""
|
|
68
|
+
data = self.parser.parse(self.filepath)
|
|
69
|
+
|
|
70
|
+
# Flatten hierarchical (nested dict) format if needed
|
|
71
|
+
if "region" not in data:
|
|
72
|
+
data = flatten_hierarchical(data, self.parser.metrics)
|
|
73
|
+
|
|
74
|
+
has_pe = "pe" in data
|
|
75
|
+
dims = ["region", "pe"] if has_pe else ["region"]
|
|
76
|
+
coords: dict = {"region": _make_unique_region_names(list(data["region"]))}
|
|
77
|
+
if has_pe:
|
|
78
|
+
coords["pe"] = data["pe"]
|
|
79
|
+
|
|
80
|
+
return xr.Dataset(
|
|
81
|
+
data_vars=dict(
|
|
82
|
+
zip(
|
|
83
|
+
self.parser.metrics,
|
|
84
|
+
[xr.DataArray(data[m], dims=dims).pint.quantify(m.units) for m in self.parser.metrics],
|
|
85
|
+
strict=True,
|
|
86
|
+
)
|
|
87
|
+
),
|
|
88
|
+
coords=coords,
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class ProfilingExperimentStatus(Enum):
|
|
93
|
+
"""Enumeration representing the status of a profiling experiment."""
|
|
94
|
+
|
|
95
|
+
NEW = 1 # Experiment has been created but not started
|
|
96
|
+
RUNNING = 2 # Experiment is running or is queued
|
|
97
|
+
DONE = 3 # Experiment has finished
|
|
98
|
+
ARCHIVED = 4 # Experiment has been archived
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def experiment_directory_walker(path: Path, arcname: Path, root: Path, follow_symlinks: bool = False):
|
|
102
|
+
"""Walks through the experiment directory, yielding files and corresponding names in the archive.
|
|
103
|
+
|
|
104
|
+
Symlinks are treated in a special manner.
|
|
105
|
+
- if the target is inside the experiment directory, the symlink itself is always returned
|
|
106
|
+
- if follow_symlinks is True and the target is a directory, then the all contents in the target directory are
|
|
107
|
+
recursively iterated
|
|
108
|
+
- if follow_symlinks is True and the target is a file, then the target file name is returned, not the symlink
|
|
109
|
+
- if follow_symlinks is False, then the symlink itself is returned for both files and directories
|
|
110
|
+
|
|
111
|
+
Args:
|
|
112
|
+
path (Path): Path to walk through.
|
|
113
|
+
arcname (Path): Archive name for the current path.
|
|
114
|
+
follow_symlinks (bool): Whether to follow symlinks. Defaults to False.
|
|
115
|
+
|
|
116
|
+
Yields:
|
|
117
|
+
Tuple[Path, Path]: A tuple containing the file path and its archive name.
|
|
118
|
+
"""
|
|
119
|
+
if path.is_symlink():
|
|
120
|
+
if not follow_symlinks:
|
|
121
|
+
# Add symlink itself without following
|
|
122
|
+
yield path, arcname
|
|
123
|
+
else:
|
|
124
|
+
target = path.resolve()
|
|
125
|
+
if target.is_dir():
|
|
126
|
+
# Recursively add target contents
|
|
127
|
+
for child in target.iterdir():
|
|
128
|
+
yield from experiment_directory_walker(
|
|
129
|
+
child, Path(arcname) / child.name, root, follow_symlinks=follow_symlinks
|
|
130
|
+
)
|
|
131
|
+
elif target.absolute().is_relative_to(root.absolute()):
|
|
132
|
+
# Target is within the experiment directory, so add symlink as is
|
|
133
|
+
yield path, arcname
|
|
134
|
+
else:
|
|
135
|
+
# Target is outside the experiment directory, add the target file instead
|
|
136
|
+
yield target, arcname
|
|
137
|
+
|
|
138
|
+
elif path.is_dir():
|
|
139
|
+
# Recursively add directory contents
|
|
140
|
+
for child in path.iterdir():
|
|
141
|
+
yield from experiment_directory_walker(
|
|
142
|
+
child, Path(arcname) / child.name, root, follow_symlinks=follow_symlinks
|
|
143
|
+
)
|
|
144
|
+
else:
|
|
145
|
+
yield path, arcname
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
class ProfilingExperiment:
|
|
149
|
+
"""Represents a profiling experiment.
|
|
150
|
+
|
|
151
|
+
Args:
|
|
152
|
+
path (Path): Path to the experiment directory.
|
|
153
|
+
run_path (Path | None): Path to a separate runs directory. If None, runs are assumed to be
|
|
154
|
+
inside path. When provided, the runs directory is also traversed during archival.
|
|
155
|
+
path contents are stored under experiment/ and run_path contents under runs/.
|
|
156
|
+
"""
|
|
157
|
+
|
|
158
|
+
path: Path # Path to the experiment directory
|
|
159
|
+
run_path: Path | None # Path to a separate runs directory, or None
|
|
160
|
+
status: ProfilingExperimentStatus = ProfilingExperimentStatus.NEW # Status of the experiment
|
|
161
|
+
|
|
162
|
+
def __init__(self, path: Path, run_path: Path | None = None) -> None:
|
|
163
|
+
self.path = path
|
|
164
|
+
self.run_path = run_path
|
|
165
|
+
if self.path.suffixes == [".tar", ".gz"]:
|
|
166
|
+
self.status = ProfilingExperimentStatus.ARCHIVED
|
|
167
|
+
|
|
168
|
+
def __repr__(self) -> str:
|
|
169
|
+
"""Returns a string representation of the ProfilingExperiment."""
|
|
170
|
+
if self.run_path is not None:
|
|
171
|
+
return f"{type(self).__name__}(path={self.path!r}, run_path={self.run_path!r}, status={self.status.name})"
|
|
172
|
+
return f"{type(self).__name__}(path={self.path!r}, status={self.status.name})"
|
|
173
|
+
|
|
174
|
+
@contextmanager
|
|
175
|
+
def directory(self):
|
|
176
|
+
"""Context manager returning the experiment and runs directories.
|
|
177
|
+
|
|
178
|
+
If the experiment has been archived, it will be extracted to a temporary directory. Otherwise, the original
|
|
179
|
+
directory paths will be used. Note that after exiting the context, the temporary directory is removed.
|
|
180
|
+
|
|
181
|
+
Returns:
|
|
182
|
+
tuple[Path, Path | None]: The experiment directory path and optional runs directory path.
|
|
183
|
+
"""
|
|
184
|
+
if self.path.suffixes == [".tar", ".gz"]:
|
|
185
|
+
with tempfile.TemporaryDirectory(prefix="access-profiling_", suffix="_data") as tmpdir:
|
|
186
|
+
with tarfile.open(self.path) as tar:
|
|
187
|
+
tar.extractall(path=Path(tmpdir), filter="data")
|
|
188
|
+
path = Path(tmpdir) / "experiment"
|
|
189
|
+
run_path = Path(tmpdir) / "runs"
|
|
190
|
+
yield path, run_path if run_path.exists() else None
|
|
191
|
+
else:
|
|
192
|
+
yield self.path, self.run_path
|
|
193
|
+
|
|
194
|
+
def archive(
|
|
195
|
+
self,
|
|
196
|
+
archive_path: Path,
|
|
197
|
+
exclude_dirs: list[str] | None = None,
|
|
198
|
+
exclude_files: list[str] | None = None,
|
|
199
|
+
follow_symlinks: bool = False,
|
|
200
|
+
overwrite: bool = False,
|
|
201
|
+
):
|
|
202
|
+
"""Archives the experiment to the specified archive path.
|
|
203
|
+
|
|
204
|
+
Only experiments with status DONE will be archived. No error will be raised if the experiment is not DONE.
|
|
205
|
+
|
|
206
|
+
Symlinks to files and directories inside the experiment directory will be include as symlinks. Symlinks to files
|
|
207
|
+
and directories outside the experiment directory will be followed if follow_symlinks is True, otherwise they
|
|
208
|
+
will be included as symlinks. path contents are stored under experiment/ in the archive. If
|
|
209
|
+
run_path is set, its contents are stored under runs/ in the archive.
|
|
210
|
+
|
|
211
|
+
Args:
|
|
212
|
+
archive_path (Path): Path to the archive destination. This should include the file name, but without
|
|
213
|
+
the .tar.gz suffix.
|
|
214
|
+
exclude_dirs (list[str] | None): Directory patterns to exclude when archiving.
|
|
215
|
+
exclude_files (list[str] | None): File patterns to exclude when archiving.
|
|
216
|
+
follow_symlinks (bool): Whether to follow symlinks when archiving. Defaults to False.
|
|
217
|
+
overwrite (bool): Whether to overwrite existing archives. Defaults to False.
|
|
218
|
+
|
|
219
|
+
Raises:
|
|
220
|
+
FileExistsError: If the archive destination already exists and overwrite is False.
|
|
221
|
+
ValueError: If the experiment status is unknown.
|
|
222
|
+
"""
|
|
223
|
+
if self.status == ProfilingExperimentStatus.NEW:
|
|
224
|
+
logger.warning(f"Experiment at {self.path} is not yet started. Skipping archiving.", stacklevel=2)
|
|
225
|
+
return
|
|
226
|
+
elif self.status == ProfilingExperimentStatus.RUNNING:
|
|
227
|
+
logger.warning(f"Experiment at {self.path} is still running. Skipping archiving.", stacklevel=2)
|
|
228
|
+
return
|
|
229
|
+
elif self.status == ProfilingExperimentStatus.DONE:
|
|
230
|
+
logger.info(f"Archiving experiment at {self.path} to {archive_path.with_suffix('.tar.gz')}")
|
|
231
|
+
elif self.status == ProfilingExperimentStatus.ARCHIVED:
|
|
232
|
+
logger.warning(f"Experiment at {self.path} is already archived. Skipping archiving.", stacklevel=2)
|
|
233
|
+
return
|
|
234
|
+
|
|
235
|
+
archive_file = archive_path.with_suffix(".tar.gz")
|
|
236
|
+
mode = "w:gz" if overwrite else "x:gz"
|
|
237
|
+
if not overwrite and archive_file.exists():
|
|
238
|
+
raise FileExistsError(f"Archive destination {archive_file} already exists.")
|
|
239
|
+
|
|
240
|
+
exclude_dirs = exclude_dirs or []
|
|
241
|
+
exclude_files = exclude_files or []
|
|
242
|
+
|
|
243
|
+
paths_to_walk = (
|
|
244
|
+
[(self.path, Path("experiment"))]
|
|
245
|
+
if self.run_path is None
|
|
246
|
+
else [(self.path, Path("experiment")), (self.run_path, Path("runs"))]
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
with tarfile.open(archive_file, mode) as tar:
|
|
250
|
+
for root, prefix in paths_to_walk:
|
|
251
|
+
for file, arcname in experiment_directory_walker(root, prefix, root, follow_symlinks=follow_symlinks):
|
|
252
|
+
# Skip if file is inside an excluded directory pattern
|
|
253
|
+
if any(any(parent.match(pat) for pat in exclude_dirs) for parent in file.parents):
|
|
254
|
+
continue
|
|
255
|
+
# Skip if the file itself matches an excluded filename pattern
|
|
256
|
+
if any(file.match(pat) for pat in exclude_files):
|
|
257
|
+
continue
|
|
258
|
+
logger.debug(f"Archiving file: {file} as {arcname}")
|
|
259
|
+
tar.add(file, arcname=arcname)
|
|
260
|
+
|
|
261
|
+
self.status = ProfilingExperimentStatus.ARCHIVED
|
|
262
|
+
self.path = archive_file
|
|
263
|
+
self.run_path = None
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
# Copyright 2025 ACCESS-NRI and contributors. See the top-level COPYRIGHT file for details.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
|
|
4
|
+
"""Parser for FMS profiling data, such as output by MOM5 and MOM6.
|
|
5
|
+
The data to be parsed is written in the following form:
|
|
6
|
+
|
|
7
|
+
hits tmin tmax tavg tstd tfrac grain pemin pemax
|
|
8
|
+
Total runtime 1 138.600364 138.600366 138.600365 0.000001 1.000 0 0 11
|
|
9
|
+
Ocean Initialization 2 2.344926 2.345701 2.345388 0.000198 0.017 11 0 11
|
|
10
|
+
Ocean 23 86.869466 86.871652 86.870450 0.000744 0.627 1 0 11
|
|
11
|
+
Ocean dynamics 96 43.721019 44.391032 43.957944 0.244785 0.317 11 0 11
|
|
12
|
+
Ocean thermodynamics and tracers 72 27.377185 33.281659 29.950144 1.792324 0.216 11 0 11
|
|
13
|
+
MPP_STACK high water mark= 0
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
import os
|
|
17
|
+
import re
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
from pint import Unit
|
|
21
|
+
|
|
22
|
+
from access.profiling.metrics import ProfilingMetric, count, pemax, pemin, tavg, tfrac, tmax, tmin, tstd
|
|
23
|
+
from access.profiling.parser import ProfilingParser, _convert_from_string, _read_text_file
|
|
24
|
+
|
|
25
|
+
grain = ProfilingMetric("grain", Unit("dimensionless"), "Grain")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class FMSProfilingParser(ProfilingParser):
|
|
29
|
+
"""FMS profiling output parser."""
|
|
30
|
+
|
|
31
|
+
has_hits: bool # whether FMS timings contains "hits" column.
|
|
32
|
+
|
|
33
|
+
def __init__(self, has_hits: bool = True):
|
|
34
|
+
"""Instantiate FMS profiling parser.
|
|
35
|
+
|
|
36
|
+
Args:
|
|
37
|
+
has_hits (bool): whether FMS timings contains "hits" column.
|
|
38
|
+
"""
|
|
39
|
+
super().__init__()
|
|
40
|
+
|
|
41
|
+
self.has_hits = has_hits
|
|
42
|
+
# FMS provides the following metrics:
|
|
43
|
+
self._metrics = [count] if self.has_hits else []
|
|
44
|
+
self._metrics += [tmin, tmax, tavg, tstd, tfrac, grain, pemin, pemax]
|
|
45
|
+
|
|
46
|
+
def parse(self, file_path: str | Path | os.PathLike) -> dict:
|
|
47
|
+
stream = _read_text_file(file_path)
|
|
48
|
+
|
|
49
|
+
labels = ["hits"] if self.has_hits else []
|
|
50
|
+
labels += ["tmin", "tmax", "tavg", "tstd", "tfrac", "grain", "pemin", "pemax"]
|
|
51
|
+
|
|
52
|
+
# Regular expression to extract the profiling section from the file
|
|
53
|
+
header = r"\s*" + r"\s*".join(labels) + r"\s*"
|
|
54
|
+
footer = r" MPP_STACK high water mark=\s*\d*"
|
|
55
|
+
profiling_section_p = re.compile(header + r"(.*)" + footer, re.DOTALL)
|
|
56
|
+
|
|
57
|
+
# Regular expression to parse the data for each region
|
|
58
|
+
profile_line = r"^\s*(?P<region>[a-zA-Z:()_/\-*&\s]+(?<!\s))"
|
|
59
|
+
for label in labels:
|
|
60
|
+
profile_line += r"\s+(?P<" + label + r">[0-9.]+)"
|
|
61
|
+
profile_line += r"$"
|
|
62
|
+
profiling_region_p = re.compile(profile_line, re.MULTILINE)
|
|
63
|
+
|
|
64
|
+
# Parse data
|
|
65
|
+
stats = {"region": []}
|
|
66
|
+
stats.update({m: [] for m in self.metrics})
|
|
67
|
+
match = profiling_section_p.search(stream)
|
|
68
|
+
if match is None:
|
|
69
|
+
raise ValueError("No FMS profiling data found")
|
|
70
|
+
else:
|
|
71
|
+
profiling_section = match.group(1)
|
|
72
|
+
for line in profiling_region_p.finditer(profiling_section):
|
|
73
|
+
stats["region"].append(line.group("region"))
|
|
74
|
+
for label, metric in zip(labels, self.metrics, strict=True):
|
|
75
|
+
stats[metric].append(_convert_from_string(line.group(label)))
|
|
76
|
+
|
|
77
|
+
# Convert time fraction to percentage
|
|
78
|
+
stats[tfrac] = [val * 100 for val in stats[tfrac]]
|
|
79
|
+
|
|
80
|
+
return stats
|