access-profiling 0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,183 @@
1
+ # Copyright 2025 ACCESS-NRI and contributors. See the top-level COPYRIGHT file for details.
2
+ # SPDX-License-Identifier: Apache-2.0
3
+
4
+ """Parser for ESMF profiling text summary data, such as output by nuopc.
5
+ The data to be parsed is written in the following form:
6
+
7
+ Region PETs PEs Count Mean (s) Min (s) Min PET Max (s) Max PET
8
+ [ESMF] 1664 1664 1 2558.5684 2555.1450 279 2559.5801 817
9
+ [ensemble] RunPhase1 1664 1664 1 1879.7292 1872.5078 376 1905.4939 1
10
+ [ESM0001] RunPhase1 1664 1664 1 1879.7286 1872.5059 858 1905.4937 1
11
+ [OCN] RunPhase1 1300 1300 960 1850.4170 1848.3905 1023 1858.6404 364
12
+ [OCN-TO-MED] RunPhase1 1664 1664 960 365.9532 0.1688 405 1673.2255 18
13
+ [ICE] RunPhase1 364 364 960 155.8202 154.7637 94 160.2443 0
14
+ cice_run_total 364 364 960 155.4648 154.3687 94 159.8980 0
15
+ cice_run_import 364 364 960 3.7565 3.5426 218 8.3892 1
16
+ cice_imp_halo 364 364 1920 2.1386 1.5405 111 6.9782 0
17
+ cice_imp_t2u 364 364 960 0.9242 0.5578 355 1.5026 111
18
+ cice_imp_atm 364 364 960 0.0579 0.0399 331 0.0834 18
19
+ cice_imp_ocn 364 364 960 0.0499 0.0419 50 0.0632 12
20
+ cice_run_export 364 364 960 1.1015 0.8846 361 1.3607 194
21
+ [MED-TO-OCN] RunPhase1 1664 1664 960 16.7498 0.4588 256 23.2875 1023
22
+ [MED] med_phases_restart_write 364 364 960 31.8234 31.8123 203 33.0513 363
23
+ MED:(med_phases_restart_write) 364 364 960 31.7651 31.7559 203 32.9919 363
24
+ ...
25
+
26
+ Where indentation depth indicates depth in the call-stack. Usually there is a header like
27
+ ********
28
+ A warning about identifying load-imbalance.
29
+ ********
30
+ Note that the profiling summary stats may contain identical region names
31
+ e.g. [ATM-TO-MED] RunPhase1 is found twice in OM3.
32
+ """
33
+
34
+ import os
35
+ from pathlib import Path
36
+
37
+ from pint import Unit
38
+
39
+ from access.profiling.metrics import (
40
+ ProfilingMetric,
41
+ count,
42
+ pemax,
43
+ pemin,
44
+ tavg,
45
+ tmax,
46
+ tmin,
47
+ )
48
+ from access.profiling.parser import ProfilingParser, _read_text_file
49
+
50
+ pets = ProfilingMetric("PETs", Unit("dimensionless"), "ESMF Virtual Machine Persistent Execution Threads")
51
+ pes = ProfilingMetric("PEs", Unit("dimensionless"), "Processing Elements")
52
+
53
+
54
+ class ESMFSummaryProfilingParser(ProfilingParser):
55
+ """ESMF text summary profiling output parser."""
56
+
57
+ hierarchical: bool # whether the call-stack hierarchy is parsed.
58
+
59
+ def __init__(self, hierarchical: bool = False):
60
+ """Instantiate ESMF profiling parser.
61
+
62
+ Args:
63
+ hierarchical (bool): Whether call-stack hierarchy is parsed.
64
+ """
65
+ super().__init__()
66
+
67
+ self.hierarchical = hierarchical
68
+
69
+ # ESMF provides the following metrics
70
+ self._metrics = [pets, pes, count, tavg, tmin, pemin, tmax, pemax]
71
+
72
+ def parse(self, file_path: str | Path | os.PathLike) -> dict:
73
+ stream = _read_text_file(file_path)
74
+
75
+ lines = stream.strip().split("\n")
76
+
77
+ if self.hierarchical:
78
+ result = {}
79
+ stack = [(result, -1)] # (current_dict, indent_level)
80
+ else:
81
+ result = {m: [] for m in self._metrics}
82
+ result["region"] = []
83
+
84
+ for line in lines:
85
+ # Split the line into region name and statistics
86
+ parts = line.split()
87
+ if len(parts) < (1 + len(self._metrics)): # Need at least 1 region + the number of stat columns
88
+ continue
89
+
90
+ # Extract region name and statistics
91
+ region = " ".join(parts[:-8])
92
+ stats = parts[-8:]
93
+
94
+ # Validate that all statistics can be parsed correctly
95
+ try:
96
+ stats_dict = {
97
+ pets: int(stats[0]),
98
+ pes: int(stats[1]),
99
+ count: int(stats[2]),
100
+ tavg: float(stats[3]),
101
+ tmin: float(stats[4]),
102
+ pemin: int(stats[5]),
103
+ tmax: float(stats[6]),
104
+ pemax: int(stats[7]),
105
+ }
106
+ except (ValueError, IndexError):
107
+ # Skip lines that don't match the expected format
108
+ continue
109
+
110
+ if self.hierarchical:
111
+ # Calculate indentation level (each level is 2 spaces)
112
+ indent = len(line) - len(line.lstrip())
113
+ indent_level = indent // 2
114
+
115
+ # Pop stack until we find the parent level
116
+ while stack and stack[-1][1] >= indent_level:
117
+ stack.pop()
118
+
119
+ # Get parent dictionary
120
+ parent_dict = stack[-1][0]
121
+
122
+ # Create entry for this region
123
+ if region not in parent_dict:
124
+ parent_dict[region] = {}
125
+
126
+ # Add statistics to this region
127
+ parent_dict[region].update(stats_dict)
128
+
129
+ # Push this level onto stack for potential children
130
+ stack.append((parent_dict[region], indent_level))
131
+ else:
132
+ _update_flat_result(result, stats_dict, region)
133
+
134
+ # fewer if statements to pass ruff checks
135
+ if (self.hierarchical and not result) or (not self.hierarchical and len(result["region"]) == 0):
136
+ raise ValueError("No ESMF summary profiling data found")
137
+
138
+ return result
139
+
140
+
141
+ def _update_flat_result(result: dict, stats_dict: dict, region: str):
142
+ """Helper function to update flat result.
143
+
144
+ Besides appending results, this function also checks whether the region already exists
145
+ and aggregates the metric values in result if possible.
146
+
147
+ Args:
148
+ result (dict): The flat result dictionary to update.
149
+ stats_dict (dict): The stats to update the result with.
150
+ region (str): The region to append the results to.
151
+
152
+ Raises:
153
+ NotImplementedError: If a stats_dict["region"] is already in result["region"],
154
+ but the PETs or PEs value aren't the same.
155
+ """
156
+ # Flat structure: just use region name as key
157
+ try:
158
+ idx = result["region"].index(region)
159
+ # only update existing region if PETs and PEs are same
160
+ if (
161
+ result[pets][idx] == stats_dict[pets]
162
+ and result[pes][idx] == stats_dict[pes]
163
+ and stats_dict[pets] == stats_dict[pes]
164
+ ):
165
+ # new avg is weighted average using count as the weight
166
+ result[tavg][idx] = (result[tavg][idx] * result[count][idx] + stats_dict[tavg] * stats_dict[count]) / (
167
+ result[count][idx] + stats_dict[count]
168
+ )
169
+ result[count][idx] += stats_dict[count]
170
+ if stats_dict[tmin] < result[tmin][idx]:
171
+ result[tmin][idx] = stats_dict[tmin]
172
+ result[pemin][idx] = stats_dict[pemin]
173
+ if stats_dict[tmax] > result[tmax][idx]:
174
+ result[tmax][idx] = stats_dict[tmax]
175
+ result[pemax][idx] = stats_dict[pemax]
176
+ else:
177
+ raise NotImplementedError(
178
+ "I don't know what to do with multiple regions with same name, but different PETs/PEs."
179
+ )
180
+ except ValueError:
181
+ result["region"].append(region)
182
+ for k, v in stats_dict.items():
183
+ result[k].append(v)
@@ -0,0 +1,263 @@
1
+ # Copyright 2025 ACCESS-NRI and contributors. See the top-level COPYRIGHT file for details.
2
+ # SPDX-License-Identifier: Apache-2.0
3
+
4
+ import logging
5
+ import tarfile
6
+ import tempfile
7
+ from contextlib import contextmanager
8
+ from enum import Enum
9
+ from pathlib import Path
10
+
11
+ import xarray as xr
12
+
13
+ from access.profiling.parser import ProfilingParser, flatten_hierarchical
14
+
15
+ logger = logging.getLogger(__name__)
16
+
17
+
18
+ def _make_unique_region_names(regions: list[object]) -> list[object]:
19
+ """Return region names with deterministic suffixes for duplicates."""
20
+
21
+ counts: dict[object, int] = {}
22
+ unique_regions: list[object] = []
23
+ for region in regions:
24
+ count = counts.get(region, 0) + 1
25
+ counts[region] = count
26
+ unique_regions.append(region if count == 1 else f"{region}_{count}")
27
+ return unique_regions
28
+
29
+
30
+ class ProfilingLog:
31
+ """Represents a profiling log file.
32
+
33
+ Args:
34
+ filepath (Path): Path to the log file.
35
+ parser (ProfilingParser): Parser to use for this log file.
36
+ optional (bool): Whether this log might be missing or does not contain parsable data. If True, no error should
37
+ be raised if the log is missing or unparsable. Defaults to False.
38
+ """
39
+
40
+ filepath: Path # Path to the log file
41
+ parser: ProfilingParser # Parser to use for this log file
42
+ _optional: bool = False # Whether this log might not be present
43
+
44
+ def __init__(self, filepath: Path, parser: ProfilingParser, optional: bool = False) -> None:
45
+ self.filepath = filepath
46
+ self.parser = parser
47
+ self._optional = optional
48
+
49
+ @property
50
+ def optional(self) -> bool:
51
+ """bool: Whether this log might not be present."""
52
+ return self._optional
53
+
54
+ def parse(self) -> xr.Dataset:
55
+ """Parses the log file and returns the profiling data as an xarray Dataset.
56
+
57
+ Accepts all three parser output formats (see parser.py module docstring):
58
+
59
+ - **Flat**: standard 1D Dataset over the ``region`` dimension.
60
+ - **Hierarchical nested dict**: automatically flattened via
61
+ :func:`flatten_hierarchical` before building the Dataset.
62
+ - **Per-PE**: produces a 2D Dataset with both ``region`` and ``pe`` dimensions.
63
+ Use :func:`aggregate_pe_data` on the result to compute summary statistics.
64
+
65
+ Returns:
66
+ xr.Dataset: Parsed profiling data.
67
+ """
68
+ data = self.parser.parse(self.filepath)
69
+
70
+ # Flatten hierarchical (nested dict) format if needed
71
+ if "region" not in data:
72
+ data = flatten_hierarchical(data, self.parser.metrics)
73
+
74
+ has_pe = "pe" in data
75
+ dims = ["region", "pe"] if has_pe else ["region"]
76
+ coords: dict = {"region": _make_unique_region_names(list(data["region"]))}
77
+ if has_pe:
78
+ coords["pe"] = data["pe"]
79
+
80
+ return xr.Dataset(
81
+ data_vars=dict(
82
+ zip(
83
+ self.parser.metrics,
84
+ [xr.DataArray(data[m], dims=dims).pint.quantify(m.units) for m in self.parser.metrics],
85
+ strict=True,
86
+ )
87
+ ),
88
+ coords=coords,
89
+ )
90
+
91
+
92
+ class ProfilingExperimentStatus(Enum):
93
+ """Enumeration representing the status of a profiling experiment."""
94
+
95
+ NEW = 1 # Experiment has been created but not started
96
+ RUNNING = 2 # Experiment is running or is queued
97
+ DONE = 3 # Experiment has finished
98
+ ARCHIVED = 4 # Experiment has been archived
99
+
100
+
101
+ def experiment_directory_walker(path: Path, arcname: Path, root: Path, follow_symlinks: bool = False):
102
+ """Walks through the experiment directory, yielding files and corresponding names in the archive.
103
+
104
+ Symlinks are treated in a special manner.
105
+ - if the target is inside the experiment directory, the symlink itself is always returned
106
+ - if follow_symlinks is True and the target is a directory, then the all contents in the target directory are
107
+ recursively iterated
108
+ - if follow_symlinks is True and the target is a file, then the target file name is returned, not the symlink
109
+ - if follow_symlinks is False, then the symlink itself is returned for both files and directories
110
+
111
+ Args:
112
+ path (Path): Path to walk through.
113
+ arcname (Path): Archive name for the current path.
114
+ follow_symlinks (bool): Whether to follow symlinks. Defaults to False.
115
+
116
+ Yields:
117
+ Tuple[Path, Path]: A tuple containing the file path and its archive name.
118
+ """
119
+ if path.is_symlink():
120
+ if not follow_symlinks:
121
+ # Add symlink itself without following
122
+ yield path, arcname
123
+ else:
124
+ target = path.resolve()
125
+ if target.is_dir():
126
+ # Recursively add target contents
127
+ for child in target.iterdir():
128
+ yield from experiment_directory_walker(
129
+ child, Path(arcname) / child.name, root, follow_symlinks=follow_symlinks
130
+ )
131
+ elif target.absolute().is_relative_to(root.absolute()):
132
+ # Target is within the experiment directory, so add symlink as is
133
+ yield path, arcname
134
+ else:
135
+ # Target is outside the experiment directory, add the target file instead
136
+ yield target, arcname
137
+
138
+ elif path.is_dir():
139
+ # Recursively add directory contents
140
+ for child in path.iterdir():
141
+ yield from experiment_directory_walker(
142
+ child, Path(arcname) / child.name, root, follow_symlinks=follow_symlinks
143
+ )
144
+ else:
145
+ yield path, arcname
146
+
147
+
148
+ class ProfilingExperiment:
149
+ """Represents a profiling experiment.
150
+
151
+ Args:
152
+ path (Path): Path to the experiment directory.
153
+ run_path (Path | None): Path to a separate runs directory. If None, runs are assumed to be
154
+ inside path. When provided, the runs directory is also traversed during archival.
155
+ path contents are stored under experiment/ and run_path contents under runs/.
156
+ """
157
+
158
+ path: Path # Path to the experiment directory
159
+ run_path: Path | None # Path to a separate runs directory, or None
160
+ status: ProfilingExperimentStatus = ProfilingExperimentStatus.NEW # Status of the experiment
161
+
162
+ def __init__(self, path: Path, run_path: Path | None = None) -> None:
163
+ self.path = path
164
+ self.run_path = run_path
165
+ if self.path.suffixes == [".tar", ".gz"]:
166
+ self.status = ProfilingExperimentStatus.ARCHIVED
167
+
168
+ def __repr__(self) -> str:
169
+ """Returns a string representation of the ProfilingExperiment."""
170
+ if self.run_path is not None:
171
+ return f"{type(self).__name__}(path={self.path!r}, run_path={self.run_path!r}, status={self.status.name})"
172
+ return f"{type(self).__name__}(path={self.path!r}, status={self.status.name})"
173
+
174
+ @contextmanager
175
+ def directory(self):
176
+ """Context manager returning the experiment and runs directories.
177
+
178
+ If the experiment has been archived, it will be extracted to a temporary directory. Otherwise, the original
179
+ directory paths will be used. Note that after exiting the context, the temporary directory is removed.
180
+
181
+ Returns:
182
+ tuple[Path, Path | None]: The experiment directory path and optional runs directory path.
183
+ """
184
+ if self.path.suffixes == [".tar", ".gz"]:
185
+ with tempfile.TemporaryDirectory(prefix="access-profiling_", suffix="_data") as tmpdir:
186
+ with tarfile.open(self.path) as tar:
187
+ tar.extractall(path=Path(tmpdir), filter="data")
188
+ path = Path(tmpdir) / "experiment"
189
+ run_path = Path(tmpdir) / "runs"
190
+ yield path, run_path if run_path.exists() else None
191
+ else:
192
+ yield self.path, self.run_path
193
+
194
+ def archive(
195
+ self,
196
+ archive_path: Path,
197
+ exclude_dirs: list[str] | None = None,
198
+ exclude_files: list[str] | None = None,
199
+ follow_symlinks: bool = False,
200
+ overwrite: bool = False,
201
+ ):
202
+ """Archives the experiment to the specified archive path.
203
+
204
+ Only experiments with status DONE will be archived. No error will be raised if the experiment is not DONE.
205
+
206
+ Symlinks to files and directories inside the experiment directory will be include as symlinks. Symlinks to files
207
+ and directories outside the experiment directory will be followed if follow_symlinks is True, otherwise they
208
+ will be included as symlinks. path contents are stored under experiment/ in the archive. If
209
+ run_path is set, its contents are stored under runs/ in the archive.
210
+
211
+ Args:
212
+ archive_path (Path): Path to the archive destination. This should include the file name, but without
213
+ the .tar.gz suffix.
214
+ exclude_dirs (list[str] | None): Directory patterns to exclude when archiving.
215
+ exclude_files (list[str] | None): File patterns to exclude when archiving.
216
+ follow_symlinks (bool): Whether to follow symlinks when archiving. Defaults to False.
217
+ overwrite (bool): Whether to overwrite existing archives. Defaults to False.
218
+
219
+ Raises:
220
+ FileExistsError: If the archive destination already exists and overwrite is False.
221
+ ValueError: If the experiment status is unknown.
222
+ """
223
+ if self.status == ProfilingExperimentStatus.NEW:
224
+ logger.warning(f"Experiment at {self.path} is not yet started. Skipping archiving.", stacklevel=2)
225
+ return
226
+ elif self.status == ProfilingExperimentStatus.RUNNING:
227
+ logger.warning(f"Experiment at {self.path} is still running. Skipping archiving.", stacklevel=2)
228
+ return
229
+ elif self.status == ProfilingExperimentStatus.DONE:
230
+ logger.info(f"Archiving experiment at {self.path} to {archive_path.with_suffix('.tar.gz')}")
231
+ elif self.status == ProfilingExperimentStatus.ARCHIVED:
232
+ logger.warning(f"Experiment at {self.path} is already archived. Skipping archiving.", stacklevel=2)
233
+ return
234
+
235
+ archive_file = archive_path.with_suffix(".tar.gz")
236
+ mode = "w:gz" if overwrite else "x:gz"
237
+ if not overwrite and archive_file.exists():
238
+ raise FileExistsError(f"Archive destination {archive_file} already exists.")
239
+
240
+ exclude_dirs = exclude_dirs or []
241
+ exclude_files = exclude_files or []
242
+
243
+ paths_to_walk = (
244
+ [(self.path, Path("experiment"))]
245
+ if self.run_path is None
246
+ else [(self.path, Path("experiment")), (self.run_path, Path("runs"))]
247
+ )
248
+
249
+ with tarfile.open(archive_file, mode) as tar:
250
+ for root, prefix in paths_to_walk:
251
+ for file, arcname in experiment_directory_walker(root, prefix, root, follow_symlinks=follow_symlinks):
252
+ # Skip if file is inside an excluded directory pattern
253
+ if any(any(parent.match(pat) for pat in exclude_dirs) for parent in file.parents):
254
+ continue
255
+ # Skip if the file itself matches an excluded filename pattern
256
+ if any(file.match(pat) for pat in exclude_files):
257
+ continue
258
+ logger.debug(f"Archiving file: {file} as {arcname}")
259
+ tar.add(file, arcname=arcname)
260
+
261
+ self.status = ProfilingExperimentStatus.ARCHIVED
262
+ self.path = archive_file
263
+ self.run_path = None
@@ -0,0 +1,80 @@
1
+ # Copyright 2025 ACCESS-NRI and contributors. See the top-level COPYRIGHT file for details.
2
+ # SPDX-License-Identifier: Apache-2.0
3
+
4
+ """Parser for FMS profiling data, such as output by MOM5 and MOM6.
5
+ The data to be parsed is written in the following form:
6
+
7
+ hits tmin tmax tavg tstd tfrac grain pemin pemax
8
+ Total runtime 1 138.600364 138.600366 138.600365 0.000001 1.000 0 0 11
9
+ Ocean Initialization 2 2.344926 2.345701 2.345388 0.000198 0.017 11 0 11
10
+ Ocean 23 86.869466 86.871652 86.870450 0.000744 0.627 1 0 11
11
+ Ocean dynamics 96 43.721019 44.391032 43.957944 0.244785 0.317 11 0 11
12
+ Ocean thermodynamics and tracers 72 27.377185 33.281659 29.950144 1.792324 0.216 11 0 11
13
+ MPP_STACK high water mark= 0
14
+ """
15
+
16
+ import os
17
+ import re
18
+ from pathlib import Path
19
+
20
+ from pint import Unit
21
+
22
+ from access.profiling.metrics import ProfilingMetric, count, pemax, pemin, tavg, tfrac, tmax, tmin, tstd
23
+ from access.profiling.parser import ProfilingParser, _convert_from_string, _read_text_file
24
+
25
+ grain = ProfilingMetric("grain", Unit("dimensionless"), "Grain")
26
+
27
+
28
+ class FMSProfilingParser(ProfilingParser):
29
+ """FMS profiling output parser."""
30
+
31
+ has_hits: bool # whether FMS timings contains "hits" column.
32
+
33
+ def __init__(self, has_hits: bool = True):
34
+ """Instantiate FMS profiling parser.
35
+
36
+ Args:
37
+ has_hits (bool): whether FMS timings contains "hits" column.
38
+ """
39
+ super().__init__()
40
+
41
+ self.has_hits = has_hits
42
+ # FMS provides the following metrics:
43
+ self._metrics = [count] if self.has_hits else []
44
+ self._metrics += [tmin, tmax, tavg, tstd, tfrac, grain, pemin, pemax]
45
+
46
+ def parse(self, file_path: str | Path | os.PathLike) -> dict:
47
+ stream = _read_text_file(file_path)
48
+
49
+ labels = ["hits"] if self.has_hits else []
50
+ labels += ["tmin", "tmax", "tavg", "tstd", "tfrac", "grain", "pemin", "pemax"]
51
+
52
+ # Regular expression to extract the profiling section from the file
53
+ header = r"\s*" + r"\s*".join(labels) + r"\s*"
54
+ footer = r" MPP_STACK high water mark=\s*\d*"
55
+ profiling_section_p = re.compile(header + r"(.*)" + footer, re.DOTALL)
56
+
57
+ # Regular expression to parse the data for each region
58
+ profile_line = r"^\s*(?P<region>[a-zA-Z:()_/\-*&\s]+(?<!\s))"
59
+ for label in labels:
60
+ profile_line += r"\s+(?P<" + label + r">[0-9.]+)"
61
+ profile_line += r"$"
62
+ profiling_region_p = re.compile(profile_line, re.MULTILINE)
63
+
64
+ # Parse data
65
+ stats = {"region": []}
66
+ stats.update({m: [] for m in self.metrics})
67
+ match = profiling_section_p.search(stream)
68
+ if match is None:
69
+ raise ValueError("No FMS profiling data found")
70
+ else:
71
+ profiling_section = match.group(1)
72
+ for line in profiling_region_p.finditer(profiling_section):
73
+ stats["region"].append(line.group("region"))
74
+ for label, metric in zip(labels, self.metrics, strict=True):
75
+ stats[metric].append(_convert_from_string(line.group(label)))
76
+
77
+ # Convert time fraction to percentage
78
+ stats[tfrac] = [val * 100 for val in stats[tfrac]]
79
+
80
+ return stats