access-profiling 0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- access/profiling/__init__.py +32 -0
- access/profiling/access_models.py +84 -0
- access/profiling/cice5_parser.py +80 -0
- access/profiling/cylc_manager.py +209 -0
- access/profiling/cylc_parser.py +146 -0
- access/profiling/esmf_parser.py +183 -0
- access/profiling/experiment.py +263 -0
- access/profiling/fms_parser.py +80 -0
- access/profiling/manager.py +559 -0
- access/profiling/metrics.py +85 -0
- access/profiling/parser.py +263 -0
- access/profiling/payu_manager.py +338 -0
- access/profiling/payujson_parser.py +75 -0
- access/profiling/plotting_utils.py +116 -0
- access/profiling/scaling.py +128 -0
- access/profiling/um_parser.py +242 -0
- access_profiling-0.1.dist-info/METADATA +100 -0
- access_profiling-0.1.dist-info/RECORD +21 -0
- access_profiling-0.1.dist-info/WHEEL +5 -0
- access_profiling-0.1.dist-info/licenses/LICENSE +201 -0
- access_profiling-0.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""
|
|
2
|
+
access-profiling package.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from contextlib import suppress
|
|
6
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
7
|
+
|
|
8
|
+
__version__ = "unknown"
|
|
9
|
+
with suppress(PackageNotFoundError):
|
|
10
|
+
__version__ = version("access-profiling")
|
|
11
|
+
|
|
12
|
+
from access.profiling.access_models import ESM16Profiling, RAM3Profiling
|
|
13
|
+
from access.profiling.cice5_parser import CICE5ProfilingParser
|
|
14
|
+
from access.profiling.cylc_parser import CylcDBReader, CylcProfilingParser
|
|
15
|
+
from access.profiling.esmf_parser import ESMFSummaryProfilingParser
|
|
16
|
+
from access.profiling.fms_parser import FMSProfilingParser
|
|
17
|
+
from access.profiling.parser import ProfilingParser
|
|
18
|
+
from access.profiling.payujson_parser import PayuJSONProfilingParser
|
|
19
|
+
from access.profiling.um_parser import UMProfilingParser
|
|
20
|
+
|
|
21
|
+
__all__ = [
|
|
22
|
+
"ProfilingParser",
|
|
23
|
+
"FMSProfilingParser",
|
|
24
|
+
"UMProfilingParser",
|
|
25
|
+
"CICE5ProfilingParser",
|
|
26
|
+
"PayuJSONProfilingParser",
|
|
27
|
+
"ESMFSummaryProfilingParser",
|
|
28
|
+
"ESM16Profiling",
|
|
29
|
+
"CylcProfilingParser",
|
|
30
|
+
"CylcDBReader",
|
|
31
|
+
"RAM3Profiling",
|
|
32
|
+
]
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
# Copyright 2025 ACCESS-NRI and contributors. See the top-level COPYRIGHT file for details.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
|
|
4
|
+
import logging
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from access.config import YAMLParser
|
|
8
|
+
from access.config.esm1p6_layout_input import (
|
|
9
|
+
LayoutSearchConfig,
|
|
10
|
+
LayoutTuple,
|
|
11
|
+
generate_esm1p6_core_layouts_from_node_count,
|
|
12
|
+
generate_esm1p6_perturb_block,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
from access.profiling.cice5_parser import CICE5ProfilingParser
|
|
16
|
+
from access.profiling.cylc_manager import CylcRoseManager
|
|
17
|
+
from access.profiling.experiment import ProfilingLog
|
|
18
|
+
from access.profiling.fms_parser import FMSProfilingParser
|
|
19
|
+
from access.profiling.payu_manager import PayuManager
|
|
20
|
+
from access.profiling.um_parser import UMProfilingParser, UMTotalRuntimeParser
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger(__name__)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class ESM16Profiling(PayuManager):
|
|
26
|
+
"""Handles profiling of ACCESS-ESM1.6 configurations."""
|
|
27
|
+
|
|
28
|
+
@property
|
|
29
|
+
def model_type(self) -> str:
|
|
30
|
+
return "access-esm1.6"
|
|
31
|
+
|
|
32
|
+
def get_component_logs(self, path: Path) -> dict[str, ProfilingLog]:
|
|
33
|
+
"""Returns available profiling logs for the components in ACCESS-ESM1.6.
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
path (Path): Path to the output directory.
|
|
37
|
+
Returns:
|
|
38
|
+
dict[str, ProfilingLog]: Dictionary mapping component names to their ProfilingLog instances.
|
|
39
|
+
"""
|
|
40
|
+
logs = {}
|
|
41
|
+
parser = YAMLParser()
|
|
42
|
+
|
|
43
|
+
um_env_path = path / "atmosphere" / "um_env.yaml"
|
|
44
|
+
um_env = parser.parse(um_env_path.read_text())
|
|
45
|
+
um_logfile = path / "atmosphere" / f"{um_env['UM_STDOUT_FILE']}0"
|
|
46
|
+
if um_logfile.is_file():
|
|
47
|
+
logger.debug(f"Found UM log file: {um_logfile}")
|
|
48
|
+
logs["UM"] = ProfilingLog(um_logfile, UMProfilingParser())
|
|
49
|
+
logs["UM_Total_Walltime"] = ProfilingLog(um_logfile, UMTotalRuntimeParser())
|
|
50
|
+
|
|
51
|
+
config_path = path / "config.yaml"
|
|
52
|
+
payu_config = parser.parse(config_path.read_text())
|
|
53
|
+
mom5_logfile = path / f"{payu_config['model']}.out"
|
|
54
|
+
if mom5_logfile.is_file():
|
|
55
|
+
logger.debug(f"Found MOM5 log file: {mom5_logfile}")
|
|
56
|
+
logs["MOM5"] = ProfilingLog(mom5_logfile, FMSProfilingParser(has_hits=False))
|
|
57
|
+
|
|
58
|
+
cice5_logfile = path / "ice" / "ice_diag.d"
|
|
59
|
+
if cice5_logfile.is_file():
|
|
60
|
+
logger.debug(f"Found CICE5 log file: {cice5_logfile}")
|
|
61
|
+
logs["CICE5"] = ProfilingLog(cice5_logfile, CICE5ProfilingParser())
|
|
62
|
+
|
|
63
|
+
return logs
|
|
64
|
+
|
|
65
|
+
def generate_core_layouts_from_node_count(
|
|
66
|
+
self, num_nodes: float, cores_per_node: int, layout_search_config: LayoutSearchConfig | None = None
|
|
67
|
+
) -> list:
|
|
68
|
+
return generate_esm1p6_core_layouts_from_node_count(
|
|
69
|
+
num_nodes, cores_per_node, layout_search_config=layout_search_config
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
def generate_perturbation_block(self, layout: LayoutTuple, branch_name_prefix: str) -> dict:
|
|
73
|
+
return generate_esm1p6_perturb_block(layout, branch_name_prefix)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
class RAM3Profiling(CylcRoseManager):
|
|
77
|
+
"""Handles profiling of ACCESS-rAM3 configurations."""
|
|
78
|
+
|
|
79
|
+
@property
|
|
80
|
+
def known_parsers(self):
|
|
81
|
+
return {
|
|
82
|
+
"UM_regions": UMProfilingParser(),
|
|
83
|
+
"UM_total": UMTotalRuntimeParser(),
|
|
84
|
+
}
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
# Copyright 2025 ACCESS-NRI and contributors. See the top-level COPYRIGHT file for details.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
|
|
4
|
+
"""Parser for CICE5 profiling data.
|
|
5
|
+
The data to be parsed is written in the following form, where block stats are discarded:
|
|
6
|
+
|
|
7
|
+
Timer 1: Total 8133.37 seconds
|
|
8
|
+
Timer stats (node): min = 8133.36 seconds
|
|
9
|
+
max = 8133.37 seconds
|
|
10
|
+
mean= 8133.36 seconds
|
|
11
|
+
Timer stats(block): min = 0.00 seconds
|
|
12
|
+
max = 0.00 seconds
|
|
13
|
+
mean= 0.00 seconds
|
|
14
|
+
Timer 2: TimeLoop 8133.00 seconds
|
|
15
|
+
Timer stats (node): min = 8132.99 seconds
|
|
16
|
+
max = 8133.00 seconds
|
|
17
|
+
mean= 8132.99 seconds
|
|
18
|
+
Timer stats(block): min = 0.00 seconds
|
|
19
|
+
max = 0.00 seconds
|
|
20
|
+
mean= 0.00 seconds
|
|
21
|
+
|
|
22
|
+
These timers are printed at the end of the CICE5 run and can be an arbitrary number of timers.
|
|
23
|
+
For example, ESM1.6 has 17 timers printed at the end of ice_diag.d output log.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
import os
|
|
27
|
+
import re
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
|
|
30
|
+
from access.profiling.metrics import tavg, tmax, tmin
|
|
31
|
+
from access.profiling.parser import ProfilingParser, _read_text_file
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class CICE5ProfilingParser(ProfilingParser):
|
|
35
|
+
"""CICE5 profiling output parser."""
|
|
36
|
+
|
|
37
|
+
_metrics = [tmin, tmax, tavg]
|
|
38
|
+
|
|
39
|
+
def parse(self, file_path: str | Path | os.PathLike) -> dict:
|
|
40
|
+
"""Implements "parse" abstract method to parse profiling data in CICE5 log output.
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
file_path (str | Path | os.PathLike): file to parse.
|
|
44
|
+
|
|
45
|
+
Returns:
|
|
46
|
+
dict: Parsed timing information.
|
|
47
|
+
|
|
48
|
+
Raises:
|
|
49
|
+
ValueError: If matching timings aren't found.
|
|
50
|
+
TypeError: If file_path cannot be converted to a valid Path object.
|
|
51
|
+
FileNotFoundError: If file_path doesn't exist or isn't a file.
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
stream = _read_text_file(file_path)
|
|
55
|
+
|
|
56
|
+
# Initialize result dictionary
|
|
57
|
+
result = {"region": [], tmin: [], tmax: [], tavg: []}
|
|
58
|
+
|
|
59
|
+
# Regex pattern to match timer blocks
|
|
60
|
+
# This captures the region name and the three node timing values
|
|
61
|
+
pattern = (
|
|
62
|
+
r"Timer\s+\d+:\s+(\w+)\s+[\d.]+\s+seconds\s+Timer stats \(node\): min =\s+([\d.]+) seconds\s+max ="
|
|
63
|
+
r"\s+([\d.]+) seconds\s+mean=\s+([\d.]+) seconds"
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
# Find all matches
|
|
67
|
+
matches = re.findall(pattern, stream, re.MULTILINE | re.DOTALL)
|
|
68
|
+
|
|
69
|
+
if not matches:
|
|
70
|
+
raise ValueError("No CICE5 profiling data found")
|
|
71
|
+
|
|
72
|
+
# Extract data from matches
|
|
73
|
+
for match in matches:
|
|
74
|
+
region, min_time, max_time, mean_time = match
|
|
75
|
+
result["region"].append(region)
|
|
76
|
+
result[tmin].append(float(min_time))
|
|
77
|
+
result[tmax].append(float(max_time))
|
|
78
|
+
result[tavg].append(float(mean_time))
|
|
79
|
+
|
|
80
|
+
return result
|
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
# Copyright 2025 ACCESS-NRI and contributors. See the top-level COPYRIGHT file for details.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
|
|
4
|
+
import logging
|
|
5
|
+
import shutil
|
|
6
|
+
import subprocess
|
|
7
|
+
from abc import ABC, abstractmethod
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from access.profiling.cylc_parser import CylcDBReader, CylcProfilingParser
|
|
11
|
+
from access.profiling.experiment import ProfilingExperiment, ProfilingExperimentStatus, ProfilingLog
|
|
12
|
+
from access.profiling.manager import ProfilingManager
|
|
13
|
+
from access.profiling.parser import ProfilingParser
|
|
14
|
+
|
|
15
|
+
logger = logging.getLogger(__name__)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class CylcRoseManager(ProfilingManager, ABC):
|
|
19
|
+
"""Abstract base class to handle profiling data for Cylc Rose configurations.
|
|
20
|
+
|
|
21
|
+
Args:
|
|
22
|
+
work_dir (Path): Working directory where profiling experiments will be generated and run.
|
|
23
|
+
archive_dir (Path): Directory where completed experiments will be archived.
|
|
24
|
+
layout_variable (str): Name of the variable in rose-suite-run.conf file that defines the layout.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
_layout_variable: str # Name of the variable in rose-suite-run.conf file that defines the layout.
|
|
28
|
+
|
|
29
|
+
def __init__(self, work_dir: Path, archive_dir: Path, layout_variable: str):
|
|
30
|
+
super().__init__(work_dir, archive_dir)
|
|
31
|
+
self._layout_variable = layout_variable
|
|
32
|
+
|
|
33
|
+
@property
|
|
34
|
+
@abstractmethod
|
|
35
|
+
def known_parsers(self) -> dict[str, ProfilingParser]:
|
|
36
|
+
"""Returns the parsers that this model configuration knows about.
|
|
37
|
+
|
|
38
|
+
Returns:
|
|
39
|
+
dict[str, ProfilingParser]: a dictionary of known parsers with names as keys.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
def parse_ncpus(self, path: Path, run_path: Path | None = None) -> int:
|
|
43
|
+
# both the run and original config will store cpu information
|
|
44
|
+
config_paths = []
|
|
45
|
+
if run_path is not None:
|
|
46
|
+
config_paths.append(run_path / "log/rose-suite-run.conf")
|
|
47
|
+
config_paths.append(path / "rose-suite.conf")
|
|
48
|
+
|
|
49
|
+
config_path = next((candidate for candidate in config_paths if candidate.is_file()), None)
|
|
50
|
+
if config_path is None:
|
|
51
|
+
tried = ", ".join(str(p) for p in config_paths)
|
|
52
|
+
raise FileNotFoundError(f"Could not find suitable config file. Tried: {tried}")
|
|
53
|
+
|
|
54
|
+
for line in config_path.read_text().splitlines():
|
|
55
|
+
if not line.startswith("!!") and "=" in line:
|
|
56
|
+
key, value = line.split("=", 1)
|
|
57
|
+
if key.strip() == self._layout_variable:
|
|
58
|
+
layout = value.split(",")
|
|
59
|
+
return int(layout[0].strip()) * int(layout[1].strip())
|
|
60
|
+
|
|
61
|
+
raise ValueError(f"Cannot find layout key, {self._layout_variable}, in {config_path}.")
|
|
62
|
+
|
|
63
|
+
def add_rose_experiment(self, rose: str, run_path: Path | None = None) -> None:
|
|
64
|
+
"""Adds the given rose as an experiment to this manager.
|
|
65
|
+
|
|
66
|
+
Args:
|
|
67
|
+
rose (str): The rose to add as an experiment.
|
|
68
|
+
run_path (Path | None): Path to the Cylc run directory holding the results. If not provided, or if the
|
|
69
|
+
provided directory does not exist, archiving will only include the experiment files.
|
|
70
|
+
|
|
71
|
+
Raises:
|
|
72
|
+
ValueError: If the experiment path does not exist.
|
|
73
|
+
"""
|
|
74
|
+
experiment_path = self.work_dir / rose
|
|
75
|
+
if not experiment_path.is_dir():
|
|
76
|
+
raise ValueError(f"Experiment path '{experiment_path}' does not exist or is not a directory.")
|
|
77
|
+
|
|
78
|
+
if run_path is not None and not run_path.is_dir():
|
|
79
|
+
logger.warning(f"Run path '{run_path}' does not exist. Archiving will only include experiment files.")
|
|
80
|
+
run_path = None
|
|
81
|
+
|
|
82
|
+
self.experiments[rose] = ProfilingExperiment(path=experiment_path, run_path=run_path)
|
|
83
|
+
self.experiments[rose].status = ProfilingExperimentStatus.DONE
|
|
84
|
+
|
|
85
|
+
def run_experiments(self) -> None:
|
|
86
|
+
"""Runs Rose Cylc experiments via `rose suite-run` for profiling data generation."""
|
|
87
|
+
|
|
88
|
+
to_run = {name: exp for name, exp in self.experiments.items() if exp.status == ProfilingExperimentStatus.NEW}
|
|
89
|
+
|
|
90
|
+
if not to_run:
|
|
91
|
+
logger.info("No new experiments to run. Will skip execution.")
|
|
92
|
+
return
|
|
93
|
+
|
|
94
|
+
for name, exp in to_run.items():
|
|
95
|
+
logger.info(f"Running experiment '{name}' via rose suite-run in '{exp.path}'.")
|
|
96
|
+
try:
|
|
97
|
+
result = subprocess.run(["rose", "suite-run"], cwd=exp.path, check=True, capture_output=True, text=True)
|
|
98
|
+
except subprocess.CalledProcessError as e:
|
|
99
|
+
for line in e.stdout.splitlines():
|
|
100
|
+
logger.info(f"[{name}] {line}")
|
|
101
|
+
for line in e.stderr.splitlines():
|
|
102
|
+
logger.error(f"[{name}] {line}")
|
|
103
|
+
raise
|
|
104
|
+
for line in result.stdout.splitlines():
|
|
105
|
+
logger.info(f"[{name}] {line}")
|
|
106
|
+
for line in result.stderr.splitlines():
|
|
107
|
+
logger.warning(f"[{name}] {line}")
|
|
108
|
+
exp.status = ProfilingExperimentStatus.RUNNING
|
|
109
|
+
|
|
110
|
+
# TODO: properly detect when running experiments have completed rather than marking them done immediately.
|
|
111
|
+
for exp_name in self.experiments:
|
|
112
|
+
if self.experiments[exp_name].status == ProfilingExperimentStatus.RUNNING:
|
|
113
|
+
self.experiments[exp_name].status = ProfilingExperimentStatus.DONE
|
|
114
|
+
|
|
115
|
+
def _delete_experiment(self, name: str, dry_run: bool) -> None:
|
|
116
|
+
"""Deletes the experiment and run directories of a single Rose Cylc experiment.
|
|
117
|
+
|
|
118
|
+
Args:
|
|
119
|
+
name (str): Name of the experiment to delete.
|
|
120
|
+
dry_run (bool): If True, logs what would be deleted without making any changes.
|
|
121
|
+
"""
|
|
122
|
+
exp = self.experiments[name]
|
|
123
|
+
exp_path = exp.path
|
|
124
|
+
run_path = exp.run_path
|
|
125
|
+
if dry_run:
|
|
126
|
+
logger.info(f"Dry run: would delete experiment directory '{exp_path}' and run directory '{run_path}'.")
|
|
127
|
+
return
|
|
128
|
+
if exp_path.is_dir():
|
|
129
|
+
logger.info(f"Deleting experiment directory '{exp_path}'.")
|
|
130
|
+
shutil.rmtree(exp_path)
|
|
131
|
+
else:
|
|
132
|
+
logger.warning(f"Experiment directory '{exp_path}' does not exist. Skipping deletion.")
|
|
133
|
+
if run_path is not None:
|
|
134
|
+
if run_path.is_dir():
|
|
135
|
+
logger.info(f"Deleting run directory '{run_path}'.")
|
|
136
|
+
shutil.rmtree(run_path)
|
|
137
|
+
else:
|
|
138
|
+
logger.warning(f"Run directory '{run_path}' does not exist. Skipping deletion.")
|
|
139
|
+
|
|
140
|
+
def archive_experiments(
|
|
141
|
+
self,
|
|
142
|
+
exclude_dirs: list[str] | None = None,
|
|
143
|
+
exclude_files: list[str] | None = None,
|
|
144
|
+
follow_symlinks: bool = False,
|
|
145
|
+
overwrite: bool = False,
|
|
146
|
+
) -> None:
|
|
147
|
+
"""Archives completed experiments to the specified archive path.
|
|
148
|
+
|
|
149
|
+
Args:
|
|
150
|
+
exclude_dirs (list[str] | None): Directory patterns to exclude when archiving. Defaults to
|
|
151
|
+
[".svn", "share"] if not provided.
|
|
152
|
+
exclude_files (list[str] | None): File patterns to exclude when archiving. Defaults to
|
|
153
|
+
["*.nc"] if not provided.
|
|
154
|
+
follow_symlinks (bool): Whether to follow symlinks when archiving. Defaults to False.
|
|
155
|
+
overwrite (bool): Whether to overwrite existing archives. Defaults to False.
|
|
156
|
+
"""
|
|
157
|
+
if exclude_dirs is None:
|
|
158
|
+
exclude_dirs = [".svn", "share"]
|
|
159
|
+
if exclude_files is None:
|
|
160
|
+
exclude_files = ["*.nc"]
|
|
161
|
+
super().archive_experiments(
|
|
162
|
+
exclude_dirs=exclude_dirs,
|
|
163
|
+
exclude_files=exclude_files,
|
|
164
|
+
follow_symlinks=follow_symlinks,
|
|
165
|
+
overwrite=overwrite,
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
def profiling_logs(self, path: Path, run_path: Path | None = None) -> dict[str, dict[int, ProfilingLog]]:
|
|
169
|
+
"""Returns all profiling logs from the specified path.
|
|
170
|
+
|
|
171
|
+
Args:
|
|
172
|
+
path (Path): Path to the experiment directory.
|
|
173
|
+
run_path (Path | None): Path to the Cylc run directory.
|
|
174
|
+
Returns:
|
|
175
|
+
dict[str, dict[int, ProfilingLog]]: Dictionary mapping log names to their logs, keyed by run number.
|
|
176
|
+
Cylc workflows have no concept of repeated runs, so every log is returned as the single run 0.
|
|
177
|
+
"""
|
|
178
|
+
if run_path is None:
|
|
179
|
+
raise ValueError("Cylc run_path is required to locate profiling logs.")
|
|
180
|
+
|
|
181
|
+
logs = {}
|
|
182
|
+
|
|
183
|
+
# setup log paths
|
|
184
|
+
suite_log = run_path / "log/suite/log" # cylc log file
|
|
185
|
+
cylcdb = run_path / "cylc-suite.db" # database with task runtimes
|
|
186
|
+
jobdir = run_path / "log/job" # where task logs are stored
|
|
187
|
+
|
|
188
|
+
logs["cylc_suite_log"] = ProfilingLog(suite_log, CylcProfilingParser())
|
|
189
|
+
# cylcdb.read_text = lambda x: x # hack to make log work
|
|
190
|
+
logs["cylc_tasks"] = ProfilingLog(cylcdb, CylcDBReader())
|
|
191
|
+
|
|
192
|
+
# Search for available profiling logs for the components in the configuration.
|
|
193
|
+
# matches <cycle> / <task> / NN / job.out e.g. 20220226T0000Z/Lismore_d1100_GAL9_um_fcst_000/NN/job.out
|
|
194
|
+
# NN is the last attempt
|
|
195
|
+
# job.out is the stdout
|
|
196
|
+
# this pattern is followed for all cylc workflows.
|
|
197
|
+
# as tasks of interest will likely have their own logging regions e.g. UM each task_cycle is
|
|
198
|
+
# treated as a "component" of the configuration.
|
|
199
|
+
possible_component_logs = list(jobdir.glob("*/*/NN/job.out"))
|
|
200
|
+
if not possible_component_logs:
|
|
201
|
+
raise RuntimeError(f"Could not find any known logs in {jobdir}")
|
|
202
|
+
|
|
203
|
+
for logfile in possible_component_logs:
|
|
204
|
+
cycle, task = logfile.parts[-4:-2]
|
|
205
|
+
for parser_name, parser in self.known_parsers.items():
|
|
206
|
+
logs[f"{task}_cycle{cycle}_{parser_name}"] = ProfilingLog(logfile, parser, optional=True)
|
|
207
|
+
|
|
208
|
+
# Cylc workflows have no concept of repeated runs, so every log is registered as the single run 0.
|
|
209
|
+
return {name: {0: log} for name, log in logs.items()}
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
# Copyright 2025 ACCESS-NRI and contributors. See the top-level COPYRIGHT file for details.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
|
|
4
|
+
"""Parser for Cylc log files. The data to be parsed is written in the following form:
|
|
5
|
+
|
|
6
|
+
2025-10-17T00:51:12Z INFO - Suite server: url=... pid=152868
|
|
7
|
+
2025-10-17T00:51:12Z INFO - Run: (re)start=0 log=1
|
|
8
|
+
2025-10-17T00:51:12Z INFO - Cylc version: 7.9.9
|
|
9
|
+
2025-10-17T00:51:12Z INFO - Run mode: live
|
|
10
|
+
2025-10-17T00:51:12Z INFO - Initial point: 20220226T0000Z
|
|
11
|
+
2025-10-17T00:51:12Z INFO - Final point: 20220226T0300Z
|
|
12
|
+
2025-10-17T00:51:12Z INFO - Cold Start 20220226T0000Z
|
|
13
|
+
...
|
|
14
|
+
2025-10-17T01:36:23Z INFO - Suite shutting down - AUTOMATIC
|
|
15
|
+
2025-10-17T01:36:30Z INFO - DONE
|
|
16
|
+
|
|
17
|
+
The differences between the first and last time-stamp are used to determine the
|
|
18
|
+
total pipeline walltime.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
import os
|
|
22
|
+
import sqlite3
|
|
23
|
+
from datetime import datetime
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
|
|
26
|
+
from access.profiling.metrics import tmax
|
|
27
|
+
from access.profiling.parser import ProfilingParser, _read_text_file, _test_file
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class CylcProfilingParser(ProfilingParser):
|
|
31
|
+
"""Cylc log profiling parser."""
|
|
32
|
+
|
|
33
|
+
_metrics = [tmax]
|
|
34
|
+
|
|
35
|
+
def parse(self, file_path: str | Path | os.PathLike) -> dict:
|
|
36
|
+
"""Implements "parse" abstract method to parse the Cycle suite run log.
|
|
37
|
+
|
|
38
|
+
Args:
|
|
39
|
+
file_path (str | Path | os.PathLike): String containing the suite run log.
|
|
40
|
+
|
|
41
|
+
Returns:
|
|
42
|
+
dict: Parsed timing information.
|
|
43
|
+
|
|
44
|
+
Raises:
|
|
45
|
+
ValueError: when the last line does not contain "DONE".
|
|
46
|
+
"""
|
|
47
|
+
lines = _read_text_file(file_path).splitlines()
|
|
48
|
+
|
|
49
|
+
first_line = lines[0]
|
|
50
|
+
last_line = lines[-1]
|
|
51
|
+
|
|
52
|
+
if "DONE" not in last_line:
|
|
53
|
+
raise ValueError("Cylc log is incomplete.")
|
|
54
|
+
|
|
55
|
+
try:
|
|
56
|
+
start_time = _extract_timestamp(first_line)
|
|
57
|
+
except Exception as e:
|
|
58
|
+
raise ValueError("First line of log doesn't contain a valid timestamp.") from e
|
|
59
|
+
try:
|
|
60
|
+
end_time = _extract_timestamp(last_line)
|
|
61
|
+
except Exception as e:
|
|
62
|
+
raise ValueError("Last line of log doesn't contain a valid timestamp.") from e
|
|
63
|
+
|
|
64
|
+
return {
|
|
65
|
+
"region": ["pipeline_elapsed_time"],
|
|
66
|
+
tmax: [int((end_time - start_time).total_seconds())],
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class CylcDBReader(ProfilingParser):
|
|
71
|
+
"""Cylc database reader."""
|
|
72
|
+
|
|
73
|
+
_table = "task_jobs"
|
|
74
|
+
_required_cols = ("cycle", "name", "time_run", "time_run_exit", "run_status")
|
|
75
|
+
_metrics = [tmax]
|
|
76
|
+
|
|
77
|
+
def parse(self, file_path: str | Path | os.PathLike) -> dict:
|
|
78
|
+
"""Implements "read" abstract method of CylcDBReader to parse the Cylc Rose task database.
|
|
79
|
+
|
|
80
|
+
Args:
|
|
81
|
+
file_path (str | Path | os.PathLike): The path to the SQLite database.
|
|
82
|
+
|
|
83
|
+
Returns:
|
|
84
|
+
dict: Read timing information.
|
|
85
|
+
|
|
86
|
+
Raises:
|
|
87
|
+
FileNotFoundError: When the provided database file doesn't exist.
|
|
88
|
+
RuntimeError: when the expected table is not present in the database or if the table doesn't have the
|
|
89
|
+
expected column names.
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
dbpath = _test_file(file_path)
|
|
93
|
+
|
|
94
|
+
with sqlite3.connect(dbpath) as con:
|
|
95
|
+
cur = con.cursor()
|
|
96
|
+
|
|
97
|
+
# collect and validate table columns . Return type: list of tuples
|
|
98
|
+
# where each list item corresponds to a column. Each tuple is (index, name, type, ?, ?, primary key)
|
|
99
|
+
col_metadata = cur.execute(f"PRAGMA table_info({self._table})").fetchall()
|
|
100
|
+
if col_metadata == []:
|
|
101
|
+
raise RuntimeError(f"Table {self._table} not found in {dbpath}!")
|
|
102
|
+
col_map = {col_data[1]: col_data[0] for col_data in col_metadata}
|
|
103
|
+
columns_missing_from_tbl = set(self._required_cols) - set(col_map.keys())
|
|
104
|
+
if columns_missing_from_tbl:
|
|
105
|
+
raise RuntimeError(f"Expected table columns: {', '.join(columns_missing_from_tbl)}")
|
|
106
|
+
|
|
107
|
+
# collect table data
|
|
108
|
+
table_data = cur.execute(f"SELECT * FROM {self._table}").fetchall()
|
|
109
|
+
|
|
110
|
+
# turn timestamps into time elapsed (seconds)
|
|
111
|
+
data = {"region": []}
|
|
112
|
+
for m in self._metrics:
|
|
113
|
+
data[m] = []
|
|
114
|
+
for row in table_data:
|
|
115
|
+
# filter out tasks that haven't completed successfully
|
|
116
|
+
if row[col_map["run_status"]] == 0:
|
|
117
|
+
# region will look like <task>_<chunk no.>_cycle<cycle timestamp>
|
|
118
|
+
region = row[col_map["name"]] + "_cycle" + row[col_map["cycle"]]
|
|
119
|
+
start = row[col_map["time_run"]]
|
|
120
|
+
end = row[col_map["time_run_exit"]]
|
|
121
|
+
runtime = (_extract_timestamp(end) - _extract_timestamp(start)).total_seconds()
|
|
122
|
+
data["region"].append(region)
|
|
123
|
+
data[self._metrics[0]].append(runtime)
|
|
124
|
+
|
|
125
|
+
return data
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _extract_timestamp(line: str) -> datetime:
|
|
129
|
+
"""Helper function to extra and convert timestamp to datetime object.
|
|
130
|
+
|
|
131
|
+
Args:
|
|
132
|
+
line (str): The line of text with the timestamp at the beginning.
|
|
133
|
+
|
|
134
|
+
Raises:
|
|
135
|
+
ValueError: When there is no timestamp or the timestamp is inavlid.
|
|
136
|
+
"""
|
|
137
|
+
|
|
138
|
+
timestamp = line.split()[0]
|
|
139
|
+
if timestamp.endswith("Z"):
|
|
140
|
+
timestamp = timestamp[:-1] + "+00:00"
|
|
141
|
+
try:
|
|
142
|
+
time = datetime.fromisoformat(timestamp)
|
|
143
|
+
except Exception as e:
|
|
144
|
+
raise ValueError("Invalid or missing timestamp") from e
|
|
145
|
+
|
|
146
|
+
return time
|