trainmeter 0.0.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- trainmeter/__init__.py +17 -0
- trainmeter/__main__.py +5 -0
- trainmeter/agent/__init__.py +1 -0
- trainmeter/agent/bootstrap/sitecustomize.py +28 -0
- trainmeter/agent/core.py +578 -0
- trainmeter/agent/sender.py +68 -0
- trainmeter/agent/taps.py +144 -0
- trainmeter/agent/wrap.py +23 -0
- trainmeter/catalog.py +100 -0
- trainmeter/cli.py +382 -0
- trainmeter/commands.py +99 -0
- trainmeter/config.py +65 -0
- trainmeter/doctor.py +118 -0
- trainmeter/emit.py +64 -0
- trainmeter/engine.py +621 -0
- trainmeter/export/__init__.py +0 -0
- trainmeter/export/files.py +24 -0
- trainmeter/export/wandb.py +87 -0
- trainmeter/facts.py +50 -0
- trainmeter/flops.py +52 -0
- trainmeter/metrics.py +68 -0
- trainmeter/passport.py +327 -0
- trainmeter/peaks.py +92 -0
- trainmeter/records.py +97 -0
- trainmeter/replay.py +110 -0
- trainmeter/report.py +47 -0
- trainmeter/sources/__init__.py +1 -0
- trainmeter/sources/gpu.py +458 -0
- trainmeter/sources/host.py +135 -0
- trainmeter/supervisor/__init__.py +1 -0
- trainmeter/supervisor/ingest.py +86 -0
- trainmeter/supervisor/launcher.py +90 -0
- trainmeter/supervisor/live.py +133 -0
- trainmeter/supervisor/store.py +118 -0
- trainmeter/timeline.py +102 -0
- trainmeter/viewer.py +273 -0
- trainmeter/web/__init__.py +0 -0
- trainmeter/web/server.py +192 -0
- trainmeter/web/static/app.js +571 -0
- trainmeter/web/static/index.html +43 -0
- trainmeter/web/static/style.css +157 -0
- trainmeter-0.0.2.dist-info/METADATA +109 -0
- trainmeter-0.0.2.dist-info/RECORD +46 -0
- trainmeter-0.0.2.dist-info/WHEEL +4 -0
- trainmeter-0.0.2.dist-info/entry_points.txt +3 -0
- trainmeter-0.0.2.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""One-way, non-blocking, lossy delivery of records to the supervisor's datagram socket.
|
|
2
|
+
|
|
3
|
+
Standard library only. It never blocks and never raises: a full buffer drops the record, and
|
|
4
|
+
after repeated failures (no supervisor any more) the sender goes quiet for good.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import os
|
|
10
|
+
import socket
|
|
11
|
+
import time
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from ..records import encode, make_record
|
|
15
|
+
|
|
16
|
+
SOCK_ENV = "TRAINMETER_SOCK"
|
|
17
|
+
MAX_FAILURES = 20
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class Sender:
|
|
21
|
+
def __init__(self, path: str) -> None:
|
|
22
|
+
self._path = path
|
|
23
|
+
self._sock: socket.socket | None = socket.socket(socket.AF_UNIX, socket.SOCK_DGRAM)
|
|
24
|
+
self._sock.setblocking(False)
|
|
25
|
+
self._failures = 0
|
|
26
|
+
|
|
27
|
+
def send(self, kind: str, src: str, d: dict[str, Any], rank: int | None = None) -> None:
|
|
28
|
+
sock = self._sock
|
|
29
|
+
if sock is None:
|
|
30
|
+
return
|
|
31
|
+
try:
|
|
32
|
+
data = encode(make_record(kind, src, time.monotonic(), d, pid=os.getpid(), rank=rank))
|
|
33
|
+
sock.sendto(data.encode("utf-8"), self._path)
|
|
34
|
+
self._failures = 0
|
|
35
|
+
except BlockingIOError:
|
|
36
|
+
return # buffer full: drop
|
|
37
|
+
except (ConnectionRefusedError, FileNotFoundError):
|
|
38
|
+
self._failures += 1
|
|
39
|
+
if self._failures >= MAX_FAILURES:
|
|
40
|
+
self.close()
|
|
41
|
+
except (OSError, TypeError, ValueError):
|
|
42
|
+
return
|
|
43
|
+
|
|
44
|
+
def close(self) -> None:
|
|
45
|
+
sock, self._sock = self._sock, None
|
|
46
|
+
if sock is not None:
|
|
47
|
+
try:
|
|
48
|
+
sock.close()
|
|
49
|
+
except OSError:
|
|
50
|
+
pass
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
_sender: Sender | None = None
|
|
54
|
+
_sender_pid: int | None = None
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def get_sender() -> Sender | None:
|
|
58
|
+
"""The process's sender, or None when not running under `tm`. Re-created after a fork."""
|
|
59
|
+
global _sender, _sender_pid
|
|
60
|
+
path = os.environ.get(SOCK_ENV)
|
|
61
|
+
if not path:
|
|
62
|
+
return None
|
|
63
|
+
if _sender is None or _sender_pid != os.getpid():
|
|
64
|
+
try:
|
|
65
|
+
_sender, _sender_pid = Sender(path), os.getpid()
|
|
66
|
+
except OSError:
|
|
67
|
+
return None
|
|
68
|
+
return _sender
|
trainmeter/agent/taps.py
ADDED
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
"""Logger taps: the series a loop already sends to wandb, TensorBoard or MLflow.
|
|
2
|
+
|
|
3
|
+
Each tap wraps the tracker's public logging call, reads the numbers, reports them and forwards the
|
|
4
|
+
call unchanged. It reads only Python and NumPy numbers and 0-dimensional CPU tensors, so a loss
|
|
5
|
+
that still lives on an accelerator is skipped and never converted. The raw key is forwarded and
|
|
6
|
+
the engine maps it to a canonical name.
|
|
7
|
+
|
|
8
|
+
The tap never changes what the tracker receives, and a failure in the tap disables that one tap
|
|
9
|
+
and warns once. An exception raised by the tracker itself is the user's and propagates untouched.
|
|
10
|
+
|
|
11
|
+
Verified against wandb 0.30.0 (offline mode), torch 2.14.0 with tensorboard 2.21.0, and
|
|
12
|
+
mlflow-skinny 3.16.1, on CPU. Trackers that bypass these entry points (tensorboardX, a raw event
|
|
13
|
+
writer) are not tapped.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import importlib
|
|
19
|
+
import threading
|
|
20
|
+
from collections.abc import Callable, Iterator
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
from ..emit import as_number
|
|
24
|
+
from .wrap import patch
|
|
25
|
+
|
|
26
|
+
Report = Callable[[str, float, "int | None", str], None]
|
|
27
|
+
Fail = Callable[[str, BaseException], None]
|
|
28
|
+
|
|
29
|
+
_busy = threading.local() # a tracker that logs through itself must not be counted twice
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _flatten(prefix: str, value: Any, sep: str) -> Iterator[tuple[str, Any]]:
|
|
33
|
+
if isinstance(value, dict):
|
|
34
|
+
for k, v in value.items():
|
|
35
|
+
yield from _flatten(f"{prefix}{sep}{k}" if prefix else str(k), v, sep)
|
|
36
|
+
else:
|
|
37
|
+
yield prefix, value
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _step(args: tuple, kwargs: dict, position: int, name: str) -> int | None:
|
|
41
|
+
value = args[position] if len(args) > position else kwargs.get(name)
|
|
42
|
+
return value if isinstance(value, int) and not isinstance(value, bool) else None
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _arg(args: tuple, kwargs: dict, position: int, name: str) -> Any:
|
|
46
|
+
return args[position] if len(args) > position else kwargs.get(name)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _scalars(pairs: Iterator[tuple[str, Any]]) -> Iterator[tuple[str, float]]:
|
|
50
|
+
for key, value in pairs:
|
|
51
|
+
number = as_number(value)
|
|
52
|
+
if number is not None:
|
|
53
|
+
yield key, number
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _tap(
|
|
57
|
+
owner: Any,
|
|
58
|
+
name: str,
|
|
59
|
+
via: str,
|
|
60
|
+
extract: Callable[[tuple, dict], tuple[Iterator[tuple[str, float]], int | None]],
|
|
61
|
+
report: Report,
|
|
62
|
+
fail: Fail,
|
|
63
|
+
disabled: Callable[[str], bool],
|
|
64
|
+
) -> bool:
|
|
65
|
+
def factory(orig: Callable) -> Callable:
|
|
66
|
+
def wrapper(*args: Any, **kwargs: Any) -> Any:
|
|
67
|
+
if getattr(_busy, "on", False):
|
|
68
|
+
return orig(*args, **kwargs) # called by the tracker itself: already counted
|
|
69
|
+
_busy.on = True
|
|
70
|
+
try:
|
|
71
|
+
if not disabled(via):
|
|
72
|
+
try:
|
|
73
|
+
pairs, step = extract(args, kwargs)
|
|
74
|
+
for key, value in pairs:
|
|
75
|
+
report(key, value, step, via)
|
|
76
|
+
except Exception as exc: # noqa: BLE001 - the tap must never cost a run
|
|
77
|
+
fail(via, exc)
|
|
78
|
+
return orig(*args, **kwargs)
|
|
79
|
+
finally:
|
|
80
|
+
_busy.on = False
|
|
81
|
+
|
|
82
|
+
return wrapper
|
|
83
|
+
|
|
84
|
+
return patch(owner, name, factory) is not None
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
# ---- one adapter per tracker --------------------------------------------------------------
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def attach_wandb(report: Report, fail: Fail, disabled: Callable[[str], bool]) -> None:
|
|
91
|
+
run_module = importlib.import_module("wandb.sdk.wandb_run")
|
|
92
|
+
|
|
93
|
+
def extract(args: tuple, kwargs: dict):
|
|
94
|
+
data = _arg(args, kwargs, 1, "data")
|
|
95
|
+
return _scalars(_flatten("", data, ".")), _step(args, kwargs, 2, "step")
|
|
96
|
+
|
|
97
|
+
_tap(run_module.Run, "log", "wandb.log", extract, report, fail, disabled)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def attach_tensorboard(report: Report, fail: Fail, disabled: Callable[[str], bool]) -> None:
|
|
101
|
+
writer = importlib.import_module("torch.utils.tensorboard.writer").SummaryWriter
|
|
102
|
+
|
|
103
|
+
def add_scalar(args: tuple, kwargs: dict):
|
|
104
|
+
pair = [(_arg(args, kwargs, 1, "tag"), _arg(args, kwargs, 2, "scalar_value"))]
|
|
105
|
+
return _scalars(iter(pair)), _step(args, kwargs, 3, "global_step")
|
|
106
|
+
|
|
107
|
+
def add_scalars(args: tuple, kwargs: dict):
|
|
108
|
+
main = _arg(args, kwargs, 1, "main_tag")
|
|
109
|
+
values = _arg(args, kwargs, 2, "tag_scalar_dict") or {}
|
|
110
|
+
pairs = ((f"{main}/{k}", v) for k, v in values.items())
|
|
111
|
+
return _scalars(pairs), _step(args, kwargs, 3, "global_step")
|
|
112
|
+
|
|
113
|
+
_tap(writer, "add_scalar", "SummaryWriter.add_scalar", add_scalar, report, fail, disabled)
|
|
114
|
+
_tap(writer, "add_scalars", "SummaryWriter.add_scalars", add_scalars, report, fail, disabled)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def attach_mlflow(report: Report, fail: Fail, disabled: Callable[[str], bool]) -> None:
|
|
118
|
+
fluent = importlib.import_module("mlflow.tracking.fluent")
|
|
119
|
+
top = importlib.import_module("mlflow")
|
|
120
|
+
|
|
121
|
+
def log_metric(args: tuple, kwargs: dict):
|
|
122
|
+
pair = [(_arg(args, kwargs, 0, "key"), _arg(args, kwargs, 1, "value"))]
|
|
123
|
+
return _scalars(iter(pair)), _step(args, kwargs, 2, "step")
|
|
124
|
+
|
|
125
|
+
def log_metrics(args: tuple, kwargs: dict):
|
|
126
|
+
values = _arg(args, kwargs, 0, "metrics") or {}
|
|
127
|
+
return _scalars(iter(values.items())), _step(args, kwargs, 1, "step")
|
|
128
|
+
|
|
129
|
+
for name, extract, via in (
|
|
130
|
+
("log_metric", log_metric, "mlflow.log_metric"),
|
|
131
|
+
("log_metrics", log_metrics, "mlflow.log_metrics"),
|
|
132
|
+
):
|
|
133
|
+
original = getattr(fluent, name)
|
|
134
|
+
if _tap(fluent, name, via, extract, report, fail, disabled):
|
|
135
|
+
if getattr(top, name, None) is original: # `mlflow.log_metric` is the same function
|
|
136
|
+
setattr(top, name, getattr(fluent, name))
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
# Module whose import triggers the adapter, and the adapter.
|
|
140
|
+
ADAPTERS: dict[str, Callable[[Report, Fail, Callable[[str], bool]], None]] = {
|
|
141
|
+
"wandb": attach_wandb,
|
|
142
|
+
"torch.utils.tensorboard.writer": attach_tensorboard,
|
|
143
|
+
"mlflow": attach_mlflow,
|
|
144
|
+
}
|
trainmeter/agent/wrap.py
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""Wrapping a callable in place, once, keeping its identity for introspection."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import functools
|
|
6
|
+
from collections.abc import Callable
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def patch(owner: Any, name: str, factory: Callable[[Callable], Callable]) -> Callable | None:
|
|
11
|
+
"""Replace `owner.name` with `factory(original)`. Returns the original, or None if already
|
|
12
|
+
wrapped or missing, so a second attach is a no-op."""
|
|
13
|
+
original = getattr(owner, name, None)
|
|
14
|
+
if original is None or getattr(original, "_trainmeter_wrapped", False):
|
|
15
|
+
return None
|
|
16
|
+
wrapped = factory(original)
|
|
17
|
+
try:
|
|
18
|
+
wrapped = functools.wraps(original)(wrapped)
|
|
19
|
+
except (AttributeError, TypeError):
|
|
20
|
+
pass
|
|
21
|
+
wrapped._trainmeter_wrapped = True # type: ignore[attr-defined]
|
|
22
|
+
setattr(owner, name, wrapped)
|
|
23
|
+
return original
|
trainmeter/catalog.py
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""Canonical names for the series a training loop reports, and the aliases that map to them.
|
|
2
|
+
|
|
3
|
+
The agent forwards the raw key the loop used. The engine maps it here, so a wrong guess never
|
|
4
|
+
changes what was recorded.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
ALIASES: dict[str, tuple[str, ...]] = {
|
|
10
|
+
"train_loss": ("train_loss", "train/loss", "loss/train", "training_loss"),
|
|
11
|
+
"val_loss": (
|
|
12
|
+
"val_loss",
|
|
13
|
+
"val/loss",
|
|
14
|
+
"loss/val",
|
|
15
|
+
"eval_loss",
|
|
16
|
+
"eval/loss",
|
|
17
|
+
"valid_loss",
|
|
18
|
+
"validation_loss",
|
|
19
|
+
),
|
|
20
|
+
"lr": ("lr", "learning_rate", "train/lr", "train/learning_rate"),
|
|
21
|
+
"grad_norm": ("grad_norm", "train/grad_norm", "gradient_norm"),
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
# A bare `loss` means train_loss, but only while no explicit train key has been seen.
|
|
25
|
+
BARE_LOSS = "loss"
|
|
26
|
+
|
|
27
|
+
_BY_KEY = {alias.lower(): name for name, aliases in ALIASES.items() for alias in aliases}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def canonical(key: str) -> str | None:
|
|
31
|
+
"""The canonical series name for a raw key, or None if no alias matches."""
|
|
32
|
+
return _BY_KEY.get(key.lower())
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
# name, label, unit, group, format. The dashboard renders any series from this list, so a new
|
|
36
|
+
# metric needs no page change. Formats: percent (a fraction shown as %), si (12.3 G), float,
|
|
37
|
+
# sci (1.2e-4), seconds, bytes.
|
|
38
|
+
METRICS: tuple[tuple[str, str, str, str, str], ...] = (
|
|
39
|
+
("train_loss", "Train loss", "", "training", "float"),
|
|
40
|
+
("val_loss", "Validation loss", "", "training", "float"),
|
|
41
|
+
("lr", "Learning rate", "", "training", "sci"),
|
|
42
|
+
("grad_norm", "Grad norm", "", "training", "float"),
|
|
43
|
+
("tokens_per_s", "Tokens per second", "tok/s", "training", "si"),
|
|
44
|
+
("step_time_s", "Step time", "s", "training", "seconds"),
|
|
45
|
+
("progress", "Progress", "", "training", "percent"),
|
|
46
|
+
("flops_spent", "FLOPs spent (modeled)", "FLOPs", "compute", "si"),
|
|
47
|
+
("flops_observed", "FLOPs observed (counters)", "FLOPs", "compute", "si"),
|
|
48
|
+
("mfu", "MFU", "", "efficiency", "percent"),
|
|
49
|
+
("ofu", "OFU", "", "efficiency", "percent"),
|
|
50
|
+
("mfu_ofu_gap", "OFU minus MFU", "", "efficiency", "percent"),
|
|
51
|
+
("arithmetic_intensity", "Arithmetic intensity", "FLOP/byte", "efficiency", "si"),
|
|
52
|
+
("goodput", "Goodput", "", "time", "percent"),
|
|
53
|
+
("host_cpu_cores", "Host CPU, process tree", "cores", "host", "float"),
|
|
54
|
+
("host_rss_b", "Host memory (RSS), process tree", "B", "host", "bytes"),
|
|
55
|
+
("gpu_util", "GPU-Util", "", "ladder", "percent"),
|
|
56
|
+
("sm_active", "SM active", "", "ladder", "percent"),
|
|
57
|
+
("tensor_active", "Tensor active", "", "ladder", "percent"),
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
# Per-GPU series are named `gpu.<index>.<field>`.
|
|
61
|
+
GPU_FIELDS: tuple[tuple[str, str, str, str], ...] = (
|
|
62
|
+
("gpu_util", "GPU-Util", "", "percent"),
|
|
63
|
+
("sm_active", "SM active", "", "percent"),
|
|
64
|
+
("tensor_active", "Tensor active", "", "percent"),
|
|
65
|
+
("dram_active", "DRAM active", "", "percent"),
|
|
66
|
+
("power_w", "Power", "W", "float"),
|
|
67
|
+
("sm_clock_mhz", "SM clock", "MHz", "float"),
|
|
68
|
+
("mem_clock_mhz", "Memory clock", "MHz", "float"),
|
|
69
|
+
("mem_used_b", "Memory used", "B", "bytes"),
|
|
70
|
+
("temp_c", "Temperature", "C", "float"),
|
|
71
|
+
("nvlink_rx_bps", "NVLink receive", "B/s", "si"),
|
|
72
|
+
("nvlink_tx_bps", "NVLink transmit", "B/s", "si"),
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def describe() -> dict:
|
|
77
|
+
"""The body of `GET /api/catalog`."""
|
|
78
|
+
return {
|
|
79
|
+
"metrics": [
|
|
80
|
+
{
|
|
81
|
+
"name": name,
|
|
82
|
+
"label": label,
|
|
83
|
+
"unit": unit,
|
|
84
|
+
"group": group,
|
|
85
|
+
"format": fmt,
|
|
86
|
+
"aliases": list(ALIASES.get(name, ())),
|
|
87
|
+
}
|
|
88
|
+
for name, label, unit, group, fmt in METRICS
|
|
89
|
+
],
|
|
90
|
+
"gpu_fields": [
|
|
91
|
+
{"name": name, "label": label, "unit": unit, "format": fmt}
|
|
92
|
+
for name, label, unit, fmt in GPU_FIELDS
|
|
93
|
+
],
|
|
94
|
+
"x_axes": [
|
|
95
|
+
{"name": "step", "label": "Step"},
|
|
96
|
+
{"name": "tokens", "label": "Tokens"},
|
|
97
|
+
{"name": "flops", "label": "FLOPs spent (modeled)"},
|
|
98
|
+
{"name": "time", "label": "Wall time"},
|
|
99
|
+
],
|
|
100
|
+
}
|