trainmeter 0.0.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. trainmeter/__init__.py +17 -0
  2. trainmeter/__main__.py +5 -0
  3. trainmeter/agent/__init__.py +1 -0
  4. trainmeter/agent/bootstrap/sitecustomize.py +28 -0
  5. trainmeter/agent/core.py +578 -0
  6. trainmeter/agent/sender.py +68 -0
  7. trainmeter/agent/taps.py +144 -0
  8. trainmeter/agent/wrap.py +23 -0
  9. trainmeter/catalog.py +100 -0
  10. trainmeter/cli.py +382 -0
  11. trainmeter/commands.py +99 -0
  12. trainmeter/config.py +65 -0
  13. trainmeter/doctor.py +118 -0
  14. trainmeter/emit.py +64 -0
  15. trainmeter/engine.py +621 -0
  16. trainmeter/export/__init__.py +0 -0
  17. trainmeter/export/files.py +24 -0
  18. trainmeter/export/wandb.py +87 -0
  19. trainmeter/facts.py +50 -0
  20. trainmeter/flops.py +52 -0
  21. trainmeter/metrics.py +68 -0
  22. trainmeter/passport.py +327 -0
  23. trainmeter/peaks.py +92 -0
  24. trainmeter/records.py +97 -0
  25. trainmeter/replay.py +110 -0
  26. trainmeter/report.py +47 -0
  27. trainmeter/sources/__init__.py +1 -0
  28. trainmeter/sources/gpu.py +458 -0
  29. trainmeter/sources/host.py +135 -0
  30. trainmeter/supervisor/__init__.py +1 -0
  31. trainmeter/supervisor/ingest.py +86 -0
  32. trainmeter/supervisor/launcher.py +90 -0
  33. trainmeter/supervisor/live.py +133 -0
  34. trainmeter/supervisor/store.py +118 -0
  35. trainmeter/timeline.py +102 -0
  36. trainmeter/viewer.py +273 -0
  37. trainmeter/web/__init__.py +0 -0
  38. trainmeter/web/server.py +192 -0
  39. trainmeter/web/static/app.js +571 -0
  40. trainmeter/web/static/index.html +43 -0
  41. trainmeter/web/static/style.css +157 -0
  42. trainmeter-0.0.2.dist-info/METADATA +109 -0
  43. trainmeter-0.0.2.dist-info/RECORD +46 -0
  44. trainmeter-0.0.2.dist-info/WHEEL +4 -0
  45. trainmeter-0.0.2.dist-info/entry_points.txt +3 -0
  46. trainmeter-0.0.2.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,68 @@
1
+ """One-way, non-blocking, lossy delivery of records to the supervisor's datagram socket.
2
+
3
+ Standard library only. It never blocks and never raises: a full buffer drops the record, and
4
+ after repeated failures (no supervisor any more) the sender goes quiet for good.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import os
10
+ import socket
11
+ import time
12
+ from typing import Any
13
+
14
+ from ..records import encode, make_record
15
+
16
+ SOCK_ENV = "TRAINMETER_SOCK"
17
+ MAX_FAILURES = 20
18
+
19
+
20
+ class Sender:
21
+ def __init__(self, path: str) -> None:
22
+ self._path = path
23
+ self._sock: socket.socket | None = socket.socket(socket.AF_UNIX, socket.SOCK_DGRAM)
24
+ self._sock.setblocking(False)
25
+ self._failures = 0
26
+
27
+ def send(self, kind: str, src: str, d: dict[str, Any], rank: int | None = None) -> None:
28
+ sock = self._sock
29
+ if sock is None:
30
+ return
31
+ try:
32
+ data = encode(make_record(kind, src, time.monotonic(), d, pid=os.getpid(), rank=rank))
33
+ sock.sendto(data.encode("utf-8"), self._path)
34
+ self._failures = 0
35
+ except BlockingIOError:
36
+ return # buffer full: drop
37
+ except (ConnectionRefusedError, FileNotFoundError):
38
+ self._failures += 1
39
+ if self._failures >= MAX_FAILURES:
40
+ self.close()
41
+ except (OSError, TypeError, ValueError):
42
+ return
43
+
44
+ def close(self) -> None:
45
+ sock, self._sock = self._sock, None
46
+ if sock is not None:
47
+ try:
48
+ sock.close()
49
+ except OSError:
50
+ pass
51
+
52
+
53
+ _sender: Sender | None = None
54
+ _sender_pid: int | None = None
55
+
56
+
57
+ def get_sender() -> Sender | None:
58
+ """The process's sender, or None when not running under `tm`. Re-created after a fork."""
59
+ global _sender, _sender_pid
60
+ path = os.environ.get(SOCK_ENV)
61
+ if not path:
62
+ return None
63
+ if _sender is None or _sender_pid != os.getpid():
64
+ try:
65
+ _sender, _sender_pid = Sender(path), os.getpid()
66
+ except OSError:
67
+ return None
68
+ return _sender
@@ -0,0 +1,144 @@
1
+ """Logger taps: the series a loop already sends to wandb, TensorBoard or MLflow.
2
+
3
+ Each tap wraps the tracker's public logging call, reads the numbers, reports them and forwards the
4
+ call unchanged. It reads only Python and NumPy numbers and 0-dimensional CPU tensors, so a loss
5
+ that still lives on an accelerator is skipped and never converted. The raw key is forwarded and
6
+ the engine maps it to a canonical name.
7
+
8
+ The tap never changes what the tracker receives, and a failure in the tap disables that one tap
9
+ and warns once. An exception raised by the tracker itself is the user's and propagates untouched.
10
+
11
+ Verified against wandb 0.30.0 (offline mode), torch 2.14.0 with tensorboard 2.21.0, and
12
+ mlflow-skinny 3.16.1, on CPU. Trackers that bypass these entry points (tensorboardX, a raw event
13
+ writer) are not tapped.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import importlib
19
+ import threading
20
+ from collections.abc import Callable, Iterator
21
+ from typing import Any
22
+
23
+ from ..emit import as_number
24
+ from .wrap import patch
25
+
26
+ Report = Callable[[str, float, "int | None", str], None]
27
+ Fail = Callable[[str, BaseException], None]
28
+
29
+ _busy = threading.local() # a tracker that logs through itself must not be counted twice
30
+
31
+
32
+ def _flatten(prefix: str, value: Any, sep: str) -> Iterator[tuple[str, Any]]:
33
+ if isinstance(value, dict):
34
+ for k, v in value.items():
35
+ yield from _flatten(f"{prefix}{sep}{k}" if prefix else str(k), v, sep)
36
+ else:
37
+ yield prefix, value
38
+
39
+
40
+ def _step(args: tuple, kwargs: dict, position: int, name: str) -> int | None:
41
+ value = args[position] if len(args) > position else kwargs.get(name)
42
+ return value if isinstance(value, int) and not isinstance(value, bool) else None
43
+
44
+
45
+ def _arg(args: tuple, kwargs: dict, position: int, name: str) -> Any:
46
+ return args[position] if len(args) > position else kwargs.get(name)
47
+
48
+
49
+ def _scalars(pairs: Iterator[tuple[str, Any]]) -> Iterator[tuple[str, float]]:
50
+ for key, value in pairs:
51
+ number = as_number(value)
52
+ if number is not None:
53
+ yield key, number
54
+
55
+
56
+ def _tap(
57
+ owner: Any,
58
+ name: str,
59
+ via: str,
60
+ extract: Callable[[tuple, dict], tuple[Iterator[tuple[str, float]], int | None]],
61
+ report: Report,
62
+ fail: Fail,
63
+ disabled: Callable[[str], bool],
64
+ ) -> bool:
65
+ def factory(orig: Callable) -> Callable:
66
+ def wrapper(*args: Any, **kwargs: Any) -> Any:
67
+ if getattr(_busy, "on", False):
68
+ return orig(*args, **kwargs) # called by the tracker itself: already counted
69
+ _busy.on = True
70
+ try:
71
+ if not disabled(via):
72
+ try:
73
+ pairs, step = extract(args, kwargs)
74
+ for key, value in pairs:
75
+ report(key, value, step, via)
76
+ except Exception as exc: # noqa: BLE001 - the tap must never cost a run
77
+ fail(via, exc)
78
+ return orig(*args, **kwargs)
79
+ finally:
80
+ _busy.on = False
81
+
82
+ return wrapper
83
+
84
+ return patch(owner, name, factory) is not None
85
+
86
+
87
+ # ---- one adapter per tracker --------------------------------------------------------------
88
+
89
+
90
+ def attach_wandb(report: Report, fail: Fail, disabled: Callable[[str], bool]) -> None:
91
+ run_module = importlib.import_module("wandb.sdk.wandb_run")
92
+
93
+ def extract(args: tuple, kwargs: dict):
94
+ data = _arg(args, kwargs, 1, "data")
95
+ return _scalars(_flatten("", data, ".")), _step(args, kwargs, 2, "step")
96
+
97
+ _tap(run_module.Run, "log", "wandb.log", extract, report, fail, disabled)
98
+
99
+
100
+ def attach_tensorboard(report: Report, fail: Fail, disabled: Callable[[str], bool]) -> None:
101
+ writer = importlib.import_module("torch.utils.tensorboard.writer").SummaryWriter
102
+
103
+ def add_scalar(args: tuple, kwargs: dict):
104
+ pair = [(_arg(args, kwargs, 1, "tag"), _arg(args, kwargs, 2, "scalar_value"))]
105
+ return _scalars(iter(pair)), _step(args, kwargs, 3, "global_step")
106
+
107
+ def add_scalars(args: tuple, kwargs: dict):
108
+ main = _arg(args, kwargs, 1, "main_tag")
109
+ values = _arg(args, kwargs, 2, "tag_scalar_dict") or {}
110
+ pairs = ((f"{main}/{k}", v) for k, v in values.items())
111
+ return _scalars(pairs), _step(args, kwargs, 3, "global_step")
112
+
113
+ _tap(writer, "add_scalar", "SummaryWriter.add_scalar", add_scalar, report, fail, disabled)
114
+ _tap(writer, "add_scalars", "SummaryWriter.add_scalars", add_scalars, report, fail, disabled)
115
+
116
+
117
+ def attach_mlflow(report: Report, fail: Fail, disabled: Callable[[str], bool]) -> None:
118
+ fluent = importlib.import_module("mlflow.tracking.fluent")
119
+ top = importlib.import_module("mlflow")
120
+
121
+ def log_metric(args: tuple, kwargs: dict):
122
+ pair = [(_arg(args, kwargs, 0, "key"), _arg(args, kwargs, 1, "value"))]
123
+ return _scalars(iter(pair)), _step(args, kwargs, 2, "step")
124
+
125
+ def log_metrics(args: tuple, kwargs: dict):
126
+ values = _arg(args, kwargs, 0, "metrics") or {}
127
+ return _scalars(iter(values.items())), _step(args, kwargs, 1, "step")
128
+
129
+ for name, extract, via in (
130
+ ("log_metric", log_metric, "mlflow.log_metric"),
131
+ ("log_metrics", log_metrics, "mlflow.log_metrics"),
132
+ ):
133
+ original = getattr(fluent, name)
134
+ if _tap(fluent, name, via, extract, report, fail, disabled):
135
+ if getattr(top, name, None) is original: # `mlflow.log_metric` is the same function
136
+ setattr(top, name, getattr(fluent, name))
137
+
138
+
139
+ # Module whose import triggers the adapter, and the adapter.
140
+ ADAPTERS: dict[str, Callable[[Report, Fail, Callable[[str], bool]], None]] = {
141
+ "wandb": attach_wandb,
142
+ "torch.utils.tensorboard.writer": attach_tensorboard,
143
+ "mlflow": attach_mlflow,
144
+ }
@@ -0,0 +1,23 @@
1
+ """Wrapping a callable in place, once, keeping its identity for introspection."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import functools
6
+ from collections.abc import Callable
7
+ from typing import Any
8
+
9
+
10
+ def patch(owner: Any, name: str, factory: Callable[[Callable], Callable]) -> Callable | None:
11
+ """Replace `owner.name` with `factory(original)`. Returns the original, or None if already
12
+ wrapped or missing, so a second attach is a no-op."""
13
+ original = getattr(owner, name, None)
14
+ if original is None or getattr(original, "_trainmeter_wrapped", False):
15
+ return None
16
+ wrapped = factory(original)
17
+ try:
18
+ wrapped = functools.wraps(original)(wrapped)
19
+ except (AttributeError, TypeError):
20
+ pass
21
+ wrapped._trainmeter_wrapped = True # type: ignore[attr-defined]
22
+ setattr(owner, name, wrapped)
23
+ return original
trainmeter/catalog.py ADDED
@@ -0,0 +1,100 @@
1
+ """Canonical names for the series a training loop reports, and the aliases that map to them.
2
+
3
+ The agent forwards the raw key the loop used. The engine maps it here, so a wrong guess never
4
+ changes what was recorded.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ ALIASES: dict[str, tuple[str, ...]] = {
10
+ "train_loss": ("train_loss", "train/loss", "loss/train", "training_loss"),
11
+ "val_loss": (
12
+ "val_loss",
13
+ "val/loss",
14
+ "loss/val",
15
+ "eval_loss",
16
+ "eval/loss",
17
+ "valid_loss",
18
+ "validation_loss",
19
+ ),
20
+ "lr": ("lr", "learning_rate", "train/lr", "train/learning_rate"),
21
+ "grad_norm": ("grad_norm", "train/grad_norm", "gradient_norm"),
22
+ }
23
+
24
+ # A bare `loss` means train_loss, but only while no explicit train key has been seen.
25
+ BARE_LOSS = "loss"
26
+
27
+ _BY_KEY = {alias.lower(): name for name, aliases in ALIASES.items() for alias in aliases}
28
+
29
+
30
+ def canonical(key: str) -> str | None:
31
+ """The canonical series name for a raw key, or None if no alias matches."""
32
+ return _BY_KEY.get(key.lower())
33
+
34
+
35
+ # name, label, unit, group, format. The dashboard renders any series from this list, so a new
36
+ # metric needs no page change. Formats: percent (a fraction shown as %), si (12.3 G), float,
37
+ # sci (1.2e-4), seconds, bytes.
38
+ METRICS: tuple[tuple[str, str, str, str, str], ...] = (
39
+ ("train_loss", "Train loss", "", "training", "float"),
40
+ ("val_loss", "Validation loss", "", "training", "float"),
41
+ ("lr", "Learning rate", "", "training", "sci"),
42
+ ("grad_norm", "Grad norm", "", "training", "float"),
43
+ ("tokens_per_s", "Tokens per second", "tok/s", "training", "si"),
44
+ ("step_time_s", "Step time", "s", "training", "seconds"),
45
+ ("progress", "Progress", "", "training", "percent"),
46
+ ("flops_spent", "FLOPs spent (modeled)", "FLOPs", "compute", "si"),
47
+ ("flops_observed", "FLOPs observed (counters)", "FLOPs", "compute", "si"),
48
+ ("mfu", "MFU", "", "efficiency", "percent"),
49
+ ("ofu", "OFU", "", "efficiency", "percent"),
50
+ ("mfu_ofu_gap", "OFU minus MFU", "", "efficiency", "percent"),
51
+ ("arithmetic_intensity", "Arithmetic intensity", "FLOP/byte", "efficiency", "si"),
52
+ ("goodput", "Goodput", "", "time", "percent"),
53
+ ("host_cpu_cores", "Host CPU, process tree", "cores", "host", "float"),
54
+ ("host_rss_b", "Host memory (RSS), process tree", "B", "host", "bytes"),
55
+ ("gpu_util", "GPU-Util", "", "ladder", "percent"),
56
+ ("sm_active", "SM active", "", "ladder", "percent"),
57
+ ("tensor_active", "Tensor active", "", "ladder", "percent"),
58
+ )
59
+
60
+ # Per-GPU series are named `gpu.<index>.<field>`.
61
+ GPU_FIELDS: tuple[tuple[str, str, str, str], ...] = (
62
+ ("gpu_util", "GPU-Util", "", "percent"),
63
+ ("sm_active", "SM active", "", "percent"),
64
+ ("tensor_active", "Tensor active", "", "percent"),
65
+ ("dram_active", "DRAM active", "", "percent"),
66
+ ("power_w", "Power", "W", "float"),
67
+ ("sm_clock_mhz", "SM clock", "MHz", "float"),
68
+ ("mem_clock_mhz", "Memory clock", "MHz", "float"),
69
+ ("mem_used_b", "Memory used", "B", "bytes"),
70
+ ("temp_c", "Temperature", "C", "float"),
71
+ ("nvlink_rx_bps", "NVLink receive", "B/s", "si"),
72
+ ("nvlink_tx_bps", "NVLink transmit", "B/s", "si"),
73
+ )
74
+
75
+
76
+ def describe() -> dict:
77
+ """The body of `GET /api/catalog`."""
78
+ return {
79
+ "metrics": [
80
+ {
81
+ "name": name,
82
+ "label": label,
83
+ "unit": unit,
84
+ "group": group,
85
+ "format": fmt,
86
+ "aliases": list(ALIASES.get(name, ())),
87
+ }
88
+ for name, label, unit, group, fmt in METRICS
89
+ ],
90
+ "gpu_fields": [
91
+ {"name": name, "label": label, "unit": unit, "format": fmt}
92
+ for name, label, unit, fmt in GPU_FIELDS
93
+ ],
94
+ "x_axes": [
95
+ {"name": "step", "label": "Step"},
96
+ {"name": "tokens", "label": "Tokens"},
97
+ {"name": "flops", "label": "FLOPs spent (modeled)"},
98
+ {"name": "time", "label": "Wall time"},
99
+ ],
100
+ }