hyperun 0.2.1__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hyperun-0.2.1 → hyperun-0.2.2}/PKG-INFO +1 -1
- {hyperun-0.2.1 → hyperun-0.2.2}/hyperun/cli.py +208 -10
- {hyperun-0.2.1 → hyperun-0.2.2}/hyperun.egg-info/PKG-INFO +1 -1
- {hyperun-0.2.1 → hyperun-0.2.2}/pyproject.toml +1 -1
- {hyperun-0.2.1 → hyperun-0.2.2}/tests/test_cli.py +271 -0
- {hyperun-0.2.1 → hyperun-0.2.2}/LICENSE +0 -0
- {hyperun-0.2.1 → hyperun-0.2.2}/README.md +0 -0
- {hyperun-0.2.1 → hyperun-0.2.2}/hyperun/__init__.py +0 -0
- {hyperun-0.2.1 → hyperun-0.2.2}/hyperun/browser_login.py +0 -0
- {hyperun-0.2.1 → hyperun-0.2.2}/hyperun/client.py +0 -0
- {hyperun-0.2.1 → hyperun-0.2.2}/hyperun/config.py +0 -0
- {hyperun-0.2.1 → hyperun-0.2.2}/hyperun.egg-info/SOURCES.txt +0 -0
- {hyperun-0.2.1 → hyperun-0.2.2}/hyperun.egg-info/dependency_links.txt +0 -0
- {hyperun-0.2.1 → hyperun-0.2.2}/hyperun.egg-info/entry_points.txt +0 -0
- {hyperun-0.2.1 → hyperun-0.2.2}/hyperun.egg-info/requires.txt +0 -0
- {hyperun-0.2.1 → hyperun-0.2.2}/hyperun.egg-info/top_level.txt +0 -0
- {hyperun-0.2.1 → hyperun-0.2.2}/setup.cfg +0 -0
- {hyperun-0.2.1 → hyperun-0.2.2}/tests/test_browser_login.py +0 -0
- {hyperun-0.2.1 → hyperun-0.2.2}/tests/test_client.py +0 -0
- {hyperun-0.2.1 → hyperun-0.2.2}/tests/test_config.py +0 -0
|
@@ -50,6 +50,7 @@ from __future__ import annotations
|
|
|
50
50
|
|
|
51
51
|
import argparse
|
|
52
52
|
import base64
|
|
53
|
+
import datetime
|
|
53
54
|
import getpass
|
|
54
55
|
import json
|
|
55
56
|
import time
|
|
@@ -387,8 +388,13 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
387
388
|
)
|
|
388
389
|
watch.add_argument("job_id")
|
|
389
390
|
watch.add_argument(
|
|
390
|
-
|
|
391
|
-
|
|
391
|
+
# default=None, not 3600, and the difference is load-bearing: it is how
|
|
392
|
+
# `cmd_watch` tells "the caller wants an hour" from "the caller said
|
|
393
|
+
# nothing", and only the second one may be widened to reach a finished
|
|
394
|
+
# job's readings. 604800 is the server's cap, not a number chosen here.
|
|
395
|
+
"--window", type=int, default=None, metavar="SECONDS",
|
|
396
|
+
help="how far back to read (default 3600, max 604800). A finished job "
|
|
397
|
+
"is widened automatically to reach its own readings.",
|
|
392
398
|
)
|
|
393
399
|
add_json_flag(watch)
|
|
394
400
|
|
|
@@ -1088,13 +1094,128 @@ def bar(percent: float, width: int = 20) -> str:
|
|
|
1088
1094
|
return "#" * filled + "-" * (width - filled)
|
|
1089
1095
|
|
|
1090
1096
|
|
|
1097
|
+
# DDPSRUN-WATCH-TERMINAL. The three phases a job never leaves. They are the
|
|
1098
|
+
# controller's own words: the phase enum in api/v1alpha1/pacsjob_types.go
|
|
1099
|
+
# carries Pending / Starting / Running / Recovering / Succeeded / Failed /
|
|
1100
|
+
# Compared and nothing else, and the screen keeps the same three in
|
|
1101
|
+
# `const TERMINAL` (ui/app.js). Recovering is NOT here -- the machine was
|
|
1102
|
+
# reclaimed and the job is being restarted, so the run is still going.
|
|
1103
|
+
TERMINAL_PHASES = ("Succeeded", "Failed", "Compared")
|
|
1104
|
+
|
|
1105
|
+
# How far back `watch` reads when the caller says nothing, and the furthest the
|
|
1106
|
+
# server will let it reach. The cap is the server's own: `window_seconds` on
|
|
1107
|
+
# `GET /v1/jobs/{id}/metrics` is declared `ge=60, le=604800`
|
|
1108
|
+
# (server/ddpsrun_server/main.py). The `--window` help said 86400 here, which was
|
|
1109
|
+
# simply wrong -- one day, against the server's seven.
|
|
1110
|
+
DEFAULT_WATCH_WINDOW = 3600
|
|
1111
|
+
MAX_WATCH_WINDOW = 604800
|
|
1112
|
+
|
|
1113
|
+
|
|
1114
|
+
def _window_reaching_back_to_start(job: dict[str, Any]) -> int:
|
|
1115
|
+
"""How wide a window must be to contain a FINISHED job's readings.
|
|
1116
|
+
|
|
1117
|
+
★ WHY THIS EXISTS. Nothing is stored: `watch` re-reads the job's own log, and
|
|
1118
|
+
the window is measured back from NOW. A running job's readings are therefore
|
|
1119
|
+
always inside the default hour. A job that finished five hours ago has NONE
|
|
1120
|
+
of its readings in the last hour, so `hyperun watch` on it printed the phase
|
|
1121
|
+
and no numbers -- and the phase is the one thing the reader already knew.
|
|
1122
|
+
|
|
1123
|
+
The fix is not a larger default, which would make every `watch` on a running
|
|
1124
|
+
job re-read a week of log. It is to widen only when the job is over AND the
|
|
1125
|
+
caller did not pick a window.
|
|
1126
|
+
|
|
1127
|
+
Args:
|
|
1128
|
+
job: the body of `GET /v1/jobs/{id}`.
|
|
1129
|
+
|
|
1130
|
+
Returns:
|
|
1131
|
+
Seconds, or 0 when the job is not finished or carries no start time, in
|
|
1132
|
+
which case the caller leaves the window as it was.
|
|
1133
|
+
"""
|
|
1134
|
+
if job.get("phase") not in TERMINAL_PHASES:
|
|
1135
|
+
return 0
|
|
1136
|
+
started = job.get("started_at") or job.get("created_at")
|
|
1137
|
+
if not started:
|
|
1138
|
+
return 0
|
|
1139
|
+
try:
|
|
1140
|
+
began = datetime.datetime.fromisoformat(started.replace("Z", "+00:00"))
|
|
1141
|
+
except ValueError:
|
|
1142
|
+
return 0
|
|
1143
|
+
elapsed = (datetime.datetime.now(datetime.timezone.utc) - began).total_seconds()
|
|
1144
|
+
# A minute of headroom so the FIRST reading falls inside the window instead
|
|
1145
|
+
# of exactly on its edge, and the server's own cap over the top.
|
|
1146
|
+
return min(MAX_WATCH_WINDOW, max(60, int(elapsed) + 60))
|
|
1147
|
+
|
|
1148
|
+
|
|
1091
1149
|
def cmd_watch(args: argparse.Namespace) -> int:
|
|
1092
|
-
"""Print a job's GPU usage and how far the training has got.
|
|
1093
|
-
|
|
1150
|
+
"""Print a job's GPU usage and how far the training has got.
|
|
1151
|
+
|
|
1152
|
+
ONE BLOCK PER CARD, because a job can rent several. baseline-c rents four
|
|
1153
|
+
A100s in one pod, and until today this printed `latest_gpu`, which is card 0
|
|
1154
|
+
alone -- three quarters of a $44 run was invisible from the CLI. The screen
|
|
1155
|
+
had the same defect and was fixed on 2026-09-08. `cards` is the server's
|
|
1156
|
+
per-card answer (CardMetricsView in server/ddpsrun_server/models.py); a
|
|
1157
|
+
server too old to send it answers with an empty list, and then the
|
|
1158
|
+
single-card block below is the whole story, exactly as before.
|
|
1159
|
+
|
|
1160
|
+
WHY THIS ASKS FOR THE PHASE AS WELL AS THE METRICS. A finished job's last
|
|
1161
|
+
reading is the idle card in the seconds before teardown -- 0%, 0 MiB -- so
|
|
1162
|
+
printing it under "GPU util" tells a reader the run was idle when it was
|
|
1163
|
+
not. Leading with the peak instead requires knowing the job is over, and
|
|
1164
|
+
THE METRICS ANSWER DOES NOT SAY: MetricsResponse carries latest_gpu,
|
|
1165
|
+
gpu_series, peak_gpu, the two utilisation figures, sample_count, cards,
|
|
1166
|
+
progress, window_seconds and note, and no phase at all
|
|
1167
|
+
(server/ddpsrun_server/models.py). Guessing from the readings fails in both
|
|
1168
|
+
directions -- a job whose framework has not allocated yet also reads 0 MiB,
|
|
1169
|
+
and a job killed mid-step leaves a final reading that is nowhere near 0 --
|
|
1170
|
+
so this asks the server instead.
|
|
1171
|
+
|
|
1172
|
+
WHAT THAT COSTS: one extra `GET /v1/jobs/{id}` per typed `watch`. `watch`
|
|
1173
|
+
prints once and exits rather than polling, so it is one extra request per
|
|
1174
|
+
command, not one per second, and `--json` pays it only when the first window
|
|
1175
|
+
came back empty. The screen pays the same price -- it reads the job and hands
|
|
1176
|
+
it to drawMetrics (ui/app.js).
|
|
1177
|
+
"""
|
|
1178
|
+
client = client_from_config()
|
|
1179
|
+
window = DEFAULT_WATCH_WINDOW if args.window is None else args.window
|
|
1180
|
+
result = client.metrics(args.job_id, window)
|
|
1181
|
+
|
|
1182
|
+
# ★ AN EMPTY ANSWER ON A FINISHED JOB IS THE WINDOW, NOT THE JOB. The window
|
|
1183
|
+
# is measured back from now, so a run that ended hours ago has every reading
|
|
1184
|
+
# outside the default hour and `watch` printed nothing. Widening is done
|
|
1185
|
+
# ONLY here: the caller did not choose a window, and the first one came back
|
|
1186
|
+
# with no readings at all. A running job never reaches this branch, so the
|
|
1187
|
+
# usual `watch` is still exactly one metrics request.
|
|
1188
|
+
#
|
|
1189
|
+
# "No readings" means BOTH are empty. `cards` is the per-card answer and
|
|
1190
|
+
# `latest_gpu` is card 0; a multi-card job fills `cards`, while a server too
|
|
1191
|
+
# old to send it fills only `latest_gpu`. Testing one alone would make
|
|
1192
|
+
# `watch --json` on a four-card job buy a second request it does not need.
|
|
1193
|
+
job: dict[str, Any] | None = None
|
|
1194
|
+
if args.window is None and not result.get("latest_gpu") and not result.get("cards"):
|
|
1195
|
+
try:
|
|
1196
|
+
job = client.status(args.job_id)
|
|
1197
|
+
except ServerError:
|
|
1198
|
+
job = None
|
|
1199
|
+
wider = _window_reaching_back_to_start(job) if job else 0
|
|
1200
|
+
if wider > window:
|
|
1201
|
+
result = client.metrics(args.job_id, wider)
|
|
1202
|
+
window = wider
|
|
1203
|
+
|
|
1094
1204
|
if args.json:
|
|
1205
|
+
# The server's answer, verbatim. It already contains `cards`, so nothing
|
|
1206
|
+
# here needs the phase.
|
|
1095
1207
|
print(json.dumps(result, indent=2, ensure_ascii=False))
|
|
1096
1208
|
return EXIT_OK
|
|
1097
1209
|
|
|
1210
|
+
# A phase lookup that fails must not lose the GPU numbers this command
|
|
1211
|
+
# exists to print, so it falls back to the running layout. The likely cause
|
|
1212
|
+
# is the job being deleted between the two calls, and the running layout is
|
|
1213
|
+
# the behaviour this command had until today rather than a wrong number.
|
|
1214
|
+
try:
|
|
1215
|
+
done = (job or client.status(args.job_id)).get("phase") in TERMINAL_PHASES
|
|
1216
|
+
except ServerError:
|
|
1217
|
+
done = False
|
|
1218
|
+
|
|
1098
1219
|
progress = result.get("progress")
|
|
1099
1220
|
if progress:
|
|
1100
1221
|
print(f" training {bar(progress['percent'])} "
|
|
@@ -1109,12 +1230,40 @@ def cmd_watch(args: argparse.Namespace) -> int:
|
|
|
1109
1230
|
|
|
1110
1231
|
gpu = result.get("latest_gpu")
|
|
1111
1232
|
if gpu:
|
|
1112
|
-
|
|
1113
|
-
|
|
1114
|
-
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
|
|
1233
|
+
# The reading with the most MEMORY in use. A server too old to send it
|
|
1234
|
+
# leaves the last reading as the only one there is, which is what the
|
|
1235
|
+
# screen falls back to as well.
|
|
1236
|
+
peak = result.get("peak_gpu") or gpu
|
|
1237
|
+
if done:
|
|
1238
|
+
# A FINISHED RUN LEADS WITH THE PEAK. Memory is what kills a run, so
|
|
1239
|
+
# the peak is the number a post-mortem came here for; the last
|
|
1240
|
+
# reading is kept, because it is still evidence, but it is named for
|
|
1241
|
+
# what it is instead of being labelled "GPU util". The screen was
|
|
1242
|
+
# changed to this shape on 2026-09-07 after the 0%, 0 MiB readout
|
|
1243
|
+
# was reported as the panel being broken.
|
|
1244
|
+
print(f" peak memory {bar(peak['memory_percent'])} "
|
|
1245
|
+
f"{peak['memory_used_mib']:,} / {peak['memory_total_mib']:,} MiB "
|
|
1246
|
+
f"({peak['memory_percent']}%)")
|
|
1247
|
+
# ★ `peak_utilization_percent`, NOT `peak['utilization_percent']`,
|
|
1248
|
+
# and the difference is the whole defect. `peak` is the sample with
|
|
1249
|
+
# the most memory in it and its utilisation is whatever the card
|
|
1250
|
+
# happened to be doing at that instant. On job-66b46719b854 all four
|
|
1251
|
+
# A100s reached 77,631 MiB and the utilisation inside those four
|
|
1252
|
+
# samples read 0, 3, 1 and 95 -- while every one of those cards
|
|
1253
|
+
# actually peaked at 99 or 100 and averaged between 37.8 and 84.6.
|
|
1254
|
+
# Reported on the screen 2026-09-11; the CLI never printed either
|
|
1255
|
+
# figure, so it is being added correct rather than fixed.
|
|
1256
|
+
print(f" peak util {_percent(result.get('peak_utilization_percent'))}")
|
|
1257
|
+
print(f" avg util {_percent(result.get('avg_utilization_percent'))}")
|
|
1258
|
+
print(f" last reading {gpu['utilization_percent']}%, "
|
|
1259
|
+
f"{gpu['memory_used_mib']:,} MiB (run ended)")
|
|
1260
|
+
else:
|
|
1261
|
+
print(f" GPU util {bar(gpu['utilization_percent'])} "
|
|
1262
|
+
f"{gpu['utilization_percent']}%")
|
|
1263
|
+
print(f" GPU memory {bar(gpu['memory_percent'])} "
|
|
1264
|
+
f"{gpu['memory_used_mib']:,} / {gpu['memory_total_mib']:,} MiB "
|
|
1265
|
+
f"({gpu['memory_percent']}%)")
|
|
1266
|
+
print(f" temp, power {gpu['temperature_c']} C, {gpu['power_w']} W")
|
|
1118
1267
|
# `sample_count`, not len(gpu_series): the server thins the series to at
|
|
1119
1268
|
# most 400 points so a chart can draw it, and job-66b46719b854 printed
|
|
1120
1269
|
# 785 readings per card while this line said 393. The screen had the same
|
|
@@ -1122,12 +1271,61 @@ def cmd_watch(args: argparse.Namespace) -> int:
|
|
|
1122
1271
|
# falls back to the thinned length, which is the only number it has.
|
|
1123
1272
|
taken = result.get("sample_count") or len(result.get("gpu_series", []))
|
|
1124
1273
|
print(f" samples {taken}, last {result['window_seconds']}s")
|
|
1274
|
+
print_card_table(result.get("cards") or [])
|
|
1125
1275
|
|
|
1126
1276
|
if result.get("note"):
|
|
1127
1277
|
print(f" note {result['note']}")
|
|
1128
1278
|
return EXIT_OK
|
|
1129
1279
|
|
|
1130
1280
|
|
|
1281
|
+
def _percent(value: float | int | None) -> str:
|
|
1282
|
+
"""One utilisation figure, or a dash when the server did not send it.
|
|
1283
|
+
|
|
1284
|
+
The two utilisation fields arrived on 2026-09-10 and a server older than
|
|
1285
|
+
that answers with them absent. A blank line reads as 0, which is the very
|
|
1286
|
+
thing the peak figures exist to stop being claimed.
|
|
1287
|
+
"""
|
|
1288
|
+
return "-" if value is None else f"{value}%"
|
|
1289
|
+
|
|
1290
|
+
|
|
1291
|
+
def print_card_table(cards: list[dict[str, Any]]) -> None:
|
|
1292
|
+
"""One row per GPU card, printed only when the job rented more than one.
|
|
1293
|
+
|
|
1294
|
+
WHY THIS EXISTS. Everything above describes card 0 -- `latest_gpu` and
|
|
1295
|
+
`peak_gpu` are the lowest-indexed card's readings, unchanged since before
|
|
1296
|
+
`cards` existed. job-66b46719b854 is four A100s in one pod, and reading its
|
|
1297
|
+
CLI output left three of them invisible.
|
|
1298
|
+
|
|
1299
|
+
WHY ONLY ABOVE ONE CARD. With a single card these four numbers repeat the
|
|
1300
|
+
block above it word for word, so the table is drawn on the same condition
|
|
1301
|
+
the screen draws its own (`cards.length > 1` in ui/app.js).
|
|
1302
|
+
|
|
1303
|
+
Args:
|
|
1304
|
+
cards: the response's `cards`, each a CardMetricsView -- `gpu_index`,
|
|
1305
|
+
`series`, `latest`, `peak`, `avg_utilization_percent`,
|
|
1306
|
+
`peak_utilization_percent`, `sample_count`.
|
|
1307
|
+
"""
|
|
1308
|
+
if len(cards) < 2:
|
|
1309
|
+
return
|
|
1310
|
+
print(f" {'card':<8}{'peak memory':>27}{'peak util':>11}"
|
|
1311
|
+
f"{'avg util':>10}{'samples':>9}")
|
|
1312
|
+
for card in cards:
|
|
1313
|
+
# `peak` is this card's highest-MEMORY reading; falling back to its last
|
|
1314
|
+
# one matches the screen's table and keeps a row rather than dropping a
|
|
1315
|
+
# card that only ever printed once.
|
|
1316
|
+
sample = card.get("peak") or card.get("latest") or {}
|
|
1317
|
+
memory = "-" if not sample else (
|
|
1318
|
+
f"{sample['memory_used_mib']:,} / {sample['memory_total_mib']:,} MiB "
|
|
1319
|
+
f"({sample['memory_percent']:.0f}%)")
|
|
1320
|
+
# Same two corrections as the block above: the card's own highest
|
|
1321
|
+
# utilisation rather than the utilisation inside its memory peak, and
|
|
1322
|
+
# the readings it printed rather than the points left after thinning.
|
|
1323
|
+
taken = card.get("sample_count") or len(card.get("series") or [])
|
|
1324
|
+
print(f" {('GPU ' + str(card['gpu_index'])):<8}{memory:>27}"
|
|
1325
|
+
f"{_percent(card.get('peak_utilization_percent')):>11}"
|
|
1326
|
+
f"{_percent(card.get('avg_utilization_percent')):>10}{taken:>9}")
|
|
1327
|
+
|
|
1328
|
+
|
|
1131
1329
|
def cmd_stats(args: argparse.Namespace) -> int:
|
|
1132
1330
|
"""Print this caller's team figures."""
|
|
1133
1331
|
result = client_from_config().stats()
|
|
@@ -34,7 +34,7 @@ build-backend = "setuptools.build_meta"
|
|
|
34
34
|
|
|
35
35
|
[project]
|
|
36
36
|
name = "hyperun"
|
|
37
|
-
version = "0.2.
|
|
37
|
+
version = "0.2.2"
|
|
38
38
|
description = "Submit a GPU job and get results back. No kubectl, no cloud account."
|
|
39
39
|
requires-python = ">=3.9"
|
|
40
40
|
dependencies = ["requests>=2.31", "PyYAML>=6.0"]
|
|
@@ -38,6 +38,15 @@ class FakeClient:
|
|
|
38
38
|
self.estimate_result = {}
|
|
39
39
|
self.validate_result = {"ok": True, "findings": [], "not_checked": []}
|
|
40
40
|
self.metrics_result = {"window_seconds": 3600, "gpu_series": [], "note": ""}
|
|
41
|
+
# Every window `watch` asked for, in order. The widening of a finished
|
|
42
|
+
# job's window is invisible in the printed output -- the numbers look the
|
|
43
|
+
# same whichever window found them -- so the only way to test it is to
|
|
44
|
+
# record what was asked.
|
|
45
|
+
self.metrics_windows: list[int] = []
|
|
46
|
+
# What to answer once the window is wider than the default hour. None
|
|
47
|
+
# means "the same answer whatever the window", which is what every test
|
|
48
|
+
# written before the widening existed expects.
|
|
49
|
+
self.metrics_wide_result = None
|
|
41
50
|
self.stats_result = {"team": "", "members": [], "jobs": 0, "gpu_hours": 0.0,
|
|
42
51
|
"cost_usd": 0.0, "unpriced_jobs": 0, "note": ""}
|
|
43
52
|
self.secrets_result = {"names": ["GITHUB_PAT", "HF_TOKEN"],
|
|
@@ -55,6 +64,9 @@ class FakeClient:
|
|
|
55
64
|
return self.validate_result
|
|
56
65
|
|
|
57
66
|
def metrics(self, job_id, window_seconds=3600):
|
|
67
|
+
self.metrics_windows.append(window_seconds)
|
|
68
|
+
if self.metrics_wide_result is not None and window_seconds > 3600:
|
|
69
|
+
return self.metrics_wide_result
|
|
58
70
|
return self.metrics_result
|
|
59
71
|
|
|
60
72
|
def stats(self):
|
|
@@ -475,6 +487,265 @@ def test_watch_prints_progress_and_gpu(fake, capsys):
|
|
|
475
487
|
assert "samples 120" in printed
|
|
476
488
|
|
|
477
489
|
|
|
490
|
+
def gpu_sample(util, used, total=81920, index=0):
|
|
491
|
+
"""One reading, for the card tests below."""
|
|
492
|
+
return {"utilization_percent": util, "memory_used_mib": used,
|
|
493
|
+
"memory_total_mib": total, "memory_percent": round(100 * used / total, 1),
|
|
494
|
+
"temperature_c": 62, "power_w": 310.0, "gpu_index": index}
|
|
495
|
+
|
|
496
|
+
|
|
497
|
+
def four_a100_cards():
|
|
498
|
+
"""job-66b46719b854 as the server reports it: four A100s in one pod.
|
|
499
|
+
|
|
500
|
+
The numbers are that run's: every card peaked at 77,631 of 81,920 MiB, and
|
|
501
|
+
the utilisation INSIDE those four highest-memory samples read 0, 3, 1 and 95
|
|
502
|
+
while the cards themselves peaked at 100, 99, 100 and 99. That gap is the
|
|
503
|
+
defect the screen carried until 2026-09-11, so the fixture keeps it.
|
|
504
|
+
"""
|
|
505
|
+
peaks = [(0, 100, 84.6), (3, 99, 37.8), (1, 100, 79.2), (95, 99, 81.3)]
|
|
506
|
+
return [
|
|
507
|
+
{"gpu_index": i,
|
|
508
|
+
"series": [{}] * 393,
|
|
509
|
+
"latest": gpu_sample(0, 0, index=i),
|
|
510
|
+
"peak": gpu_sample(at_peak, 77631, index=i),
|
|
511
|
+
"avg_utilization_percent": avg,
|
|
512
|
+
"peak_utilization_percent": real_peak,
|
|
513
|
+
"sample_count": 785}
|
|
514
|
+
for i, (at_peak, real_peak, avg) in enumerate(peaks)
|
|
515
|
+
]
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
def test_every_card_is_printed_not_only_the_first(fake, capsys):
|
|
519
|
+
# ★ THE DEFECT. `latest_gpu` is card 0, so a four-A100 job printed one card
|
|
520
|
+
# and three quarters of a $44 run was invisible from the CLI. The screen was
|
|
521
|
+
# fixed on 2026-09-08 and this surface was not.
|
|
522
|
+
fake.metrics_result = {
|
|
523
|
+
"latest_gpu": gpu_sample(94, 38200),
|
|
524
|
+
"peak_gpu": gpu_sample(0, 77631),
|
|
525
|
+
"gpu_series": [{}] * 393,
|
|
526
|
+
"sample_count": 785,
|
|
527
|
+
"peak_utilization_percent": 100,
|
|
528
|
+
"avg_utilization_percent": 84.6,
|
|
529
|
+
"cards": four_a100_cards(),
|
|
530
|
+
"window_seconds": 604800, "note": "",
|
|
531
|
+
}
|
|
532
|
+
assert run(["watch", "job-66b46719b854"]) == cli.EXIT_OK
|
|
533
|
+
printed = capsys.readouterr().out
|
|
534
|
+
# `GPU 0` is a table row; `GPU util` and `GPU memory` are the headline above
|
|
535
|
+
# it, so the card index is what tells the two apart.
|
|
536
|
+
rows = {line.split()[1]: line.split()
|
|
537
|
+
for line in printed.splitlines()
|
|
538
|
+
if line.startswith(" GPU ") and line.split()[1].isdigit()}
|
|
539
|
+
assert sorted(rows) == ["0", "1", "2", "3"]
|
|
540
|
+
|
|
541
|
+
# A row reads: GPU <n> <used> / <total> MiB (<pct>%) <peak util> <avg> <n>
|
|
542
|
+
# so the last three fields are the three that were wrong or missing.
|
|
543
|
+
expected = {"0": ("100%", "84.6%"), "1": ("99%", "37.8%"),
|
|
544
|
+
"2": ("100%", "79.2%"), "3": ("99%", "81.3%")}
|
|
545
|
+
for index, (peak_util, avg_util) in expected.items():
|
|
546
|
+
fields = rows[index]
|
|
547
|
+
# Each card's own peak memory, not card 0's repeated four times.
|
|
548
|
+
assert fields[2:7] == ["77,631", "/", "81,920", "MiB", "(95%)"]
|
|
549
|
+
# ★ `peak_utilization_percent`, NOT the utilisation inside the memory
|
|
550
|
+
# peak. The two differ on this very run: the four highest-memory samples
|
|
551
|
+
# read 0%, 3%, 1% and 95% while the cards peaked at 100, 99, 100 and 99.
|
|
552
|
+
assert fields[7] == peak_util
|
|
553
|
+
assert fields[8] == avg_util
|
|
554
|
+
# The readings this card printed, not the 393 points left after the
|
|
555
|
+
# server thinned the series for the chart.
|
|
556
|
+
assert fields[9] == "785"
|
|
557
|
+
|
|
558
|
+
|
|
559
|
+
def test_one_card_prints_no_table_because_it_would_repeat_the_block(fake, capsys):
|
|
560
|
+
# With a single card the table's four numbers are the same four printed
|
|
561
|
+
# directly above it, so it is drawn on the same condition the screen uses.
|
|
562
|
+
fake.metrics_result = {
|
|
563
|
+
"latest_gpu": gpu_sample(94, 38200, total=45440),
|
|
564
|
+
"gpu_series": [{}] * 120,
|
|
565
|
+
"cards": [{"gpu_index": 0, "series": [{}] * 120,
|
|
566
|
+
"latest": gpu_sample(94, 38200, total=45440),
|
|
567
|
+
"peak": gpu_sample(94, 38200, total=45440),
|
|
568
|
+
"avg_utilization_percent": 90.0,
|
|
569
|
+
"peak_utilization_percent": 99,
|
|
570
|
+
"sample_count": 120}],
|
|
571
|
+
"window_seconds": 3600, "note": "",
|
|
572
|
+
}
|
|
573
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
574
|
+
printed = capsys.readouterr().out
|
|
575
|
+
assert "peak memory" not in printed # the table's header row
|
|
576
|
+
assert "38,200 / 45,440 MiB" in printed
|
|
577
|
+
|
|
578
|
+
|
|
579
|
+
def test_a_finished_job_leads_with_the_peak_not_the_idle_last_reading(fake, capsys):
|
|
580
|
+
# ★ THE DEFECT. A finished job's last reading is the idle card in the
|
|
581
|
+
# seconds before teardown -- 0%, 0 MiB -- and printing it as "GPU util" says
|
|
582
|
+
# the run was idle. What a post-mortem asks for is the peak, because running
|
|
583
|
+
# out of memory is what kills a run. The screen was changed to this shape on
|
|
584
|
+
# 2026-09-07.
|
|
585
|
+
fake.status_result = {"job_id": "job-66b46719b854", "name": "baseline-c",
|
|
586
|
+
"phase": "Succeeded"}
|
|
587
|
+
fake.metrics_result = {
|
|
588
|
+
"latest_gpu": gpu_sample(0, 0),
|
|
589
|
+
"peak_gpu": gpu_sample(0, 77631),
|
|
590
|
+
"gpu_series": [{}] * 393,
|
|
591
|
+
"sample_count": 785,
|
|
592
|
+
"peak_utilization_percent": 100,
|
|
593
|
+
"avg_utilization_percent": 84.6,
|
|
594
|
+
"window_seconds": 604800, "note": "",
|
|
595
|
+
}
|
|
596
|
+
assert run(["watch", "job-66b46719b854"]) == cli.EXIT_OK
|
|
597
|
+
printed = capsys.readouterr().out
|
|
598
|
+
assert "peak memory" in printed
|
|
599
|
+
assert "77,631 / 81,920 MiB" in printed
|
|
600
|
+
# The peak of the CARD, not the utilisation inside the highest-memory
|
|
601
|
+
# sample, which on this run was 0 while the card reached 100.
|
|
602
|
+
assert "peak util 100%" in printed
|
|
603
|
+
assert "avg util 84.6%" in printed
|
|
604
|
+
# The last reading is still shown, and named for what it is.
|
|
605
|
+
assert "last reading 0%, 0 MiB (run ended)" in printed
|
|
606
|
+
# The running job's labels are gone: leaving "GPU util 0%" on the screen is
|
|
607
|
+
# exactly the claim this fix removes.
|
|
608
|
+
assert "GPU util" not in printed
|
|
609
|
+
|
|
610
|
+
|
|
611
|
+
def test_a_running_job_keeps_the_live_readings_as_the_headline(fake, capsys):
|
|
612
|
+
# The mirror of the test above. `peak_gpu` being present is not on its own a
|
|
613
|
+
# finished job -- the server sends it for a running one too -- so the phase
|
|
614
|
+
# is what decides, and a Running job still leads with what the card is doing
|
|
615
|
+
# NOW.
|
|
616
|
+
fake.metrics_result = {
|
|
617
|
+
"latest_gpu": gpu_sample(94, 38200, total=45440),
|
|
618
|
+
"peak_gpu": gpu_sample(99, 40000, total=45440),
|
|
619
|
+
"gpu_series": [{}] * 120,
|
|
620
|
+
"peak_utilization_percent": 99,
|
|
621
|
+
"avg_utilization_percent": 90.0,
|
|
622
|
+
"window_seconds": 3600, "note": "",
|
|
623
|
+
}
|
|
624
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
625
|
+
printed = capsys.readouterr().out
|
|
626
|
+
assert "GPU util" in printed
|
|
627
|
+
assert "38,200 / 45,440 MiB" in printed
|
|
628
|
+
assert "last reading" not in printed
|
|
629
|
+
|
|
630
|
+
|
|
631
|
+
def test_a_phase_lookup_that_fails_still_prints_the_gpu_numbers(fake, capsys):
|
|
632
|
+
# The numbers are what the command exists for, so a job deleted between the
|
|
633
|
+
# metrics call and the phase call falls back to the running layout rather
|
|
634
|
+
# than exiting with nothing printed.
|
|
635
|
+
def refuse(job_id):
|
|
636
|
+
raise ServerError("404: no such job")
|
|
637
|
+
|
|
638
|
+
fake.status = refuse
|
|
639
|
+
fake.metrics_result = {
|
|
640
|
+
"latest_gpu": gpu_sample(94, 38200, total=45440),
|
|
641
|
+
"gpu_series": [{}] * 120,
|
|
642
|
+
"window_seconds": 3600, "note": "",
|
|
643
|
+
}
|
|
644
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
645
|
+
assert "38,200 / 45,440 MiB" in capsys.readouterr().out
|
|
646
|
+
|
|
647
|
+
|
|
648
|
+
def _hours_ago(hours):
|
|
649
|
+
"""An RFC 3339 timestamp that many hours in the past, as the server writes it."""
|
|
650
|
+
import datetime as _dt
|
|
651
|
+
t = _dt.datetime.now(_dt.timezone.utc) - _dt.timedelta(hours=hours)
|
|
652
|
+
return t.strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
def test_watch_widens_the_window_to_reach_a_finished_jobs_readings(fake, capsys):
|
|
656
|
+
# ★ THE DEFECT: nothing is stored, so `watch` re-reads the log and the window
|
|
657
|
+
# is measured back from NOW. A run that ended five hours ago has every
|
|
658
|
+
# reading outside the default hour, and `watch` printed the phase and no
|
|
659
|
+
# numbers -- the phase being the one thing the reader already knew.
|
|
660
|
+
fake.status_result = {"job_id": "job-a8acdef80a07", "name": "bank-exp2",
|
|
661
|
+
"phase": "Succeeded", "started_at": _hours_ago(9),
|
|
662
|
+
"finished_at": _hours_ago(5)}
|
|
663
|
+
fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
|
|
664
|
+
fake.metrics_wide_result = {
|
|
665
|
+
"latest_gpu": gpu_sample(94, 38200, total=45440),
|
|
666
|
+
"peak_gpu": gpu_sample(94, 38200, total=45440),
|
|
667
|
+
"peak_utilization_percent": 94,
|
|
668
|
+
"gpu_series": [{}] * 120, "window_seconds": 36000, "note": "",
|
|
669
|
+
}
|
|
670
|
+
|
|
671
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
672
|
+
assert len(fake.metrics_windows) == 2, "the empty first window was not retried"
|
|
673
|
+
assert fake.metrics_windows[0] == 3600
|
|
674
|
+
# Nine hours of run plus a minute of headroom, so the FIRST reading is inside
|
|
675
|
+
# the window rather than exactly on its edge.
|
|
676
|
+
assert fake.metrics_windows[1] >= 9 * 3600
|
|
677
|
+
assert "38,200 / 45,440 MiB" in capsys.readouterr().out
|
|
678
|
+
|
|
679
|
+
|
|
680
|
+
def test_a_window_the_caller_chose_is_never_widened(fake):
|
|
681
|
+
# An explicit --window is an instruction, not a default. Widening it would
|
|
682
|
+
# answer a question the caller did not ask, and the answer would be a bigger
|
|
683
|
+
# read of the log than they wanted.
|
|
684
|
+
fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Succeeded",
|
|
685
|
+
"started_at": _hours_ago(9)}
|
|
686
|
+
fake.metrics_result = {"gpu_series": [], "window_seconds": 600, "note": ""}
|
|
687
|
+
|
|
688
|
+
assert run(["watch", "job-a8acdef80a07", "--window", "600"]) == cli.EXIT_OK
|
|
689
|
+
assert fake.metrics_windows == [600]
|
|
690
|
+
|
|
691
|
+
|
|
692
|
+
def test_a_running_job_is_not_widened_however_quiet_it_is(fake):
|
|
693
|
+
# A running job's readings ARE inside the last hour, so an empty answer means
|
|
694
|
+
# the job has printed nothing yet. Re-reading a week of its log would buy a
|
|
695
|
+
# second empty answer and a much larger response.
|
|
696
|
+
fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Running",
|
|
697
|
+
"started_at": _hours_ago(9)}
|
|
698
|
+
fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
|
|
699
|
+
|
|
700
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
701
|
+
assert fake.metrics_windows == [3600]
|
|
702
|
+
|
|
703
|
+
|
|
704
|
+
def test_the_widened_window_stops_at_the_servers_own_cap(fake):
|
|
705
|
+
# `window_seconds` is declared ge=60, le=604800 on the metrics route. Asking
|
|
706
|
+
# for more is a 422, so a job that started a month ago must be clamped here
|
|
707
|
+
# rather than turned into an error the reader cannot act on.
|
|
708
|
+
fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Failed",
|
|
709
|
+
"started_at": _hours_ago(24 * 30)}
|
|
710
|
+
fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
|
|
711
|
+
|
|
712
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
713
|
+
assert fake.metrics_windows[1] == cli.MAX_WATCH_WINDOW == 604800
|
|
714
|
+
|
|
715
|
+
|
|
716
|
+
def test_a_finished_job_with_no_start_time_leaves_the_window_alone(fake):
|
|
717
|
+
# Nothing to compute a width from. Widening to the cap "just in case" would
|
|
718
|
+
# read seven days of log off the back of a missing field.
|
|
719
|
+
fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Succeeded"}
|
|
720
|
+
fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
|
|
721
|
+
|
|
722
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
723
|
+
assert fake.metrics_windows == [3600]
|
|
724
|
+
|
|
725
|
+
|
|
726
|
+
def test_the_window_help_quotes_the_servers_real_cap(capsys):
|
|
727
|
+
# It said 86400 and the server accepts 604800, so the help talked a reader
|
|
728
|
+
# out of a window the server would have answered.
|
|
729
|
+
with pytest.raises(SystemExit):
|
|
730
|
+
cli.main(["watch", "--help"])
|
|
731
|
+
printed = capsys.readouterr().out
|
|
732
|
+
assert "604800" in printed
|
|
733
|
+
assert "86400" not in printed
|
|
734
|
+
|
|
735
|
+
|
|
736
|
+
def test_watch_json_prints_the_cards_and_asks_for_no_phase(fake, capsys):
|
|
737
|
+
# --json is the server's answer verbatim, so the second request would buy
|
|
738
|
+
# nothing. It is one GET per typed `watch` and worth not making.
|
|
739
|
+
def refuse(job_id):
|
|
740
|
+
raise AssertionError("--json must not ask for the phase")
|
|
741
|
+
|
|
742
|
+
fake.status = refuse
|
|
743
|
+
fake.metrics_result = {"cards": four_a100_cards(), "gpu_series": [],
|
|
744
|
+
"window_seconds": 3600, "note": ""}
|
|
745
|
+
assert run(["watch", "job-66b46719b854", "--json"]) == cli.EXIT_OK
|
|
746
|
+
assert len(json.loads(capsys.readouterr().out)["cards"]) == 4
|
|
747
|
+
|
|
748
|
+
|
|
478
749
|
def test_samples_counts_the_readings_taken_not_the_chart_points(fake, capsys):
|
|
479
750
|
# ★ THE SAME DEFECT THE SCREEN HAD. The server thins the series to at most
|
|
480
751
|
# 400 points so a chart can draw it; job-66b46719b854 printed 785 readings
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|