hyperun 0.2.1__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: hyperun
3
- Version: 0.2.1
3
+ Version: 0.2.2
4
4
  Summary: Submit a GPU job and get results back. No kubectl, no cloud account.
5
5
  Author: DDPS Lab
6
6
  License-Expression: Apache-2.0
@@ -50,6 +50,7 @@ from __future__ import annotations
50
50
 
51
51
  import argparse
52
52
  import base64
53
+ import datetime
53
54
  import getpass
54
55
  import json
55
56
  import time
@@ -387,8 +388,13 @@ def build_parser() -> argparse.ArgumentParser:
387
388
  )
388
389
  watch.add_argument("job_id")
389
390
  watch.add_argument(
390
- "--window", type=int, default=3600, metavar="SECONDS",
391
- help="how far back to read (default 3600, max 86400)",
391
+ # default=None, not 3600, and the difference is load-bearing: it is how
392
+ # `cmd_watch` tells "the caller wants an hour" from "the caller said
393
+ # nothing", and only the second one may be widened to reach a finished
394
+ # job's readings. 604800 is the server's cap, not a number chosen here.
395
+ "--window", type=int, default=None, metavar="SECONDS",
396
+ help="how far back to read (default 3600, max 604800). A finished job "
397
+ "is widened automatically to reach its own readings.",
392
398
  )
393
399
  add_json_flag(watch)
394
400
 
@@ -1088,13 +1094,128 @@ def bar(percent: float, width: int = 20) -> str:
1088
1094
  return "#" * filled + "-" * (width - filled)
1089
1095
 
1090
1096
 
1097
+ # DDPSRUN-WATCH-TERMINAL. The three phases a job never leaves. They are the
1098
+ # controller's own words: the phase enum in api/v1alpha1/pacsjob_types.go
1099
+ # carries Pending / Starting / Running / Recovering / Succeeded / Failed /
1100
+ # Compared and nothing else, and the screen keeps the same three in
1101
+ # `const TERMINAL` (ui/app.js). Recovering is NOT here -- the machine was
1102
+ # reclaimed and the job is being restarted, so the run is still going.
1103
+ TERMINAL_PHASES = ("Succeeded", "Failed", "Compared")
1104
+
1105
+ # How far back `watch` reads when the caller says nothing, and the furthest the
1106
+ # server will let it reach. The cap is the server's own: `window_seconds` on
1107
+ # `GET /v1/jobs/{id}/metrics` is declared `ge=60, le=604800`
1108
+ # (server/ddpsrun_server/main.py). The `--window` help said 86400 here, which was
1109
+ # simply wrong -- one day, against the server's seven.
1110
+ DEFAULT_WATCH_WINDOW = 3600
1111
+ MAX_WATCH_WINDOW = 604800
1112
+
1113
+
1114
+ def _window_reaching_back_to_start(job: dict[str, Any]) -> int:
1115
+ """How wide a window must be to contain a FINISHED job's readings.
1116
+
1117
+ ★ WHY THIS EXISTS. Nothing is stored: `watch` re-reads the job's own log, and
1118
+ the window is measured back from NOW. A running job's readings are therefore
1119
+ always inside the default hour. A job that finished five hours ago has NONE
1120
+ of its readings in the last hour, so `hyperun watch` on it printed the phase
1121
+ and no numbers -- and the phase is the one thing the reader already knew.
1122
+
1123
+ The fix is not a larger default, which would make every `watch` on a running
1124
+ job re-read a week of log. It is to widen only when the job is over AND the
1125
+ caller did not pick a window.
1126
+
1127
+ Args:
1128
+ job: the body of `GET /v1/jobs/{id}`.
1129
+
1130
+ Returns:
1131
+ Seconds, or 0 when the job is not finished or carries no start time, in
1132
+ which case the caller leaves the window as it was.
1133
+ """
1134
+ if job.get("phase") not in TERMINAL_PHASES:
1135
+ return 0
1136
+ started = job.get("started_at") or job.get("created_at")
1137
+ if not started:
1138
+ return 0
1139
+ try:
1140
+ began = datetime.datetime.fromisoformat(started.replace("Z", "+00:00"))
1141
+ except ValueError:
1142
+ return 0
1143
+ elapsed = (datetime.datetime.now(datetime.timezone.utc) - began).total_seconds()
1144
+ # A minute of headroom so the FIRST reading falls inside the window instead
1145
+ # of exactly on its edge, and the server's own cap over the top.
1146
+ return min(MAX_WATCH_WINDOW, max(60, int(elapsed) + 60))
1147
+
1148
+
1091
1149
  def cmd_watch(args: argparse.Namespace) -> int:
1092
- """Print a job's GPU usage and how far the training has got."""
1093
- result = client_from_config().metrics(args.job_id, args.window)
1150
+ """Print a job's GPU usage and how far the training has got.
1151
+
1152
+ ONE BLOCK PER CARD, because a job can rent several. baseline-c rents four
1153
+ A100s in one pod, and until today this printed `latest_gpu`, which is card 0
1154
+ alone -- three quarters of a $44 run was invisible from the CLI. The screen
1155
+ had the same defect and was fixed on 2026-09-08. `cards` is the server's
1156
+ per-card answer (CardMetricsView in server/ddpsrun_server/models.py); a
1157
+ server too old to send it answers with an empty list, and then the
1158
+ single-card block below is the whole story, exactly as before.
1159
+
1160
+ WHY THIS ASKS FOR THE PHASE AS WELL AS THE METRICS. A finished job's last
1161
+ reading is the idle card in the seconds before teardown -- 0%, 0 MiB -- so
1162
+ printing it under "GPU util" tells a reader the run was idle when it was
1163
+ not. Leading with the peak instead requires knowing the job is over, and
1164
+ THE METRICS ANSWER DOES NOT SAY: MetricsResponse carries latest_gpu,
1165
+ gpu_series, peak_gpu, the two utilisation figures, sample_count, cards,
1166
+ progress, window_seconds and note, and no phase at all
1167
+ (server/ddpsrun_server/models.py). Guessing from the readings fails in both
1168
+ directions -- a job whose framework has not allocated yet also reads 0 MiB,
1169
+ and a job killed mid-step leaves a final reading that is nowhere near 0 --
1170
+ so this asks the server instead.
1171
+
1172
+ WHAT THAT COSTS: one extra `GET /v1/jobs/{id}` per typed `watch`. `watch`
1173
+ prints once and exits rather than polling, so it is one extra request per
1174
+ command, not one per second, and `--json` pays it only when the first window
1175
+ came back empty. The screen pays the same price -- it reads the job and hands
1176
+ it to drawMetrics (ui/app.js).
1177
+ """
1178
+ client = client_from_config()
1179
+ window = DEFAULT_WATCH_WINDOW if args.window is None else args.window
1180
+ result = client.metrics(args.job_id, window)
1181
+
1182
+ # ★ AN EMPTY ANSWER ON A FINISHED JOB IS THE WINDOW, NOT THE JOB. The window
1183
+ # is measured back from now, so a run that ended hours ago has every reading
1184
+ # outside the default hour and `watch` printed nothing. Widening is done
1185
+ # ONLY here: the caller did not choose a window, and the first one came back
1186
+ # with no readings at all. A running job never reaches this branch, so the
1187
+ # usual `watch` is still exactly one metrics request.
1188
+ #
1189
+ # "No readings" means BOTH are empty. `cards` is the per-card answer and
1190
+ # `latest_gpu` is card 0; a multi-card job fills `cards`, while a server too
1191
+ # old to send it fills only `latest_gpu`. Testing one alone would make
1192
+ # `watch --json` on a four-card job buy a second request it does not need.
1193
+ job: dict[str, Any] | None = None
1194
+ if args.window is None and not result.get("latest_gpu") and not result.get("cards"):
1195
+ try:
1196
+ job = client.status(args.job_id)
1197
+ except ServerError:
1198
+ job = None
1199
+ wider = _window_reaching_back_to_start(job) if job else 0
1200
+ if wider > window:
1201
+ result = client.metrics(args.job_id, wider)
1202
+ window = wider
1203
+
1094
1204
  if args.json:
1205
+ # The server's answer, verbatim. It already contains `cards`, so nothing
1206
+ # here needs the phase.
1095
1207
  print(json.dumps(result, indent=2, ensure_ascii=False))
1096
1208
  return EXIT_OK
1097
1209
 
1210
+ # A phase lookup that fails must not lose the GPU numbers this command
1211
+ # exists to print, so it falls back to the running layout. The likely cause
1212
+ # is the job being deleted between the two calls, and the running layout is
1213
+ # the behaviour this command had until today rather than a wrong number.
1214
+ try:
1215
+ done = (job or client.status(args.job_id)).get("phase") in TERMINAL_PHASES
1216
+ except ServerError:
1217
+ done = False
1218
+
1098
1219
  progress = result.get("progress")
1099
1220
  if progress:
1100
1221
  print(f" training {bar(progress['percent'])} "
@@ -1109,12 +1230,40 @@ def cmd_watch(args: argparse.Namespace) -> int:
1109
1230
 
1110
1231
  gpu = result.get("latest_gpu")
1111
1232
  if gpu:
1112
- print(f" GPU util {bar(gpu['utilization_percent'])} "
1113
- f"{gpu['utilization_percent']}%")
1114
- print(f" GPU memory {bar(gpu['memory_percent'])} "
1115
- f"{gpu['memory_used_mib']:,} / {gpu['memory_total_mib']:,} MiB "
1116
- f"({gpu['memory_percent']}%)")
1117
- print(f" temp, power {gpu['temperature_c']} C, {gpu['power_w']} W")
1233
+ # The reading with the most MEMORY in use. A server too old to send it
1234
+ # leaves the last reading as the only one there is, which is what the
1235
+ # screen falls back to as well.
1236
+ peak = result.get("peak_gpu") or gpu
1237
+ if done:
1238
+ # A FINISHED RUN LEADS WITH THE PEAK. Memory is what kills a run, so
1239
+ # the peak is the number a post-mortem came here for; the last
1240
+ # reading is kept, because it is still evidence, but it is named for
1241
+ # what it is instead of being labelled "GPU util". The screen was
1242
+ # changed to this shape on 2026-09-07 after the 0%, 0 MiB readout
1243
+ # was reported as the panel being broken.
1244
+ print(f" peak memory {bar(peak['memory_percent'])} "
1245
+ f"{peak['memory_used_mib']:,} / {peak['memory_total_mib']:,} MiB "
1246
+ f"({peak['memory_percent']}%)")
1247
+ # ★ `peak_utilization_percent`, NOT `peak['utilization_percent']`,
1248
+ # and the difference is the whole defect. `peak` is the sample with
1249
+ # the most memory in it and its utilisation is whatever the card
1250
+ # happened to be doing at that instant. On job-66b46719b854 all four
1251
+ # A100s reached 77,631 MiB and the utilisation inside those four
1252
+ # samples read 0, 3, 1 and 95 -- while every one of those cards
1253
+ # actually peaked at 99 or 100 and averaged between 37.8 and 84.6.
1254
+ # Reported on the screen 2026-09-11; the CLI never printed either
1255
+ # figure, so it is being added correct rather than fixed.
1256
+ print(f" peak util {_percent(result.get('peak_utilization_percent'))}")
1257
+ print(f" avg util {_percent(result.get('avg_utilization_percent'))}")
1258
+ print(f" last reading {gpu['utilization_percent']}%, "
1259
+ f"{gpu['memory_used_mib']:,} MiB (run ended)")
1260
+ else:
1261
+ print(f" GPU util {bar(gpu['utilization_percent'])} "
1262
+ f"{gpu['utilization_percent']}%")
1263
+ print(f" GPU memory {bar(gpu['memory_percent'])} "
1264
+ f"{gpu['memory_used_mib']:,} / {gpu['memory_total_mib']:,} MiB "
1265
+ f"({gpu['memory_percent']}%)")
1266
+ print(f" temp, power {gpu['temperature_c']} C, {gpu['power_w']} W")
1118
1267
  # `sample_count`, not len(gpu_series): the server thins the series to at
1119
1268
  # most 400 points so a chart can draw it, and job-66b46719b854 printed
1120
1269
  # 785 readings per card while this line said 393. The screen had the same
@@ -1122,12 +1271,61 @@ def cmd_watch(args: argparse.Namespace) -> int:
1122
1271
  # falls back to the thinned length, which is the only number it has.
1123
1272
  taken = result.get("sample_count") or len(result.get("gpu_series", []))
1124
1273
  print(f" samples {taken}, last {result['window_seconds']}s")
1274
+ print_card_table(result.get("cards") or [])
1125
1275
 
1126
1276
  if result.get("note"):
1127
1277
  print(f" note {result['note']}")
1128
1278
  return EXIT_OK
1129
1279
 
1130
1280
 
1281
+ def _percent(value: float | int | None) -> str:
1282
+ """One utilisation figure, or a dash when the server did not send it.
1283
+
1284
+ The two utilisation fields arrived on 2026-09-10 and a server older than
1285
+ that answers with them absent. A blank line reads as 0, which is the very
1286
+ thing the peak figures exist to stop being claimed.
1287
+ """
1288
+ return "-" if value is None else f"{value}%"
1289
+
1290
+
1291
+ def print_card_table(cards: list[dict[str, Any]]) -> None:
1292
+ """One row per GPU card, printed only when the job rented more than one.
1293
+
1294
+ WHY THIS EXISTS. Everything above describes card 0 -- `latest_gpu` and
1295
+ `peak_gpu` are the lowest-indexed card's readings, unchanged since before
1296
+ `cards` existed. job-66b46719b854 is four A100s in one pod, and reading its
1297
+ CLI output left three of them invisible.
1298
+
1299
+ WHY ONLY ABOVE ONE CARD. With a single card these four numbers repeat the
1300
+ block above it word for word, so the table is drawn on the same condition
1301
+ the screen draws its own (`cards.length > 1` in ui/app.js).
1302
+
1303
+ Args:
1304
+ cards: the response's `cards`, each a CardMetricsView -- `gpu_index`,
1305
+ `series`, `latest`, `peak`, `avg_utilization_percent`,
1306
+ `peak_utilization_percent`, `sample_count`.
1307
+ """
1308
+ if len(cards) < 2:
1309
+ return
1310
+ print(f" {'card':<8}{'peak memory':>27}{'peak util':>11}"
1311
+ f"{'avg util':>10}{'samples':>9}")
1312
+ for card in cards:
1313
+ # `peak` is this card's highest-MEMORY reading; falling back to its last
1314
+ # one matches the screen's table and keeps a row rather than dropping a
1315
+ # card that only ever printed once.
1316
+ sample = card.get("peak") or card.get("latest") or {}
1317
+ memory = "-" if not sample else (
1318
+ f"{sample['memory_used_mib']:,} / {sample['memory_total_mib']:,} MiB "
1319
+ f"({sample['memory_percent']:.0f}%)")
1320
+ # Same two corrections as the block above: the card's own highest
1321
+ # utilisation rather than the utilisation inside its memory peak, and
1322
+ # the readings it printed rather than the points left after thinning.
1323
+ taken = card.get("sample_count") or len(card.get("series") or [])
1324
+ print(f" {('GPU ' + str(card['gpu_index'])):<8}{memory:>27}"
1325
+ f"{_percent(card.get('peak_utilization_percent')):>11}"
1326
+ f"{_percent(card.get('avg_utilization_percent')):>10}{taken:>9}")
1327
+
1328
+
1131
1329
  def cmd_stats(args: argparse.Namespace) -> int:
1132
1330
  """Print this caller's team figures."""
1133
1331
  result = client_from_config().stats()
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: hyperun
3
- Version: 0.2.1
3
+ Version: 0.2.2
4
4
  Summary: Submit a GPU job and get results back. No kubectl, no cloud account.
5
5
  Author: DDPS Lab
6
6
  License-Expression: Apache-2.0
@@ -34,7 +34,7 @@ build-backend = "setuptools.build_meta"
34
34
 
35
35
  [project]
36
36
  name = "hyperun"
37
- version = "0.2.1"
37
+ version = "0.2.2"
38
38
  description = "Submit a GPU job and get results back. No kubectl, no cloud account."
39
39
  requires-python = ">=3.9"
40
40
  dependencies = ["requests>=2.31", "PyYAML>=6.0"]
@@ -38,6 +38,15 @@ class FakeClient:
38
38
  self.estimate_result = {}
39
39
  self.validate_result = {"ok": True, "findings": [], "not_checked": []}
40
40
  self.metrics_result = {"window_seconds": 3600, "gpu_series": [], "note": ""}
41
+ # Every window `watch` asked for, in order. The widening of a finished
42
+ # job's window is invisible in the printed output -- the numbers look the
43
+ # same whichever window found them -- so the only way to test it is to
44
+ # record what was asked.
45
+ self.metrics_windows: list[int] = []
46
+ # What to answer once the window is wider than the default hour. None
47
+ # means "the same answer whatever the window", which is what every test
48
+ # written before the widening existed expects.
49
+ self.metrics_wide_result = None
41
50
  self.stats_result = {"team": "", "members": [], "jobs": 0, "gpu_hours": 0.0,
42
51
  "cost_usd": 0.0, "unpriced_jobs": 0, "note": ""}
43
52
  self.secrets_result = {"names": ["GITHUB_PAT", "HF_TOKEN"],
@@ -55,6 +64,9 @@ class FakeClient:
55
64
  return self.validate_result
56
65
 
57
66
  def metrics(self, job_id, window_seconds=3600):
67
+ self.metrics_windows.append(window_seconds)
68
+ if self.metrics_wide_result is not None and window_seconds > 3600:
69
+ return self.metrics_wide_result
58
70
  return self.metrics_result
59
71
 
60
72
  def stats(self):
@@ -475,6 +487,265 @@ def test_watch_prints_progress_and_gpu(fake, capsys):
475
487
  assert "samples 120" in printed
476
488
 
477
489
 
490
+ def gpu_sample(util, used, total=81920, index=0):
491
+ """One reading, for the card tests below."""
492
+ return {"utilization_percent": util, "memory_used_mib": used,
493
+ "memory_total_mib": total, "memory_percent": round(100 * used / total, 1),
494
+ "temperature_c": 62, "power_w": 310.0, "gpu_index": index}
495
+
496
+
497
+ def four_a100_cards():
498
+ """job-66b46719b854 as the server reports it: four A100s in one pod.
499
+
500
+ The numbers are that run's: every card peaked at 77,631 of 81,920 MiB, and
501
+ the utilisation INSIDE those four highest-memory samples read 0, 3, 1 and 95
502
+ while the cards themselves peaked at 100, 99, 100 and 99. That gap is the
503
+ defect the screen carried until 2026-09-11, so the fixture keeps it.
504
+ """
505
+ peaks = [(0, 100, 84.6), (3, 99, 37.8), (1, 100, 79.2), (95, 99, 81.3)]
506
+ return [
507
+ {"gpu_index": i,
508
+ "series": [{}] * 393,
509
+ "latest": gpu_sample(0, 0, index=i),
510
+ "peak": gpu_sample(at_peak, 77631, index=i),
511
+ "avg_utilization_percent": avg,
512
+ "peak_utilization_percent": real_peak,
513
+ "sample_count": 785}
514
+ for i, (at_peak, real_peak, avg) in enumerate(peaks)
515
+ ]
516
+
517
+
518
+ def test_every_card_is_printed_not_only_the_first(fake, capsys):
519
+ # ★ THE DEFECT. `latest_gpu` is card 0, so a four-A100 job printed one card
520
+ # and three quarters of a $44 run was invisible from the CLI. The screen was
521
+ # fixed on 2026-09-08 and this surface was not.
522
+ fake.metrics_result = {
523
+ "latest_gpu": gpu_sample(94, 38200),
524
+ "peak_gpu": gpu_sample(0, 77631),
525
+ "gpu_series": [{}] * 393,
526
+ "sample_count": 785,
527
+ "peak_utilization_percent": 100,
528
+ "avg_utilization_percent": 84.6,
529
+ "cards": four_a100_cards(),
530
+ "window_seconds": 604800, "note": "",
531
+ }
532
+ assert run(["watch", "job-66b46719b854"]) == cli.EXIT_OK
533
+ printed = capsys.readouterr().out
534
+ # `GPU 0` is a table row; `GPU util` and `GPU memory` are the headline above
535
+ # it, so the card index is what tells the two apart.
536
+ rows = {line.split()[1]: line.split()
537
+ for line in printed.splitlines()
538
+ if line.startswith(" GPU ") and line.split()[1].isdigit()}
539
+ assert sorted(rows) == ["0", "1", "2", "3"]
540
+
541
+ # A row reads: GPU <n> <used> / <total> MiB (<pct>%) <peak util> <avg> <n>
542
+ # so the last three fields are the three that were wrong or missing.
543
+ expected = {"0": ("100%", "84.6%"), "1": ("99%", "37.8%"),
544
+ "2": ("100%", "79.2%"), "3": ("99%", "81.3%")}
545
+ for index, (peak_util, avg_util) in expected.items():
546
+ fields = rows[index]
547
+ # Each card's own peak memory, not card 0's repeated four times.
548
+ assert fields[2:7] == ["77,631", "/", "81,920", "MiB", "(95%)"]
549
+ # ★ `peak_utilization_percent`, NOT the utilisation inside the memory
550
+ # peak. The two differ on this very run: the four highest-memory samples
551
+ # read 0%, 3%, 1% and 95% while the cards peaked at 100, 99, 100 and 99.
552
+ assert fields[7] == peak_util
553
+ assert fields[8] == avg_util
554
+ # The readings this card printed, not the 393 points left after the
555
+ # server thinned the series for the chart.
556
+ assert fields[9] == "785"
557
+
558
+
559
+ def test_one_card_prints_no_table_because_it_would_repeat_the_block(fake, capsys):
560
+ # With a single card the table's four numbers are the same four printed
561
+ # directly above it, so it is drawn on the same condition the screen uses.
562
+ fake.metrics_result = {
563
+ "latest_gpu": gpu_sample(94, 38200, total=45440),
564
+ "gpu_series": [{}] * 120,
565
+ "cards": [{"gpu_index": 0, "series": [{}] * 120,
566
+ "latest": gpu_sample(94, 38200, total=45440),
567
+ "peak": gpu_sample(94, 38200, total=45440),
568
+ "avg_utilization_percent": 90.0,
569
+ "peak_utilization_percent": 99,
570
+ "sample_count": 120}],
571
+ "window_seconds": 3600, "note": "",
572
+ }
573
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
574
+ printed = capsys.readouterr().out
575
+ assert "peak memory" not in printed # the table's header row
576
+ assert "38,200 / 45,440 MiB" in printed
577
+
578
+
579
+ def test_a_finished_job_leads_with_the_peak_not_the_idle_last_reading(fake, capsys):
580
+ # ★ THE DEFECT. A finished job's last reading is the idle card in the
581
+ # seconds before teardown -- 0%, 0 MiB -- and printing it as "GPU util" says
582
+ # the run was idle. What a post-mortem asks for is the peak, because running
583
+ # out of memory is what kills a run. The screen was changed to this shape on
584
+ # 2026-09-07.
585
+ fake.status_result = {"job_id": "job-66b46719b854", "name": "baseline-c",
586
+ "phase": "Succeeded"}
587
+ fake.metrics_result = {
588
+ "latest_gpu": gpu_sample(0, 0),
589
+ "peak_gpu": gpu_sample(0, 77631),
590
+ "gpu_series": [{}] * 393,
591
+ "sample_count": 785,
592
+ "peak_utilization_percent": 100,
593
+ "avg_utilization_percent": 84.6,
594
+ "window_seconds": 604800, "note": "",
595
+ }
596
+ assert run(["watch", "job-66b46719b854"]) == cli.EXIT_OK
597
+ printed = capsys.readouterr().out
598
+ assert "peak memory" in printed
599
+ assert "77,631 / 81,920 MiB" in printed
600
+ # The peak of the CARD, not the utilisation inside the highest-memory
601
+ # sample, which on this run was 0 while the card reached 100.
602
+ assert "peak util 100%" in printed
603
+ assert "avg util 84.6%" in printed
604
+ # The last reading is still shown, and named for what it is.
605
+ assert "last reading 0%, 0 MiB (run ended)" in printed
606
+ # The running job's labels are gone: leaving "GPU util 0%" on the screen is
607
+ # exactly the claim this fix removes.
608
+ assert "GPU util" not in printed
609
+
610
+
611
+ def test_a_running_job_keeps_the_live_readings_as_the_headline(fake, capsys):
612
+ # The mirror of the test above. `peak_gpu` being present is not on its own a
613
+ # finished job -- the server sends it for a running one too -- so the phase
614
+ # is what decides, and a Running job still leads with what the card is doing
615
+ # NOW.
616
+ fake.metrics_result = {
617
+ "latest_gpu": gpu_sample(94, 38200, total=45440),
618
+ "peak_gpu": gpu_sample(99, 40000, total=45440),
619
+ "gpu_series": [{}] * 120,
620
+ "peak_utilization_percent": 99,
621
+ "avg_utilization_percent": 90.0,
622
+ "window_seconds": 3600, "note": "",
623
+ }
624
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
625
+ printed = capsys.readouterr().out
626
+ assert "GPU util" in printed
627
+ assert "38,200 / 45,440 MiB" in printed
628
+ assert "last reading" not in printed
629
+
630
+
631
+ def test_a_phase_lookup_that_fails_still_prints_the_gpu_numbers(fake, capsys):
632
+ # The numbers are what the command exists for, so a job deleted between the
633
+ # metrics call and the phase call falls back to the running layout rather
634
+ # than exiting with nothing printed.
635
+ def refuse(job_id):
636
+ raise ServerError("404: no such job")
637
+
638
+ fake.status = refuse
639
+ fake.metrics_result = {
640
+ "latest_gpu": gpu_sample(94, 38200, total=45440),
641
+ "gpu_series": [{}] * 120,
642
+ "window_seconds": 3600, "note": "",
643
+ }
644
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
645
+ assert "38,200 / 45,440 MiB" in capsys.readouterr().out
646
+
647
+
648
+ def _hours_ago(hours):
649
+ """An RFC 3339 timestamp that many hours in the past, as the server writes it."""
650
+ import datetime as _dt
651
+ t = _dt.datetime.now(_dt.timezone.utc) - _dt.timedelta(hours=hours)
652
+ return t.strftime("%Y-%m-%dT%H:%M:%SZ")
653
+
654
+
655
+ def test_watch_widens_the_window_to_reach_a_finished_jobs_readings(fake, capsys):
656
+ # ★ THE DEFECT: nothing is stored, so `watch` re-reads the log and the window
657
+ # is measured back from NOW. A run that ended five hours ago has every
658
+ # reading outside the default hour, and `watch` printed the phase and no
659
+ # numbers -- the phase being the one thing the reader already knew.
660
+ fake.status_result = {"job_id": "job-a8acdef80a07", "name": "bank-exp2",
661
+ "phase": "Succeeded", "started_at": _hours_ago(9),
662
+ "finished_at": _hours_ago(5)}
663
+ fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
664
+ fake.metrics_wide_result = {
665
+ "latest_gpu": gpu_sample(94, 38200, total=45440),
666
+ "peak_gpu": gpu_sample(94, 38200, total=45440),
667
+ "peak_utilization_percent": 94,
668
+ "gpu_series": [{}] * 120, "window_seconds": 36000, "note": "",
669
+ }
670
+
671
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
672
+ assert len(fake.metrics_windows) == 2, "the empty first window was not retried"
673
+ assert fake.metrics_windows[0] == 3600
674
+ # Nine hours of run plus a minute of headroom, so the FIRST reading is inside
675
+ # the window rather than exactly on its edge.
676
+ assert fake.metrics_windows[1] >= 9 * 3600
677
+ assert "38,200 / 45,440 MiB" in capsys.readouterr().out
678
+
679
+
680
+ def test_a_window_the_caller_chose_is_never_widened(fake):
681
+ # An explicit --window is an instruction, not a default. Widening it would
682
+ # answer a question the caller did not ask, and the answer would be a bigger
683
+ # read of the log than they wanted.
684
+ fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Succeeded",
685
+ "started_at": _hours_ago(9)}
686
+ fake.metrics_result = {"gpu_series": [], "window_seconds": 600, "note": ""}
687
+
688
+ assert run(["watch", "job-a8acdef80a07", "--window", "600"]) == cli.EXIT_OK
689
+ assert fake.metrics_windows == [600]
690
+
691
+
692
+ def test_a_running_job_is_not_widened_however_quiet_it_is(fake):
693
+ # A running job's readings ARE inside the last hour, so an empty answer means
694
+ # the job has printed nothing yet. Re-reading a week of its log would buy a
695
+ # second empty answer and a much larger response.
696
+ fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Running",
697
+ "started_at": _hours_ago(9)}
698
+ fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
699
+
700
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
701
+ assert fake.metrics_windows == [3600]
702
+
703
+
704
+ def test_the_widened_window_stops_at_the_servers_own_cap(fake):
705
+ # `window_seconds` is declared ge=60, le=604800 on the metrics route. Asking
706
+ # for more is a 422, so a job that started a month ago must be clamped here
707
+ # rather than turned into an error the reader cannot act on.
708
+ fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Failed",
709
+ "started_at": _hours_ago(24 * 30)}
710
+ fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
711
+
712
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
713
+ assert fake.metrics_windows[1] == cli.MAX_WATCH_WINDOW == 604800
714
+
715
+
716
+ def test_a_finished_job_with_no_start_time_leaves_the_window_alone(fake):
717
+ # Nothing to compute a width from. Widening to the cap "just in case" would
718
+ # read seven days of log off the back of a missing field.
719
+ fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Succeeded"}
720
+ fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
721
+
722
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
723
+ assert fake.metrics_windows == [3600]
724
+
725
+
726
+ def test_the_window_help_quotes_the_servers_real_cap(capsys):
727
+ # It said 86400 and the server accepts 604800, so the help talked a reader
728
+ # out of a window the server would have answered.
729
+ with pytest.raises(SystemExit):
730
+ cli.main(["watch", "--help"])
731
+ printed = capsys.readouterr().out
732
+ assert "604800" in printed
733
+ assert "86400" not in printed
734
+
735
+
736
+ def test_watch_json_prints_the_cards_and_asks_for_no_phase(fake, capsys):
737
+ # --json is the server's answer verbatim, so the second request would buy
738
+ # nothing. It is one GET per typed `watch` and worth not making.
739
+ def refuse(job_id):
740
+ raise AssertionError("--json must not ask for the phase")
741
+
742
+ fake.status = refuse
743
+ fake.metrics_result = {"cards": four_a100_cards(), "gpu_series": [],
744
+ "window_seconds": 3600, "note": ""}
745
+ assert run(["watch", "job-66b46719b854", "--json"]) == cli.EXIT_OK
746
+ assert len(json.loads(capsys.readouterr().out)["cards"]) == 4
747
+
748
+
478
749
  def test_samples_counts_the_readings_taken_not_the_chart_points(fake, capsys):
479
750
  # ★ THE SAME DEFECT THE SCREEN HAD. The server thins the series to at most
480
751
  # 400 points so a chart can draw it; job-66b46719b854 printed 785 readings
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes