hyperun 0.2.0__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: hyperun
3
- Version: 0.2.0
3
+ Version: 0.2.2
4
4
  Summary: Submit a GPU job and get results back. No kubectl, no cloud account.
5
5
  Author: DDPS Lab
6
6
  License-Expression: Apache-2.0
@@ -50,6 +50,7 @@ from __future__ import annotations
50
50
 
51
51
  import argparse
52
52
  import base64
53
+ import datetime
53
54
  import getpass
54
55
  import json
55
56
  import time
@@ -365,9 +366,10 @@ def build_parser() -> argparse.ArgumentParser:
365
366
  "through the job's driver pod into the workload container on the "
366
367
  "rented machine and brings the exit code back like ssh. It is NOT a "
367
368
  "TTY — no vim, no top, about 25 seconds per command — because the "
368
- "server is a Lambda and cannot hold a terminal open. AWS and GCP "
369
- "machine rentals only: a RunPod job is a rented container with no "
370
- "machine behind it, and the relay refuses it. ★ PUT OPTIONS BEFORE THE "
369
+ "server is a Lambda and cannot hold a terminal open. AWS, GCP and "
370
+ "Shadeform machine rentals only: a RunPod job is a rented container "
371
+ "with no machine behind it, so there is no cluster to attach to and "
372
+ "the relay refuses it. ★ PUT OPTIONS BEFORE THE "
371
373
  "JOB ID -- everything after it is sent to the workload as-is, so "
372
374
  "`shell job-x --slot 2` asks pod 0 and passes `--slot 2` to the shell. "
373
375
  "That is refused rather than obeyed.",
@@ -386,8 +388,13 @@ def build_parser() -> argparse.ArgumentParser:
386
388
  )
387
389
  watch.add_argument("job_id")
388
390
  watch.add_argument(
389
- "--window", type=int, default=3600, metavar="SECONDS",
390
- help="how far back to read (default 3600, max 86400)",
391
+ # default=None, not 3600, and the difference is load-bearing: it is how
392
+ # `cmd_watch` tells "the caller wants an hour" from "the caller said
393
+ # nothing", and only the second one may be widened to reach a finished
394
+ # job's readings. 604800 is the server's cap, not a number chosen here.
395
+ "--window", type=int, default=None, metavar="SECONDS",
396
+ help="how far back to read (default 3600, max 604800). A finished job "
397
+ "is widened automatically to reach its own readings.",
391
398
  )
392
399
  add_json_flag(watch)
393
400
 
@@ -826,11 +833,26 @@ def cmd_shell(args: argparse.Namespace) -> int:
826
833
  pod onto the rented machine and returns the exit code like ssh. With a
827
834
  trailing `-- command` it runs once and exits with that code; without one
828
835
  it prompts, which FEELS like a slow shell and is honestly a request loop.
836
+
837
+ ★ THE PROMPT USES A SESSION AND THE ONE-SHOT FORM DOES NOT, since
838
+ 2026-09-10. The prompt types into a shell that stays running in the driver
839
+ pod, so `cd`, an exported variable and an activated venv survive from one
840
+ line to the next -- which is what a person at a prompt assumes and what
841
+ every earlier version quietly did not do. A `-- command` invocation is a
842
+ script's shape and gets the stateless form: it wants no history and should
843
+ not leave a session behind for the idle timer to close.
829
844
  """
830
845
  client = client_from_config()
831
-
832
- def run_once(line: str) -> int:
833
- answer = client.exec_in_job(args.job, line, slot=args.slot)
846
+ # The output sequence the session has already shown us. It is here rather
847
+ # than inside run_once because it has to survive between lines, which is the
848
+ # whole point.
849
+ state = {"seq": 0}
850
+
851
+ def run_once(line: str, session: bool = False) -> int:
852
+ answer = client.exec_in_job(args.job, line, slot=args.slot,
853
+ session=session, seq=state["seq"])
854
+ if session:
855
+ state["seq"] = int(answer.get("seq", state["seq"]))
834
856
  output = answer.get("output") or ""
835
857
  if output:
836
858
  print(output, end="" if output.endswith("\n") else "\n")
@@ -873,7 +895,8 @@ def cmd_shell(args: argparse.Namespace) -> int:
873
895
  return run_once(" ".join(words))
874
896
 
875
897
  print(
876
- f"one command per line, ~25s each; 'exit' or Ctrl-D leaves. Not a TTY.",
898
+ f"one command per line, ~25s each; 'exit' or Ctrl-D leaves. Not a TTY, "
899
+ f"but `cd` and exported variables DO survive between lines.",
877
900
  file=sys.stderr,
878
901
  )
879
902
  while True:
@@ -888,10 +911,29 @@ def cmd_shell(args: argparse.Namespace) -> int:
888
911
  if line in ("exit", "quit"):
889
912
  return EXIT_OK
890
913
  try:
891
- code = run_once(line)
914
+ code = run_once(line, session=True)
892
915
  if code:
893
916
  print(f"(exit {code})", file=sys.stderr)
894
917
  except ServerError as exc:
918
+ # A SESSION THAT IS GONE IS THE ONE ERROR WORTH ACTING ON. The
919
+ # server answers 409 when the driver pod has no session for this
920
+ # slot -- it timed out after ten minutes unread, or the workload
921
+ # restarted. Reopening is right there and only there: everywhere
922
+ # else a new shell would silently lose the `cd` the person is
923
+ # relying on, which is the failure this whole feature exists to
924
+ # end. The sequence resets with it, because the new shell's
925
+ # output starts from nothing.
926
+ if "open one first" in str(exc) or "no session" in str(exc):
927
+ state["seq"] = 0
928
+ print("(the session had closed; reopening — `cd` and variables "
929
+ "from before are gone)", file=sys.stderr)
930
+ try:
931
+ code = run_once(line, session=True)
932
+ if code:
933
+ print(f"(exit {code})", file=sys.stderr)
934
+ continue
935
+ except ServerError as retry_exc:
936
+ exc = retry_exc
895
937
  # One failed command must not end the session: say what the server
896
938
  # said and keep the prompt.
897
939
  print(f"error: {exc}", file=sys.stderr)
@@ -1052,13 +1094,128 @@ def bar(percent: float, width: int = 20) -> str:
1052
1094
  return "#" * filled + "-" * (width - filled)
1053
1095
 
1054
1096
 
1097
+ # DDPSRUN-WATCH-TERMINAL. The three phases a job never leaves. They are the
1098
+ # controller's own words: the phase enum in api/v1alpha1/pacsjob_types.go
1099
+ # carries Pending / Starting / Running / Recovering / Succeeded / Failed /
1100
+ # Compared and nothing else, and the screen keeps the same three in
1101
+ # `const TERMINAL` (ui/app.js). Recovering is NOT here -- the machine was
1102
+ # reclaimed and the job is being restarted, so the run is still going.
1103
+ TERMINAL_PHASES = ("Succeeded", "Failed", "Compared")
1104
+
1105
+ # How far back `watch` reads when the caller says nothing, and the furthest the
1106
+ # server will let it reach. The cap is the server's own: `window_seconds` on
1107
+ # `GET /v1/jobs/{id}/metrics` is declared `ge=60, le=604800`
1108
+ # (server/ddpsrun_server/main.py). The `--window` help said 86400 here, which was
1109
+ # simply wrong -- one day, against the server's seven.
1110
+ DEFAULT_WATCH_WINDOW = 3600
1111
+ MAX_WATCH_WINDOW = 604800
1112
+
1113
+
1114
+ def _window_reaching_back_to_start(job: dict[str, Any]) -> int:
1115
+ """How wide a window must be to contain a FINISHED job's readings.
1116
+
1117
+ ★ WHY THIS EXISTS. Nothing is stored: `watch` re-reads the job's own log, and
1118
+ the window is measured back from NOW. A running job's readings are therefore
1119
+ always inside the default hour. A job that finished five hours ago has NONE
1120
+ of its readings in the last hour, so `hyperun watch` on it printed the phase
1121
+ and no numbers -- and the phase is the one thing the reader already knew.
1122
+
1123
+ The fix is not a larger default, which would make every `watch` on a running
1124
+ job re-read a week of log. It is to widen only when the job is over AND the
1125
+ caller did not pick a window.
1126
+
1127
+ Args:
1128
+ job: the body of `GET /v1/jobs/{id}`.
1129
+
1130
+ Returns:
1131
+ Seconds, or 0 when the job is not finished or carries no start time, in
1132
+ which case the caller leaves the window as it was.
1133
+ """
1134
+ if job.get("phase") not in TERMINAL_PHASES:
1135
+ return 0
1136
+ started = job.get("started_at") or job.get("created_at")
1137
+ if not started:
1138
+ return 0
1139
+ try:
1140
+ began = datetime.datetime.fromisoformat(started.replace("Z", "+00:00"))
1141
+ except ValueError:
1142
+ return 0
1143
+ elapsed = (datetime.datetime.now(datetime.timezone.utc) - began).total_seconds()
1144
+ # A minute of headroom so the FIRST reading falls inside the window instead
1145
+ # of exactly on its edge, and the server's own cap over the top.
1146
+ return min(MAX_WATCH_WINDOW, max(60, int(elapsed) + 60))
1147
+
1148
+
1055
1149
  def cmd_watch(args: argparse.Namespace) -> int:
1056
- """Print a job's GPU usage and how far the training has got."""
1057
- result = client_from_config().metrics(args.job_id, args.window)
1150
+ """Print a job's GPU usage and how far the training has got.
1151
+
1152
+ ONE BLOCK PER CARD, because a job can rent several. baseline-c rents four
1153
+ A100s in one pod, and until today this printed `latest_gpu`, which is card 0
1154
+ alone -- three quarters of a $44 run was invisible from the CLI. The screen
1155
+ had the same defect and was fixed on 2026-09-08. `cards` is the server's
1156
+ per-card answer (CardMetricsView in server/ddpsrun_server/models.py); a
1157
+ server too old to send it answers with an empty list, and then the
1158
+ single-card block below is the whole story, exactly as before.
1159
+
1160
+ WHY THIS ASKS FOR THE PHASE AS WELL AS THE METRICS. A finished job's last
1161
+ reading is the idle card in the seconds before teardown -- 0%, 0 MiB -- so
1162
+ printing it under "GPU util" tells a reader the run was idle when it was
1163
+ not. Leading with the peak instead requires knowing the job is over, and
1164
+ THE METRICS ANSWER DOES NOT SAY: MetricsResponse carries latest_gpu,
1165
+ gpu_series, peak_gpu, the two utilisation figures, sample_count, cards,
1166
+ progress, window_seconds and note, and no phase at all
1167
+ (server/ddpsrun_server/models.py). Guessing from the readings fails in both
1168
+ directions -- a job whose framework has not allocated yet also reads 0 MiB,
1169
+ and a job killed mid-step leaves a final reading that is nowhere near 0 --
1170
+ so this asks the server instead.
1171
+
1172
+ WHAT THAT COSTS: one extra `GET /v1/jobs/{id}` per typed `watch`. `watch`
1173
+ prints once and exits rather than polling, so it is one extra request per
1174
+ command, not one per second, and `--json` pays it only when the first window
1175
+ came back empty. The screen pays the same price -- it reads the job and hands
1176
+ it to drawMetrics (ui/app.js).
1177
+ """
1178
+ client = client_from_config()
1179
+ window = DEFAULT_WATCH_WINDOW if args.window is None else args.window
1180
+ result = client.metrics(args.job_id, window)
1181
+
1182
+ # ★ AN EMPTY ANSWER ON A FINISHED JOB IS THE WINDOW, NOT THE JOB. The window
1183
+ # is measured back from now, so a run that ended hours ago has every reading
1184
+ # outside the default hour and `watch` printed nothing. Widening is done
1185
+ # ONLY here: the caller did not choose a window, and the first one came back
1186
+ # with no readings at all. A running job never reaches this branch, so the
1187
+ # usual `watch` is still exactly one metrics request.
1188
+ #
1189
+ # "No readings" means BOTH are empty. `cards` is the per-card answer and
1190
+ # `latest_gpu` is card 0; a multi-card job fills `cards`, while a server too
1191
+ # old to send it fills only `latest_gpu`. Testing one alone would make
1192
+ # `watch --json` on a four-card job buy a second request it does not need.
1193
+ job: dict[str, Any] | None = None
1194
+ if args.window is None and not result.get("latest_gpu") and not result.get("cards"):
1195
+ try:
1196
+ job = client.status(args.job_id)
1197
+ except ServerError:
1198
+ job = None
1199
+ wider = _window_reaching_back_to_start(job) if job else 0
1200
+ if wider > window:
1201
+ result = client.metrics(args.job_id, wider)
1202
+ window = wider
1203
+
1058
1204
  if args.json:
1205
+ # The server's answer, verbatim. It already contains `cards`, so nothing
1206
+ # here needs the phase.
1059
1207
  print(json.dumps(result, indent=2, ensure_ascii=False))
1060
1208
  return EXIT_OK
1061
1209
 
1210
+ # A phase lookup that fails must not lose the GPU numbers this command
1211
+ # exists to print, so it falls back to the running layout. The likely cause
1212
+ # is the job being deleted between the two calls, and the running layout is
1213
+ # the behaviour this command had until today rather than a wrong number.
1214
+ try:
1215
+ done = (job or client.status(args.job_id)).get("phase") in TERMINAL_PHASES
1216
+ except ServerError:
1217
+ done = False
1218
+
1062
1219
  progress = result.get("progress")
1063
1220
  if progress:
1064
1221
  print(f" training {bar(progress['percent'])} "
@@ -1073,20 +1230,102 @@ def cmd_watch(args: argparse.Namespace) -> int:
1073
1230
 
1074
1231
  gpu = result.get("latest_gpu")
1075
1232
  if gpu:
1076
- print(f" GPU util {bar(gpu['utilization_percent'])} "
1077
- f"{gpu['utilization_percent']}%")
1078
- print(f" GPU memory {bar(gpu['memory_percent'])} "
1079
- f"{gpu['memory_used_mib']:,} / {gpu['memory_total_mib']:,} MiB "
1080
- f"({gpu['memory_percent']}%)")
1081
- print(f" temp, power {gpu['temperature_c']} C, {gpu['power_w']} W")
1082
- print(f" samples {len(result.get('gpu_series', []))}, "
1083
- f"last {result['window_seconds']}s")
1233
+ # The reading with the most MEMORY in use. A server too old to send it
1234
+ # leaves the last reading as the only one there is, which is what the
1235
+ # screen falls back to as well.
1236
+ peak = result.get("peak_gpu") or gpu
1237
+ if done:
1238
+ # A FINISHED RUN LEADS WITH THE PEAK. Memory is what kills a run, so
1239
+ # the peak is the number a post-mortem came here for; the last
1240
+ # reading is kept, because it is still evidence, but it is named for
1241
+ # what it is instead of being labelled "GPU util". The screen was
1242
+ # changed to this shape on 2026-09-07 after the 0%, 0 MiB readout
1243
+ # was reported as the panel being broken.
1244
+ print(f" peak memory {bar(peak['memory_percent'])} "
1245
+ f"{peak['memory_used_mib']:,} / {peak['memory_total_mib']:,} MiB "
1246
+ f"({peak['memory_percent']}%)")
1247
+ # ★ `peak_utilization_percent`, NOT `peak['utilization_percent']`,
1248
+ # and the difference is the whole defect. `peak` is the sample with
1249
+ # the most memory in it and its utilisation is whatever the card
1250
+ # happened to be doing at that instant. On job-66b46719b854 all four
1251
+ # A100s reached 77,631 MiB and the utilisation inside those four
1252
+ # samples read 0, 3, 1 and 95 -- while every one of those cards
1253
+ # actually peaked at 99 or 100 and averaged between 37.8 and 84.6.
1254
+ # Reported on the screen 2026-09-11; the CLI never printed either
1255
+ # figure, so it is being added correct rather than fixed.
1256
+ print(f" peak util {_percent(result.get('peak_utilization_percent'))}")
1257
+ print(f" avg util {_percent(result.get('avg_utilization_percent'))}")
1258
+ print(f" last reading {gpu['utilization_percent']}%, "
1259
+ f"{gpu['memory_used_mib']:,} MiB (run ended)")
1260
+ else:
1261
+ print(f" GPU util {bar(gpu['utilization_percent'])} "
1262
+ f"{gpu['utilization_percent']}%")
1263
+ print(f" GPU memory {bar(gpu['memory_percent'])} "
1264
+ f"{gpu['memory_used_mib']:,} / {gpu['memory_total_mib']:,} MiB "
1265
+ f"({gpu['memory_percent']}%)")
1266
+ print(f" temp, power {gpu['temperature_c']} C, {gpu['power_w']} W")
1267
+ # `sample_count`, not len(gpu_series): the server thins the series to at
1268
+ # most 400 points so a chart can draw it, and job-66b46719b854 printed
1269
+ # 785 readings per card while this line said 393. The screen had the same
1270
+ # defect and was fixed on 2026-09-12. A server too old to send the count
1271
+ # falls back to the thinned length, which is the only number it has.
1272
+ taken = result.get("sample_count") or len(result.get("gpu_series", []))
1273
+ print(f" samples {taken}, last {result['window_seconds']}s")
1274
+ print_card_table(result.get("cards") or [])
1084
1275
 
1085
1276
  if result.get("note"):
1086
1277
  print(f" note {result['note']}")
1087
1278
  return EXIT_OK
1088
1279
 
1089
1280
 
1281
+ def _percent(value: float | int | None) -> str:
1282
+ """One utilisation figure, or a dash when the server did not send it.
1283
+
1284
+ The two utilisation fields arrived on 2026-09-10 and a server older than
1285
+ that answers with them absent. A blank line reads as 0, which is the very
1286
+ thing the peak figures exist to stop being claimed.
1287
+ """
1288
+ return "-" if value is None else f"{value}%"
1289
+
1290
+
1291
+ def print_card_table(cards: list[dict[str, Any]]) -> None:
1292
+ """One row per GPU card, printed only when the job rented more than one.
1293
+
1294
+ WHY THIS EXISTS. Everything above describes card 0 -- `latest_gpu` and
1295
+ `peak_gpu` are the lowest-indexed card's readings, unchanged since before
1296
+ `cards` existed. job-66b46719b854 is four A100s in one pod, and reading its
1297
+ CLI output left three of them invisible.
1298
+
1299
+ WHY ONLY ABOVE ONE CARD. With a single card these four numbers repeat the
1300
+ block above it word for word, so the table is drawn on the same condition
1301
+ the screen draws its own (`cards.length > 1` in ui/app.js).
1302
+
1303
+ Args:
1304
+ cards: the response's `cards`, each a CardMetricsView -- `gpu_index`,
1305
+ `series`, `latest`, `peak`, `avg_utilization_percent`,
1306
+ `peak_utilization_percent`, `sample_count`.
1307
+ """
1308
+ if len(cards) < 2:
1309
+ return
1310
+ print(f" {'card':<8}{'peak memory':>27}{'peak util':>11}"
1311
+ f"{'avg util':>10}{'samples':>9}")
1312
+ for card in cards:
1313
+ # `peak` is this card's highest-MEMORY reading; falling back to its last
1314
+ # one matches the screen's table and keeps a row rather than dropping a
1315
+ # card that only ever printed once.
1316
+ sample = card.get("peak") or card.get("latest") or {}
1317
+ memory = "-" if not sample else (
1318
+ f"{sample['memory_used_mib']:,} / {sample['memory_total_mib']:,} MiB "
1319
+ f"({sample['memory_percent']:.0f}%)")
1320
+ # Same two corrections as the block above: the card's own highest
1321
+ # utilisation rather than the utilisation inside its memory peak, and
1322
+ # the readings it printed rather than the points left after thinning.
1323
+ taken = card.get("sample_count") or len(card.get("series") or [])
1324
+ print(f" {('GPU ' + str(card['gpu_index'])):<8}{memory:>27}"
1325
+ f"{_percent(card.get('peak_utilization_percent')):>11}"
1326
+ f"{_percent(card.get('avg_utilization_percent')):>10}{taken:>9}")
1327
+
1328
+
1090
1329
  def cmd_stats(args: argparse.Namespace) -> int:
1091
1330
  """Print this caller's team figures."""
1092
1331
  result = client_from_config().stats()
@@ -193,7 +193,8 @@ class Client:
193
193
  return self._call("GET", "/v1/stats").json()
194
194
 
195
195
  def exec_in_job(
196
- self, job: str, command: str, slot: int = 0, timeout_seconds: int = 20
196
+ self, job: str, command: str, slot: int = 0, timeout_seconds: int = 20,
197
+ session: bool = False, seq: int = 0,
197
198
  ) -> dict[str, Any]:
198
199
  """Run one shell line inside a running job's workload container.
199
200
 
@@ -206,6 +207,13 @@ class Client:
206
207
  command: one shell line, run as `sh -lc <command>`.
207
208
  slot: which pod of a parallel job.
208
209
  timeout_seconds: server-side wait, capped at 25 by the server.
210
+ session: type the line into a shell ALREADY RUNNING in the driver
211
+ pod, so `cd` and exported variables survive to the next call.
212
+ False starts a fresh `sh -lc`, which is what a script wants.
213
+ seq: with `session`, the output sequence this caller last saw. The
214
+ reply carries the next one. It exists because the caller is a
215
+ different process on the server's side every time and cannot
216
+ hold a position in the output.
209
217
  """
210
218
  return self._call(
211
219
  "POST",
@@ -214,6 +222,8 @@ class Client:
214
222
  "command": command,
215
223
  "slot": slot,
216
224
  "timeout_seconds": timeout_seconds,
225
+ "session": session,
226
+ "seq": seq,
217
227
  },
218
228
  # The server may hold the request for timeout_seconds before
219
229
  # answering; the read timeout has to outlive that on purpose.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: hyperun
3
- Version: 0.2.0
3
+ Version: 0.2.2
4
4
  Summary: Submit a GPU job and get results back. No kubectl, no cloud account.
5
5
  Author: DDPS Lab
6
6
  License-Expression: Apache-2.0
@@ -34,7 +34,7 @@ build-backend = "setuptools.build_meta"
34
34
 
35
35
  [project]
36
36
  name = "hyperun"
37
- version = "0.2.0"
37
+ version = "0.2.2"
38
38
  description = "Submit a GPU job and get results back. No kubectl, no cloud account."
39
39
  requires-python = ">=3.9"
40
40
  dependencies = ["requests>=2.31", "PyYAML>=6.0"]
@@ -38,6 +38,15 @@ class FakeClient:
38
38
  self.estimate_result = {}
39
39
  self.validate_result = {"ok": True, "findings": [], "not_checked": []}
40
40
  self.metrics_result = {"window_seconds": 3600, "gpu_series": [], "note": ""}
41
+ # Every window `watch` asked for, in order. The widening of a finished
42
+ # job's window is invisible in the printed output -- the numbers look the
43
+ # same whichever window found them -- so the only way to test it is to
44
+ # record what was asked.
45
+ self.metrics_windows: list[int] = []
46
+ # What to answer once the window is wider than the default hour. None
47
+ # means "the same answer whatever the window", which is what every test
48
+ # written before the widening existed expects.
49
+ self.metrics_wide_result = None
41
50
  self.stats_result = {"team": "", "members": [], "jobs": 0, "gpu_hours": 0.0,
42
51
  "cost_usd": 0.0, "unpriced_jobs": 0, "note": ""}
43
52
  self.secrets_result = {"names": ["GITHUB_PAT", "HF_TOKEN"],
@@ -55,6 +64,9 @@ class FakeClient:
55
64
  return self.validate_result
56
65
 
57
66
  def metrics(self, job_id, window_seconds=3600):
67
+ self.metrics_windows.append(window_seconds)
68
+ if self.metrics_wide_result is not None and window_seconds > 3600:
69
+ return self.metrics_wide_result
58
70
  return self.metrics_result
59
71
 
60
72
  def stats(self):
@@ -100,8 +112,8 @@ class FakeClient:
100
112
  # DDPSRUN-SHELL. Absent until 2026-09-10, which is exactly why the dispatch
101
113
  # bug below survived every release: with no double for this call, no test
102
114
  # could reach cmd_shell at all.
103
- def exec_in_job(self, job, line, slot=0):
104
- self.execed.append((job, line, slot))
115
+ def exec_in_job(self, job, line, slot=0, session=False, seq=0):
116
+ self.execed.append((job, line, slot, session, seq))
105
117
  return self.exec_result
106
118
 
107
119
 
@@ -475,6 +487,298 @@ def test_watch_prints_progress_and_gpu(fake, capsys):
475
487
  assert "samples 120" in printed
476
488
 
477
489
 
490
+ def gpu_sample(util, used, total=81920, index=0):
491
+ """One reading, for the card tests below."""
492
+ return {"utilization_percent": util, "memory_used_mib": used,
493
+ "memory_total_mib": total, "memory_percent": round(100 * used / total, 1),
494
+ "temperature_c": 62, "power_w": 310.0, "gpu_index": index}
495
+
496
+
497
+ def four_a100_cards():
498
+ """job-66b46719b854 as the server reports it: four A100s in one pod.
499
+
500
+ The numbers are that run's: every card peaked at 77,631 of 81,920 MiB, and
501
+ the utilisation INSIDE those four highest-memory samples read 0, 3, 1 and 95
502
+ while the cards themselves peaked at 100, 99, 100 and 99. That gap is the
503
+ defect the screen carried until 2026-09-11, so the fixture keeps it.
504
+ """
505
+ peaks = [(0, 100, 84.6), (3, 99, 37.8), (1, 100, 79.2), (95, 99, 81.3)]
506
+ return [
507
+ {"gpu_index": i,
508
+ "series": [{}] * 393,
509
+ "latest": gpu_sample(0, 0, index=i),
510
+ "peak": gpu_sample(at_peak, 77631, index=i),
511
+ "avg_utilization_percent": avg,
512
+ "peak_utilization_percent": real_peak,
513
+ "sample_count": 785}
514
+ for i, (at_peak, real_peak, avg) in enumerate(peaks)
515
+ ]
516
+
517
+
518
+ def test_every_card_is_printed_not_only_the_first(fake, capsys):
519
+ # ★ THE DEFECT. `latest_gpu` is card 0, so a four-A100 job printed one card
520
+ # and three quarters of a $44 run was invisible from the CLI. The screen was
521
+ # fixed on 2026-09-08 and this surface was not.
522
+ fake.metrics_result = {
523
+ "latest_gpu": gpu_sample(94, 38200),
524
+ "peak_gpu": gpu_sample(0, 77631),
525
+ "gpu_series": [{}] * 393,
526
+ "sample_count": 785,
527
+ "peak_utilization_percent": 100,
528
+ "avg_utilization_percent": 84.6,
529
+ "cards": four_a100_cards(),
530
+ "window_seconds": 604800, "note": "",
531
+ }
532
+ assert run(["watch", "job-66b46719b854"]) == cli.EXIT_OK
533
+ printed = capsys.readouterr().out
534
+ # `GPU 0` is a table row; `GPU util` and `GPU memory` are the headline above
535
+ # it, so the card index is what tells the two apart.
536
+ rows = {line.split()[1]: line.split()
537
+ for line in printed.splitlines()
538
+ if line.startswith(" GPU ") and line.split()[1].isdigit()}
539
+ assert sorted(rows) == ["0", "1", "2", "3"]
540
+
541
+ # A row reads: GPU <n> <used> / <total> MiB (<pct>%) <peak util> <avg> <n>
542
+ # so the last three fields are the three that were wrong or missing.
543
+ expected = {"0": ("100%", "84.6%"), "1": ("99%", "37.8%"),
544
+ "2": ("100%", "79.2%"), "3": ("99%", "81.3%")}
545
+ for index, (peak_util, avg_util) in expected.items():
546
+ fields = rows[index]
547
+ # Each card's own peak memory, not card 0's repeated four times.
548
+ assert fields[2:7] == ["77,631", "/", "81,920", "MiB", "(95%)"]
549
+ # ★ `peak_utilization_percent`, NOT the utilisation inside the memory
550
+ # peak. The two differ on this very run: the four highest-memory samples
551
+ # read 0%, 3%, 1% and 95% while the cards peaked at 100, 99, 100 and 99.
552
+ assert fields[7] == peak_util
553
+ assert fields[8] == avg_util
554
+ # The readings this card printed, not the 393 points left after the
555
+ # server thinned the series for the chart.
556
+ assert fields[9] == "785"
557
+
558
+
559
+ def test_one_card_prints_no_table_because_it_would_repeat_the_block(fake, capsys):
560
+ # With a single card the table's four numbers are the same four printed
561
+ # directly above it, so it is drawn on the same condition the screen uses.
562
+ fake.metrics_result = {
563
+ "latest_gpu": gpu_sample(94, 38200, total=45440),
564
+ "gpu_series": [{}] * 120,
565
+ "cards": [{"gpu_index": 0, "series": [{}] * 120,
566
+ "latest": gpu_sample(94, 38200, total=45440),
567
+ "peak": gpu_sample(94, 38200, total=45440),
568
+ "avg_utilization_percent": 90.0,
569
+ "peak_utilization_percent": 99,
570
+ "sample_count": 120}],
571
+ "window_seconds": 3600, "note": "",
572
+ }
573
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
574
+ printed = capsys.readouterr().out
575
+ assert "peak memory" not in printed # the table's header row
576
+ assert "38,200 / 45,440 MiB" in printed
577
+
578
+
579
+ def test_a_finished_job_leads_with_the_peak_not_the_idle_last_reading(fake, capsys):
580
+ # ★ THE DEFECT. A finished job's last reading is the idle card in the
581
+ # seconds before teardown -- 0%, 0 MiB -- and printing it as "GPU util" says
582
+ # the run was idle. What a post-mortem asks for is the peak, because running
583
+ # out of memory is what kills a run. The screen was changed to this shape on
584
+ # 2026-09-07.
585
+ fake.status_result = {"job_id": "job-66b46719b854", "name": "baseline-c",
586
+ "phase": "Succeeded"}
587
+ fake.metrics_result = {
588
+ "latest_gpu": gpu_sample(0, 0),
589
+ "peak_gpu": gpu_sample(0, 77631),
590
+ "gpu_series": [{}] * 393,
591
+ "sample_count": 785,
592
+ "peak_utilization_percent": 100,
593
+ "avg_utilization_percent": 84.6,
594
+ "window_seconds": 604800, "note": "",
595
+ }
596
+ assert run(["watch", "job-66b46719b854"]) == cli.EXIT_OK
597
+ printed = capsys.readouterr().out
598
+ assert "peak memory" in printed
599
+ assert "77,631 / 81,920 MiB" in printed
600
+ # The peak of the CARD, not the utilisation inside the highest-memory
601
+ # sample, which on this run was 0 while the card reached 100.
602
+ assert "peak util 100%" in printed
603
+ assert "avg util 84.6%" in printed
604
+ # The last reading is still shown, and named for what it is.
605
+ assert "last reading 0%, 0 MiB (run ended)" in printed
606
+ # The running job's labels are gone: leaving "GPU util 0%" on the screen is
607
+ # exactly the claim this fix removes.
608
+ assert "GPU util" not in printed
609
+
610
+
611
+ def test_a_running_job_keeps_the_live_readings_as_the_headline(fake, capsys):
612
+ # The mirror of the test above. `peak_gpu` being present is not on its own a
613
+ # finished job -- the server sends it for a running one too -- so the phase
614
+ # is what decides, and a Running job still leads with what the card is doing
615
+ # NOW.
616
+ fake.metrics_result = {
617
+ "latest_gpu": gpu_sample(94, 38200, total=45440),
618
+ "peak_gpu": gpu_sample(99, 40000, total=45440),
619
+ "gpu_series": [{}] * 120,
620
+ "peak_utilization_percent": 99,
621
+ "avg_utilization_percent": 90.0,
622
+ "window_seconds": 3600, "note": "",
623
+ }
624
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
625
+ printed = capsys.readouterr().out
626
+ assert "GPU util" in printed
627
+ assert "38,200 / 45,440 MiB" in printed
628
+ assert "last reading" not in printed
629
+
630
+
631
+ def test_a_phase_lookup_that_fails_still_prints_the_gpu_numbers(fake, capsys):
632
+ # The numbers are what the command exists for, so a job deleted between the
633
+ # metrics call and the phase call falls back to the running layout rather
634
+ # than exiting with nothing printed.
635
+ def refuse(job_id):
636
+ raise ServerError("404: no such job")
637
+
638
+ fake.status = refuse
639
+ fake.metrics_result = {
640
+ "latest_gpu": gpu_sample(94, 38200, total=45440),
641
+ "gpu_series": [{}] * 120,
642
+ "window_seconds": 3600, "note": "",
643
+ }
644
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
645
+ assert "38,200 / 45,440 MiB" in capsys.readouterr().out
646
+
647
+
648
+ def _hours_ago(hours):
649
+ """An RFC 3339 timestamp that many hours in the past, as the server writes it."""
650
+ import datetime as _dt
651
+ t = _dt.datetime.now(_dt.timezone.utc) - _dt.timedelta(hours=hours)
652
+ return t.strftime("%Y-%m-%dT%H:%M:%SZ")
653
+
654
+
655
+ def test_watch_widens_the_window_to_reach_a_finished_jobs_readings(fake, capsys):
656
+ # ★ THE DEFECT: nothing is stored, so `watch` re-reads the log and the window
657
+ # is measured back from NOW. A run that ended five hours ago has every
658
+ # reading outside the default hour, and `watch` printed the phase and no
659
+ # numbers -- the phase being the one thing the reader already knew.
660
+ fake.status_result = {"job_id": "job-a8acdef80a07", "name": "bank-exp2",
661
+ "phase": "Succeeded", "started_at": _hours_ago(9),
662
+ "finished_at": _hours_ago(5)}
663
+ fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
664
+ fake.metrics_wide_result = {
665
+ "latest_gpu": gpu_sample(94, 38200, total=45440),
666
+ "peak_gpu": gpu_sample(94, 38200, total=45440),
667
+ "peak_utilization_percent": 94,
668
+ "gpu_series": [{}] * 120, "window_seconds": 36000, "note": "",
669
+ }
670
+
671
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
672
+ assert len(fake.metrics_windows) == 2, "the empty first window was not retried"
673
+ assert fake.metrics_windows[0] == 3600
674
+ # Nine hours of run plus a minute of headroom, so the FIRST reading is inside
675
+ # the window rather than exactly on its edge.
676
+ assert fake.metrics_windows[1] >= 9 * 3600
677
+ assert "38,200 / 45,440 MiB" in capsys.readouterr().out
678
+
679
+
680
+ def test_a_window_the_caller_chose_is_never_widened(fake):
681
+ # An explicit --window is an instruction, not a default. Widening it would
682
+ # answer a question the caller did not ask, and the answer would be a bigger
683
+ # read of the log than they wanted.
684
+ fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Succeeded",
685
+ "started_at": _hours_ago(9)}
686
+ fake.metrics_result = {"gpu_series": [], "window_seconds": 600, "note": ""}
687
+
688
+ assert run(["watch", "job-a8acdef80a07", "--window", "600"]) == cli.EXIT_OK
689
+ assert fake.metrics_windows == [600]
690
+
691
+
692
+ def test_a_running_job_is_not_widened_however_quiet_it_is(fake):
693
+ # A running job's readings ARE inside the last hour, so an empty answer means
694
+ # the job has printed nothing yet. Re-reading a week of its log would buy a
695
+ # second empty answer and a much larger response.
696
+ fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Running",
697
+ "started_at": _hours_ago(9)}
698
+ fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
699
+
700
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
701
+ assert fake.metrics_windows == [3600]
702
+
703
+
704
+ def test_the_widened_window_stops_at_the_servers_own_cap(fake):
705
+ # `window_seconds` is declared ge=60, le=604800 on the metrics route. Asking
706
+ # for more is a 422, so a job that started a month ago must be clamped here
707
+ # rather than turned into an error the reader cannot act on.
708
+ fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Failed",
709
+ "started_at": _hours_ago(24 * 30)}
710
+ fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
711
+
712
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
713
+ assert fake.metrics_windows[1] == cli.MAX_WATCH_WINDOW == 604800
714
+
715
+
716
+ def test_a_finished_job_with_no_start_time_leaves_the_window_alone(fake):
717
+ # Nothing to compute a width from. Widening to the cap "just in case" would
718
+ # read seven days of log off the back of a missing field.
719
+ fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Succeeded"}
720
+ fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
721
+
722
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
723
+ assert fake.metrics_windows == [3600]
724
+
725
+
726
+ def test_the_window_help_quotes_the_servers_real_cap(capsys):
727
+ # It said 86400 and the server accepts 604800, so the help talked a reader
728
+ # out of a window the server would have answered.
729
+ with pytest.raises(SystemExit):
730
+ cli.main(["watch", "--help"])
731
+ printed = capsys.readouterr().out
732
+ assert "604800" in printed
733
+ assert "86400" not in printed
734
+
735
+
736
+ def test_watch_json_prints_the_cards_and_asks_for_no_phase(fake, capsys):
737
+ # --json is the server's answer verbatim, so the second request would buy
738
+ # nothing. It is one GET per typed `watch` and worth not making.
739
+ def refuse(job_id):
740
+ raise AssertionError("--json must not ask for the phase")
741
+
742
+ fake.status = refuse
743
+ fake.metrics_result = {"cards": four_a100_cards(), "gpu_series": [],
744
+ "window_seconds": 3600, "note": ""}
745
+ assert run(["watch", "job-66b46719b854", "--json"]) == cli.EXIT_OK
746
+ assert len(json.loads(capsys.readouterr().out)["cards"]) == 4
747
+
748
+
749
+ def test_samples_counts_the_readings_taken_not_the_chart_points(fake, capsys):
750
+ # ★ THE SAME DEFECT THE SCREEN HAD. The server thins the series to at most
751
+ # 400 points so a chart can draw it; job-66b46719b854 printed 785 readings
752
+ # per card and both surfaces counted the 393 that survived. The mean and the
753
+ # peak are computed over all 785, so the count beside them has to be 785.
754
+ fake.metrics_result = {
755
+ "latest_gpu": {"utilization_percent": 94, "memory_used_mib": 38200,
756
+ "memory_total_mib": 45440, "memory_percent": 84.1,
757
+ "temperature_c": 71, "power_w": 298.5},
758
+ "gpu_series": [{}] * 393,
759
+ "sample_count": 785,
760
+ "window_seconds": 604800, "note": "",
761
+ }
762
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
763
+ printed = capsys.readouterr().out
764
+ assert "samples 785" in printed
765
+ assert "samples 393" not in printed
766
+
767
+
768
+ def test_a_server_too_old_to_send_the_count_still_prints_one(fake, capsys):
769
+ # The field arrived on 2026-09-12. Against a server that predates it the
770
+ # thinned length is the only number there is, and a blank is worse.
771
+ fake.metrics_result = {
772
+ "latest_gpu": {"utilization_percent": 10, "memory_used_mib": 1,
773
+ "memory_total_mib": 2, "memory_percent": 50.0,
774
+ "temperature_c": 40, "power_w": 50.0},
775
+ "gpu_series": [{}] * 12,
776
+ "window_seconds": 3600, "note": "",
777
+ }
778
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
779
+ assert "samples 12" in capsys.readouterr().out
780
+
781
+
478
782
  def test_an_unsettled_projection_is_labelled_rather_than_stated(fake, capsys):
479
783
  fake.metrics_result = {
480
784
  "progress": {"step": 5, "total_steps": 556, "percent": 0.9,
@@ -820,7 +1124,8 @@ def test_the_subcommand_name_survives_the_shell_positional():
820
1124
 
821
1125
  def test_shell_with_a_command_runs_it_once_and_returns_its_exit_code(fake, capsys):
822
1126
  assert run(["shell", "job-a8acdef80a07", "--", "date"]) == 0
823
- assert fake.execed == [("job-a8acdef80a07", "date", 0)]
1127
+ assert fake.execed == [("job-a8acdef80a07", "date", 0, False, 0)], (
1128
+ "`-- command` 형태는 stateless 다 — script 는 남길 session 이 없다")
824
1129
  assert "Sep 10" in capsys.readouterr().out
825
1130
 
826
1131
 
@@ -858,3 +1163,62 @@ def test_every_other_subcommand_still_dispatches(fake):
858
1163
  """dest 를 바꾼 것이 나머지를 깨지 않았다는 확인."""
859
1164
  assert run(["status", "job-a8acdef80a07"]) == 0
860
1165
  assert run(["secrets"]) == 0
1166
+
1167
+
1168
+ def test_the_prompt_uses_a_session_and_carries_the_sequence(fake, monkeypatch, capsys):
1169
+ """★ WHY THE PROMPT DIFFERS FROM `-- command`. A person at a prompt assumes
1170
+ `cd` sticks; a script does not want a session left behind for the idle timer.
1171
+ So the prompt sends session=True and threads the sequence through, and the
1172
+ one-shot form stays stateless."""
1173
+ typed = iter(["cd /workspace", "pwd", "exit"])
1174
+ monkeypatch.setattr("builtins.input", lambda _prompt: next(typed))
1175
+ fake.exec_result = {"output": "/workspace\n", "exit_code": 0, "seq": 11, "lost": False}
1176
+
1177
+ assert run(["shell", "job-a8acdef80a07"]) == 0
1178
+ assert [(line, sess) for _job, line, _slot, sess, _seq in fake.execed] == [
1179
+ ("cd /workspace", True), ("pwd", True)]
1180
+ # The FIRST line starts at 0 and the second carries what the first returned.
1181
+ assert [seq for *_rest, seq in fake.execed] == [0, 11]
1182
+
1183
+
1184
+ def test_a_session_that_closed_is_reopened_once_and_the_person_is_told(fake, monkeypatch,
1185
+ capsys):
1186
+ """The one case where starting a new shell is right: the old one is provably
1187
+ gone (ten minutes unread, or the workload restarted). Everywhere else a new
1188
+ shell silently loses the `cd`, which is the failure this feature exists to end
1189
+ -- so the message says what was lost."""
1190
+ typed = iter(["pwd", "exit"])
1191
+ monkeypatch.setattr("builtins.input", lambda _prompt: next(typed))
1192
+
1193
+ calls = {"n": 0}
1194
+ original = fake.exec_in_job
1195
+
1196
+ def flaky(job, line, slot=0, session=False, seq=0):
1197
+ calls["n"] += 1
1198
+ if calls["n"] == 1:
1199
+ raise cli.ServerError("no session for this slot; open one first")
1200
+ return original(job, line, slot=slot, session=session, seq=seq)
1201
+
1202
+ fake.exec_in_job = flaky
1203
+ assert run(["shell", "job-a8acdef80a07"]) == 0
1204
+ assert calls["n"] == 2, "it retried once rather than giving up or looping"
1205
+ err = capsys.readouterr().err
1206
+ assert "reopening" in err and "are gone" in err, (
1207
+ "a silent reopen would let somebody keep typing paths relative to a cd that no "
1208
+ "longer applies")
1209
+
1210
+
1211
+ def test_a_failure_that_is_not_a_lost_session_is_not_retried(fake, monkeypatch, capsys):
1212
+ typed = iter(["pwd", "exit"])
1213
+ monkeypatch.setattr("builtins.input", lambda _prompt: next(typed))
1214
+
1215
+ calls = {"n": 0}
1216
+
1217
+ def always_bad(job, line, slot=0, session=False, seq=0):
1218
+ calls["n"] += 1
1219
+ raise cli.ServerError("the driver pod did not answer in JSON")
1220
+
1221
+ fake.exec_in_job = always_bad
1222
+ assert run(["shell", "job-a8acdef80a07"]) == 0
1223
+ assert calls["n"] == 1, "retrying an unrelated failure would double every bad command"
1224
+ assert "did not answer in JSON" in capsys.readouterr().err
@@ -65,7 +65,9 @@ def test_shell_sends_the_command_and_waits_longer_than_the_server():
65
65
  result = client_with(session).exec_in_job("baseline-c", "nvidia-smi -L", slot=1)
66
66
  call = session.calls[0]
67
67
  assert call["url"] == "https://run.example/v1/jobs/baseline-c/exec"
68
- assert call["json"] == {"command": "nvidia-smi -L", "slot": 1, "timeout_seconds": 20}
68
+ assert call["json"] == {"command": "nvidia-smi -L", "slot": 1, "timeout_seconds": 20,
69
+ "session": False, "seq": 0}, (
70
+ "PACSRUN-SHELL-SESSION 이후에도 기본은 한 줄짜리 stateless 형태다")
69
71
  # The server may hold the request for its whole window, so the client's
70
72
  # read timeout must outlive it.
71
73
  assert call["timeout"] > 20
@@ -155,3 +157,15 @@ def test_a_submit_is_given_more_patience_than_a_read():
155
157
  read_session = FakeSession(FakeResponse(200, {}))
156
158
  client_with(read_session).status("job-a8acdef80a07")
157
159
  assert submit_session.calls[0]["timeout"] > read_session.calls[0]["timeout"]
160
+
161
+
162
+ def test_shell_can_ask_for_the_persistent_session_and_carries_its_sequence():
163
+ """The prompt loop's shape: the same route, with `session` and the output
164
+ sequence the caller last saw. Without the sequence the caller would be handed
165
+ everything the shell has ever printed on every line."""
166
+ session = FakeSession(FakeResponse(200, {"output": "/workspace\n", "exit_code": 0,
167
+ "seq": 42, "lost": False, "note": ""}))
168
+ result = client_with(session).exec_in_job("baseline-c", "pwd", session=True, seq=7)
169
+ assert session.calls[0]["json"]["session"] is True
170
+ assert session.calls[0]["json"]["seq"] == 7
171
+ assert result["seq"] == 42, "the reply carries the next one back"
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes