hyperun 0.2.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hyperun-0.2.0 → hyperun-0.2.2}/PKG-INFO +1 -1
- {hyperun-0.2.0 → hyperun-0.2.2}/hyperun/cli.py +259 -20
- {hyperun-0.2.0 → hyperun-0.2.2}/hyperun/client.py +11 -1
- {hyperun-0.2.0 → hyperun-0.2.2}/hyperun.egg-info/PKG-INFO +1 -1
- {hyperun-0.2.0 → hyperun-0.2.2}/pyproject.toml +1 -1
- {hyperun-0.2.0 → hyperun-0.2.2}/tests/test_cli.py +367 -3
- {hyperun-0.2.0 → hyperun-0.2.2}/tests/test_client.py +15 -1
- {hyperun-0.2.0 → hyperun-0.2.2}/LICENSE +0 -0
- {hyperun-0.2.0 → hyperun-0.2.2}/README.md +0 -0
- {hyperun-0.2.0 → hyperun-0.2.2}/hyperun/__init__.py +0 -0
- {hyperun-0.2.0 → hyperun-0.2.2}/hyperun/browser_login.py +0 -0
- {hyperun-0.2.0 → hyperun-0.2.2}/hyperun/config.py +0 -0
- {hyperun-0.2.0 → hyperun-0.2.2}/hyperun.egg-info/SOURCES.txt +0 -0
- {hyperun-0.2.0 → hyperun-0.2.2}/hyperun.egg-info/dependency_links.txt +0 -0
- {hyperun-0.2.0 → hyperun-0.2.2}/hyperun.egg-info/entry_points.txt +0 -0
- {hyperun-0.2.0 → hyperun-0.2.2}/hyperun.egg-info/requires.txt +0 -0
- {hyperun-0.2.0 → hyperun-0.2.2}/hyperun.egg-info/top_level.txt +0 -0
- {hyperun-0.2.0 → hyperun-0.2.2}/setup.cfg +0 -0
- {hyperun-0.2.0 → hyperun-0.2.2}/tests/test_browser_login.py +0 -0
- {hyperun-0.2.0 → hyperun-0.2.2}/tests/test_config.py +0 -0
|
@@ -50,6 +50,7 @@ from __future__ import annotations
|
|
|
50
50
|
|
|
51
51
|
import argparse
|
|
52
52
|
import base64
|
|
53
|
+
import datetime
|
|
53
54
|
import getpass
|
|
54
55
|
import json
|
|
55
56
|
import time
|
|
@@ -365,9 +366,10 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
365
366
|
"through the job's driver pod into the workload container on the "
|
|
366
367
|
"rented machine and brings the exit code back like ssh. It is NOT a "
|
|
367
368
|
"TTY — no vim, no top, about 25 seconds per command — because the "
|
|
368
|
-
"server is a Lambda and cannot hold a terminal open. AWS and
|
|
369
|
-
"machine rentals only: a RunPod job is a rented container
|
|
370
|
-
"machine behind it,
|
|
369
|
+
"server is a Lambda and cannot hold a terminal open. AWS, GCP and "
|
|
370
|
+
"Shadeform machine rentals only: a RunPod job is a rented container "
|
|
371
|
+
"with no machine behind it, so there is no cluster to attach to and "
|
|
372
|
+
"the relay refuses it. ★ PUT OPTIONS BEFORE THE "
|
|
371
373
|
"JOB ID -- everything after it is sent to the workload as-is, so "
|
|
372
374
|
"`shell job-x --slot 2` asks pod 0 and passes `--slot 2` to the shell. "
|
|
373
375
|
"That is refused rather than obeyed.",
|
|
@@ -386,8 +388,13 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
386
388
|
)
|
|
387
389
|
watch.add_argument("job_id")
|
|
388
390
|
watch.add_argument(
|
|
389
|
-
|
|
390
|
-
|
|
391
|
+
# default=None, not 3600, and the difference is load-bearing: it is how
|
|
392
|
+
# `cmd_watch` tells "the caller wants an hour" from "the caller said
|
|
393
|
+
# nothing", and only the second one may be widened to reach a finished
|
|
394
|
+
# job's readings. 604800 is the server's cap, not a number chosen here.
|
|
395
|
+
"--window", type=int, default=None, metavar="SECONDS",
|
|
396
|
+
help="how far back to read (default 3600, max 604800). A finished job "
|
|
397
|
+
"is widened automatically to reach its own readings.",
|
|
391
398
|
)
|
|
392
399
|
add_json_flag(watch)
|
|
393
400
|
|
|
@@ -826,11 +833,26 @@ def cmd_shell(args: argparse.Namespace) -> int:
|
|
|
826
833
|
pod onto the rented machine and returns the exit code like ssh. With a
|
|
827
834
|
trailing `-- command` it runs once and exits with that code; without one
|
|
828
835
|
it prompts, which FEELS like a slow shell and is honestly a request loop.
|
|
836
|
+
|
|
837
|
+
★ THE PROMPT USES A SESSION AND THE ONE-SHOT FORM DOES NOT, since
|
|
838
|
+
2026-09-10. The prompt types into a shell that stays running in the driver
|
|
839
|
+
pod, so `cd`, an exported variable and an activated venv survive from one
|
|
840
|
+
line to the next -- which is what a person at a prompt assumes and what
|
|
841
|
+
every earlier version quietly did not do. A `-- command` invocation is a
|
|
842
|
+
script's shape and gets the stateless form: it wants no history and should
|
|
843
|
+
not leave a session behind for the idle timer to close.
|
|
829
844
|
"""
|
|
830
845
|
client = client_from_config()
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
846
|
+
# The output sequence the session has already shown us. It is here rather
|
|
847
|
+
# than inside run_once because it has to survive between lines, which is the
|
|
848
|
+
# whole point.
|
|
849
|
+
state = {"seq": 0}
|
|
850
|
+
|
|
851
|
+
def run_once(line: str, session: bool = False) -> int:
|
|
852
|
+
answer = client.exec_in_job(args.job, line, slot=args.slot,
|
|
853
|
+
session=session, seq=state["seq"])
|
|
854
|
+
if session:
|
|
855
|
+
state["seq"] = int(answer.get("seq", state["seq"]))
|
|
834
856
|
output = answer.get("output") or ""
|
|
835
857
|
if output:
|
|
836
858
|
print(output, end="" if output.endswith("\n") else "\n")
|
|
@@ -873,7 +895,8 @@ def cmd_shell(args: argparse.Namespace) -> int:
|
|
|
873
895
|
return run_once(" ".join(words))
|
|
874
896
|
|
|
875
897
|
print(
|
|
876
|
-
f"one command per line, ~25s each; 'exit' or Ctrl-D leaves. Not a TTY
|
|
898
|
+
f"one command per line, ~25s each; 'exit' or Ctrl-D leaves. Not a TTY, "
|
|
899
|
+
f"but `cd` and exported variables DO survive between lines.",
|
|
877
900
|
file=sys.stderr,
|
|
878
901
|
)
|
|
879
902
|
while True:
|
|
@@ -888,10 +911,29 @@ def cmd_shell(args: argparse.Namespace) -> int:
|
|
|
888
911
|
if line in ("exit", "quit"):
|
|
889
912
|
return EXIT_OK
|
|
890
913
|
try:
|
|
891
|
-
code = run_once(line)
|
|
914
|
+
code = run_once(line, session=True)
|
|
892
915
|
if code:
|
|
893
916
|
print(f"(exit {code})", file=sys.stderr)
|
|
894
917
|
except ServerError as exc:
|
|
918
|
+
# A SESSION THAT IS GONE IS THE ONE ERROR WORTH ACTING ON. The
|
|
919
|
+
# server answers 409 when the driver pod has no session for this
|
|
920
|
+
# slot -- it timed out after ten minutes unread, or the workload
|
|
921
|
+
# restarted. Reopening is right there and only there: everywhere
|
|
922
|
+
# else a new shell would silently lose the `cd` the person is
|
|
923
|
+
# relying on, which is the failure this whole feature exists to
|
|
924
|
+
# end. The sequence resets with it, because the new shell's
|
|
925
|
+
# output starts from nothing.
|
|
926
|
+
if "open one first" in str(exc) or "no session" in str(exc):
|
|
927
|
+
state["seq"] = 0
|
|
928
|
+
print("(the session had closed; reopening — `cd` and variables "
|
|
929
|
+
"from before are gone)", file=sys.stderr)
|
|
930
|
+
try:
|
|
931
|
+
code = run_once(line, session=True)
|
|
932
|
+
if code:
|
|
933
|
+
print(f"(exit {code})", file=sys.stderr)
|
|
934
|
+
continue
|
|
935
|
+
except ServerError as retry_exc:
|
|
936
|
+
exc = retry_exc
|
|
895
937
|
# One failed command must not end the session: say what the server
|
|
896
938
|
# said and keep the prompt.
|
|
897
939
|
print(f"error: {exc}", file=sys.stderr)
|
|
@@ -1052,13 +1094,128 @@ def bar(percent: float, width: int = 20) -> str:
|
|
|
1052
1094
|
return "#" * filled + "-" * (width - filled)
|
|
1053
1095
|
|
|
1054
1096
|
|
|
1097
|
+
# DDPSRUN-WATCH-TERMINAL. The three phases a job never leaves. They are the
|
|
1098
|
+
# controller's own words: the phase enum in api/v1alpha1/pacsjob_types.go
|
|
1099
|
+
# carries Pending / Starting / Running / Recovering / Succeeded / Failed /
|
|
1100
|
+
# Compared and nothing else, and the screen keeps the same three in
|
|
1101
|
+
# `const TERMINAL` (ui/app.js). Recovering is NOT here -- the machine was
|
|
1102
|
+
# reclaimed and the job is being restarted, so the run is still going.
|
|
1103
|
+
TERMINAL_PHASES = ("Succeeded", "Failed", "Compared")
|
|
1104
|
+
|
|
1105
|
+
# How far back `watch` reads when the caller says nothing, and the furthest the
|
|
1106
|
+
# server will let it reach. The cap is the server's own: `window_seconds` on
|
|
1107
|
+
# `GET /v1/jobs/{id}/metrics` is declared `ge=60, le=604800`
|
|
1108
|
+
# (server/ddpsrun_server/main.py). The `--window` help said 86400 here, which was
|
|
1109
|
+
# simply wrong -- one day, against the server's seven.
|
|
1110
|
+
DEFAULT_WATCH_WINDOW = 3600
|
|
1111
|
+
MAX_WATCH_WINDOW = 604800
|
|
1112
|
+
|
|
1113
|
+
|
|
1114
|
+
def _window_reaching_back_to_start(job: dict[str, Any]) -> int:
|
|
1115
|
+
"""How wide a window must be to contain a FINISHED job's readings.
|
|
1116
|
+
|
|
1117
|
+
★ WHY THIS EXISTS. Nothing is stored: `watch` re-reads the job's own log, and
|
|
1118
|
+
the window is measured back from NOW. A running job's readings are therefore
|
|
1119
|
+
always inside the default hour. A job that finished five hours ago has NONE
|
|
1120
|
+
of its readings in the last hour, so `hyperun watch` on it printed the phase
|
|
1121
|
+
and no numbers -- and the phase is the one thing the reader already knew.
|
|
1122
|
+
|
|
1123
|
+
The fix is not a larger default, which would make every `watch` on a running
|
|
1124
|
+
job re-read a week of log. It is to widen only when the job is over AND the
|
|
1125
|
+
caller did not pick a window.
|
|
1126
|
+
|
|
1127
|
+
Args:
|
|
1128
|
+
job: the body of `GET /v1/jobs/{id}`.
|
|
1129
|
+
|
|
1130
|
+
Returns:
|
|
1131
|
+
Seconds, or 0 when the job is not finished or carries no start time, in
|
|
1132
|
+
which case the caller leaves the window as it was.
|
|
1133
|
+
"""
|
|
1134
|
+
if job.get("phase") not in TERMINAL_PHASES:
|
|
1135
|
+
return 0
|
|
1136
|
+
started = job.get("started_at") or job.get("created_at")
|
|
1137
|
+
if not started:
|
|
1138
|
+
return 0
|
|
1139
|
+
try:
|
|
1140
|
+
began = datetime.datetime.fromisoformat(started.replace("Z", "+00:00"))
|
|
1141
|
+
except ValueError:
|
|
1142
|
+
return 0
|
|
1143
|
+
elapsed = (datetime.datetime.now(datetime.timezone.utc) - began).total_seconds()
|
|
1144
|
+
# A minute of headroom so the FIRST reading falls inside the window instead
|
|
1145
|
+
# of exactly on its edge, and the server's own cap over the top.
|
|
1146
|
+
return min(MAX_WATCH_WINDOW, max(60, int(elapsed) + 60))
|
|
1147
|
+
|
|
1148
|
+
|
|
1055
1149
|
def cmd_watch(args: argparse.Namespace) -> int:
|
|
1056
|
-
"""Print a job's GPU usage and how far the training has got.
|
|
1057
|
-
|
|
1150
|
+
"""Print a job's GPU usage and how far the training has got.
|
|
1151
|
+
|
|
1152
|
+
ONE BLOCK PER CARD, because a job can rent several. baseline-c rents four
|
|
1153
|
+
A100s in one pod, and until today this printed `latest_gpu`, which is card 0
|
|
1154
|
+
alone -- three quarters of a $44 run was invisible from the CLI. The screen
|
|
1155
|
+
had the same defect and was fixed on 2026-09-08. `cards` is the server's
|
|
1156
|
+
per-card answer (CardMetricsView in server/ddpsrun_server/models.py); a
|
|
1157
|
+
server too old to send it answers with an empty list, and then the
|
|
1158
|
+
single-card block below is the whole story, exactly as before.
|
|
1159
|
+
|
|
1160
|
+
WHY THIS ASKS FOR THE PHASE AS WELL AS THE METRICS. A finished job's last
|
|
1161
|
+
reading is the idle card in the seconds before teardown -- 0%, 0 MiB -- so
|
|
1162
|
+
printing it under "GPU util" tells a reader the run was idle when it was
|
|
1163
|
+
not. Leading with the peak instead requires knowing the job is over, and
|
|
1164
|
+
THE METRICS ANSWER DOES NOT SAY: MetricsResponse carries latest_gpu,
|
|
1165
|
+
gpu_series, peak_gpu, the two utilisation figures, sample_count, cards,
|
|
1166
|
+
progress, window_seconds and note, and no phase at all
|
|
1167
|
+
(server/ddpsrun_server/models.py). Guessing from the readings fails in both
|
|
1168
|
+
directions -- a job whose framework has not allocated yet also reads 0 MiB,
|
|
1169
|
+
and a job killed mid-step leaves a final reading that is nowhere near 0 --
|
|
1170
|
+
so this asks the server instead.
|
|
1171
|
+
|
|
1172
|
+
WHAT THAT COSTS: one extra `GET /v1/jobs/{id}` per typed `watch`. `watch`
|
|
1173
|
+
prints once and exits rather than polling, so it is one extra request per
|
|
1174
|
+
command, not one per second, and `--json` pays it only when the first window
|
|
1175
|
+
came back empty. The screen pays the same price -- it reads the job and hands
|
|
1176
|
+
it to drawMetrics (ui/app.js).
|
|
1177
|
+
"""
|
|
1178
|
+
client = client_from_config()
|
|
1179
|
+
window = DEFAULT_WATCH_WINDOW if args.window is None else args.window
|
|
1180
|
+
result = client.metrics(args.job_id, window)
|
|
1181
|
+
|
|
1182
|
+
# ★ AN EMPTY ANSWER ON A FINISHED JOB IS THE WINDOW, NOT THE JOB. The window
|
|
1183
|
+
# is measured back from now, so a run that ended hours ago has every reading
|
|
1184
|
+
# outside the default hour and `watch` printed nothing. Widening is done
|
|
1185
|
+
# ONLY here: the caller did not choose a window, and the first one came back
|
|
1186
|
+
# with no readings at all. A running job never reaches this branch, so the
|
|
1187
|
+
# usual `watch` is still exactly one metrics request.
|
|
1188
|
+
#
|
|
1189
|
+
# "No readings" means BOTH are empty. `cards` is the per-card answer and
|
|
1190
|
+
# `latest_gpu` is card 0; a multi-card job fills `cards`, while a server too
|
|
1191
|
+
# old to send it fills only `latest_gpu`. Testing one alone would make
|
|
1192
|
+
# `watch --json` on a four-card job buy a second request it does not need.
|
|
1193
|
+
job: dict[str, Any] | None = None
|
|
1194
|
+
if args.window is None and not result.get("latest_gpu") and not result.get("cards"):
|
|
1195
|
+
try:
|
|
1196
|
+
job = client.status(args.job_id)
|
|
1197
|
+
except ServerError:
|
|
1198
|
+
job = None
|
|
1199
|
+
wider = _window_reaching_back_to_start(job) if job else 0
|
|
1200
|
+
if wider > window:
|
|
1201
|
+
result = client.metrics(args.job_id, wider)
|
|
1202
|
+
window = wider
|
|
1203
|
+
|
|
1058
1204
|
if args.json:
|
|
1205
|
+
# The server's answer, verbatim. It already contains `cards`, so nothing
|
|
1206
|
+
# here needs the phase.
|
|
1059
1207
|
print(json.dumps(result, indent=2, ensure_ascii=False))
|
|
1060
1208
|
return EXIT_OK
|
|
1061
1209
|
|
|
1210
|
+
# A phase lookup that fails must not lose the GPU numbers this command
|
|
1211
|
+
# exists to print, so it falls back to the running layout. The likely cause
|
|
1212
|
+
# is the job being deleted between the two calls, and the running layout is
|
|
1213
|
+
# the behaviour this command had until today rather than a wrong number.
|
|
1214
|
+
try:
|
|
1215
|
+
done = (job or client.status(args.job_id)).get("phase") in TERMINAL_PHASES
|
|
1216
|
+
except ServerError:
|
|
1217
|
+
done = False
|
|
1218
|
+
|
|
1062
1219
|
progress = result.get("progress")
|
|
1063
1220
|
if progress:
|
|
1064
1221
|
print(f" training {bar(progress['percent'])} "
|
|
@@ -1073,20 +1230,102 @@ def cmd_watch(args: argparse.Namespace) -> int:
|
|
|
1073
1230
|
|
|
1074
1231
|
gpu = result.get("latest_gpu")
|
|
1075
1232
|
if gpu:
|
|
1076
|
-
|
|
1077
|
-
|
|
1078
|
-
|
|
1079
|
-
|
|
1080
|
-
|
|
1081
|
-
|
|
1082
|
-
|
|
1083
|
-
|
|
1233
|
+
# The reading with the most MEMORY in use. A server too old to send it
|
|
1234
|
+
# leaves the last reading as the only one there is, which is what the
|
|
1235
|
+
# screen falls back to as well.
|
|
1236
|
+
peak = result.get("peak_gpu") or gpu
|
|
1237
|
+
if done:
|
|
1238
|
+
# A FINISHED RUN LEADS WITH THE PEAK. Memory is what kills a run, so
|
|
1239
|
+
# the peak is the number a post-mortem came here for; the last
|
|
1240
|
+
# reading is kept, because it is still evidence, but it is named for
|
|
1241
|
+
# what it is instead of being labelled "GPU util". The screen was
|
|
1242
|
+
# changed to this shape on 2026-09-07 after the 0%, 0 MiB readout
|
|
1243
|
+
# was reported as the panel being broken.
|
|
1244
|
+
print(f" peak memory {bar(peak['memory_percent'])} "
|
|
1245
|
+
f"{peak['memory_used_mib']:,} / {peak['memory_total_mib']:,} MiB "
|
|
1246
|
+
f"({peak['memory_percent']}%)")
|
|
1247
|
+
# ★ `peak_utilization_percent`, NOT `peak['utilization_percent']`,
|
|
1248
|
+
# and the difference is the whole defect. `peak` is the sample with
|
|
1249
|
+
# the most memory in it and its utilisation is whatever the card
|
|
1250
|
+
# happened to be doing at that instant. On job-66b46719b854 all four
|
|
1251
|
+
# A100s reached 77,631 MiB and the utilisation inside those four
|
|
1252
|
+
# samples read 0, 3, 1 and 95 -- while every one of those cards
|
|
1253
|
+
# actually peaked at 99 or 100 and averaged between 37.8 and 84.6.
|
|
1254
|
+
# Reported on the screen 2026-09-11; the CLI never printed either
|
|
1255
|
+
# figure, so it is being added correct rather than fixed.
|
|
1256
|
+
print(f" peak util {_percent(result.get('peak_utilization_percent'))}")
|
|
1257
|
+
print(f" avg util {_percent(result.get('avg_utilization_percent'))}")
|
|
1258
|
+
print(f" last reading {gpu['utilization_percent']}%, "
|
|
1259
|
+
f"{gpu['memory_used_mib']:,} MiB (run ended)")
|
|
1260
|
+
else:
|
|
1261
|
+
print(f" GPU util {bar(gpu['utilization_percent'])} "
|
|
1262
|
+
f"{gpu['utilization_percent']}%")
|
|
1263
|
+
print(f" GPU memory {bar(gpu['memory_percent'])} "
|
|
1264
|
+
f"{gpu['memory_used_mib']:,} / {gpu['memory_total_mib']:,} MiB "
|
|
1265
|
+
f"({gpu['memory_percent']}%)")
|
|
1266
|
+
print(f" temp, power {gpu['temperature_c']} C, {gpu['power_w']} W")
|
|
1267
|
+
# `sample_count`, not len(gpu_series): the server thins the series to at
|
|
1268
|
+
# most 400 points so a chart can draw it, and job-66b46719b854 printed
|
|
1269
|
+
# 785 readings per card while this line said 393. The screen had the same
|
|
1270
|
+
# defect and was fixed on 2026-09-12. A server too old to send the count
|
|
1271
|
+
# falls back to the thinned length, which is the only number it has.
|
|
1272
|
+
taken = result.get("sample_count") or len(result.get("gpu_series", []))
|
|
1273
|
+
print(f" samples {taken}, last {result['window_seconds']}s")
|
|
1274
|
+
print_card_table(result.get("cards") or [])
|
|
1084
1275
|
|
|
1085
1276
|
if result.get("note"):
|
|
1086
1277
|
print(f" note {result['note']}")
|
|
1087
1278
|
return EXIT_OK
|
|
1088
1279
|
|
|
1089
1280
|
|
|
1281
|
+
def _percent(value: float | int | None) -> str:
|
|
1282
|
+
"""One utilisation figure, or a dash when the server did not send it.
|
|
1283
|
+
|
|
1284
|
+
The two utilisation fields arrived on 2026-09-10 and a server older than
|
|
1285
|
+
that answers with them absent. A blank line reads as 0, which is the very
|
|
1286
|
+
thing the peak figures exist to stop being claimed.
|
|
1287
|
+
"""
|
|
1288
|
+
return "-" if value is None else f"{value}%"
|
|
1289
|
+
|
|
1290
|
+
|
|
1291
|
+
def print_card_table(cards: list[dict[str, Any]]) -> None:
|
|
1292
|
+
"""One row per GPU card, printed only when the job rented more than one.
|
|
1293
|
+
|
|
1294
|
+
WHY THIS EXISTS. Everything above describes card 0 -- `latest_gpu` and
|
|
1295
|
+
`peak_gpu` are the lowest-indexed card's readings, unchanged since before
|
|
1296
|
+
`cards` existed. job-66b46719b854 is four A100s in one pod, and reading its
|
|
1297
|
+
CLI output left three of them invisible.
|
|
1298
|
+
|
|
1299
|
+
WHY ONLY ABOVE ONE CARD. With a single card these four numbers repeat the
|
|
1300
|
+
block above it word for word, so the table is drawn on the same condition
|
|
1301
|
+
the screen draws its own (`cards.length > 1` in ui/app.js).
|
|
1302
|
+
|
|
1303
|
+
Args:
|
|
1304
|
+
cards: the response's `cards`, each a CardMetricsView -- `gpu_index`,
|
|
1305
|
+
`series`, `latest`, `peak`, `avg_utilization_percent`,
|
|
1306
|
+
`peak_utilization_percent`, `sample_count`.
|
|
1307
|
+
"""
|
|
1308
|
+
if len(cards) < 2:
|
|
1309
|
+
return
|
|
1310
|
+
print(f" {'card':<8}{'peak memory':>27}{'peak util':>11}"
|
|
1311
|
+
f"{'avg util':>10}{'samples':>9}")
|
|
1312
|
+
for card in cards:
|
|
1313
|
+
# `peak` is this card's highest-MEMORY reading; falling back to its last
|
|
1314
|
+
# one matches the screen's table and keeps a row rather than dropping a
|
|
1315
|
+
# card that only ever printed once.
|
|
1316
|
+
sample = card.get("peak") or card.get("latest") or {}
|
|
1317
|
+
memory = "-" if not sample else (
|
|
1318
|
+
f"{sample['memory_used_mib']:,} / {sample['memory_total_mib']:,} MiB "
|
|
1319
|
+
f"({sample['memory_percent']:.0f}%)")
|
|
1320
|
+
# Same two corrections as the block above: the card's own highest
|
|
1321
|
+
# utilisation rather than the utilisation inside its memory peak, and
|
|
1322
|
+
# the readings it printed rather than the points left after thinning.
|
|
1323
|
+
taken = card.get("sample_count") or len(card.get("series") or [])
|
|
1324
|
+
print(f" {('GPU ' + str(card['gpu_index'])):<8}{memory:>27}"
|
|
1325
|
+
f"{_percent(card.get('peak_utilization_percent')):>11}"
|
|
1326
|
+
f"{_percent(card.get('avg_utilization_percent')):>10}{taken:>9}")
|
|
1327
|
+
|
|
1328
|
+
|
|
1090
1329
|
def cmd_stats(args: argparse.Namespace) -> int:
|
|
1091
1330
|
"""Print this caller's team figures."""
|
|
1092
1331
|
result = client_from_config().stats()
|
|
@@ -193,7 +193,8 @@ class Client:
|
|
|
193
193
|
return self._call("GET", "/v1/stats").json()
|
|
194
194
|
|
|
195
195
|
def exec_in_job(
|
|
196
|
-
self, job: str, command: str, slot: int = 0, timeout_seconds: int = 20
|
|
196
|
+
self, job: str, command: str, slot: int = 0, timeout_seconds: int = 20,
|
|
197
|
+
session: bool = False, seq: int = 0,
|
|
197
198
|
) -> dict[str, Any]:
|
|
198
199
|
"""Run one shell line inside a running job's workload container.
|
|
199
200
|
|
|
@@ -206,6 +207,13 @@ class Client:
|
|
|
206
207
|
command: one shell line, run as `sh -lc <command>`.
|
|
207
208
|
slot: which pod of a parallel job.
|
|
208
209
|
timeout_seconds: server-side wait, capped at 25 by the server.
|
|
210
|
+
session: type the line into a shell ALREADY RUNNING in the driver
|
|
211
|
+
pod, so `cd` and exported variables survive to the next call.
|
|
212
|
+
False starts a fresh `sh -lc`, which is what a script wants.
|
|
213
|
+
seq: with `session`, the output sequence this caller last saw. The
|
|
214
|
+
reply carries the next one. It exists because the caller is a
|
|
215
|
+
different process on the server's side every time and cannot
|
|
216
|
+
hold a position in the output.
|
|
209
217
|
"""
|
|
210
218
|
return self._call(
|
|
211
219
|
"POST",
|
|
@@ -214,6 +222,8 @@ class Client:
|
|
|
214
222
|
"command": command,
|
|
215
223
|
"slot": slot,
|
|
216
224
|
"timeout_seconds": timeout_seconds,
|
|
225
|
+
"session": session,
|
|
226
|
+
"seq": seq,
|
|
217
227
|
},
|
|
218
228
|
# The server may hold the request for timeout_seconds before
|
|
219
229
|
# answering; the read timeout has to outlive that on purpose.
|
|
@@ -34,7 +34,7 @@ build-backend = "setuptools.build_meta"
|
|
|
34
34
|
|
|
35
35
|
[project]
|
|
36
36
|
name = "hyperun"
|
|
37
|
-
version = "0.2.
|
|
37
|
+
version = "0.2.2"
|
|
38
38
|
description = "Submit a GPU job and get results back. No kubectl, no cloud account."
|
|
39
39
|
requires-python = ">=3.9"
|
|
40
40
|
dependencies = ["requests>=2.31", "PyYAML>=6.0"]
|
|
@@ -38,6 +38,15 @@ class FakeClient:
|
|
|
38
38
|
self.estimate_result = {}
|
|
39
39
|
self.validate_result = {"ok": True, "findings": [], "not_checked": []}
|
|
40
40
|
self.metrics_result = {"window_seconds": 3600, "gpu_series": [], "note": ""}
|
|
41
|
+
# Every window `watch` asked for, in order. The widening of a finished
|
|
42
|
+
# job's window is invisible in the printed output -- the numbers look the
|
|
43
|
+
# same whichever window found them -- so the only way to test it is to
|
|
44
|
+
# record what was asked.
|
|
45
|
+
self.metrics_windows: list[int] = []
|
|
46
|
+
# What to answer once the window is wider than the default hour. None
|
|
47
|
+
# means "the same answer whatever the window", which is what every test
|
|
48
|
+
# written before the widening existed expects.
|
|
49
|
+
self.metrics_wide_result = None
|
|
41
50
|
self.stats_result = {"team": "", "members": [], "jobs": 0, "gpu_hours": 0.0,
|
|
42
51
|
"cost_usd": 0.0, "unpriced_jobs": 0, "note": ""}
|
|
43
52
|
self.secrets_result = {"names": ["GITHUB_PAT", "HF_TOKEN"],
|
|
@@ -55,6 +64,9 @@ class FakeClient:
|
|
|
55
64
|
return self.validate_result
|
|
56
65
|
|
|
57
66
|
def metrics(self, job_id, window_seconds=3600):
|
|
67
|
+
self.metrics_windows.append(window_seconds)
|
|
68
|
+
if self.metrics_wide_result is not None and window_seconds > 3600:
|
|
69
|
+
return self.metrics_wide_result
|
|
58
70
|
return self.metrics_result
|
|
59
71
|
|
|
60
72
|
def stats(self):
|
|
@@ -100,8 +112,8 @@ class FakeClient:
|
|
|
100
112
|
# DDPSRUN-SHELL. Absent until 2026-09-10, which is exactly why the dispatch
|
|
101
113
|
# bug below survived every release: with no double for this call, no test
|
|
102
114
|
# could reach cmd_shell at all.
|
|
103
|
-
def exec_in_job(self, job, line, slot=0):
|
|
104
|
-
self.execed.append((job, line, slot))
|
|
115
|
+
def exec_in_job(self, job, line, slot=0, session=False, seq=0):
|
|
116
|
+
self.execed.append((job, line, slot, session, seq))
|
|
105
117
|
return self.exec_result
|
|
106
118
|
|
|
107
119
|
|
|
@@ -475,6 +487,298 @@ def test_watch_prints_progress_and_gpu(fake, capsys):
|
|
|
475
487
|
assert "samples 120" in printed
|
|
476
488
|
|
|
477
489
|
|
|
490
|
+
def gpu_sample(util, used, total=81920, index=0):
|
|
491
|
+
"""One reading, for the card tests below."""
|
|
492
|
+
return {"utilization_percent": util, "memory_used_mib": used,
|
|
493
|
+
"memory_total_mib": total, "memory_percent": round(100 * used / total, 1),
|
|
494
|
+
"temperature_c": 62, "power_w": 310.0, "gpu_index": index}
|
|
495
|
+
|
|
496
|
+
|
|
497
|
+
def four_a100_cards():
|
|
498
|
+
"""job-66b46719b854 as the server reports it: four A100s in one pod.
|
|
499
|
+
|
|
500
|
+
The numbers are that run's: every card peaked at 77,631 of 81,920 MiB, and
|
|
501
|
+
the utilisation INSIDE those four highest-memory samples read 0, 3, 1 and 95
|
|
502
|
+
while the cards themselves peaked at 100, 99, 100 and 99. That gap is the
|
|
503
|
+
defect the screen carried until 2026-09-11, so the fixture keeps it.
|
|
504
|
+
"""
|
|
505
|
+
peaks = [(0, 100, 84.6), (3, 99, 37.8), (1, 100, 79.2), (95, 99, 81.3)]
|
|
506
|
+
return [
|
|
507
|
+
{"gpu_index": i,
|
|
508
|
+
"series": [{}] * 393,
|
|
509
|
+
"latest": gpu_sample(0, 0, index=i),
|
|
510
|
+
"peak": gpu_sample(at_peak, 77631, index=i),
|
|
511
|
+
"avg_utilization_percent": avg,
|
|
512
|
+
"peak_utilization_percent": real_peak,
|
|
513
|
+
"sample_count": 785}
|
|
514
|
+
for i, (at_peak, real_peak, avg) in enumerate(peaks)
|
|
515
|
+
]
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
def test_every_card_is_printed_not_only_the_first(fake, capsys):
|
|
519
|
+
# ★ THE DEFECT. `latest_gpu` is card 0, so a four-A100 job printed one card
|
|
520
|
+
# and three quarters of a $44 run was invisible from the CLI. The screen was
|
|
521
|
+
# fixed on 2026-09-08 and this surface was not.
|
|
522
|
+
fake.metrics_result = {
|
|
523
|
+
"latest_gpu": gpu_sample(94, 38200),
|
|
524
|
+
"peak_gpu": gpu_sample(0, 77631),
|
|
525
|
+
"gpu_series": [{}] * 393,
|
|
526
|
+
"sample_count": 785,
|
|
527
|
+
"peak_utilization_percent": 100,
|
|
528
|
+
"avg_utilization_percent": 84.6,
|
|
529
|
+
"cards": four_a100_cards(),
|
|
530
|
+
"window_seconds": 604800, "note": "",
|
|
531
|
+
}
|
|
532
|
+
assert run(["watch", "job-66b46719b854"]) == cli.EXIT_OK
|
|
533
|
+
printed = capsys.readouterr().out
|
|
534
|
+
# `GPU 0` is a table row; `GPU util` and `GPU memory` are the headline above
|
|
535
|
+
# it, so the card index is what tells the two apart.
|
|
536
|
+
rows = {line.split()[1]: line.split()
|
|
537
|
+
for line in printed.splitlines()
|
|
538
|
+
if line.startswith(" GPU ") and line.split()[1].isdigit()}
|
|
539
|
+
assert sorted(rows) == ["0", "1", "2", "3"]
|
|
540
|
+
|
|
541
|
+
# A row reads: GPU <n> <used> / <total> MiB (<pct>%) <peak util> <avg> <n>
|
|
542
|
+
# so the last three fields are the three that were wrong or missing.
|
|
543
|
+
expected = {"0": ("100%", "84.6%"), "1": ("99%", "37.8%"),
|
|
544
|
+
"2": ("100%", "79.2%"), "3": ("99%", "81.3%")}
|
|
545
|
+
for index, (peak_util, avg_util) in expected.items():
|
|
546
|
+
fields = rows[index]
|
|
547
|
+
# Each card's own peak memory, not card 0's repeated four times.
|
|
548
|
+
assert fields[2:7] == ["77,631", "/", "81,920", "MiB", "(95%)"]
|
|
549
|
+
# ★ `peak_utilization_percent`, NOT the utilisation inside the memory
|
|
550
|
+
# peak. The two differ on this very run: the four highest-memory samples
|
|
551
|
+
# read 0%, 3%, 1% and 95% while the cards peaked at 100, 99, 100 and 99.
|
|
552
|
+
assert fields[7] == peak_util
|
|
553
|
+
assert fields[8] == avg_util
|
|
554
|
+
# The readings this card printed, not the 393 points left after the
|
|
555
|
+
# server thinned the series for the chart.
|
|
556
|
+
assert fields[9] == "785"
|
|
557
|
+
|
|
558
|
+
|
|
559
|
+
def test_one_card_prints_no_table_because_it_would_repeat_the_block(fake, capsys):
|
|
560
|
+
# With a single card the table's four numbers are the same four printed
|
|
561
|
+
# directly above it, so it is drawn on the same condition the screen uses.
|
|
562
|
+
fake.metrics_result = {
|
|
563
|
+
"latest_gpu": gpu_sample(94, 38200, total=45440),
|
|
564
|
+
"gpu_series": [{}] * 120,
|
|
565
|
+
"cards": [{"gpu_index": 0, "series": [{}] * 120,
|
|
566
|
+
"latest": gpu_sample(94, 38200, total=45440),
|
|
567
|
+
"peak": gpu_sample(94, 38200, total=45440),
|
|
568
|
+
"avg_utilization_percent": 90.0,
|
|
569
|
+
"peak_utilization_percent": 99,
|
|
570
|
+
"sample_count": 120}],
|
|
571
|
+
"window_seconds": 3600, "note": "",
|
|
572
|
+
}
|
|
573
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
574
|
+
printed = capsys.readouterr().out
|
|
575
|
+
assert "peak memory" not in printed # the table's header row
|
|
576
|
+
assert "38,200 / 45,440 MiB" in printed
|
|
577
|
+
|
|
578
|
+
|
|
579
|
+
def test_a_finished_job_leads_with_the_peak_not_the_idle_last_reading(fake, capsys):
|
|
580
|
+
# ★ THE DEFECT. A finished job's last reading is the idle card in the
|
|
581
|
+
# seconds before teardown -- 0%, 0 MiB -- and printing it as "GPU util" says
|
|
582
|
+
# the run was idle. What a post-mortem asks for is the peak, because running
|
|
583
|
+
# out of memory is what kills a run. The screen was changed to this shape on
|
|
584
|
+
# 2026-09-07.
|
|
585
|
+
fake.status_result = {"job_id": "job-66b46719b854", "name": "baseline-c",
|
|
586
|
+
"phase": "Succeeded"}
|
|
587
|
+
fake.metrics_result = {
|
|
588
|
+
"latest_gpu": gpu_sample(0, 0),
|
|
589
|
+
"peak_gpu": gpu_sample(0, 77631),
|
|
590
|
+
"gpu_series": [{}] * 393,
|
|
591
|
+
"sample_count": 785,
|
|
592
|
+
"peak_utilization_percent": 100,
|
|
593
|
+
"avg_utilization_percent": 84.6,
|
|
594
|
+
"window_seconds": 604800, "note": "",
|
|
595
|
+
}
|
|
596
|
+
assert run(["watch", "job-66b46719b854"]) == cli.EXIT_OK
|
|
597
|
+
printed = capsys.readouterr().out
|
|
598
|
+
assert "peak memory" in printed
|
|
599
|
+
assert "77,631 / 81,920 MiB" in printed
|
|
600
|
+
# The peak of the CARD, not the utilisation inside the highest-memory
|
|
601
|
+
# sample, which on this run was 0 while the card reached 100.
|
|
602
|
+
assert "peak util 100%" in printed
|
|
603
|
+
assert "avg util 84.6%" in printed
|
|
604
|
+
# The last reading is still shown, and named for what it is.
|
|
605
|
+
assert "last reading 0%, 0 MiB (run ended)" in printed
|
|
606
|
+
# The running job's labels are gone: leaving "GPU util 0%" on the screen is
|
|
607
|
+
# exactly the claim this fix removes.
|
|
608
|
+
assert "GPU util" not in printed
|
|
609
|
+
|
|
610
|
+
|
|
611
|
+
def test_a_running_job_keeps_the_live_readings_as_the_headline(fake, capsys):
|
|
612
|
+
# The mirror of the test above. `peak_gpu` being present is not on its own a
|
|
613
|
+
# finished job -- the server sends it for a running one too -- so the phase
|
|
614
|
+
# is what decides, and a Running job still leads with what the card is doing
|
|
615
|
+
# NOW.
|
|
616
|
+
fake.metrics_result = {
|
|
617
|
+
"latest_gpu": gpu_sample(94, 38200, total=45440),
|
|
618
|
+
"peak_gpu": gpu_sample(99, 40000, total=45440),
|
|
619
|
+
"gpu_series": [{}] * 120,
|
|
620
|
+
"peak_utilization_percent": 99,
|
|
621
|
+
"avg_utilization_percent": 90.0,
|
|
622
|
+
"window_seconds": 3600, "note": "",
|
|
623
|
+
}
|
|
624
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
625
|
+
printed = capsys.readouterr().out
|
|
626
|
+
assert "GPU util" in printed
|
|
627
|
+
assert "38,200 / 45,440 MiB" in printed
|
|
628
|
+
assert "last reading" not in printed
|
|
629
|
+
|
|
630
|
+
|
|
631
|
+
def test_a_phase_lookup_that_fails_still_prints_the_gpu_numbers(fake, capsys):
|
|
632
|
+
# The numbers are what the command exists for, so a job deleted between the
|
|
633
|
+
# metrics call and the phase call falls back to the running layout rather
|
|
634
|
+
# than exiting with nothing printed.
|
|
635
|
+
def refuse(job_id):
|
|
636
|
+
raise ServerError("404: no such job")
|
|
637
|
+
|
|
638
|
+
fake.status = refuse
|
|
639
|
+
fake.metrics_result = {
|
|
640
|
+
"latest_gpu": gpu_sample(94, 38200, total=45440),
|
|
641
|
+
"gpu_series": [{}] * 120,
|
|
642
|
+
"window_seconds": 3600, "note": "",
|
|
643
|
+
}
|
|
644
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
645
|
+
assert "38,200 / 45,440 MiB" in capsys.readouterr().out
|
|
646
|
+
|
|
647
|
+
|
|
648
|
+
def _hours_ago(hours):
|
|
649
|
+
"""An RFC 3339 timestamp that many hours in the past, as the server writes it."""
|
|
650
|
+
import datetime as _dt
|
|
651
|
+
t = _dt.datetime.now(_dt.timezone.utc) - _dt.timedelta(hours=hours)
|
|
652
|
+
return t.strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
def test_watch_widens_the_window_to_reach_a_finished_jobs_readings(fake, capsys):
|
|
656
|
+
# ★ THE DEFECT: nothing is stored, so `watch` re-reads the log and the window
|
|
657
|
+
# is measured back from NOW. A run that ended five hours ago has every
|
|
658
|
+
# reading outside the default hour, and `watch` printed the phase and no
|
|
659
|
+
# numbers -- the phase being the one thing the reader already knew.
|
|
660
|
+
fake.status_result = {"job_id": "job-a8acdef80a07", "name": "bank-exp2",
|
|
661
|
+
"phase": "Succeeded", "started_at": _hours_ago(9),
|
|
662
|
+
"finished_at": _hours_ago(5)}
|
|
663
|
+
fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
|
|
664
|
+
fake.metrics_wide_result = {
|
|
665
|
+
"latest_gpu": gpu_sample(94, 38200, total=45440),
|
|
666
|
+
"peak_gpu": gpu_sample(94, 38200, total=45440),
|
|
667
|
+
"peak_utilization_percent": 94,
|
|
668
|
+
"gpu_series": [{}] * 120, "window_seconds": 36000, "note": "",
|
|
669
|
+
}
|
|
670
|
+
|
|
671
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
672
|
+
assert len(fake.metrics_windows) == 2, "the empty first window was not retried"
|
|
673
|
+
assert fake.metrics_windows[0] == 3600
|
|
674
|
+
# Nine hours of run plus a minute of headroom, so the FIRST reading is inside
|
|
675
|
+
# the window rather than exactly on its edge.
|
|
676
|
+
assert fake.metrics_windows[1] >= 9 * 3600
|
|
677
|
+
assert "38,200 / 45,440 MiB" in capsys.readouterr().out
|
|
678
|
+
|
|
679
|
+
|
|
680
|
+
def test_a_window_the_caller_chose_is_never_widened(fake):
|
|
681
|
+
# An explicit --window is an instruction, not a default. Widening it would
|
|
682
|
+
# answer a question the caller did not ask, and the answer would be a bigger
|
|
683
|
+
# read of the log than they wanted.
|
|
684
|
+
fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Succeeded",
|
|
685
|
+
"started_at": _hours_ago(9)}
|
|
686
|
+
fake.metrics_result = {"gpu_series": [], "window_seconds": 600, "note": ""}
|
|
687
|
+
|
|
688
|
+
assert run(["watch", "job-a8acdef80a07", "--window", "600"]) == cli.EXIT_OK
|
|
689
|
+
assert fake.metrics_windows == [600]
|
|
690
|
+
|
|
691
|
+
|
|
692
|
+
def test_a_running_job_is_not_widened_however_quiet_it_is(fake):
|
|
693
|
+
# A running job's readings ARE inside the last hour, so an empty answer means
|
|
694
|
+
# the job has printed nothing yet. Re-reading a week of its log would buy a
|
|
695
|
+
# second empty answer and a much larger response.
|
|
696
|
+
fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Running",
|
|
697
|
+
"started_at": _hours_ago(9)}
|
|
698
|
+
fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
|
|
699
|
+
|
|
700
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
701
|
+
assert fake.metrics_windows == [3600]
|
|
702
|
+
|
|
703
|
+
|
|
704
|
+
def test_the_widened_window_stops_at_the_servers_own_cap(fake):
|
|
705
|
+
# `window_seconds` is declared ge=60, le=604800 on the metrics route. Asking
|
|
706
|
+
# for more is a 422, so a job that started a month ago must be clamped here
|
|
707
|
+
# rather than turned into an error the reader cannot act on.
|
|
708
|
+
fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Failed",
|
|
709
|
+
"started_at": _hours_ago(24 * 30)}
|
|
710
|
+
fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
|
|
711
|
+
|
|
712
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
713
|
+
assert fake.metrics_windows[1] == cli.MAX_WATCH_WINDOW == 604800
|
|
714
|
+
|
|
715
|
+
|
|
716
|
+
def test_a_finished_job_with_no_start_time_leaves_the_window_alone(fake):
|
|
717
|
+
# Nothing to compute a width from. Widening to the cap "just in case" would
|
|
718
|
+
# read seven days of log off the back of a missing field.
|
|
719
|
+
fake.status_result = {"job_id": "job-a8acdef80a07", "phase": "Succeeded"}
|
|
720
|
+
fake.metrics_result = {"gpu_series": [], "window_seconds": 3600, "note": ""}
|
|
721
|
+
|
|
722
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
723
|
+
assert fake.metrics_windows == [3600]
|
|
724
|
+
|
|
725
|
+
|
|
726
|
+
def test_the_window_help_quotes_the_servers_real_cap(capsys):
|
|
727
|
+
# It said 86400 and the server accepts 604800, so the help talked a reader
|
|
728
|
+
# out of a window the server would have answered.
|
|
729
|
+
with pytest.raises(SystemExit):
|
|
730
|
+
cli.main(["watch", "--help"])
|
|
731
|
+
printed = capsys.readouterr().out
|
|
732
|
+
assert "604800" in printed
|
|
733
|
+
assert "86400" not in printed
|
|
734
|
+
|
|
735
|
+
|
|
736
|
+
def test_watch_json_prints_the_cards_and_asks_for_no_phase(fake, capsys):
|
|
737
|
+
# --json is the server's answer verbatim, so the second request would buy
|
|
738
|
+
# nothing. It is one GET per typed `watch` and worth not making.
|
|
739
|
+
def refuse(job_id):
|
|
740
|
+
raise AssertionError("--json must not ask for the phase")
|
|
741
|
+
|
|
742
|
+
fake.status = refuse
|
|
743
|
+
fake.metrics_result = {"cards": four_a100_cards(), "gpu_series": [],
|
|
744
|
+
"window_seconds": 3600, "note": ""}
|
|
745
|
+
assert run(["watch", "job-66b46719b854", "--json"]) == cli.EXIT_OK
|
|
746
|
+
assert len(json.loads(capsys.readouterr().out)["cards"]) == 4
|
|
747
|
+
|
|
748
|
+
|
|
749
|
+
def test_samples_counts_the_readings_taken_not_the_chart_points(fake, capsys):
|
|
750
|
+
# ★ THE SAME DEFECT THE SCREEN HAD. The server thins the series to at most
|
|
751
|
+
# 400 points so a chart can draw it; job-66b46719b854 printed 785 readings
|
|
752
|
+
# per card and both surfaces counted the 393 that survived. The mean and the
|
|
753
|
+
# peak are computed over all 785, so the count beside them has to be 785.
|
|
754
|
+
fake.metrics_result = {
|
|
755
|
+
"latest_gpu": {"utilization_percent": 94, "memory_used_mib": 38200,
|
|
756
|
+
"memory_total_mib": 45440, "memory_percent": 84.1,
|
|
757
|
+
"temperature_c": 71, "power_w": 298.5},
|
|
758
|
+
"gpu_series": [{}] * 393,
|
|
759
|
+
"sample_count": 785,
|
|
760
|
+
"window_seconds": 604800, "note": "",
|
|
761
|
+
}
|
|
762
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
763
|
+
printed = capsys.readouterr().out
|
|
764
|
+
assert "samples 785" in printed
|
|
765
|
+
assert "samples 393" not in printed
|
|
766
|
+
|
|
767
|
+
|
|
768
|
+
def test_a_server_too_old_to_send_the_count_still_prints_one(fake, capsys):
|
|
769
|
+
# The field arrived on 2026-09-12. Against a server that predates it the
|
|
770
|
+
# thinned length is the only number there is, and a blank is worse.
|
|
771
|
+
fake.metrics_result = {
|
|
772
|
+
"latest_gpu": {"utilization_percent": 10, "memory_used_mib": 1,
|
|
773
|
+
"memory_total_mib": 2, "memory_percent": 50.0,
|
|
774
|
+
"temperature_c": 40, "power_w": 50.0},
|
|
775
|
+
"gpu_series": [{}] * 12,
|
|
776
|
+
"window_seconds": 3600, "note": "",
|
|
777
|
+
}
|
|
778
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
779
|
+
assert "samples 12" in capsys.readouterr().out
|
|
780
|
+
|
|
781
|
+
|
|
478
782
|
def test_an_unsettled_projection_is_labelled_rather_than_stated(fake, capsys):
|
|
479
783
|
fake.metrics_result = {
|
|
480
784
|
"progress": {"step": 5, "total_steps": 556, "percent": 0.9,
|
|
@@ -820,7 +1124,8 @@ def test_the_subcommand_name_survives_the_shell_positional():
|
|
|
820
1124
|
|
|
821
1125
|
def test_shell_with_a_command_runs_it_once_and_returns_its_exit_code(fake, capsys):
|
|
822
1126
|
assert run(["shell", "job-a8acdef80a07", "--", "date"]) == 0
|
|
823
|
-
assert fake.execed == [("job-a8acdef80a07", "date", 0)]
|
|
1127
|
+
assert fake.execed == [("job-a8acdef80a07", "date", 0, False, 0)], (
|
|
1128
|
+
"`-- command` 형태는 stateless 다 — script 는 남길 session 이 없다")
|
|
824
1129
|
assert "Sep 10" in capsys.readouterr().out
|
|
825
1130
|
|
|
826
1131
|
|
|
@@ -858,3 +1163,62 @@ def test_every_other_subcommand_still_dispatches(fake):
|
|
|
858
1163
|
"""dest 를 바꾼 것이 나머지를 깨지 않았다는 확인."""
|
|
859
1164
|
assert run(["status", "job-a8acdef80a07"]) == 0
|
|
860
1165
|
assert run(["secrets"]) == 0
|
|
1166
|
+
|
|
1167
|
+
|
|
1168
|
+
def test_the_prompt_uses_a_session_and_carries_the_sequence(fake, monkeypatch, capsys):
|
|
1169
|
+
"""★ WHY THE PROMPT DIFFERS FROM `-- command`. A person at a prompt assumes
|
|
1170
|
+
`cd` sticks; a script does not want a session left behind for the idle timer.
|
|
1171
|
+
So the prompt sends session=True and threads the sequence through, and the
|
|
1172
|
+
one-shot form stays stateless."""
|
|
1173
|
+
typed = iter(["cd /workspace", "pwd", "exit"])
|
|
1174
|
+
monkeypatch.setattr("builtins.input", lambda _prompt: next(typed))
|
|
1175
|
+
fake.exec_result = {"output": "/workspace\n", "exit_code": 0, "seq": 11, "lost": False}
|
|
1176
|
+
|
|
1177
|
+
assert run(["shell", "job-a8acdef80a07"]) == 0
|
|
1178
|
+
assert [(line, sess) for _job, line, _slot, sess, _seq in fake.execed] == [
|
|
1179
|
+
("cd /workspace", True), ("pwd", True)]
|
|
1180
|
+
# The FIRST line starts at 0 and the second carries what the first returned.
|
|
1181
|
+
assert [seq for *_rest, seq in fake.execed] == [0, 11]
|
|
1182
|
+
|
|
1183
|
+
|
|
1184
|
+
def test_a_session_that_closed_is_reopened_once_and_the_person_is_told(fake, monkeypatch,
|
|
1185
|
+
capsys):
|
|
1186
|
+
"""The one case where starting a new shell is right: the old one is provably
|
|
1187
|
+
gone (ten minutes unread, or the workload restarted). Everywhere else a new
|
|
1188
|
+
shell silently loses the `cd`, which is the failure this feature exists to end
|
|
1189
|
+
-- so the message says what was lost."""
|
|
1190
|
+
typed = iter(["pwd", "exit"])
|
|
1191
|
+
monkeypatch.setattr("builtins.input", lambda _prompt: next(typed))
|
|
1192
|
+
|
|
1193
|
+
calls = {"n": 0}
|
|
1194
|
+
original = fake.exec_in_job
|
|
1195
|
+
|
|
1196
|
+
def flaky(job, line, slot=0, session=False, seq=0):
|
|
1197
|
+
calls["n"] += 1
|
|
1198
|
+
if calls["n"] == 1:
|
|
1199
|
+
raise cli.ServerError("no session for this slot; open one first")
|
|
1200
|
+
return original(job, line, slot=slot, session=session, seq=seq)
|
|
1201
|
+
|
|
1202
|
+
fake.exec_in_job = flaky
|
|
1203
|
+
assert run(["shell", "job-a8acdef80a07"]) == 0
|
|
1204
|
+
assert calls["n"] == 2, "it retried once rather than giving up or looping"
|
|
1205
|
+
err = capsys.readouterr().err
|
|
1206
|
+
assert "reopening" in err and "are gone" in err, (
|
|
1207
|
+
"a silent reopen would let somebody keep typing paths relative to a cd that no "
|
|
1208
|
+
"longer applies")
|
|
1209
|
+
|
|
1210
|
+
|
|
1211
|
+
def test_a_failure_that_is_not_a_lost_session_is_not_retried(fake, monkeypatch, capsys):
|
|
1212
|
+
typed = iter(["pwd", "exit"])
|
|
1213
|
+
monkeypatch.setattr("builtins.input", lambda _prompt: next(typed))
|
|
1214
|
+
|
|
1215
|
+
calls = {"n": 0}
|
|
1216
|
+
|
|
1217
|
+
def always_bad(job, line, slot=0, session=False, seq=0):
|
|
1218
|
+
calls["n"] += 1
|
|
1219
|
+
raise cli.ServerError("the driver pod did not answer in JSON")
|
|
1220
|
+
|
|
1221
|
+
fake.exec_in_job = always_bad
|
|
1222
|
+
assert run(["shell", "job-a8acdef80a07"]) == 0
|
|
1223
|
+
assert calls["n"] == 1, "retrying an unrelated failure would double every bad command"
|
|
1224
|
+
assert "did not answer in JSON" in capsys.readouterr().err
|
|
@@ -65,7 +65,9 @@ def test_shell_sends_the_command_and_waits_longer_than_the_server():
|
|
|
65
65
|
result = client_with(session).exec_in_job("baseline-c", "nvidia-smi -L", slot=1)
|
|
66
66
|
call = session.calls[0]
|
|
67
67
|
assert call["url"] == "https://run.example/v1/jobs/baseline-c/exec"
|
|
68
|
-
assert call["json"] == {"command": "nvidia-smi -L", "slot": 1, "timeout_seconds": 20
|
|
68
|
+
assert call["json"] == {"command": "nvidia-smi -L", "slot": 1, "timeout_seconds": 20,
|
|
69
|
+
"session": False, "seq": 0}, (
|
|
70
|
+
"PACSRUN-SHELL-SESSION 이후에도 기본은 한 줄짜리 stateless 형태다")
|
|
69
71
|
# The server may hold the request for its whole window, so the client's
|
|
70
72
|
# read timeout must outlive it.
|
|
71
73
|
assert call["timeout"] > 20
|
|
@@ -155,3 +157,15 @@ def test_a_submit_is_given_more_patience_than_a_read():
|
|
|
155
157
|
read_session = FakeSession(FakeResponse(200, {}))
|
|
156
158
|
client_with(read_session).status("job-a8acdef80a07")
|
|
157
159
|
assert submit_session.calls[0]["timeout"] > read_session.calls[0]["timeout"]
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def test_shell_can_ask_for_the_persistent_session_and_carries_its_sequence():
|
|
163
|
+
"""The prompt loop's shape: the same route, with `session` and the output
|
|
164
|
+
sequence the caller last saw. Without the sequence the caller would be handed
|
|
165
|
+
everything the shell has ever printed on every line."""
|
|
166
|
+
session = FakeSession(FakeResponse(200, {"output": "/workspace\n", "exit_code": 0,
|
|
167
|
+
"seq": 42, "lost": False, "note": ""}))
|
|
168
|
+
result = client_with(session).exec_in_job("baseline-c", "pwd", session=True, seq=7)
|
|
169
|
+
assert session.calls[0]["json"]["session"] is True
|
|
170
|
+
assert session.calls[0]["json"]["seq"] == 7
|
|
171
|
+
assert result["seq"] == 42, "the reply carries the next one back"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|