hyperun 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
hyperun/cli.py ADDED
@@ -0,0 +1,1202 @@
1
+ """The `hyperun` command: argument parsing, and deciding what to print.
2
+
3
+ END-TO-END FLOW of one `hyperun submit -f job.yaml`:
4
+
5
+ 1. `main()` parses the arguments and dispatches to `cmd_submit`.
6
+ 2. `config.load()` finds the server URL and the token, from the environment or
7
+ from `~/.config/hyperun/config.json`.
8
+ 3. `build_submit_body()` reads the file, applies any flag overrides on top,
9
+ and produces the request body. It does NOT validate: the server owns that
10
+ judgement, and a second copy here would eventually disagree with it.
11
+ 4. `Client.submit()` POSTs it.
12
+ 5. The job id, the phase and the result path are printed, or the whole JSON
13
+ when `--json` was given.
14
+
15
+ THE COMMANDS, AND WHY EACH EXISTS
16
+
17
+ login / logout store and remove the token. Stage 1 authentication is a
18
+ static token an operator hands out, so `login` takes one
19
+ rather than opening a browser. When Cognito lands
20
+ (`docs/08-plan.md` open item 4) only this command changes.
21
+ explain / schema ask the SERVER what it is and what it accepts, rather than
22
+ printing something baked in here that can go stale. This is
23
+ what makes the tool usable by a coding agent that has read
24
+ no documentation (`docs/07-agent-skill.md`).
25
+ estimate how long, how much, which GPU. Submits nothing.
26
+ validate what is wrong with the job. Submits nothing. Exits 1 when
27
+ something would actually stop it, so a script can gate a
28
+ submit on it.
29
+ submit the point of the whole thing.
30
+ status phase, message, which GPU, how many restarts.
31
+ watch GPU usage and training progress, read out of the job's own
32
+ log. Nothing is stored anywhere for this.
33
+ stats what your team has spent. Aggregate only: being on a team
34
+ does not let you read a member's jobs.
35
+ logs output, optionally followed.
36
+
37
+ WHAT IS NOT HERE. No `gpus`. Answering it needs a vendor API key in the server
38
+ and a catalogue cache, and a CLI that answered it locally would be the second
39
+ source of truth this design exists to avoid.
40
+
41
+ EXIT CODES, because a script will read them:
42
+ 0 it worked
43
+ 1 the server refused, or could not be reached
44
+ 2 the command was wrong, or there are no credentials
45
+
46
+ Grep anchor: DDPSRUN-CLI
47
+ """
48
+
49
+ from __future__ import annotations
50
+
51
+ import argparse
52
+ import base64
53
+ import getpass
54
+ import json
55
+ import time
56
+ import sys
57
+ from typing import Any
58
+
59
+ from . import __version__
60
+ from . import browser_login
61
+ from . import config
62
+ from .client import Client, ServerError
63
+
64
+ EXIT_OK = 0
65
+ EXIT_SERVER = 1
66
+ EXIT_USAGE = 2
67
+
68
+
69
+ def job_arguments() -> argparse.ArgumentParser:
70
+ """The arguments that describe a job, shared by three subcommands.
71
+
72
+ WHY THIS IS A PARENT PARSER RATHER THAN THREE COPIES. `estimate`,
73
+ `validate` and `submit` must accept exactly the same job description, so
74
+ that a user checks a job and then submits THAT job with nothing rewritten
75
+ in between. Three copies would drift, and the drift would show up as a
76
+ validate that passes something the submit refuses.
77
+
78
+ Returns:
79
+ A parser with `add_help=False`, to be passed as a `parents=` entry.
80
+ """
81
+ shared = argparse.ArgumentParser(add_help=False)
82
+ shared.add_argument("-f", "--file", help="a YAML or JSON file describing the job")
83
+ shared.add_argument("--name", help="a name for your own benefit")
84
+ shared.add_argument("--image", help="container image to run")
85
+ shared.add_argument(
86
+ "--arg", action="append", default=[], dest="args_", metavar="ARG",
87
+ help="one argument to the container. Repeat it, in order.",
88
+ )
89
+ shared.add_argument(
90
+ "--env", action="append", default=[], metavar="KEY=VALUE",
91
+ help="a non-secret environment variable. Repeatable.",
92
+ )
93
+ shared.add_argument(
94
+ "--secret", action="append", default=[], metavar="NAME",
95
+ help="the NAME of a stored secret to inject. Never the value. Repeatable.",
96
+ )
97
+ shared.add_argument("--gpu-vram", type=int, metavar="GB", help="minimum GPU memory, e.g. 48")
98
+ shared.add_argument("--gpu-name", metavar="MODEL", help="exact GPU model, e.g. L40S")
99
+ shared.add_argument(
100
+ "--gpu-count", type=int, metavar="N", help="how many GPUs PER POD (default 1)"
101
+ )
102
+ shared.add_argument(
103
+ "--capacity-type", choices=["on-demand", "spot"],
104
+ help="how the machine is bought. YOU decide this. on-demand costs more and is "
105
+ "not taken away; spot is cheaper and can be reclaimed mid-run. Run "
106
+ "`hyperun estimate` first — it recommends one and says why. submit refuses "
107
+ "without it rather than choosing for you.",
108
+ )
109
+ # DDPSRUN-VENDOR-CHOICE. The two placement fields that are not capacity type.
110
+ # The valid names are NOT listed as argparse `choices` on purpose: the server
111
+ # owns that list and refuses an unknown one with a message naming every name
112
+ # that would have worked, and a second copy in this file would go stale the
113
+ # first time a vendor is added.
114
+ shared.add_argument(
115
+ "--vendor", action="append", metavar="NAME",
116
+ help="who the machine may be bought from. Repeat it to allow several; "
117
+ "omit it entirely for no restriction, which is what every job did before "
118
+ "this flag existed. aws and runpod can actually run a job; gcp, azure, "
119
+ "lambda and nebius can only be PRICED, so name one of those only with "
120
+ "--placement-mode compare.",
121
+ )
122
+ # DDPSRUN-REGIONS. Missing until 2026-09-08, so every job this CLI submitted
123
+ # ran in the operator's one default region and there was no way to say
124
+ # otherwise -- the same hole `--vendor` filled a day earlier.
125
+ shared.add_argument(
126
+ "--region", action="append", metavar="NAME",
127
+ help="where the machine may be bought, as PACSrun spells it: a bare "
128
+ "vendor ('gcp') or a vendor and region ('aws/us-east-1'). Repeat it to "
129
+ "allow several. OMITTING IT IS NOT 'anywhere' -- an AWS ask that names "
130
+ "no region gets the operator's ONE default region, us-west-2 on this "
131
+ "deployment. It matters: the H100 is $6.88/hour in us-west-2 and $8.60 "
132
+ "in ap-northeast-1. `hyperun schema` lists every region on offer.",
133
+ )
134
+ shared.add_argument(
135
+ "--placement-mode", choices=["ordered", "cheapest", "compare"],
136
+ help="what to do with the candidates. ordered (the default) asks them in "
137
+ "order and stops at the first that answers, comparing nothing. cheapest "
138
+ "asks every candidate and buys the cheapest answer. compare asks every "
139
+ "candidate, ranks them and then STOPS -- nothing is bought and the job "
140
+ "ends in the phase Compared with the winner and the margin in its "
141
+ "message. compare is the only mode that costs nothing to run.",
142
+ )
143
+ shared.add_argument(
144
+ "--parallelism", type=int, metavar="N",
145
+ help="how many pods run at once (default 1). They are independent workers that never "
146
+ "talk to each other. With --gpu-count this is how a job fills a multi-GPU machine: "
147
+ "--parallelism 8 --gpu-count 1 may land 4 pods on each of two 4-GPU boxes.",
148
+ )
149
+ # DDPSRUN-GROUP. Distributed training. Two flags and not one, because the
150
+ # size alone says nothing: `--group-size 2` with the default independent
151
+ # mode is identical to no group at all, and PACSrun treats the two shapes
152
+ # with opposite scheduling rules.
153
+ shared.add_argument(
154
+ "--group-size", type=int, metavar="N",
155
+ help="how many pods form ONE distributed group. --parallelism divided by "
156
+ "this is the number of groups, so --parallelism 6 --group-size 2 is three "
157
+ "groups of two. Needs --group-mode distributed to mean anything.",
158
+ )
159
+ shared.add_argument(
160
+ "--group-mode", choices=("independent", "distributed"),
161
+ help="independent (default) is pods that never talk. distributed makes "
162
+ "each group one process group: every pod is given PACSRUN_MASTER_ADDR, "
163
+ "PACSRUN_MASTER_PORT, PACSRUN_GROUP_RANK and PACSRUN_GROUP_SIZE, and NONE "
164
+ "starts until the whole group has a machine. Your script passes those to "
165
+ "its launcher -- nothing translates them for you.",
166
+ )
167
+ shared.add_argument("--cpus", help='CPU request, e.g. "4"')
168
+ shared.add_argument("--memory", help='memory request, e.g. "16Gi"')
169
+ shared.add_argument(
170
+ "--expected-hours", type=float, help="your own guess at the runtime, in hours"
171
+ )
172
+ # DDPSRUN-CONTINUE-FROM. Chain this job onto a previous one's result path.
173
+ shared.add_argument(
174
+ "--continue-from", metavar="JOB_ID",
175
+ help="the job id of a previous run of YOURS whose result path this job "
176
+ "should reuse, e.g. job-3e1e34cb042c. Use it to continue a "
177
+ "multi-iteration run: without it every submit writes to a fresh prefix "
178
+ "and the resume step finds nothing. Must be a job of yours in the same "
179
+ "namespace; anything else is refused with 404.",
180
+ )
181
+ # The facts the server cannot read out of a container image. Only estimate
182
+ # and validate use them today, but submit accepts them too so that one file
183
+ # works for all three.
184
+ shared.add_argument("--pairs", type=int, help="how many training pairs your dataset holds")
185
+ shared.add_argument("--epochs", type=int, help="how many passes over the dataset")
186
+ shared.add_argument(
187
+ "--row-tokens", type=int, metavar="N",
188
+ help="average length of ONE response, in tokens. Without it there is no "
189
+ "runtime estimate.",
190
+ )
191
+ shared.add_argument("--cap", type=int, help="--max-len, which decides peak memory")
192
+ shared.add_argument("--batch-size", type=int, help="per_device_train_batch_size (default 1)")
193
+ shared.add_argument("--grad-accum", type=int, help="gradient_accumulation_steps (default 8)")
194
+ shared.add_argument(
195
+ "--resumable", action="store_true",
196
+ help="your job can restart from a checkpoint",
197
+ )
198
+ shared.add_argument(
199
+ "--script", metavar="PATH",
200
+ help="your run.sh. THIS IS WHAT RUNS when you pass no --arg: the job gets "
201
+ "args ['bash','-lc',<the file's text>]. It also unlocks four more validate "
202
+ "checks. The text is read and sent; the path is not, and nothing is stored.",
203
+ )
204
+ return shared
205
+
206
+
207
+ def build_parser() -> argparse.ArgumentParser:
208
+ """Assemble the command line.
209
+
210
+ Returns:
211
+ A parser whose every `help` string is written for someone who has not
212
+ read anything else. This is the only documentation most users will see.
213
+ """
214
+ parser = argparse.ArgumentParser(
215
+ prog="hyperun",
216
+ description="Submit a batch job to a GPU we rent for you, and get the results back.",
217
+ epilog="Run `hyperun explain` for the full description, straight from the server.",
218
+ )
219
+ # ON THE TOP-LEVEL PARSER ON PURPOSE, unlike --json below. "which version am
220
+ # I running" is a question about the INSTALL and not about any one command,
221
+ # and `hyperun --version` is where every other tool puts it. There was no way
222
+ # to ask at all before 2026-09-08, which is awkward the moment somebody
223
+ # reports a bug against a CLI they installed from PyPI.
224
+ parser.add_argument(
225
+ "--version", action="version", version=f"hyperun {__version__}",
226
+ help="print the installed version and exit",
227
+ )
228
+ # ★ THE dest IS `subcommand`, NOT `command`, AND THAT IS NOT COSMETIC. The
229
+ # `shell` subparser has its own positional called `command` -- the shell line
230
+ # to run -- and argparse writes both into the SAME namespace. With both named
231
+ # `command` the subparser's REMAINDER overwrote the subcommand name, so
232
+ # `ddpsrun shell job-x` set args.command to [] and main() printed the
233
+ # top-level help, while `ddpsrun shell job-x -- nvidia-smi` set it to
234
+ # ['nvidia-smi'] and main() died on `COMMANDS[['nvidia-smi']]` with
235
+ # "TypeError: unhashable type: 'list'". Both forms, every version: `shell`
236
+ # had never once dispatched. Found 2026-09-10 from a user's paste of the
237
+ # help text where a command should have run.
238
+ sub = parser.add_subparsers(dest="subcommand", metavar="<command>")
239
+
240
+ def add_json_flag(target: argparse.ArgumentParser) -> None:
241
+ """`--json` goes on each subcommand that has something to print.
242
+
243
+ Not on the top-level parser: argparse would require it BEFORE the
244
+ subcommand (`hyperun --json status X`), which is not the order anyone
245
+ types, and a subparser redefining it would silently reset it to False.
246
+ """
247
+ target.add_argument(
248
+ "--json", action="store_true",
249
+ help="print raw JSON instead of a human summary. Use this from a script.",
250
+ )
251
+
252
+ login = sub.add_parser("login", help="sign in and store the result")
253
+ login.add_argument("--server", required=True, help="the gateway URL, e.g. https://run.example")
254
+ login.add_argument(
255
+ "--token",
256
+ help="skip the browser and use this token. For CI and scripts, which "
257
+ "have nobody to sign in. Omitting it opens a browser when the server "
258
+ "supports that, and prompts otherwise so it stays out of shell history.",
259
+ )
260
+
261
+ # ★ THE COMMAND IS `delete` AND `cancel` IS THE OLD SPELLING. It was called
262
+ # cancel and that name described an intention rather than the action: there
263
+ # is no cancelled state to move a job into, because the CRD's only stop is
264
+ # deleting the PacsJob (config/deploy/rbac.yaml, and the phase enum in
265
+ # api/v1alpha1/pacsjob_types.go carries Pending / Starting / Running /
266
+ # Recovering / Succeeded / Failed / Compared and nothing else). A person
267
+ # who reads "cancel" expects the job to still be listed afterwards, and on
268
+ # 2026-09-09 one asked why a cancelled job showed no cancelled state. The
269
+ # alias stays because scripts and every document written before today say
270
+ # `cancel`, and breaking those to rename a verb is not a trade worth making.
271
+ cancel = sub.add_parser(
272
+ "delete", aliases=["cancel"],
273
+ help="delete a job -- it stops and disappears from the list",
274
+ description="Deletes the PacsJob. That is the only stop the CRD offers: "
275
+ "PACSrun watches for the object going away and gives back whatever the "
276
+ "job had rented. There is no cancelled state to look at afterwards, "
277
+ "because there is no object left to carry one. Files already written to "
278
+ "the job's result path are NOT deleted.",
279
+ )
280
+ cancel.add_argument("job_id", help="the id `submit` printed")
281
+ cancel.add_argument(
282
+ "--yes", "-y", action="store_true",
283
+ help="skip the confirmation. For scripts, which have nobody to answer it.",
284
+ )
285
+
286
+ sub.add_parser("logout", help="delete the stored token")
287
+ sub.add_parser("explain", help="what this tool is and how to use it (asks the server)")
288
+ sub.add_parser("schema", help="the exact shape of a submit request (asks the server)")
289
+
290
+ job = job_arguments()
291
+
292
+ estimate = sub.add_parser(
293
+ "estimate", parents=[job],
294
+ help="how long it will take, what it will cost, which GPU. Submits nothing",
295
+ description="Nothing is submitted. An answer of `unknown` is a real answer: "
296
+ "the last time we estimated a combination we had never measured, we were 96% out.",
297
+ )
298
+ add_json_flag(estimate)
299
+
300
+ validate = sub.add_parser(
301
+ "validate", parents=[job],
302
+ help="what is wrong with this job. Submits nothing",
303
+ description="Nothing is submitted. Pass --script to unlock four more checks.",
304
+ )
305
+ add_json_flag(validate)
306
+
307
+ submit = sub.add_parser(
308
+ "submit", parents=[job], help="submit a job",
309
+ description="Give a YAML or JSON file, or build the request from flags, or both. "
310
+ "Flags win over the file.",
311
+ )
312
+ add_json_flag(submit)
313
+
314
+ sub.add_parser(
315
+ "secrets", help="which secret names you can use",
316
+ description="`secrets: [NAME]` on a submit request is a word that opens "
317
+ "the server's vault, not a field you fill with a value. This prints the "
318
+ "words that work: the deployment's own, plus whatever your namespace "
319
+ "registered with `secret set`. Values are never shown by any command.",
320
+ )
321
+
322
+ # DDPSRUN-USER-SECRET. Registering a value for your own namespace.
323
+ secret_set = sub.add_parser(
324
+ "secret-set", help="store a value in your namespace under a name",
325
+ description="Store one value so your jobs can ask for it by name. It is "
326
+ "kept in a Kubernetes Secret in YOUR namespace; jobs elsewhere cannot "
327
+ "name it, and no command ever prints it back.\n\n"
328
+ "★ THE VALUE IS NEVER AN ARGUMENT. It comes from a file or from stdin, "
329
+ "because anything on a command line is in your shell history, in `ps` "
330
+ "output for every user on the machine, and in any terminal recording.",
331
+ )
332
+ secret_set.add_argument(
333
+ "name", help="the environment variable name your script reads, e.g. HF_TOKEN"
334
+ )
335
+ secret_set.add_argument(
336
+ "--from-file", metavar="PATH",
337
+ help="read the value from this file. Use - for stdin, which is also the "
338
+ "default when this is omitted.",
339
+ )
340
+ secret_set.add_argument(
341
+ "--expires-at", metavar="WHEN",
342
+ help="when this value stops working, ISO-8601 (e.g. 2026-09-10T02:27:00Z). "
343
+ "Worth sending for a TEMPORARY credential -- a federation token, an "
344
+ "assumed-role session. `validate` then refuses a job that asks for it "
345
+ "after that time, so the expiry is found before a GPU is rented instead "
346
+ "of an hour into the run.",
347
+ )
348
+
349
+ secret_rm = sub.add_parser(
350
+ "secret-rm", help="forget a name your namespace registered",
351
+ description="Remove one registered value. Do this the moment a "
352
+ "credential leaks: until it is gone, every job in the namespace can "
353
+ "still ask for it. An operator's deployment-wide binding cannot be "
354
+ "removed from here.",
355
+ )
356
+ secret_rm.add_argument("name")
357
+
358
+ status = sub.add_parser("status", help="how a job is doing")
359
+ status.add_argument("job_id")
360
+ add_json_flag(status)
361
+
362
+ shell = sub.add_parser(
363
+ "shell", help="run commands inside a running job's workload",
364
+ description="Each line is one HTTPS round trip: the server relays it "
365
+ "through the job's driver pod into the workload container on the "
366
+ "rented machine and brings the exit code back like ssh. It is NOT a "
367
+ "TTY — no vim, no top, about 25 seconds per command — because the "
368
+ "server is a Lambda and cannot hold a terminal open. AWS and GCP "
369
+ "machine rentals only: a RunPod job is a rented container with no "
370
+ "machine behind it, and the relay refuses it. ★ PUT OPTIONS BEFORE THE "
371
+ "JOB ID -- everything after it is sent to the workload as-is, so "
372
+ "`shell job-x --slot 2` asks pod 0 and passes `--slot 2` to the shell. "
373
+ "That is refused rather than obeyed.",
374
+ )
375
+ shell.add_argument("job", help="the job id, or the PacsJob's Kubernetes name")
376
+ shell.add_argument("--slot", type=int, default=0, help="which pod of a parallel job")
377
+ shell.add_argument(
378
+ "command", nargs=argparse.REMAINDER, metavar="-- COMMAND",
379
+ help="one shell line to run and exit; leave it off for a prompt",
380
+ )
381
+
382
+ watch = sub.add_parser(
383
+ "watch", help="GPU usage and training progress",
384
+ description="Read out of the job's own log. Nothing is stored, so what "
385
+ "you can see goes back as far as the log does.",
386
+ )
387
+ watch.add_argument("job_id")
388
+ watch.add_argument(
389
+ "--window", type=int, default=3600, metavar="SECONDS",
390
+ help="how far back to read (default 3600, max 86400)",
391
+ )
392
+ add_json_flag(watch)
393
+
394
+ team_stats = sub.add_parser(
395
+ "stats", help="what your team has spent",
396
+ description="Aggregate only. Being on a team does not let you read a member's jobs.",
397
+ )
398
+ add_json_flag(team_stats)
399
+
400
+ logs = sub.add_parser("logs", help="a job's output")
401
+ logs.add_argument("job_id")
402
+ logs.add_argument(
403
+ "-f", "--follow", action="store_true",
404
+ help="keep printing as new lines arrive. The server cannot stream, so this "
405
+ "asks again every few seconds and prints what is new.",
406
+ )
407
+ logs.add_argument(
408
+ "--interval", type=float, default=6.0, metavar="SECONDS",
409
+ help="how often to ask, with --follow (default 6)",
410
+ )
411
+
412
+ return parser
413
+
414
+
415
+ def load_job_file(path: str) -> dict[str, Any]:
416
+ """Read a job description from YAML or JSON.
417
+
418
+ Args:
419
+ path: the file to read. `.json` is parsed as JSON, anything else as
420
+ YAML — and YAML is a superset of JSON, so a `.txt` holding JSON
421
+ works too.
422
+
423
+ Returns:
424
+ The parsed mapping.
425
+
426
+ Raises:
427
+ SystemExit: the file is unreadable or is not a mapping. Exiting here
428
+ rather than raising keeps the error one line instead of a traceback.
429
+ """
430
+ try:
431
+ with open(path, "r", encoding="utf-8") as handle:
432
+ text = handle.read()
433
+ except OSError as exc:
434
+ raise SystemExit(f"cannot read {path}: {exc}") from exc
435
+
436
+ try:
437
+ if path.endswith(".json"):
438
+ document = json.loads(text)
439
+ else:
440
+ import yaml
441
+
442
+ document = yaml.safe_load(text)
443
+ except Exception as exc: # noqa: BLE001 - json and yaml raise different types
444
+ raise SystemExit(f"{path} does not parse: {exc}") from exc
445
+
446
+ if not isinstance(document, dict):
447
+ raise SystemExit(f"{path} must contain a mapping, not a {type(document).__name__}")
448
+ return document
449
+
450
+
451
+ def parse_env_pairs(pairs: list[str]) -> dict[str, str]:
452
+ """Turn `--env KEY=VALUE` arguments into a mapping.
453
+
454
+ Args:
455
+ pairs: the raw strings.
456
+
457
+ Returns:
458
+ A dict. A value containing `=` is kept whole, because
459
+ `--env ARGS=--lr=1e-5` is a real thing people write.
460
+
461
+ Raises:
462
+ SystemExit: a pair has no `=` at all.
463
+ """
464
+ result: dict[str, str] = {}
465
+ for pair in pairs:
466
+ key, separator, value = pair.partition("=")
467
+ if not separator or not key:
468
+ raise SystemExit(f"--env {pair!r} must be KEY=VALUE")
469
+ result[key] = value
470
+ return result
471
+
472
+
473
+ def build_submit_body(args: argparse.Namespace) -> dict[str, Any]:
474
+ """Assemble the request body from a file and the flags on top of it.
475
+
476
+ Args:
477
+ args: the parsed `submit` arguments.
478
+
479
+ Returns:
480
+ The body to POST. Nothing is validated here — the server decides what is
481
+ acceptable, and a second copy of that judgement in the CLI would drift
482
+ away from it. The one exception is the shape of `--env`, which has to be
483
+ parsed before it can be sent at all.
484
+
485
+ Raises:
486
+ SystemExit: neither a file nor the two fields the server requires.
487
+ """
488
+ body: dict[str, Any] = load_job_file(args.file) if args.file else {}
489
+
490
+ if args.name:
491
+ body["name"] = args.name
492
+ if args.image:
493
+ body["image"] = args.image
494
+ if args.args_:
495
+ body["args"] = args.args_
496
+ if args.env:
497
+ # Merged, not replaced: a file may carry the ten stable values and a
498
+ # flag override the one that changes between runs.
499
+ merged = dict(body.get("env") or {})
500
+ merged.update(parse_env_pairs(args.env))
501
+ body["env"] = merged
502
+ if args.secret:
503
+ body["secrets"] = sorted(set(body.get("secrets") or []) | set(args.secret))
504
+ if args.cpus:
505
+ body["cpus"] = args.cpus
506
+ if args.memory:
507
+ body["memory"] = args.memory
508
+ if args.expected_hours is not None:
509
+ body["expected_hours"] = args.expected_hours
510
+ if getattr(args, "parallelism", None) is not None:
511
+ body["parallelism"] = args.parallelism
512
+ if getattr(args, "continue_from", None):
513
+ body["continue_from"] = args.continue_from
514
+ if (getattr(args, "group_size", None) is not None
515
+ or getattr(args, "group_mode", None) is not None):
516
+ # Either flag alone is meaningful: a size with the default mode is
517
+ # `independent`, which the server drops, and a mode with no size is a
518
+ # group of one, which it also drops. Sending what was asked for and
519
+ # letting the server decide what is worth writing keeps one rule in one
520
+ # place.
521
+ body["group"] = {
522
+ "size": args.group_size if args.group_size is not None else 1,
523
+ "mode": args.group_mode or "independent",
524
+ }
525
+ if getattr(args, "capacity_type", None):
526
+ body["capacity_type"] = args.capacity_type
527
+ # DDPSRUN-VENDOR-CHOICE. `--vendor` REPLACES the file's list rather than
528
+ # adding to it, which is the opposite of how `--secret` behaves one screen
529
+ # up. The two are different in kind: a secret is one more thing to inject and
530
+ # a union is the obvious reading, while a vendor list is a RESTRICTION, so a
531
+ # union would silently WIDEN what the file allowed -- `--vendor runpod`
532
+ # against a file saying `[aws]` would run on either, which is not what
533
+ # anybody typing that means.
534
+ if getattr(args, "vendor", None):
535
+ body["vendors"] = list(dict.fromkeys(args.vendor))
536
+ if getattr(args, "placement_mode", None):
537
+ body["placement_mode"] = args.placement_mode
538
+ # REPLACES the file's list rather than adding to it, for the same reason
539
+ # --vendor does: a region list is a RESTRICTION, so a union would silently
540
+ # widen what the file allowed.
541
+ if getattr(args, "region", None):
542
+ body["regions"] = list(dict.fromkeys(args.region))
543
+
544
+ # A GPU is asked for in exactly one of two styles, by memory or by model.
545
+ # TWO CASES THAT LOOK ALIKE AND ARE NOT:
546
+ # file says one style, a flag says the other -> the flag is the newer
547
+ # intent, so it replaces the file's style.
548
+ # BOTH flags on the same command line -> the user contradicted
549
+ # themselves. Picking one would send a job to a GPU they did not ask
550
+ # for, so this stops. Found 2026-08-31 by running the CLI against the
551
+ # server: the two branches used to erase each other and whichever ran
552
+ # last silently won.
553
+ if args.gpu_vram and args.gpu_name:
554
+ raise SystemExit(
555
+ "--gpu-vram and --gpu-name ask for a GPU in two different ways. "
556
+ "Give one: --gpu-vram 48 for 'at least 48 GB', --gpu-name L40S for "
557
+ "that exact model."
558
+ )
559
+ if args.gpu_vram or args.gpu_name or args.gpu_count:
560
+ gpu = dict(body.get("gpu") or {})
561
+ if args.gpu_vram:
562
+ gpu["vram_gb"] = args.gpu_vram
563
+ gpu.pop("name", None)
564
+ if args.gpu_name:
565
+ gpu["name"] = args.gpu_name
566
+ gpu.pop("vram_gb", None)
567
+ if args.gpu_count:
568
+ gpu["count"] = args.gpu_count
569
+ body["gpu"] = gpu
570
+
571
+ # The facts the server cannot read out of a container image. They live under
572
+ # `training` in the request, and a flag overrides the file the same way
573
+ # everything else does.
574
+ training = dict(body.get("training") or {})
575
+ for flag, field in (
576
+ ("pairs", "pairs"), ("epochs", "epochs"), ("row_tokens", "row_tokens"),
577
+ ("cap", "cap"), ("batch_size", "batch_size"), ("grad_accum", "grad_accum"),
578
+ ):
579
+ value = getattr(args, flag, None)
580
+ if value is not None:
581
+ training[field] = value
582
+ if getattr(args, "resumable", False):
583
+ training["resumable"] = True
584
+ if training:
585
+ body["training"] = training
586
+
587
+ # `--script run.sh` sends the file's TEXT, not its path. The server has no
588
+ # way to read a file on the user's laptop.
589
+ script_path = getattr(args, "script", None)
590
+ if script_path:
591
+ try:
592
+ with open(script_path, "r", encoding="utf-8") as handle:
593
+ body["script"] = handle.read()
594
+ except OSError as exc:
595
+ raise SystemExit(f"cannot read {script_path}: {exc}") from exc
596
+
597
+ if not body.get("name") or not body.get("image"):
598
+ raise SystemExit(
599
+ "a job needs at least a name and an image. Give them with --name and "
600
+ "--image, or in a file with -f. `hyperun schema` prints every field."
601
+ )
602
+ return body
603
+
604
+
605
+ def refreshed(credentials: config.Credentials) -> config.Credentials:
606
+ """Renew the stored id_token when it is about to expire.
607
+
608
+ DDPSRUN-CLI-REFRESH. A Cognito id_token lives an hour. Without this, a
609
+ command run 61 minutes after `hyperun login` fails with 401 and the person
610
+ has no idea why. With a refresh token stored, the renewal is silent.
611
+
612
+ Args:
613
+ credentials: what `config.load` returned.
614
+
615
+ Returns:
616
+ The same credentials, or ones carrying a fresh id_token. A static token
617
+ has no `exp` and comes back untouched, as does anything that cannot be
618
+ renewed — the request then fails on its own and says so, which is a
619
+ better error than one from in here.
620
+ """
621
+ if not credentials.refresh_token:
622
+ return credentials
623
+ try:
624
+ payload = credentials.token.split(".")[1]
625
+ payload += "=" * (-len(payload) % 4)
626
+ claims = json.loads(base64.urlsafe_b64decode(payload).decode())
627
+ expiry = float(claims["exp"])
628
+ except Exception:
629
+ return credentials
630
+ # 60 seconds of margin, so a token does not expire between this check and
631
+ # the request it is about to be used on.
632
+ if time.time() < expiry - 60:
633
+ return credentials
634
+
635
+ try:
636
+ settings = browser_login.login_config(credentials.server)
637
+ if not settings.get("enabled"):
638
+ return credentials
639
+ tokens = browser_login.exchange(settings, {
640
+ "grant_type": "refresh_token",
641
+ "refresh_token": credentials.refresh_token,
642
+ })
643
+ except browser_login.LoginError:
644
+ return credentials
645
+
646
+ renewed = config.Credentials(
647
+ server=credentials.server,
648
+ token=tokens.id_token or credentials.token,
649
+ # A refresh does not mint a new refresh token, so keep the one we have.
650
+ refresh_token=tokens.refresh_token or credentials.refresh_token,
651
+ )
652
+ config.save(renewed)
653
+ return renewed
654
+
655
+
656
+ def client_from_config() -> Client:
657
+ """Build a client, or exit 2 saying how to log in."""
658
+ try:
659
+ credentials = config.load()
660
+ except config.NotLoggedIn as exc:
661
+ print(exc, file=sys.stderr)
662
+ raise SystemExit(EXIT_USAGE) from exc
663
+ credentials = refreshed(credentials)
664
+ return Client(credentials.server, credentials.token)
665
+
666
+
667
+ def cmd_login(args: argparse.Namespace) -> int:
668
+ """Store the server address and the token.
669
+
670
+ The token is read with `getpass` when it was not given on the command line,
671
+ so it does not end up in the shell's history file.
672
+ """
673
+ server = args.server.rstrip("/")
674
+ refresh = ""
675
+
676
+ if args.token:
677
+ token = args.token.strip()
678
+ else:
679
+ # DDPSRUN-CLI-LOGIN. With no --token, try the browser first. A server
680
+ # that has no user pool says so and we fall back to the prompt, which is
681
+ # exactly what this command did before Cognito existed.
682
+ try:
683
+ tokens = browser_login.login(server)
684
+ token, refresh = tokens.id_token, tokens.refresh_token
685
+ except browser_login.LoginError as exc:
686
+ if "does not have browser sign-in" not in str(exc):
687
+ print(str(exc), file=sys.stderr)
688
+ return EXIT_SERVER
689
+ token = getpass.getpass("token: ").strip()
690
+
691
+ if not token:
692
+ print("no token given", file=sys.stderr)
693
+ return EXIT_USAGE
694
+
695
+ path = config.save(config.Credentials(
696
+ server=server, token=token, refresh_token=refresh))
697
+ print(f"saved to {path}")
698
+
699
+ # Prove the credentials work now, rather than at the user's first real
700
+ # command. `status` on an id that cannot exist returns 404 when the token is
701
+ # good and 401 when it is not, which is exactly the distinction we want.
702
+ #
703
+ # The id has letters in it on purpose: twelve digits in a row is what an AWS
704
+ # account number looks like, and this repository's CI refuses those on
705
+ # sight (`.github/workflows/ci.yml`).
706
+ try:
707
+ Client(args.server, token).status("job-ffffffffffff")
708
+ except ServerError as exc:
709
+ message = str(exc)
710
+ if "no such job" in message:
711
+ print("token accepted")
712
+ else:
713
+ print(f"warning: the server did not accept this: {message}", file=sys.stderr)
714
+ return EXIT_OK
715
+
716
+
717
+ def cmd_logout(_: argparse.Namespace) -> int:
718
+ """Delete the stored credentials."""
719
+ print("logged out" if config.forget() else "nothing stored")
720
+ return EXIT_OK
721
+
722
+
723
+ def cmd_explain(_: argparse.Namespace) -> int:
724
+ """Print the server's own description of itself."""
725
+ print(client_from_config().explain(), end="")
726
+ return EXIT_OK
727
+
728
+
729
+ def cmd_schema(args: argparse.Namespace) -> int:
730
+ """Print the JSON Schema of a submit request."""
731
+ print(json.dumps(client_from_config().schema(), indent=2, ensure_ascii=False))
732
+ return EXIT_OK
733
+
734
+
735
+ def cmd_submit(args: argparse.Namespace) -> int:
736
+ """Submit a job and print where its output will go."""
737
+ body = build_submit_body(args)
738
+ result = client_from_config().submit(body)
739
+ if args.json:
740
+ print(json.dumps(result, indent=2, ensure_ascii=False))
741
+ else:
742
+ print(f"submitted {result['job_id']}")
743
+ print(f"results {result['result_path']}")
744
+ print(f"follow hyperun logs {result['job_id']} --follow")
745
+ return EXIT_OK
746
+
747
+
748
+ def cmd_estimate(args: argparse.Namespace) -> int:
749
+ """Print how long a job will take, what it will cost, and which GPU it wants."""
750
+ body = build_submit_body(args)
751
+ result = client_from_config().estimate(body)
752
+ if args.json:
753
+ print(json.dumps(result, indent=2, ensure_ascii=False))
754
+ return EXIT_OK
755
+
756
+ hours, cost, gpu = result["hours"], result["cost_usd"], result["gpu"]
757
+ rate = result.get("rate") or {}
758
+ if result.get("steps"):
759
+ print(f" steps {result['steps']:,}")
760
+ if hours.get("low") is not None:
761
+ print(f" time {hours['low']} - {hours['high']} h [{hours['confidence']}]")
762
+ else:
763
+ # `unknown` is a real answer, and printing a blank instead of saying so
764
+ # is how a user ends up assuming zero.
765
+ print(f" time unknown [{hours['confidence']}]")
766
+ # THE RATE IS PRINTED WHETHER OR NOT THE HOURS ARE KNOWN. Twelve of the
767
+ # fourteen choosable cards have no throughput measurement, so their `time`
768
+ # is `unknown` -- and this block used to skip the money entirely for them,
769
+ # which is the same blank-instead-of-saying-so hazard the `time` line above
770
+ # was already fixed for. The rate needs no measurement of ours.
771
+ if rate.get("usd_per_hour_low") is not None:
772
+ span = ("" if rate["usd_per_hour_high"] == rate["usd_per_hour_low"]
773
+ else f" - ${rate['usd_per_hour_high']}")
774
+ machines = f" ({rate['machines']} machine(s))" if rate.get("machines", 1) > 1 else ""
775
+ print(f" rate ${rate['usd_per_hour_low']}{span} /h"
776
+ f" [{rate.get('vendor') or 'unpriced'}]{machines}")
777
+ if cost.get("low") is not None:
778
+ print(f" cost ${cost['low']} - ${cost['high']}")
779
+ elif rate.get("usd_per_hour_low") is not None:
780
+ print(" cost unknown -- the rate above is known, the hours are not")
781
+ if rate.get("basis"):
782
+ print(f" price basis {rate['basis']}")
783
+ print(f" basis {result['basis']}")
784
+ print(f" GPU {gpu.get('recommended') or 'none'}"
785
+ f" ({gpu.get('recommended_vram_gb')} GB), logits peak {gpu['peak_logits_gib']} GiB")
786
+ print(f" {gpu['reason']}")
787
+ print(f" capacity type {result['capacity_type']} "
788
+ f"<- pass --capacity-type {result['capacity_type']} when you submit")
789
+ print(f" {result['capacity_reason']}")
790
+ for warning in result.get("warnings", []):
791
+ print(f" note {warning}")
792
+ return EXIT_OK
793
+
794
+
795
+ def cmd_validate(args: argparse.Namespace) -> int:
796
+ """Print what is wrong with a job, without submitting it.
797
+
798
+ Returns:
799
+ 0 when nothing is an error, 1 when something is. A script can gate a
800
+ submit on this.
801
+ """
802
+ body = build_submit_body(args)
803
+ result = client_from_config().validate(body)
804
+ if args.json:
805
+ print(json.dumps(result, indent=2, ensure_ascii=False))
806
+ return EXIT_OK if result["ok"] else EXIT_SERVER
807
+
808
+ marker = {"error": "[error] ", "warning": "[warning]", "info": "[info] "}
809
+ for finding in result.get("findings", []):
810
+ print(f"{marker.get(finding['level'], '[?] ')} {finding['code']}")
811
+ print(f" {finding['message']}")
812
+ if finding.get("fix"):
813
+ print(f" fix: {finding['fix']}")
814
+ print()
815
+ for line in result.get("not_checked", []):
816
+ print(f"[not checked] {line}")
817
+ print()
818
+ print("Nothing blocking." if result["ok"] else "Something is blocking this job.")
819
+ return EXIT_OK if result["ok"] else EXIT_SERVER
820
+
821
+
822
+ def cmd_shell(args: argparse.Namespace) -> int:
823
+ """Run commands inside a running job's workload container.
824
+
825
+ One line, one HTTPS round trip: the server relays it through the driver
826
+ pod onto the rented machine and returns the exit code like ssh. With a
827
+ trailing `-- command` it runs once and exits with that code; without one
828
+ it prompts, which FEELS like a slow shell and is honestly a request loop.
829
+ """
830
+ client = client_from_config()
831
+
832
+ def run_once(line: str) -> int:
833
+ answer = client.exec_in_job(args.job, line, slot=args.slot)
834
+ output = answer.get("output") or ""
835
+ if output:
836
+ print(output, end="" if output.endswith("\n") else "\n")
837
+ if answer.get("note"):
838
+ print(f"({answer['note']})", file=sys.stderr)
839
+ code = answer.get("exit_code")
840
+ return 1 if code is None else int(code)
841
+
842
+ # ★ A FLAG AFTER THE JOB ID IS SWALLOWED, so refuse it rather than run the
843
+ # wrong thing quietly. argparse.REMAINDER starts consuming at the token
844
+ # right after the `job` positional, so
845
+ #
846
+ # hyperun shell job-x --slot 2 -- hostname
847
+ #
848
+ # parses as slot=0 with the command line "--slot 2 -- hostname": the user
849
+ # asked for pod 2 and would have got pod 0, with no message. Found
850
+ # 2026-09-10 while writing the first test that ever reached this function.
851
+ # The separator is the tell: argparse consumes a `--` that immediately
852
+ # follows the job id, so a `--` still sitting in the remainder -- or a
853
+ # remainder that begins with a dash -- means options were written late.
854
+ raw = list(args.command or [])
855
+ if raw and (raw[0].startswith("-") or "--" in raw):
856
+ # THE SHAPE, NOT A REBUILT LINE. A first draft printed a corrected
857
+ # command and got it wrong: `--slot` had already been swallowed, so it
858
+ # echoed the default slot and left the 2 in the workload's line
859
+ # ("--slot 0 job-x -- 2 hostname"). Suggesting a wrong command is worse
860
+ # than suggesting none, so this states the rule and the shape and lets
861
+ # the reader place their own flags.
862
+ print("error: options go BEFORE the job id. Everything after the job id "
863
+ "is sent to the workload as-is, so a flag written there became "
864
+ "part of the command line instead.", file=sys.stderr)
865
+ print(" shape: hyperun shell [--slot N] <job> -- <command>",
866
+ file=sys.stderr)
867
+ print(f" you wrote: shell {args.job} " + " ".join(raw),
868
+ file=sys.stderr)
869
+ return EXIT_USAGE
870
+ # argparse.REMAINDER keeps the "--" separator itself; drop it.
871
+ words = [w for w in raw if w != "--"]
872
+ if words:
873
+ return run_once(" ".join(words))
874
+
875
+ print(
876
+ f"one command per line, ~25s each; 'exit' or Ctrl-D leaves. Not a TTY.",
877
+ file=sys.stderr,
878
+ )
879
+ while True:
880
+ try:
881
+ line = input(f"{args.job}$ ")
882
+ except (EOFError, KeyboardInterrupt):
883
+ print()
884
+ return EXIT_OK
885
+ line = line.strip()
886
+ if not line:
887
+ continue
888
+ if line in ("exit", "quit"):
889
+ return EXIT_OK
890
+ try:
891
+ code = run_once(line)
892
+ if code:
893
+ print(f"(exit {code})", file=sys.stderr)
894
+ except ServerError as exc:
895
+ # One failed command must not end the session: say what the server
896
+ # said and keep the prompt.
897
+ print(f"error: {exc}", file=sys.stderr)
898
+
899
+
900
+ def cmd_secrets(args: argparse.Namespace) -> int:
901
+ """Print the secret names this caller can use, saying which are their own."""
902
+ answer = client_from_config().secrets()
903
+ names = answer.get("names") or []
904
+ own = set(answer.get("own") or [])
905
+ for name in names:
906
+ # Marking them matters: only the second kind can be changed from here,
907
+ # and only the second kind is invisible to the rest of the lab.
908
+ print(f"{name} (yours)" if name in own else f"{name} (deployment)")
909
+ if answer.get("note"):
910
+ print(f"\n{answer['note']}")
911
+ return EXIT_OK
912
+
913
+
914
+ def cmd_secret_set(args: argparse.Namespace) -> int:
915
+ """Store one value under one name, for this caller's own namespace.
916
+
917
+ THE VALUE IS READ FROM A FILE OR STDIN AND NEVER FROM argv. A secret on a
918
+ command line is in the shell's history file, in `ps` output for every other
919
+ user on the machine, and in any terminal recording of the session -- three
920
+ copies nobody meant to make, in places nobody thinks to clear.
921
+
922
+ Returns:
923
+ EXIT_OK, or EXIT_USAGE when the file could not be read or is empty.
924
+ """
925
+ source = args.from_file or "-"
926
+ if source == "-":
927
+ if sys.stdin.isatty():
928
+ # A bare `hyperun secret-set NAME` on a terminal would sit there
929
+ # looking hung. Say what it is waiting for.
930
+ print(
931
+ f"reading the value for {args.name} from stdin. Paste it and press "
932
+ f"Ctrl-D, or re-run with --from-file PATH.",
933
+ file=sys.stderr,
934
+ )
935
+ value = sys.stdin.read()
936
+ else:
937
+ try:
938
+ with open(source, encoding="utf-8") as handle:
939
+ value = handle.read()
940
+ except OSError as exc:
941
+ print(f"cannot read {source}: {exc}", file=sys.stderr)
942
+ return EXIT_USAGE
943
+ # A trailing newline is what `echo` and every editor add, and it is not part
944
+ # of a token -- a newline inside an Authorization header is a 400 from the
945
+ # far end, an hour into a job. Only the trailing whitespace goes; anything
946
+ # else could be part of a private key or a JSON blob.
947
+ value = value.rstrip("\r\n")
948
+ if not value:
949
+ print(
950
+ f"nothing to store: the value for {args.name} is empty.", file=sys.stderr
951
+ )
952
+ return EXIT_USAGE
953
+
954
+ answer = client_from_config().put_secret(args.name, value,
955
+ expires_at=args.expires_at)
956
+ where = answer.get("namespace", "your namespace")
957
+ what = "stored" if answer.get("created") else "replaced"
958
+ print(f"{what} {answer.get('name', args.name)} in {where}.")
959
+ if answer.get("expires_at"):
960
+ print(f"it stops working at {answer['expires_at']}; after that a job asking "
961
+ f"for it is refused by `hyperun validate`.")
962
+ print(f"a job can now ask for it with `--secret {args.name}`.")
963
+ return EXIT_OK
964
+
965
+
966
+ def cmd_secret_rm(args: argparse.Namespace) -> int:
967
+ """Forget one registered name."""
968
+ client_from_config().delete_secret(args.name)
969
+ print(f"{args.name} is no longer registered. Jobs asking for it are refused.")
970
+ return EXIT_OK
971
+
972
+
973
+ def cmd_status(args: argparse.Namespace) -> int:
974
+ """Print one job's state."""
975
+ view = client_from_config().status(args.job_id)
976
+ if args.json:
977
+ print(json.dumps(view, indent=2, ensure_ascii=False))
978
+ return EXIT_OK
979
+
980
+ # A job the controller has not looked at yet has an empty phase. Saying
981
+ # "accepted" is truer than printing a blank.
982
+ print(f"{view['job_id']} {view.get('name', '')}")
983
+ print(f" phase {view.get('phase') or 'accepted, not yet started'}")
984
+ if view.get("gpu"):
985
+ vendor = f" ({view['vendor']})" if view.get("vendor") else ""
986
+ print(f" running on {view['gpu']}{vendor}")
987
+ if view.get("recovery_count"):
988
+ # Not a failure. Rented capacity is taken back, and the job is restarted.
989
+ print(f" restarts {view['recovery_count']} (the machine was reclaimed)")
990
+ if view.get("message"):
991
+ print(f" message {view['message']}")
992
+ if view.get("result_path"):
993
+ print(f" results {view['result_path']}")
994
+ return EXIT_OK
995
+
996
+
997
+ def cmd_delete(args: argparse.Namespace) -> int:
998
+ """Delete a job. It stops, and it disappears.
999
+
1000
+ DDPSRUN-CANCEL. A job can sit in Pending forever with nothing to do about
1001
+ it: on 2026-09-02 one asked for an L40S on spot, which RunPod refuses before
1002
+ reading the catalogue and which no AWS row matched, and the controller
1003
+ retried that same failure every eleven minutes. Until this existed the only
1004
+ way to stop it was kubectl, which is the thing this tool exists to remove.
1005
+
1006
+ Asks first unless --yes is given. The delete is not undoable and a job id is
1007
+ twelve hex characters, which is easy enough to mistype.
1008
+
1009
+ THE PROMPT SAYS `delete`, NOT `cancel`, since 2026-09-10. The old wording
1010
+ let a reader expect the job to survive in some cancelled state; what
1011
+ actually happens is that the object is removed and the row is gone. What
1012
+ survives is the result path in S3, which the prompt now says out loud so
1013
+ nobody deletes a job expecting to lose its output, or keeps one expecting
1014
+ to protect it.
1015
+
1016
+ Returns:
1017
+ 0 when the job is gone, 1 when the server refused, 2 when the user said
1018
+ no at the prompt.
1019
+ """
1020
+ if not args.yes:
1021
+ answer = input(
1022
+ f"delete {args.job_id}? it stops and disappears from the list; "
1023
+ f"files already in its result path stay [y/N] ").strip().lower()
1024
+ if answer not in ("y", "yes"):
1025
+ print("left alone")
1026
+ return EXIT_USAGE
1027
+
1028
+ try:
1029
+ client_from_config().cancel(args.job_id)
1030
+ except ServerError as exc:
1031
+ print(str(exc), file=sys.stderr)
1032
+ return EXIT_SERVER
1033
+ print(f"cancelled {args.job_id}")
1034
+ return EXIT_OK
1035
+
1036
+
1037
+ def bar(percent: float, width: int = 20) -> str:
1038
+ """Draw a progress bar.
1039
+
1040
+ Args:
1041
+ percent: 0 to 100.
1042
+ width: how many characters wide.
1043
+
1044
+ Returns:
1045
+ A string of filled and empty blocks.
1046
+
1047
+ Example:
1048
+ >>> bar(50, width=10)
1049
+ '#####-----'
1050
+ """
1051
+ filled = int(round(width * max(0.0, min(100.0, percent)) / 100))
1052
+ return "#" * filled + "-" * (width - filled)
1053
+
1054
+
1055
+ def cmd_watch(args: argparse.Namespace) -> int:
1056
+ """Print a job's GPU usage and how far the training has got."""
1057
+ result = client_from_config().metrics(args.job_id, args.window)
1058
+ if args.json:
1059
+ print(json.dumps(result, indent=2, ensure_ascii=False))
1060
+ return EXIT_OK
1061
+
1062
+ progress = result.get("progress")
1063
+ if progress:
1064
+ print(f" training {bar(progress['percent'])} "
1065
+ f"{progress['step']:,} / {progress['total_steps']:,} steps "
1066
+ f"({progress['percent']}%)")
1067
+ print(f" {progress['seconds_per_step']} s/step, "
1068
+ f"{progress['elapsed']} elapsed, {progress['remaining']} remaining")
1069
+ # `steady` False means too few steps have run for this to be worth
1070
+ # quoting, so it is labelled rather than printed as a fact.
1071
+ settled = "" if progress["steady"] else " (rate has not settled yet)"
1072
+ print(f" projected {progress['projected_total_hours']} h{settled}")
1073
+
1074
+ gpu = result.get("latest_gpu")
1075
+ if gpu:
1076
+ print(f" GPU util {bar(gpu['utilization_percent'])} "
1077
+ f"{gpu['utilization_percent']}%")
1078
+ print(f" GPU memory {bar(gpu['memory_percent'])} "
1079
+ f"{gpu['memory_used_mib']:,} / {gpu['memory_total_mib']:,} MiB "
1080
+ f"({gpu['memory_percent']}%)")
1081
+ print(f" temp, power {gpu['temperature_c']} C, {gpu['power_w']} W")
1082
+ print(f" samples {len(result.get('gpu_series', []))}, "
1083
+ f"last {result['window_seconds']}s")
1084
+
1085
+ if result.get("note"):
1086
+ print(f" note {result['note']}")
1087
+ return EXIT_OK
1088
+
1089
+
1090
+ def cmd_stats(args: argparse.Namespace) -> int:
1091
+ """Print this caller's team figures."""
1092
+ result = client_from_config().stats()
1093
+ if args.json:
1094
+ print(json.dumps(result, indent=2, ensure_ascii=False))
1095
+ return EXIT_OK
1096
+
1097
+ print(f"team {result['team'] or '(none)'}")
1098
+ if result.get("members"):
1099
+ print(f" {'member':<14}{'jobs':>6}{'ok':>5}{'failed':>8}{'running':>9}"
1100
+ f"{'GPU h':>9}{'spend':>10}")
1101
+ for m in result["members"]:
1102
+ print(f" {m['user']:<14}{m['jobs']:>6}{m['succeeded']:>5}{m['failed']:>8}"
1103
+ f"{m['running']:>9}{m['gpu_hours']:>9}{('$' + str(m['cost_usd'])):>10}")
1104
+ print(f" {'total':<14}{result['jobs']:>6}{'':>5}{'':>8}{'':>9}"
1105
+ f"{result['gpu_hours']:>9}{('$' + str(result['cost_usd'])):>10}")
1106
+ if result.get("note"):
1107
+ print(f" note {result['note']}")
1108
+ return EXIT_OK
1109
+
1110
+
1111
+ def cmd_logs(args: argparse.Namespace) -> int:
1112
+ """Print a job's output, once or repeatedly.
1113
+
1114
+ WHY THIS POLLS INSTEAD OF STREAMING. The server runs as a Lambda function
1115
+ and one execution is capped at 15 minutes, while a training run is thirty
1116
+ hours. So a window is read, printed, and asked for again.
1117
+
1118
+ The only state kept is `since`, the timestamp of the last line printed.
1119
+ The server remembers nothing, which is what lets a fresh execution answer
1120
+ every request.
1121
+ """
1122
+ client = client_from_config()
1123
+ # A window several times the interval, so one slow round trip does not lose
1124
+ # lines. Capped at the server's own limit.
1125
+ window = min(3600, max(5, int(args.interval * 5)))
1126
+ since: str | None = None
1127
+
1128
+ while True:
1129
+ result = client.log_window(args.job_id, since=since, window_seconds=window)
1130
+ for line in result["lines"]:
1131
+ # The timestamp is bookkeeping for the next request, not something a
1132
+ # user asked to read, so it is dropped on the way to the terminal.
1133
+ print(line.split(" ", 1)[1] if " " in line else line)
1134
+ if result.get("last_timestamp"):
1135
+ since = result["last_timestamp"]
1136
+ if not args.follow:
1137
+ return EXIT_OK
1138
+ time.sleep(args.interval)
1139
+
1140
+
1141
+ COMMANDS = {
1142
+ "login": cmd_login,
1143
+ "delete": cmd_delete,
1144
+ # The old spelling, kept as an alias -- see the comment on the parser.
1145
+ "cancel": cmd_delete,
1146
+ "logout": cmd_logout,
1147
+ "explain": cmd_explain,
1148
+ "schema": cmd_schema,
1149
+ "estimate": cmd_estimate,
1150
+ "validate": cmd_validate,
1151
+ "submit": cmd_submit,
1152
+ "secrets": cmd_secrets,
1153
+ "secret-set": cmd_secret_set,
1154
+ "secret-rm": cmd_secret_rm,
1155
+ "shell": cmd_shell,
1156
+ "status": cmd_status,
1157
+ "watch": cmd_watch,
1158
+ "stats": cmd_stats,
1159
+ "logs": cmd_logs,
1160
+ }
1161
+
1162
+
1163
+ def main(argv: list[str] | None = None) -> int:
1164
+ """The entry point `pip install hyperun` puts on the PATH.
1165
+
1166
+ Args:
1167
+ argv: arguments without the program name. Defaults to `sys.argv[1:]`.
1168
+
1169
+ Returns:
1170
+ The exit code.
1171
+ """
1172
+ parser = build_parser()
1173
+ args = parser.parse_args(argv)
1174
+ if not args.subcommand:
1175
+ parser.print_help()
1176
+ return EXIT_USAGE
1177
+
1178
+ try:
1179
+ return COMMANDS[args.subcommand](args)
1180
+ except SystemExit as exc:
1181
+ # `build_submit_body` and `client_from_config` raise SystemExit with a
1182
+ # one-line message already printed, because a traceback for "you forgot
1183
+ # --image" helps nobody. Catching it here keeps `main`'s contract — it
1184
+ # RETURNS an exit code — true for every path, which is what lets a test
1185
+ # call it directly and what lets another program import it.
1186
+ code = exc.code
1187
+ if isinstance(code, str):
1188
+ print(code, file=sys.stderr)
1189
+ return EXIT_USAGE
1190
+ return EXIT_USAGE if code is None else int(code)
1191
+ except ServerError as exc:
1192
+ print(f"error: {exc}", file=sys.stderr)
1193
+ return EXIT_SERVER
1194
+ except KeyboardInterrupt:
1195
+ # Ctrl-C during `logs --follow` is how a user stops watching. It is not
1196
+ # a failure and it does not touch the job, which keeps running.
1197
+ print()
1198
+ return EXIT_OK
1199
+
1200
+
1201
+ if __name__ == "__main__":
1202
+ sys.exit(main())