hyperun 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hyperun-0.2.0 → hyperun-0.2.1}/PKG-INFO +1 -1
- {hyperun-0.2.0 → hyperun-0.2.1}/hyperun/cli.py +51 -10
- {hyperun-0.2.0 → hyperun-0.2.1}/hyperun/client.py +11 -1
- {hyperun-0.2.0 → hyperun-0.2.1}/hyperun.egg-info/PKG-INFO +1 -1
- {hyperun-0.2.0 → hyperun-0.2.1}/pyproject.toml +1 -1
- {hyperun-0.2.0 → hyperun-0.2.1}/tests/test_cli.py +96 -3
- {hyperun-0.2.0 → hyperun-0.2.1}/tests/test_client.py +15 -1
- {hyperun-0.2.0 → hyperun-0.2.1}/LICENSE +0 -0
- {hyperun-0.2.0 → hyperun-0.2.1}/README.md +0 -0
- {hyperun-0.2.0 → hyperun-0.2.1}/hyperun/__init__.py +0 -0
- {hyperun-0.2.0 → hyperun-0.2.1}/hyperun/browser_login.py +0 -0
- {hyperun-0.2.0 → hyperun-0.2.1}/hyperun/config.py +0 -0
- {hyperun-0.2.0 → hyperun-0.2.1}/hyperun.egg-info/SOURCES.txt +0 -0
- {hyperun-0.2.0 → hyperun-0.2.1}/hyperun.egg-info/dependency_links.txt +0 -0
- {hyperun-0.2.0 → hyperun-0.2.1}/hyperun.egg-info/entry_points.txt +0 -0
- {hyperun-0.2.0 → hyperun-0.2.1}/hyperun.egg-info/requires.txt +0 -0
- {hyperun-0.2.0 → hyperun-0.2.1}/hyperun.egg-info/top_level.txt +0 -0
- {hyperun-0.2.0 → hyperun-0.2.1}/setup.cfg +0 -0
- {hyperun-0.2.0 → hyperun-0.2.1}/tests/test_browser_login.py +0 -0
- {hyperun-0.2.0 → hyperun-0.2.1}/tests/test_config.py +0 -0
|
@@ -365,9 +365,10 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
365
365
|
"through the job's driver pod into the workload container on the "
|
|
366
366
|
"rented machine and brings the exit code back like ssh. It is NOT a "
|
|
367
367
|
"TTY — no vim, no top, about 25 seconds per command — because the "
|
|
368
|
-
"server is a Lambda and cannot hold a terminal open. AWS and
|
|
369
|
-
"machine rentals only: a RunPod job is a rented container
|
|
370
|
-
"machine behind it,
|
|
368
|
+
"server is a Lambda and cannot hold a terminal open. AWS, GCP and "
|
|
369
|
+
"Shadeform machine rentals only: a RunPod job is a rented container "
|
|
370
|
+
"with no machine behind it, so there is no cluster to attach to and "
|
|
371
|
+
"the relay refuses it. ★ PUT OPTIONS BEFORE THE "
|
|
371
372
|
"JOB ID -- everything after it is sent to the workload as-is, so "
|
|
372
373
|
"`shell job-x --slot 2` asks pod 0 and passes `--slot 2` to the shell. "
|
|
373
374
|
"That is refused rather than obeyed.",
|
|
@@ -826,11 +827,26 @@ def cmd_shell(args: argparse.Namespace) -> int:
|
|
|
826
827
|
pod onto the rented machine and returns the exit code like ssh. With a
|
|
827
828
|
trailing `-- command` it runs once and exits with that code; without one
|
|
828
829
|
it prompts, which FEELS like a slow shell and is honestly a request loop.
|
|
830
|
+
|
|
831
|
+
★ THE PROMPT USES A SESSION AND THE ONE-SHOT FORM DOES NOT, since
|
|
832
|
+
2026-09-10. The prompt types into a shell that stays running in the driver
|
|
833
|
+
pod, so `cd`, an exported variable and an activated venv survive from one
|
|
834
|
+
line to the next -- which is what a person at a prompt assumes and what
|
|
835
|
+
every earlier version quietly did not do. A `-- command` invocation is a
|
|
836
|
+
script's shape and gets the stateless form: it wants no history and should
|
|
837
|
+
not leave a session behind for the idle timer to close.
|
|
829
838
|
"""
|
|
830
839
|
client = client_from_config()
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
840
|
+
# The output sequence the session has already shown us. It is here rather
|
|
841
|
+
# than inside run_once because it has to survive between lines, which is the
|
|
842
|
+
# whole point.
|
|
843
|
+
state = {"seq": 0}
|
|
844
|
+
|
|
845
|
+
def run_once(line: str, session: bool = False) -> int:
|
|
846
|
+
answer = client.exec_in_job(args.job, line, slot=args.slot,
|
|
847
|
+
session=session, seq=state["seq"])
|
|
848
|
+
if session:
|
|
849
|
+
state["seq"] = int(answer.get("seq", state["seq"]))
|
|
834
850
|
output = answer.get("output") or ""
|
|
835
851
|
if output:
|
|
836
852
|
print(output, end="" if output.endswith("\n") else "\n")
|
|
@@ -873,7 +889,8 @@ def cmd_shell(args: argparse.Namespace) -> int:
|
|
|
873
889
|
return run_once(" ".join(words))
|
|
874
890
|
|
|
875
891
|
print(
|
|
876
|
-
f"one command per line, ~25s each; 'exit' or Ctrl-D leaves. Not a TTY
|
|
892
|
+
f"one command per line, ~25s each; 'exit' or Ctrl-D leaves. Not a TTY, "
|
|
893
|
+
f"but `cd` and exported variables DO survive between lines.",
|
|
877
894
|
file=sys.stderr,
|
|
878
895
|
)
|
|
879
896
|
while True:
|
|
@@ -888,10 +905,29 @@ def cmd_shell(args: argparse.Namespace) -> int:
|
|
|
888
905
|
if line in ("exit", "quit"):
|
|
889
906
|
return EXIT_OK
|
|
890
907
|
try:
|
|
891
|
-
code = run_once(line)
|
|
908
|
+
code = run_once(line, session=True)
|
|
892
909
|
if code:
|
|
893
910
|
print(f"(exit {code})", file=sys.stderr)
|
|
894
911
|
except ServerError as exc:
|
|
912
|
+
# A SESSION THAT IS GONE IS THE ONE ERROR WORTH ACTING ON. The
|
|
913
|
+
# server answers 409 when the driver pod has no session for this
|
|
914
|
+
# slot -- it timed out after ten minutes unread, or the workload
|
|
915
|
+
# restarted. Reopening is right there and only there: everywhere
|
|
916
|
+
# else a new shell would silently lose the `cd` the person is
|
|
917
|
+
# relying on, which is the failure this whole feature exists to
|
|
918
|
+
# end. The sequence resets with it, because the new shell's
|
|
919
|
+
# output starts from nothing.
|
|
920
|
+
if "open one first" in str(exc) or "no session" in str(exc):
|
|
921
|
+
state["seq"] = 0
|
|
922
|
+
print("(the session had closed; reopening — `cd` and variables "
|
|
923
|
+
"from before are gone)", file=sys.stderr)
|
|
924
|
+
try:
|
|
925
|
+
code = run_once(line, session=True)
|
|
926
|
+
if code:
|
|
927
|
+
print(f"(exit {code})", file=sys.stderr)
|
|
928
|
+
continue
|
|
929
|
+
except ServerError as retry_exc:
|
|
930
|
+
exc = retry_exc
|
|
895
931
|
# One failed command must not end the session: say what the server
|
|
896
932
|
# said and keep the prompt.
|
|
897
933
|
print(f"error: {exc}", file=sys.stderr)
|
|
@@ -1079,8 +1115,13 @@ def cmd_watch(args: argparse.Namespace) -> int:
|
|
|
1079
1115
|
f"{gpu['memory_used_mib']:,} / {gpu['memory_total_mib']:,} MiB "
|
|
1080
1116
|
f"({gpu['memory_percent']}%)")
|
|
1081
1117
|
print(f" temp, power {gpu['temperature_c']} C, {gpu['power_w']} W")
|
|
1082
|
-
|
|
1083
|
-
|
|
1118
|
+
# `sample_count`, not len(gpu_series): the server thins the series to at
|
|
1119
|
+
# most 400 points so a chart can draw it, and job-66b46719b854 printed
|
|
1120
|
+
# 785 readings per card while this line said 393. The screen had the same
|
|
1121
|
+
# defect and was fixed on 2026-09-12. A server too old to send the count
|
|
1122
|
+
# falls back to the thinned length, which is the only number it has.
|
|
1123
|
+
taken = result.get("sample_count") or len(result.get("gpu_series", []))
|
|
1124
|
+
print(f" samples {taken}, last {result['window_seconds']}s")
|
|
1084
1125
|
|
|
1085
1126
|
if result.get("note"):
|
|
1086
1127
|
print(f" note {result['note']}")
|
|
@@ -193,7 +193,8 @@ class Client:
|
|
|
193
193
|
return self._call("GET", "/v1/stats").json()
|
|
194
194
|
|
|
195
195
|
def exec_in_job(
|
|
196
|
-
self, job: str, command: str, slot: int = 0, timeout_seconds: int = 20
|
|
196
|
+
self, job: str, command: str, slot: int = 0, timeout_seconds: int = 20,
|
|
197
|
+
session: bool = False, seq: int = 0,
|
|
197
198
|
) -> dict[str, Any]:
|
|
198
199
|
"""Run one shell line inside a running job's workload container.
|
|
199
200
|
|
|
@@ -206,6 +207,13 @@ class Client:
|
|
|
206
207
|
command: one shell line, run as `sh -lc <command>`.
|
|
207
208
|
slot: which pod of a parallel job.
|
|
208
209
|
timeout_seconds: server-side wait, capped at 25 by the server.
|
|
210
|
+
session: type the line into a shell ALREADY RUNNING in the driver
|
|
211
|
+
pod, so `cd` and exported variables survive to the next call.
|
|
212
|
+
False starts a fresh `sh -lc`, which is what a script wants.
|
|
213
|
+
seq: with `session`, the output sequence this caller last saw. The
|
|
214
|
+
reply carries the next one. It exists because the caller is a
|
|
215
|
+
different process on the server's side every time and cannot
|
|
216
|
+
hold a position in the output.
|
|
209
217
|
"""
|
|
210
218
|
return self._call(
|
|
211
219
|
"POST",
|
|
@@ -214,6 +222,8 @@ class Client:
|
|
|
214
222
|
"command": command,
|
|
215
223
|
"slot": slot,
|
|
216
224
|
"timeout_seconds": timeout_seconds,
|
|
225
|
+
"session": session,
|
|
226
|
+
"seq": seq,
|
|
217
227
|
},
|
|
218
228
|
# The server may hold the request for timeout_seconds before
|
|
219
229
|
# answering; the read timeout has to outlive that on purpose.
|
|
@@ -34,7 +34,7 @@ build-backend = "setuptools.build_meta"
|
|
|
34
34
|
|
|
35
35
|
[project]
|
|
36
36
|
name = "hyperun"
|
|
37
|
-
version = "0.2.
|
|
37
|
+
version = "0.2.1"
|
|
38
38
|
description = "Submit a GPU job and get results back. No kubectl, no cloud account."
|
|
39
39
|
requires-python = ">=3.9"
|
|
40
40
|
dependencies = ["requests>=2.31", "PyYAML>=6.0"]
|
|
@@ -100,8 +100,8 @@ class FakeClient:
|
|
|
100
100
|
# DDPSRUN-SHELL. Absent until 2026-09-10, which is exactly why the dispatch
|
|
101
101
|
# bug below survived every release: with no double for this call, no test
|
|
102
102
|
# could reach cmd_shell at all.
|
|
103
|
-
def exec_in_job(self, job, line, slot=0):
|
|
104
|
-
self.execed.append((job, line, slot))
|
|
103
|
+
def exec_in_job(self, job, line, slot=0, session=False, seq=0):
|
|
104
|
+
self.execed.append((job, line, slot, session, seq))
|
|
105
105
|
return self.exec_result
|
|
106
106
|
|
|
107
107
|
|
|
@@ -475,6 +475,39 @@ def test_watch_prints_progress_and_gpu(fake, capsys):
|
|
|
475
475
|
assert "samples 120" in printed
|
|
476
476
|
|
|
477
477
|
|
|
478
|
+
def test_samples_counts_the_readings_taken_not_the_chart_points(fake, capsys):
|
|
479
|
+
# ★ THE SAME DEFECT THE SCREEN HAD. The server thins the series to at most
|
|
480
|
+
# 400 points so a chart can draw it; job-66b46719b854 printed 785 readings
|
|
481
|
+
# per card and both surfaces counted the 393 that survived. The mean and the
|
|
482
|
+
# peak are computed over all 785, so the count beside them has to be 785.
|
|
483
|
+
fake.metrics_result = {
|
|
484
|
+
"latest_gpu": {"utilization_percent": 94, "memory_used_mib": 38200,
|
|
485
|
+
"memory_total_mib": 45440, "memory_percent": 84.1,
|
|
486
|
+
"temperature_c": 71, "power_w": 298.5},
|
|
487
|
+
"gpu_series": [{}] * 393,
|
|
488
|
+
"sample_count": 785,
|
|
489
|
+
"window_seconds": 604800, "note": "",
|
|
490
|
+
}
|
|
491
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
492
|
+
printed = capsys.readouterr().out
|
|
493
|
+
assert "samples 785" in printed
|
|
494
|
+
assert "samples 393" not in printed
|
|
495
|
+
|
|
496
|
+
|
|
497
|
+
def test_a_server_too_old_to_send_the_count_still_prints_one(fake, capsys):
|
|
498
|
+
# The field arrived on 2026-09-12. Against a server that predates it the
|
|
499
|
+
# thinned length is the only number there is, and a blank is worse.
|
|
500
|
+
fake.metrics_result = {
|
|
501
|
+
"latest_gpu": {"utilization_percent": 10, "memory_used_mib": 1,
|
|
502
|
+
"memory_total_mib": 2, "memory_percent": 50.0,
|
|
503
|
+
"temperature_c": 40, "power_w": 50.0},
|
|
504
|
+
"gpu_series": [{}] * 12,
|
|
505
|
+
"window_seconds": 3600, "note": "",
|
|
506
|
+
}
|
|
507
|
+
assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
|
|
508
|
+
assert "samples 12" in capsys.readouterr().out
|
|
509
|
+
|
|
510
|
+
|
|
478
511
|
def test_an_unsettled_projection_is_labelled_rather_than_stated(fake, capsys):
|
|
479
512
|
fake.metrics_result = {
|
|
480
513
|
"progress": {"step": 5, "total_steps": 556, "percent": 0.9,
|
|
@@ -820,7 +853,8 @@ def test_the_subcommand_name_survives_the_shell_positional():
|
|
|
820
853
|
|
|
821
854
|
def test_shell_with_a_command_runs_it_once_and_returns_its_exit_code(fake, capsys):
|
|
822
855
|
assert run(["shell", "job-a8acdef80a07", "--", "date"]) == 0
|
|
823
|
-
assert fake.execed == [("job-a8acdef80a07", "date", 0)]
|
|
856
|
+
assert fake.execed == [("job-a8acdef80a07", "date", 0, False, 0)], (
|
|
857
|
+
"`-- command` 형태는 stateless 다 — script 는 남길 session 이 없다")
|
|
824
858
|
assert "Sep 10" in capsys.readouterr().out
|
|
825
859
|
|
|
826
860
|
|
|
@@ -858,3 +892,62 @@ def test_every_other_subcommand_still_dispatches(fake):
|
|
|
858
892
|
"""dest 를 바꾼 것이 나머지를 깨지 않았다는 확인."""
|
|
859
893
|
assert run(["status", "job-a8acdef80a07"]) == 0
|
|
860
894
|
assert run(["secrets"]) == 0
|
|
895
|
+
|
|
896
|
+
|
|
897
|
+
def test_the_prompt_uses_a_session_and_carries_the_sequence(fake, monkeypatch, capsys):
|
|
898
|
+
"""★ WHY THE PROMPT DIFFERS FROM `-- command`. A person at a prompt assumes
|
|
899
|
+
`cd` sticks; a script does not want a session left behind for the idle timer.
|
|
900
|
+
So the prompt sends session=True and threads the sequence through, and the
|
|
901
|
+
one-shot form stays stateless."""
|
|
902
|
+
typed = iter(["cd /workspace", "pwd", "exit"])
|
|
903
|
+
monkeypatch.setattr("builtins.input", lambda _prompt: next(typed))
|
|
904
|
+
fake.exec_result = {"output": "/workspace\n", "exit_code": 0, "seq": 11, "lost": False}
|
|
905
|
+
|
|
906
|
+
assert run(["shell", "job-a8acdef80a07"]) == 0
|
|
907
|
+
assert [(line, sess) for _job, line, _slot, sess, _seq in fake.execed] == [
|
|
908
|
+
("cd /workspace", True), ("pwd", True)]
|
|
909
|
+
# The FIRST line starts at 0 and the second carries what the first returned.
|
|
910
|
+
assert [seq for *_rest, seq in fake.execed] == [0, 11]
|
|
911
|
+
|
|
912
|
+
|
|
913
|
+
def test_a_session_that_closed_is_reopened_once_and_the_person_is_told(fake, monkeypatch,
|
|
914
|
+
capsys):
|
|
915
|
+
"""The one case where starting a new shell is right: the old one is provably
|
|
916
|
+
gone (ten minutes unread, or the workload restarted). Everywhere else a new
|
|
917
|
+
shell silently loses the `cd`, which is the failure this feature exists to end
|
|
918
|
+
-- so the message says what was lost."""
|
|
919
|
+
typed = iter(["pwd", "exit"])
|
|
920
|
+
monkeypatch.setattr("builtins.input", lambda _prompt: next(typed))
|
|
921
|
+
|
|
922
|
+
calls = {"n": 0}
|
|
923
|
+
original = fake.exec_in_job
|
|
924
|
+
|
|
925
|
+
def flaky(job, line, slot=0, session=False, seq=0):
|
|
926
|
+
calls["n"] += 1
|
|
927
|
+
if calls["n"] == 1:
|
|
928
|
+
raise cli.ServerError("no session for this slot; open one first")
|
|
929
|
+
return original(job, line, slot=slot, session=session, seq=seq)
|
|
930
|
+
|
|
931
|
+
fake.exec_in_job = flaky
|
|
932
|
+
assert run(["shell", "job-a8acdef80a07"]) == 0
|
|
933
|
+
assert calls["n"] == 2, "it retried once rather than giving up or looping"
|
|
934
|
+
err = capsys.readouterr().err
|
|
935
|
+
assert "reopening" in err and "are gone" in err, (
|
|
936
|
+
"a silent reopen would let somebody keep typing paths relative to a cd that no "
|
|
937
|
+
"longer applies")
|
|
938
|
+
|
|
939
|
+
|
|
940
|
+
def test_a_failure_that_is_not_a_lost_session_is_not_retried(fake, monkeypatch, capsys):
|
|
941
|
+
typed = iter(["pwd", "exit"])
|
|
942
|
+
monkeypatch.setattr("builtins.input", lambda _prompt: next(typed))
|
|
943
|
+
|
|
944
|
+
calls = {"n": 0}
|
|
945
|
+
|
|
946
|
+
def always_bad(job, line, slot=0, session=False, seq=0):
|
|
947
|
+
calls["n"] += 1
|
|
948
|
+
raise cli.ServerError("the driver pod did not answer in JSON")
|
|
949
|
+
|
|
950
|
+
fake.exec_in_job = always_bad
|
|
951
|
+
assert run(["shell", "job-a8acdef80a07"]) == 0
|
|
952
|
+
assert calls["n"] == 1, "retrying an unrelated failure would double every bad command"
|
|
953
|
+
assert "did not answer in JSON" in capsys.readouterr().err
|
|
@@ -65,7 +65,9 @@ def test_shell_sends_the_command_and_waits_longer_than_the_server():
|
|
|
65
65
|
result = client_with(session).exec_in_job("baseline-c", "nvidia-smi -L", slot=1)
|
|
66
66
|
call = session.calls[0]
|
|
67
67
|
assert call["url"] == "https://run.example/v1/jobs/baseline-c/exec"
|
|
68
|
-
assert call["json"] == {"command": "nvidia-smi -L", "slot": 1, "timeout_seconds": 20
|
|
68
|
+
assert call["json"] == {"command": "nvidia-smi -L", "slot": 1, "timeout_seconds": 20,
|
|
69
|
+
"session": False, "seq": 0}, (
|
|
70
|
+
"PACSRUN-SHELL-SESSION 이후에도 기본은 한 줄짜리 stateless 형태다")
|
|
69
71
|
# The server may hold the request for its whole window, so the client's
|
|
70
72
|
# read timeout must outlive it.
|
|
71
73
|
assert call["timeout"] > 20
|
|
@@ -155,3 +157,15 @@ def test_a_submit_is_given_more_patience_than_a_read():
|
|
|
155
157
|
read_session = FakeSession(FakeResponse(200, {}))
|
|
156
158
|
client_with(read_session).status("job-a8acdef80a07")
|
|
157
159
|
assert submit_session.calls[0]["timeout"] > read_session.calls[0]["timeout"]
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def test_shell_can_ask_for_the_persistent_session_and_carries_its_sequence():
|
|
163
|
+
"""The prompt loop's shape: the same route, with `session` and the output
|
|
164
|
+
sequence the caller last saw. Without the sequence the caller would be handed
|
|
165
|
+
everything the shell has ever printed on every line."""
|
|
166
|
+
session = FakeSession(FakeResponse(200, {"output": "/workspace\n", "exit_code": 0,
|
|
167
|
+
"seq": 42, "lost": False, "note": ""}))
|
|
168
|
+
result = client_with(session).exec_in_job("baseline-c", "pwd", session=True, seq=7)
|
|
169
|
+
assert session.calls[0]["json"]["session"] is True
|
|
170
|
+
assert session.calls[0]["json"]["seq"] == 7
|
|
171
|
+
assert result["seq"] == 42, "the reply carries the next one back"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|