hyperun 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: hyperun
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: Submit a GPU job and get results back. No kubectl, no cloud account.
5
5
  Author: DDPS Lab
6
6
  License-Expression: Apache-2.0
@@ -365,9 +365,10 @@ def build_parser() -> argparse.ArgumentParser:
365
365
  "through the job's driver pod into the workload container on the "
366
366
  "rented machine and brings the exit code back like ssh. It is NOT a "
367
367
  "TTY — no vim, no top, about 25 seconds per command — because the "
368
- "server is a Lambda and cannot hold a terminal open. AWS and GCP "
369
- "machine rentals only: a RunPod job is a rented container with no "
370
- "machine behind it, and the relay refuses it. ★ PUT OPTIONS BEFORE THE "
368
+ "server is a Lambda and cannot hold a terminal open. AWS, GCP and "
369
+ "Shadeform machine rentals only: a RunPod job is a rented container "
370
+ "with no machine behind it, so there is no cluster to attach to and "
371
+ "the relay refuses it. ★ PUT OPTIONS BEFORE THE "
371
372
  "JOB ID -- everything after it is sent to the workload as-is, so "
372
373
  "`shell job-x --slot 2` asks pod 0 and passes `--slot 2` to the shell. "
373
374
  "That is refused rather than obeyed.",
@@ -826,11 +827,26 @@ def cmd_shell(args: argparse.Namespace) -> int:
826
827
  pod onto the rented machine and returns the exit code like ssh. With a
827
828
  trailing `-- command` it runs once and exits with that code; without one
828
829
  it prompts, which FEELS like a slow shell and is honestly a request loop.
830
+
831
+ ★ THE PROMPT USES A SESSION AND THE ONE-SHOT FORM DOES NOT, since
832
+ 2026-09-10. The prompt types into a shell that stays running in the driver
833
+ pod, so `cd`, an exported variable and an activated venv survive from one
834
+ line to the next -- which is what a person at a prompt assumes and what
835
+ every earlier version quietly did not do. A `-- command` invocation is a
836
+ script's shape and gets the stateless form: it wants no history and should
837
+ not leave a session behind for the idle timer to close.
829
838
  """
830
839
  client = client_from_config()
831
-
832
- def run_once(line: str) -> int:
833
- answer = client.exec_in_job(args.job, line, slot=args.slot)
840
+ # The output sequence the session has already shown us. It is here rather
841
+ # than inside run_once because it has to survive between lines, which is the
842
+ # whole point.
843
+ state = {"seq": 0}
844
+
845
+ def run_once(line: str, session: bool = False) -> int:
846
+ answer = client.exec_in_job(args.job, line, slot=args.slot,
847
+ session=session, seq=state["seq"])
848
+ if session:
849
+ state["seq"] = int(answer.get("seq", state["seq"]))
834
850
  output = answer.get("output") or ""
835
851
  if output:
836
852
  print(output, end="" if output.endswith("\n") else "\n")
@@ -873,7 +889,8 @@ def cmd_shell(args: argparse.Namespace) -> int:
873
889
  return run_once(" ".join(words))
874
890
 
875
891
  print(
876
- f"one command per line, ~25s each; 'exit' or Ctrl-D leaves. Not a TTY.",
892
+ f"one command per line, ~25s each; 'exit' or Ctrl-D leaves. Not a TTY, "
893
+ f"but `cd` and exported variables DO survive between lines.",
877
894
  file=sys.stderr,
878
895
  )
879
896
  while True:
@@ -888,10 +905,29 @@ def cmd_shell(args: argparse.Namespace) -> int:
888
905
  if line in ("exit", "quit"):
889
906
  return EXIT_OK
890
907
  try:
891
- code = run_once(line)
908
+ code = run_once(line, session=True)
892
909
  if code:
893
910
  print(f"(exit {code})", file=sys.stderr)
894
911
  except ServerError as exc:
912
+ # A SESSION THAT IS GONE IS THE ONE ERROR WORTH ACTING ON. The
913
+ # server answers 409 when the driver pod has no session for this
914
+ # slot -- it timed out after ten minutes unread, or the workload
915
+ # restarted. Reopening is right there and only there: everywhere
916
+ # else a new shell would silently lose the `cd` the person is
917
+ # relying on, which is the failure this whole feature exists to
918
+ # end. The sequence resets with it, because the new shell's
919
+ # output starts from nothing.
920
+ if "open one first" in str(exc) or "no session" in str(exc):
921
+ state["seq"] = 0
922
+ print("(the session had closed; reopening — `cd` and variables "
923
+ "from before are gone)", file=sys.stderr)
924
+ try:
925
+ code = run_once(line, session=True)
926
+ if code:
927
+ print(f"(exit {code})", file=sys.stderr)
928
+ continue
929
+ except ServerError as retry_exc:
930
+ exc = retry_exc
895
931
  # One failed command must not end the session: say what the server
896
932
  # said and keep the prompt.
897
933
  print(f"error: {exc}", file=sys.stderr)
@@ -1079,8 +1115,13 @@ def cmd_watch(args: argparse.Namespace) -> int:
1079
1115
  f"{gpu['memory_used_mib']:,} / {gpu['memory_total_mib']:,} MiB "
1080
1116
  f"({gpu['memory_percent']}%)")
1081
1117
  print(f" temp, power {gpu['temperature_c']} C, {gpu['power_w']} W")
1082
- print(f" samples {len(result.get('gpu_series', []))}, "
1083
- f"last {result['window_seconds']}s")
1118
+ # `sample_count`, not len(gpu_series): the server thins the series to at
1119
+ # most 400 points so a chart can draw it, and job-66b46719b854 printed
1120
+ # 785 readings per card while this line said 393. The screen had the same
1121
+ # defect and was fixed on 2026-09-12. A server too old to send the count
1122
+ # falls back to the thinned length, which is the only number it has.
1123
+ taken = result.get("sample_count") or len(result.get("gpu_series", []))
1124
+ print(f" samples {taken}, last {result['window_seconds']}s")
1084
1125
 
1085
1126
  if result.get("note"):
1086
1127
  print(f" note {result['note']}")
@@ -193,7 +193,8 @@ class Client:
193
193
  return self._call("GET", "/v1/stats").json()
194
194
 
195
195
  def exec_in_job(
196
- self, job: str, command: str, slot: int = 0, timeout_seconds: int = 20
196
+ self, job: str, command: str, slot: int = 0, timeout_seconds: int = 20,
197
+ session: bool = False, seq: int = 0,
197
198
  ) -> dict[str, Any]:
198
199
  """Run one shell line inside a running job's workload container.
199
200
 
@@ -206,6 +207,13 @@ class Client:
206
207
  command: one shell line, run as `sh -lc <command>`.
207
208
  slot: which pod of a parallel job.
208
209
  timeout_seconds: server-side wait, capped at 25 by the server.
210
+ session: type the line into a shell ALREADY RUNNING in the driver
211
+ pod, so `cd` and exported variables survive to the next call.
212
+ False starts a fresh `sh -lc`, which is what a script wants.
213
+ seq: with `session`, the output sequence this caller last saw. The
214
+ reply carries the next one. It exists because the caller is a
215
+ different process on the server's side every time and cannot
216
+ hold a position in the output.
209
217
  """
210
218
  return self._call(
211
219
  "POST",
@@ -214,6 +222,8 @@ class Client:
214
222
  "command": command,
215
223
  "slot": slot,
216
224
  "timeout_seconds": timeout_seconds,
225
+ "session": session,
226
+ "seq": seq,
217
227
  },
218
228
  # The server may hold the request for timeout_seconds before
219
229
  # answering; the read timeout has to outlive that on purpose.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: hyperun
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: Submit a GPU job and get results back. No kubectl, no cloud account.
5
5
  Author: DDPS Lab
6
6
  License-Expression: Apache-2.0
@@ -34,7 +34,7 @@ build-backend = "setuptools.build_meta"
34
34
 
35
35
  [project]
36
36
  name = "hyperun"
37
- version = "0.2.0"
37
+ version = "0.2.1"
38
38
  description = "Submit a GPU job and get results back. No kubectl, no cloud account."
39
39
  requires-python = ">=3.9"
40
40
  dependencies = ["requests>=2.31", "PyYAML>=6.0"]
@@ -100,8 +100,8 @@ class FakeClient:
100
100
  # DDPSRUN-SHELL. Absent until 2026-09-10, which is exactly why the dispatch
101
101
  # bug below survived every release: with no double for this call, no test
102
102
  # could reach cmd_shell at all.
103
- def exec_in_job(self, job, line, slot=0):
104
- self.execed.append((job, line, slot))
103
+ def exec_in_job(self, job, line, slot=0, session=False, seq=0):
104
+ self.execed.append((job, line, slot, session, seq))
105
105
  return self.exec_result
106
106
 
107
107
 
@@ -475,6 +475,39 @@ def test_watch_prints_progress_and_gpu(fake, capsys):
475
475
  assert "samples 120" in printed
476
476
 
477
477
 
478
+ def test_samples_counts_the_readings_taken_not_the_chart_points(fake, capsys):
479
+ # ★ THE SAME DEFECT THE SCREEN HAD. The server thins the series to at most
480
+ # 400 points so a chart can draw it; job-66b46719b854 printed 785 readings
481
+ # per card and both surfaces counted the 393 that survived. The mean and the
482
+ # peak are computed over all 785, so the count beside them has to be 785.
483
+ fake.metrics_result = {
484
+ "latest_gpu": {"utilization_percent": 94, "memory_used_mib": 38200,
485
+ "memory_total_mib": 45440, "memory_percent": 84.1,
486
+ "temperature_c": 71, "power_w": 298.5},
487
+ "gpu_series": [{}] * 393,
488
+ "sample_count": 785,
489
+ "window_seconds": 604800, "note": "",
490
+ }
491
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
492
+ printed = capsys.readouterr().out
493
+ assert "samples 785" in printed
494
+ assert "samples 393" not in printed
495
+
496
+
497
+ def test_a_server_too_old_to_send_the_count_still_prints_one(fake, capsys):
498
+ # The field arrived on 2026-09-12. Against a server that predates it the
499
+ # thinned length is the only number there is, and a blank is worse.
500
+ fake.metrics_result = {
501
+ "latest_gpu": {"utilization_percent": 10, "memory_used_mib": 1,
502
+ "memory_total_mib": 2, "memory_percent": 50.0,
503
+ "temperature_c": 40, "power_w": 50.0},
504
+ "gpu_series": [{}] * 12,
505
+ "window_seconds": 3600, "note": "",
506
+ }
507
+ assert run(["watch", "job-a8acdef80a07"]) == cli.EXIT_OK
508
+ assert "samples 12" in capsys.readouterr().out
509
+
510
+
478
511
  def test_an_unsettled_projection_is_labelled_rather_than_stated(fake, capsys):
479
512
  fake.metrics_result = {
480
513
  "progress": {"step": 5, "total_steps": 556, "percent": 0.9,
@@ -820,7 +853,8 @@ def test_the_subcommand_name_survives_the_shell_positional():
820
853
 
821
854
  def test_shell_with_a_command_runs_it_once_and_returns_its_exit_code(fake, capsys):
822
855
  assert run(["shell", "job-a8acdef80a07", "--", "date"]) == 0
823
- assert fake.execed == [("job-a8acdef80a07", "date", 0)]
856
+ assert fake.execed == [("job-a8acdef80a07", "date", 0, False, 0)], (
857
+ "`-- command` 형태는 stateless 다 — script 는 남길 session 이 없다")
824
858
  assert "Sep 10" in capsys.readouterr().out
825
859
 
826
860
 
@@ -858,3 +892,62 @@ def test_every_other_subcommand_still_dispatches(fake):
858
892
  """dest 를 바꾼 것이 나머지를 깨지 않았다는 확인."""
859
893
  assert run(["status", "job-a8acdef80a07"]) == 0
860
894
  assert run(["secrets"]) == 0
895
+
896
+
897
+ def test_the_prompt_uses_a_session_and_carries_the_sequence(fake, monkeypatch, capsys):
898
+ """★ WHY THE PROMPT DIFFERS FROM `-- command`. A person at a prompt assumes
899
+ `cd` sticks; a script does not want a session left behind for the idle timer.
900
+ So the prompt sends session=True and threads the sequence through, and the
901
+ one-shot form stays stateless."""
902
+ typed = iter(["cd /workspace", "pwd", "exit"])
903
+ monkeypatch.setattr("builtins.input", lambda _prompt: next(typed))
904
+ fake.exec_result = {"output": "/workspace\n", "exit_code": 0, "seq": 11, "lost": False}
905
+
906
+ assert run(["shell", "job-a8acdef80a07"]) == 0
907
+ assert [(line, sess) for _job, line, _slot, sess, _seq in fake.execed] == [
908
+ ("cd /workspace", True), ("pwd", True)]
909
+ # The FIRST line starts at 0 and the second carries what the first returned.
910
+ assert [seq for *_rest, seq in fake.execed] == [0, 11]
911
+
912
+
913
+ def test_a_session_that_closed_is_reopened_once_and_the_person_is_told(fake, monkeypatch,
914
+ capsys):
915
+ """The one case where starting a new shell is right: the old one is provably
916
+ gone (ten minutes unread, or the workload restarted). Everywhere else a new
917
+ shell silently loses the `cd`, which is the failure this feature exists to end
918
+ -- so the message says what was lost."""
919
+ typed = iter(["pwd", "exit"])
920
+ monkeypatch.setattr("builtins.input", lambda _prompt: next(typed))
921
+
922
+ calls = {"n": 0}
923
+ original = fake.exec_in_job
924
+
925
+ def flaky(job, line, slot=0, session=False, seq=0):
926
+ calls["n"] += 1
927
+ if calls["n"] == 1:
928
+ raise cli.ServerError("no session for this slot; open one first")
929
+ return original(job, line, slot=slot, session=session, seq=seq)
930
+
931
+ fake.exec_in_job = flaky
932
+ assert run(["shell", "job-a8acdef80a07"]) == 0
933
+ assert calls["n"] == 2, "it retried once rather than giving up or looping"
934
+ err = capsys.readouterr().err
935
+ assert "reopening" in err and "are gone" in err, (
936
+ "a silent reopen would let somebody keep typing paths relative to a cd that no "
937
+ "longer applies")
938
+
939
+
940
+ def test_a_failure_that_is_not_a_lost_session_is_not_retried(fake, monkeypatch, capsys):
941
+ typed = iter(["pwd", "exit"])
942
+ monkeypatch.setattr("builtins.input", lambda _prompt: next(typed))
943
+
944
+ calls = {"n": 0}
945
+
946
+ def always_bad(job, line, slot=0, session=False, seq=0):
947
+ calls["n"] += 1
948
+ raise cli.ServerError("the driver pod did not answer in JSON")
949
+
950
+ fake.exec_in_job = always_bad
951
+ assert run(["shell", "job-a8acdef80a07"]) == 0
952
+ assert calls["n"] == 1, "retrying an unrelated failure would double every bad command"
953
+ assert "did not answer in JSON" in capsys.readouterr().err
@@ -65,7 +65,9 @@ def test_shell_sends_the_command_and_waits_longer_than_the_server():
65
65
  result = client_with(session).exec_in_job("baseline-c", "nvidia-smi -L", slot=1)
66
66
  call = session.calls[0]
67
67
  assert call["url"] == "https://run.example/v1/jobs/baseline-c/exec"
68
- assert call["json"] == {"command": "nvidia-smi -L", "slot": 1, "timeout_seconds": 20}
68
+ assert call["json"] == {"command": "nvidia-smi -L", "slot": 1, "timeout_seconds": 20,
69
+ "session": False, "seq": 0}, (
70
+ "PACSRUN-SHELL-SESSION 이후에도 기본은 한 줄짜리 stateless 형태다")
69
71
  # The server may hold the request for its whole window, so the client's
70
72
  # read timeout must outlive it.
71
73
  assert call["timeout"] > 20
@@ -155,3 +157,15 @@ def test_a_submit_is_given_more_patience_than_a_read():
155
157
  read_session = FakeSession(FakeResponse(200, {}))
156
158
  client_with(read_session).status("job-a8acdef80a07")
157
159
  assert submit_session.calls[0]["timeout"] > read_session.calls[0]["timeout"]
160
+
161
+
162
+ def test_shell_can_ask_for_the_persistent_session_and_carries_its_sequence():
163
+ """The prompt loop's shape: the same route, with `session` and the output
164
+ sequence the caller last saw. Without the sequence the caller would be handed
165
+ everything the shell has ever printed on every line."""
166
+ session = FakeSession(FakeResponse(200, {"output": "/workspace\n", "exit_code": 0,
167
+ "seq": 42, "lost": False, "note": ""}))
168
+ result = client_with(session).exec_in_job("baseline-c", "pwd", session=True, seq=7)
169
+ assert session.calls[0]["json"]["session"] is True
170
+ assert session.calls[0]["json"]["seq"] == 7
171
+ assert result["seq"] == 42, "the reply carries the next one back"
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes