outerloop-science 0.1.0.dev0__py3-none-any.whl → 0.1.0.dev2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
outerloop/climbboard.py CHANGED
@@ -391,6 +391,22 @@ def render_html(
391
391
  "border-radius:2px;display:inline-block;box-sizing:border-box}\n"
392
392
  ".pill{border-radius:.6rem;padding:.05rem .55rem;font-size:.72rem;\n"
393
393
  "font-weight:600;color:#fff}\n"
394
+ ".run.queue{width:100%;padding-bottom:.55rem}\n"
395
+ ".run.queue table{width:100%;border-collapse:collapse;margin-top:.35rem;\n"
396
+ " font-size:.72rem;font-variant-numeric:tabular-nums}\n"
397
+ ".run.queue th{text-align:left;font-weight:500;color:var(--muted);\n"
398
+ " padding:.1rem .6rem .2rem 0;border-bottom:1px solid var(--line)}\n"
399
+ ".run.queue td{padding:.17rem .6rem .17rem 0;border-bottom:1px solid var(--line);\n"
400
+ " white-space:nowrap;vertical-align:middle;color:var(--ink)}\n"
401
+ ".run.queue tr:last-child td{border-bottom:0}\n"
402
+ ".run.queue .mono{font-family:ui-monospace,SFMono-Regular,Menlo,Consolas,monospace}\n"
403
+ ".run.queue td.name{max-width:24rem;overflow:hidden;text-overflow:ellipsis}\n"
404
+ ".run.queue td.part,.run.queue td.time{color:var(--muted)}\n"
405
+ ".run.queue .st{display:inline-block;min-width:1.5rem;text-align:center;\n"
406
+ " border-radius:.4rem;padding:0 .35rem;font-size:.66rem;font-weight:600}\n"
407
+ ".run.queue .st-R{background:var(--win);color:#fff}\n"
408
+ ".run.queue .st-PD{background:var(--line);color:var(--muted)}\n"
409
+ ".run.queue .st-X{background:var(--near);color:#fff}\n"
394
410
  ".card{background:var(--card);border:1px solid var(--line);border-radius:.5rem;\n"
395
411
  "padding:1rem}\n"
396
412
  ".card label{font-size:.8rem;color:var(--muted);margin-right:1rem}\n"
@@ -773,6 +789,55 @@ def render_html(
773
789
  " now.append(card);\n"
774
790
  " }\n"
775
791
  " if (!strip.runs.length) now.textContent = 'no active runs';\n"
792
+ " // the kernel's own Slurm jobs, as squeue would list them (when\n"
793
+ " // published): one full-width card in the strip, running jobs first\n"
794
+ " if (Array.isArray(strip.queue) && strip.queue.length) {\n"
795
+ " const ST = {RUNNING: 'R', PENDING: 'PD', COMPLETING: 'CG', CONFIGURING: 'CF'};\n"
796
+ " // squeue's reading order: running, finishing, starting, waiting, other\n"
797
+ " const RANK = {RUNNING: 0, COMPLETING: 1, CONFIGURING: 2, PENDING: 3};\n"
798
+ " const rank = j => RANK[j.state] ?? 4;\n"
799
+ " const jobs = [...strip.queue].sort((a, b) =>\n"
800
+ " rank(a) - rank(b) || String(a.id).localeCompare(String(b.id)));\n"
801
+ " const card = document.createElement('span'); card.className = 'run queue';\n"
802
+ " const top = document.createElement('span'); top.className = 'top';\n"
803
+ " const who = document.createElement('b'); who.textContent = 'squeue';\n"
804
+ " const meta = document.createElement('span');\n"
805
+ " const running = jobs.filter(j => j.state === 'RUNNING').length;\n"
806
+ " meta.textContent = jobs.length + ' jobs · ' + running + ' running';\n"
807
+ ' meta.title = "the kernel\'s own Slurm jobs, as squeue --me lists them";\n'
808
+ " top.append(who, meta); card.append(top);\n"
809
+ " const table = document.createElement('table');\n"
810
+ " const thead = table.createTHead().insertRow();\n"
811
+ " for (const h of ['JOBID', 'NAME', 'ST', 'TIME', 'PARTITION', 'AGENT']) {\n"
812
+ " const th = document.createElement('th'); th.textContent = h; thead.append(th);\n"
813
+ " }\n"
814
+ " const body = table.createTBody();\n"
815
+ " const td = (tr, text, cls) => {\n"
816
+ " const c = tr.insertCell(); c.textContent = text ?? '';\n"
817
+ " if (cls) c.className = cls; return c;\n"
818
+ " };\n"
819
+ " for (const j of jobs) {\n"
820
+ " const tr = body.insertRow();\n"
821
+ " td(tr, j.id, 'mono');\n"
822
+ " td(tr, j.name, 'mono name').title = j.name;\n"
823
+ " const st = document.createElement('span');\n"
824
+ " const code = ST[j.state] || j.state;\n"
825
+ " st.className = 'st st-' + (code === 'R' || code === 'PD' ? code : 'X');\n"
826
+ " st.textContent = code; st.title = j.state;\n"
827
+ " tr.insertCell().append(st);\n"
828
+ " td(tr, j.elapsed, 'mono time');\n"
829
+ " // a job eligible for several partitions lists them comma-joined;\n"
830
+ " // show the first and count the rest, full list on hover\n"
831
+ " const parts = String(j.partition || '').split(',').filter(Boolean);\n"
832
+ " const pc = td(tr, parts.length > 1 ? parts[0] + ' +' + (parts.length - 1)\n"
833
+ " : parts[0] || '', 'part');\n"
834
+ " pc.title = parts.join(', ');\n"
835
+ " const a = td(tr, j.agent || '—', 'agent');\n"
836
+ " if (j.agent) { a.style.color = agentColor(j.agent); a.style.fontWeight = '600'; }\n"
837
+ " }\n"
838
+ " card.append(table);\n"
839
+ " now.append(card);\n"
840
+ " }\n"
776
841
  "};\n"
777
842
  "refresh();\n"
778
843
  "// the strip re-FETCHES every few minutes (new/left/changed runs) and\n"
@@ -784,6 +849,110 @@ def render_html(
784
849
 
785
850
 
786
851
  STATUS_PATH = "climb/status.json"
852
+ # The queue view shows the kernel's own jobs only. Fixed-name jobs (the tick
853
+ # chain, issue sessions, climb sessions) are matched by shape; per-run jobs
854
+ # (wake, followup, launch) are matched against the names the kernel itself
855
+ # derives from this target's run ids, with the same 60-character cut Slurm
856
+ # forces on them; an eval's name is a liveness hash and is claimed only
857
+ # through its run's marker. Anything else on the account is the operator's
858
+ # and never leaves the cluster.
859
+ JOB_NAME_LIMIT = 60
860
+ FIXED_JOB_PATTERNS = tuple(
861
+ re.compile(p)
862
+ for p in (
863
+ r"^(autoresearch|outerloop)-(resident|tick)$",
864
+ r"^climb-[\w.-]+-agent-\d+$",
865
+ r"^(steward|climb)-issue-\d+$",
866
+ )
867
+ )
868
+ _AGENT_RE = re.compile(r"agent-\d+")
869
+
870
+
871
+ def is_fixed_kernel_job(name: str) -> bool:
872
+ return any(p.match(name) for p in FIXED_JOB_PATTERNS)
873
+
874
+
875
+ def run_job_names(root: Path, target: str) -> dict[str, tuple[str, str, bool]]:
876
+ """Expected per-run job names for `target`: name -> (run_id, agent,
877
+ is_prefix). Exact for wake and followup; a prefix for launches, whose
878
+ names end in the experiment's own label. Every run of the target counts,
879
+ ended ones too: a finishing job can outlive its record's state."""
880
+ out: dict[str, tuple[str, str, bool]] = {}
881
+ for record in list_runs(root):
882
+ if record.target != target:
883
+ continue
884
+ rid, agent = record.run_id, record.agent_id
885
+ for name, prefix in (
886
+ (f"wake-{rid}", False),
887
+ (f"followup-{rid}", False),
888
+ (f"{rid}-launch-", True),
889
+ ):
890
+ key = name[:JOB_NAME_LIMIT]
891
+ if key in out and out[key][0] != rid:
892
+ # two runs whose names collide under the cut: the job is still
893
+ # the kernel's, but nobody can say whose, so it is claimed by none
894
+ out[key] = ("", "", prefix)
895
+ else:
896
+ out[key] = (rid, agent, prefix)
897
+ return out
898
+
899
+
900
+ def eval_job_owners(root: Path, target: str) -> dict[str, tuple[str, str]]:
901
+ """Slurm job id -> (run_id, agent) for every eval a live run of `target`
902
+ has dispatched: the measurer leaves the id in eval-*/submitted. Eval job
903
+ names are liveness hashes with no agent in them, so this is how the queue
904
+ view knows whose eval is whose."""
905
+ out: dict[str, tuple[str, str]] = {}
906
+ for record in list_runs(root):
907
+ if record.target != target or record.state == ENDED:
908
+ continue
909
+ for submitted in run_dir(root, record.run_id).glob("eval-*/submitted"):
910
+ try:
911
+ job_id = submitted.read_text().strip()
912
+ except OSError:
913
+ continue
914
+ if job_id:
915
+ out[job_id] = (record.run_id, record.agent_id)
916
+ return out
917
+
918
+
919
+ def queue_rows(root: Path, target: str, snapshot: list[dict[str, str]]) -> list[dict[str, str]]:
920
+ """The published queue: kernel jobs from a `Compute.queue_snapshot()`,
921
+ each attributed to an agent (by eval marker, else by the agent id every
922
+ other kernel job carries in its name)."""
923
+ owners = eval_job_owners(root, target)
924
+ expected = run_job_names(root, target)
925
+ rows: list[dict[str, str]] = []
926
+ for job in snapshot:
927
+ name = str(job.get("name", ""))
928
+ job_id = str(job.get("id", ""))
929
+ run_id = agent = ""
930
+ if job_id in owners: # an eval of one of this target's live runs
931
+ run_id, agent = owners[job_id]
932
+ elif name in expected and not expected[name][2]: # wake / followup, exact
933
+ run_id, agent = expected[name][:2]
934
+ elif any(name.startswith(k) for k, v in expected.items() if v[2]): # a launch
935
+ run_id, agent = next(v[:2] for k, v in expected.items() if v[2] and name.startswith(k))
936
+ elif is_fixed_kernel_job(name):
937
+ found = _AGENT_RE.search(name)
938
+ agent = found.group(0) if found else ""
939
+ else:
940
+ continue # an eval whose marker is not written yet, or the operator's own job
941
+ rows.append(
942
+ {
943
+ "id": str(job.get("id", "")),
944
+ "name": name,
945
+ "state": str(job.get("state", "")),
946
+ "elapsed": str(job.get("elapsed", "")),
947
+ "partition": str(job.get("partition", "")),
948
+ "submitted": str(job.get("submitted", "")),
949
+ "agent": agent,
950
+ "run_id": run_id,
951
+ }
952
+ )
953
+ return rows
954
+
955
+
787
956
  _LIVE_STATES = ("implementing", "waiting", "in-review", "concluding")
788
957
 
789
958
 
@@ -823,7 +992,13 @@ def _experiment_progress(root: Path, record: Any) -> tuple[int, int, int]:
823
992
  return (done, len(names), minutes)
824
993
 
825
994
 
826
- def collect_status(root: Path, target: str, now: float, contract: Any = None) -> dict[str, Any]:
995
+ def collect_status(
996
+ root: Path,
997
+ target: str,
998
+ now: float,
999
+ contract: Any = None,
1000
+ queue: list[dict[str, str]] | None = None,
1001
+ ) -> dict[str, Any]:
827
1002
  """The fleet's live picture for `target`: one entry per non-terminal run.
828
1003
  Timestamps, not durations — the page computes elapsed time client-side,
829
1004
  so the strip feels live between pushes."""
@@ -878,15 +1053,25 @@ def collect_status(root: Path, target: str, now: float, contract: Any = None) ->
878
1053
  }
879
1054
  )
880
1055
  runs.sort(key=lambda r: str(r.get("run_id")))
881
- return {"target": target, "published": now, "runs": runs}
1056
+ status: dict[str, Any] = {"target": target, "published": now, "runs": runs}
1057
+ if queue is not None: # a snapshot was taken (Slurm); the local loop has no queue
1058
+ status["queue"] = queue_rows(root, target, queue)
1059
+ return status
882
1060
 
883
1061
 
884
- def service_status(root: Path, github: Any, target: str, now: float, contract: Any = None) -> bool:
1062
+ def service_status(
1063
+ root: Path,
1064
+ github: Any,
1065
+ target: str,
1066
+ now: float,
1067
+ contract: Any = None,
1068
+ queue: list[dict[str, str]] | None = None,
1069
+ ) -> bool:
885
1070
  """Publish the strip when the fleet's SHAPE changed — a run appearing,
886
1071
  leaving, or changing state/phase — never on every tick: the page shows
887
1072
  elapsed time client-side, so timestamp-only drift is not worth a commit.
888
1073
  Advisory like the board; True when a write happened."""
889
- status = collect_status(root, target, now, contract)
1074
+ status = collect_status(root, target, now, contract, queue)
890
1075
  try:
891
1076
  existing_raw: str | None = github.get_file(target, STATUS_PATH, BOARD_BRANCH)
892
1077
  except Exception as exc:
@@ -920,10 +1105,30 @@ def service_status(root: Path, github: Any, target: str, now: float, contract: A
920
1105
  "sleep_k",
921
1106
  )
922
1107
  shape = lambda runs: [{k: r.get(k) for k in keys} for r in runs]
1108
+ # the queue's shape is which jobs exist, their state, partition and
1109
+ # attribution (an eval claimed by its marker one tick later must
1110
+ # republish); elapsed time drifts every tick and is left out
1111
+ # an absent queue (no snapshot) and an empty one (nothing running)
1112
+ # are different shapes: recovering from a blind tick must publish
1113
+ qshape = lambda q: (
1114
+ None
1115
+ if q is None
1116
+ else [
1117
+ (
1118
+ j.get("id"),
1119
+ j.get("state"),
1120
+ j.get("partition"),
1121
+ j.get("agent"),
1122
+ j.get("run_id"),
1123
+ )
1124
+ for j in q
1125
+ ]
1126
+ )
923
1127
  if (
924
1128
  isinstance(existing, dict)
925
1129
  and isinstance(existing.get("runs"), list)
926
1130
  and shape(existing["runs"]) == shape(status["runs"])
1131
+ and qshape(existing.get("queue")) == qshape(status.get("queue"))
927
1132
  ):
928
1133
  return False
929
1134
  except (ValueError, TypeError, AttributeError):
outerloop/compute.py CHANGED
@@ -95,6 +95,10 @@ class JobSpec:
95
95
  # Slurm scheduling controls
96
96
  dependency: str = "" # e.g. "afterany:12345" or "singleton"
97
97
  begin: str = "" # e.g. "now+30" or an absolute "YYYY-MM-DDTHH:MM:SS"
98
+ # submitted held (PENDING, reason JobHeldUser) until `release`: the
99
+ # tick's launch admission lets GPU launches into the queue in order,
100
+ # under the per-user cap, instead of queueing them all at once
101
+ hold: bool = False
98
102
  extra: tuple[str, ...] = ()
99
103
 
100
104
  def to_argv(self) -> list[str]:
@@ -104,12 +108,13 @@ class JobSpec:
104
108
  "sbatch",
105
109
  "--parsable",
106
110
  f"--job-name={self.job_name}",
107
- f"--account={self.account}",
108
111
  f"--time={self.time_minutes}",
109
112
  f"--cpus-per-task={self.cpus}",
110
113
  f"--mem={self.mem}",
111
114
  f"--output={self.output}",
112
115
  ]
116
+ if self.account: # unset bills the caller's default Slurm association
117
+ argv.append(f"--account={self.account}")
113
118
  if self.partition: # unset lets Slurm pick its default partition
114
119
  argv.append(f"--partition={self.partition}")
115
120
  if self.gpus:
@@ -120,6 +125,8 @@ class JobSpec:
120
125
  argv.append(f"--gpus-per-node={self.gpus}")
121
126
  if self.qos:
122
127
  argv.append(f"--qos={self.qos}")
128
+ if self.hold:
129
+ argv.append("--hold")
123
130
  if self.dependency:
124
131
  argv.append(f"--dependency={self.dependency}")
125
132
  if self.begin:
@@ -133,6 +140,11 @@ class JobSpec:
133
140
  return argv
134
141
 
135
142
 
143
+ # `reason` and `gres` feed launch admission (why a job waits, how many GPUs it
144
+ # asks for); the board reads the first six by key and ignores the rest
145
+ QUEUE_FIELDS = ("id", "name", "state", "elapsed", "partition", "submitted", "reason", "gres")
146
+
147
+
136
148
  class Compute(Protocol):
137
149
  """The verbs every compute backend implements. Callers (the measurer, the
138
150
  launcher, the wake dispatcher) depend on this, never on a backend."""
@@ -142,17 +154,19 @@ class Compute(Protocol):
142
154
  def pending_reason(self, job_id: str) -> str: ...
143
155
  def job_partition(self, job_id: str) -> str: ...
144
156
  def active_job_names(self) -> list[str]: ...
157
+ def queue_snapshot(self) -> list[dict[str, str]]: ...
145
158
  def job_id_for_name(self, name: str) -> str: ...
146
159
  def cancel(self, job_id: str) -> None: ...
160
+ def release(self, job_id: str) -> None: ...
147
161
 
148
162
 
149
163
  def local_mode() -> bool:
150
- """AUTORESEARCH_COMPUTE=local selects the monolith: every job a
164
+ """OUTERLOOP_COMPUTE=local selects the monolith: every job a
151
165
  synchronous subprocess of the caller (docs/design/onboarding.md) — the
152
166
  zero-cluster on-ramp and the paper's serialized-baseline ablation. Any
153
167
  other value (or none) is Slurm. This helper is the only reader of the
154
168
  env var, so mode checks cannot drift."""
155
- return os.environ.get("AUTORESEARCH_COMPUTE", "").strip().lower() == "local"
169
+ return os.environ.get("OUTERLOOP_COMPUTE", "").strip().lower() == "local"
156
170
 
157
171
 
158
172
  def compute_from_env() -> SlurmCompute | LocalCompute:
@@ -257,6 +271,26 @@ class SlurmCompute:
257
271
  raise SlurmQueryError(f"squeue failed ({result.returncode}): {result.stderr.strip()}")
258
272
  return [line.strip() for line in result.stdout.splitlines() if line.strip()]
259
273
 
274
+ def queue_snapshot(self) -> list[dict[str, str]]:
275
+ """This user's PENDING and RUNNING jobs as rows of QUEUE_FIELDS — what
276
+ a queue view needs, nothing a caller acts on. Raises SlurmQueryError
277
+ on failure, like active_job_names."""
278
+ try:
279
+ result = self.runner(
280
+ ["squeue", "--me", "--noheader", "-o", "%i|%j|%T|%M|%P|%V|%r|%b"],
281
+ self.command_timeout_s,
282
+ )
283
+ except (OSError, subprocess.TimeoutExpired) as exc:
284
+ raise SlurmQueryError(f"squeue did not run: {exc}") from exc
285
+ if result.returncode != 0:
286
+ raise SlurmQueryError(f"squeue failed ({result.returncode}): {result.stderr.strip()}")
287
+ rows: list[dict[str, str]] = []
288
+ for line in result.stdout.splitlines():
289
+ parts = line.strip().split("|")
290
+ if len(parts) == len(QUEUE_FIELDS) and parts[0]:
291
+ rows.append(dict(zip(QUEUE_FIELDS, parts, strict=True)))
292
+ return rows
293
+
260
294
  def job_id_for_name(self, name: str) -> str:
261
295
  """The id of this user's PENDING/RUNNING job with exactly `name`, or
262
296
  "" if none. Authoritative for "is this still live" independent of any
@@ -283,6 +317,15 @@ class SlurmCompute:
283
317
  if result.returncode != 0:
284
318
  log.warning("scancel %s: %s", job_id, result.stderr.strip())
285
319
 
320
+ def release(self, job_id: str) -> None:
321
+ """Release a job submitted with `hold` so the scheduler may start it.
322
+ Releasing a job that is not held is not an error."""
323
+ if not job_id.isdigit():
324
+ raise ValueError(f"not a job id: {job_id!r}")
325
+ result = self.runner(["scontrol", "release", job_id], self.command_timeout_s)
326
+ if result.returncode != 0:
327
+ log.warning("scontrol release %s: %s", job_id, result.stderr.strip())
328
+
286
329
 
287
330
  # Local job ids start far above any real Slurm id so the two can never be
288
331
  # confused in a record; they stay numeric because callers validate isdigit.
@@ -294,7 +337,7 @@ def _local_state_dir() -> Path | None:
294
337
  attempts it spawns each hold their own LocalCompute): under the state
295
338
  root when the deployment names one, else nowhere (memory-only — tests).
296
339
  Local jobs are synchronous, so only TERMINAL states ever need sharing."""
297
- root = os.environ.get("AUTORESEARCH_ROOT", "").strip()
340
+ root = os.environ.get("OUTERLOOP_ROOT", "").strip()
298
341
  return Path(root) / "local_jobs" if root else None
299
342
 
300
343
 
@@ -330,7 +373,7 @@ class LocalCompute:
330
373
  # the submitting process holds live keys (and any inherited
331
374
  # APPTAINERENV_* would cross --cleanenv into the container), so the
332
375
  # job script starts from a minimal environment and sets its own.
333
- # AUTORESEARCH_* / REVIEW_HERMES_* pass through as a PREFIX rule:
376
+ # OUTERLOOP_* / REVIEW_HERMES_* pass through as a PREFIX rule:
334
377
  # Slurm jobs inherit the tick's whole environment, and local jobs
335
378
  # need the same config surface (compute mode, author backend, panel,
336
379
  # key-file PATHS). Enumerating allowed names is how a mode flag dies
@@ -344,7 +387,7 @@ class LocalCompute:
344
387
  k: v
345
388
  for k, v in os.environ.items()
346
389
  if k in ("PATH", "HOME", "LANG", "TMPDIR", "SLURM_TMPDIR", "USER", "LOGNAME")
347
- or (k.startswith(("AUTORESEARCH_", "REVIEW_HERMES_")) and not _secret_name(k))
390
+ or (k.startswith(("OUTERLOOP_", "REVIEW_HERMES_")) and not _secret_name(k))
348
391
  }
349
392
  try:
350
393
  # the job runs in its OWN session (= process group), so the
@@ -389,6 +432,13 @@ class LocalCompute:
389
432
  tmp = state_dir / f".{job_id}.{os.getpid()}.tmp"
390
433
  tmp.write_text(state)
391
434
  os.replace(tmp, state_dir / job_id)
435
+ # the job's combined stdout/stderr beside its state: a failed local
436
+ # job otherwise leaves nothing to read (#295)
437
+ out_fd = os.open(
438
+ state_dir / f"{job_id}.out", os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600
439
+ )
440
+ with os.fdopen(out_fd, "w") as fh:
441
+ fh.write(output)
392
442
  # opportunistic prune: one entry per job would leak forever
393
443
  # on a long-running loop; anything the sweep could still want
394
444
  # is far younger than a day
@@ -408,7 +458,12 @@ class LocalCompute:
408
458
  except OSError as exc:
409
459
  log.warning("local job %s: output write failed: %s", spec.job_name, exc)
410
460
  self._states[job_id] = state
411
- log.info("ran %s locally as job %s: %s", spec.job_name, job_id, state)
461
+ where = (
462
+ f"; output in {state_dir / (job_id + '.out')}"
463
+ if state_dir and state != "COMPLETED"
464
+ else ""
465
+ )
466
+ log.info("ran %s locally as job %s: %s%s", spec.job_name, job_id, state, where)
412
467
  return job_id
413
468
 
414
469
  def status(self, job_id: str) -> str:
@@ -436,6 +491,9 @@ class LocalCompute:
436
491
  def active_job_names(self) -> list[str]:
437
492
  return [] # synchronous: nothing is ever pending or running
438
493
 
494
+ def queue_snapshot(self) -> list[dict[str, str]]:
495
+ return []
496
+
439
497
  def job_id_for_name(self, name: str) -> str:
440
498
  return ""
441
499
 
@@ -444,6 +502,9 @@ class LocalCompute:
444
502
  raise ValueError(f"not a job id: {job_id!r}")
445
503
  # already terminal; cancelling a finished job is not an error
446
504
 
505
+ def release(self, job_id: str) -> None:
506
+ """Local jobs run synchronously at submit; nothing is ever held."""
507
+
447
508
 
448
509
  def parse_elapsed(text: str) -> int | None:
449
510
  """Seconds in a sacct Elapsed field: `MM:SS`, `HH:MM:SS` or `D-HH:MM:SS`.
@@ -477,6 +538,22 @@ def is_pending(state: str) -> bool:
477
538
  return state.startswith("PENDING")
478
539
 
479
540
 
541
+ def gpus_in_gres(gres: str) -> int:
542
+ """GPUs a queue row asks for, from squeue's %b (`gres/gpu:8`,
543
+ `gres/gpu:h200:2`, `gres:gpu:4`; `N/A` or anything else = 0)."""
544
+ total = 0
545
+ for part in gres.replace(";", ",").split(","):
546
+ fields = part.strip().split(":")
547
+ if (len(fields) >= 2 and fields[0] in ("gres/gpu", "gpu")) or (
548
+ len(fields) >= 3 and fields[0] == "gres" and fields[1] == "gpu"
549
+ ):
550
+ try:
551
+ total += int(fields[-1])
552
+ except ValueError:
553
+ continue
554
+ return total
555
+
556
+
480
557
  def quote_command(parts: Sequence[str]) -> str:
481
558
  """Shell-quote a command for JobSpec.command (--wrap takes a string)."""
482
559
  return " ".join(shlex.quote(p) for p in parts)
outerloop/contract.py CHANGED
@@ -26,6 +26,7 @@ SELF_REPO = "outerloop-science/outerloop"
26
26
  # `.autoresearch.yaml` (targets written before the rename) is still honored. Every
27
27
  # read goes through `find_contract`, which tries the new name first. Neither is
28
28
  # ever a writable path for the agent.
29
+ # The pre-rename name is dropped in the release after 0.1.
29
30
  CONTRACT_NAMES: tuple[str, ...] = (".outerloop.yaml", ".autoresearch.yaml")
30
31
  CONTRACT_NAME = CONTRACT_NAMES[0] # what the docs and new contracts use
31
32
  ALWAYS_FORBIDDEN: tuple[str, ...] = (".github", *CONTRACT_NAMES)
@@ -126,8 +127,8 @@ class Benchmark(_StrictModel):
126
127
  eval_minutes: int | None = Field(default=None, ge=1)
127
128
  # GPUs for every dispatched job of this benchmark — gate measures and
128
129
  # author launches alike. 0 (default) = CPU. A GPU benchmark needs the
129
- # deployment to name a GPU lane (AUTORESEARCH_GPU_PARTITION, optionally
130
- # AUTORESEARCH_GPU_ACCOUNT); without one the tick refuses to launch
130
+ # deployment to name a GPU lane (OUTERLOOP_GPU_PARTITION, optionally
131
+ # OUTERLOOP_GPU_ACCOUNT); without one the tick refuses to launch
131
132
  # attempts on it rather than queue evals that can never run. Bounded at
132
133
  # one node's worth: multi-node evals are not a shape the jail supports.
133
134
  gpus: int = Field(default=0, ge=0, le=8)
outerloop/followup.py CHANGED
@@ -37,7 +37,7 @@ from outerloop.github import (
37
37
  contract_at,
38
38
  is_own_login,
39
39
  )
40
- from outerloop.harness import Harness, outage, redact
40
+ from outerloop.harness import Harness, default_binary, outage, redact
41
41
  from outerloop.markers import has_marker, marker
42
42
  from outerloop.orchestrator import (
43
43
  Evaluator,
@@ -958,6 +958,7 @@ def _respond(
958
958
  changed=changed,
959
959
  conflict_head=conflict_head if conflict_wake else "",
960
960
  base_synced=base_synced,
961
+ pr_head=str((pr.get("head") or {}).get("sha", "")),
961
962
  panel_wake=panel_wake,
962
963
  now=now,
963
964
  secrets=secrets,
@@ -1593,6 +1594,7 @@ def _park_remeasure(
1593
1594
  changed: list[str],
1594
1595
  conflict_head: str,
1595
1596
  base_synced: bool,
1597
+ pr_head: str,
1596
1598
  panel_wake: bool,
1597
1599
  now: float,
1598
1600
  secrets: tuple[str, ...],
@@ -1606,6 +1608,11 @@ def _park_remeasure(
1606
1608
  "candidate_sha": snap.commit,
1607
1609
  "candidate_ref": snap.ref,
1608
1610
  "parent": ws.git("rev-parse", "HEAD").strip(),
1611
+ # the PR's head on GitHub at park. After a base sync `parent` is the
1612
+ # session's local merge, which GitHub never saw: the resume compares
1613
+ # the live head against THIS, not against parent (speedrun agent-03,
1614
+ # 2026-09-06: every base-synced re-measure was abandoned as "moved")
1615
+ "pr_head": pr_head,
1609
1616
  "job_ids": list(pend.job_ids),
1610
1617
  "afterany": pend.afterany(),
1611
1618
  "seed": run_seed,
@@ -1729,7 +1736,13 @@ def _resume_measure(
1729
1736
  ws.git("clean", "-fdq")
1730
1737
  drop_snapshot(ws, snapshot)
1731
1738
  github.comment(record.target, number, f"{REPLY_MARKER}\n{note}")
1732
- save_record(run_root, replace(load_record(run_root, run_id), followup_stage={}), now)
1739
+ # the thread was told and invited to ask again: that next ask needs a
1740
+ # follow-up, so the landed measure counts as progress for the cap
1741
+ save_record(
1742
+ run_root,
1743
+ replace(load_record(run_root, run_id), followup_stage={}, wake_attempts=0),
1744
+ now,
1745
+ )
1733
1746
  return FollowupOutcome(run_id, "replied", "dispatched re-measure abandoned")
1734
1747
 
1735
1748
  try:
@@ -1764,12 +1777,16 @@ def _resume_measure(
1764
1777
  )
1765
1778
  return FollowupOutcome(run_id, "replied", "dispatched re-measure already landed")
1766
1779
 
1767
- # the sealed commit is parented on the head the PR had at park: a push
1768
- # since (a maintainer's) makes it unpushable AND measured on a tree that
1769
- # is no longer the PR's — abandon honestly rather than force or rebuild
1770
- if head_now and parent and head_now != parent:
1780
+ # the sealed commit descends from the head the PR had at park (directly,
1781
+ # or through the session's local base-sync merge): a push since (a
1782
+ # maintainer's) makes it unpushable AND measured on a tree that is no
1783
+ # longer the PR's — abandon honestly rather than force or rebuild. Stages
1784
+ # parked before `pr_head` was recorded fall back to conflict_head, then
1785
+ # parent.
1786
+ parked_head = str(stage.get("pr_head") or stage.get("conflict_head") or parent)
1787
+ if head_now and parked_head and head_now != parked_head:
1771
1788
  return _abandon(
1772
- f"_(The PR's head moved while the re-measure ran (`{parent[:12]}` → "
1789
+ f"_(The PR's head moved while the re-measure ran (`{parked_head[:12]}` → "
1773
1790
  f"`{head_now[:12]}`), so the measured change no longer applies to this "
1774
1791
  "branch and was not pushed. Ask again and it will be redone on the new head.)_"
1775
1792
  )
@@ -1946,16 +1963,14 @@ def main() -> int:
1946
1963
  parser.add_argument("--run-id", required=True)
1947
1964
  parser.add_argument("--image", default="")
1948
1965
  parser.add_argument("--uncontained", action="store_true")
1949
- parser.add_argument("--claude-bin", default=os.path.expanduser("~/.local/bin/claude"))
1966
+ parser.add_argument("--claude-bin", default=default_binary("claude"))
1950
1967
  parser.add_argument(
1951
1968
  "--codex-bin",
1952
- default=os.path.expanduser(
1953
- os.environ.get("AUTORESEARCH_CODEX_BIN") or "~/.local/bin/codex"
1954
- ),
1969
+ default=default_binary("codex"),
1955
1970
  )
1956
1971
  parser.add_argument(
1957
1972
  "--model",
1958
- default=os.environ.get("AUTORESEARCH_AUTHOR_MODEL") or "claude-opus-5",
1973
+ default=os.environ.get("OUTERLOOP_AUTHOR_MODEL") or "claude-opus-5",
1959
1974
  help="fallback model only; a run's OWN (backend, model) from its record wins",
1960
1975
  )
1961
1976
  # No --author-backend: a follow-up services ONE run, whose backend+model are
@@ -1981,7 +1996,7 @@ def main() -> int:
1981
1996
  parser.add_argument("--pat-file", default=str(CONFIG_DIR / "bot_pat"))
1982
1997
  parser.add_argument(
1983
1998
  "--github-app-file",
1984
- default=os.environ.get("AUTORESEARCH_GITHUB_APP_FILE", ""),
1999
+ default=os.environ.get("OUTERLOOP_GITHUB_APP_FILE", ""),
1985
2000
  help="GitHub App config (JSON: app_id, installation_id, private_key); "
1986
2001
  "when set, installation tokens replace the PAT",
1987
2002
  )
@@ -1989,7 +2004,7 @@ def main() -> int:
1989
2004
  "--key-file",
1990
2005
  default="",
1991
2006
  help="author key file; default resolves per backend (config-driven): "
1992
- "AUTORESEARCH_HARNESS_KEY_FILE for claude, AUTORESEARCH_CODEX_KEY_FILE for codex",
2007
+ "OUTERLOOP_CLAUDE_KEY_FILE for claude, OUTERLOOP_CODEX_KEY_FILE for codex",
1993
2008
  )
1994
2009
  parser.add_argument(
1995
2010
  "--panel",
@@ -1998,10 +2013,10 @@ def main() -> int:
1998
2013
  "re-read a pushed code change; '' = no re-read, a changed PR stays human-merged",
1999
2014
  )
2000
2015
  parser.add_argument("--panel-key-file", default="", help="the claude panel lenses' key file")
2001
- parser.add_argument("--account", default=os.environ.get("AUTORESEARCH_ACCOUNT", ""))
2002
- parser.add_argument("--partition", default=os.environ.get("AUTORESEARCH_PARTITION", ""))
2003
- parser.add_argument("--gpu-partition", default=os.environ.get("AUTORESEARCH_GPU_PARTITION", ""))
2004
- parser.add_argument("--gpu-account", default=os.environ.get("AUTORESEARCH_GPU_ACCOUNT", ""))
2016
+ parser.add_argument("--account", default=os.environ.get("OUTERLOOP_ACCOUNT", ""))
2017
+ parser.add_argument("--partition", default=os.environ.get("OUTERLOOP_PARTITION", ""))
2018
+ parser.add_argument("--gpu-partition", default=os.environ.get("OUTERLOOP_GPU_PARTITION", ""))
2019
+ parser.add_argument("--gpu-account", default=os.environ.get("OUTERLOOP_GPU_ACCOUNT", ""))
2005
2020
  parser.add_argument(
2006
2021
  "--panel-minutes",
2007
2022
  type=int,
@@ -2026,11 +2041,10 @@ def main() -> int:
2026
2041
  )
2027
2042
  from outerloop.panel import panel_read_minutes
2028
2043
 
2029
- # cluster coordinates (the climb's own resolver): a GPU benchmark's
2030
- # re-measure is dispatched to the GPU lane; with none, evals run inline
2031
- dispatch = (
2032
- _dispatch_settings(args) if (args.account or args.partition or args.gpu_partition) else None
2033
- )
2044
+ # cluster coordinates (the climb's own resolver, the climb's own rule):
2045
+ # a real image dispatches evals, a GPU benchmark's to the GPU lane; with
2046
+ # no image, evals run inline. Account and partition are optional (#300).
2047
+ dispatch = _dispatch_settings(args) if (args.image and Path(args.image).is_file()) else None
2034
2048
 
2035
2049
  # a follow-up services ONE run: reproduce THAT run's author (the persisted
2036
2050
  # (backend, model) PAIR), not the current fleet default, so a codex-authored
outerloop/github.py CHANGED
@@ -43,21 +43,21 @@ DEFAULT_BOT_LOGIN = "agentic-learning-bot"
43
43
  def bot_login_from_env(default: str = DEFAULT_BOT_LOGIN) -> str:
44
44
  """The login the kernel posts and pushes as — the bot ACCOUNT under a PAT,
45
45
  the App's `<slug>[bot]` under App auth (docs/design/github-app-auth.md).
46
- `AUTORESEARCH_BOT_LOGIN` sets it for every role at once (the tick exports
46
+ `OUTERLOOP_BOT_LOGIN` sets it for every role at once (the tick exports
47
47
  it, jobs inherit it); every own-comment filter, alarm-issue lookup and
48
48
  intake-claim scan keys on this string, so it must follow the credential."""
49
- return os.environ.get("AUTORESEARCH_BOT_LOGIN", "").strip() or default
49
+ return os.environ.get("OUTERLOOP_BOT_LOGIN", "").strip() or default
50
50
 
51
51
 
52
52
  def bot_aliases_from_env() -> tuple[str, ...]:
53
53
  """Former logins the kernel posted as (comma-separated
54
- `AUTORESEARCH_BOT_ALIASES`). Every issue, claim, alarm and PR created
54
+ `OUTERLOOP_BOT_ALIASES`). Every issue, claim, alarm and PR created
55
55
  before an identity flip carries the OLD login; recognition stays keyed on
56
56
  login — never on the public markers, which anyone can paste — so the set
57
57
  of our logins is what a flip must widen (live: after the App flip the
58
58
  intake lane claimed the kernel's own research-log issue, authored by the
59
59
  PAT account)."""
60
- raw = os.environ.get("AUTORESEARCH_BOT_ALIASES", "")
60
+ raw = os.environ.get("OUTERLOOP_BOT_ALIASES", "")
61
61
  return tuple(dict.fromkeys(a.strip() for a in raw.split(",") if a.strip()))
62
62
 
63
63