outerloop-science 0.1.0.dev2__py3-none-any.whl → 0.1.0.dev4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. outerloop/__init__.py +2 -2
  2. outerloop/appauth.py +17 -0
  3. outerloop/attempt.py +376 -101
  4. outerloop/brief.py +38 -25
  5. outerloop/cli.py +104 -6
  6. outerloop/climbboard.py +67 -22
  7. outerloop/compute.py +148 -53
  8. outerloop/contract.py +8 -0
  9. outerloop/dispatch.py +63 -18
  10. outerloop/evalcache.py +147 -0
  11. outerloop/followup.py +40 -25
  12. outerloop/github.py +67 -22
  13. outerloop/harness.py +22 -47
  14. outerloop/housekeeping.py +1 -17
  15. outerloop/image.py +0 -4
  16. outerloop/init.py +45 -2
  17. outerloop/intake.py +4 -7
  18. outerloop/launchlog.py +239 -0
  19. outerloop/maintain.py +353 -0
  20. outerloop/maintain_agent_cli.py +81 -0
  21. outerloop/maintain_post_cli.py +140 -0
  22. outerloop/measure.py +6 -0
  23. outerloop/orchestrator.py +141 -31
  24. outerloop/panel.py +3 -3
  25. outerloop/review.py +4 -0
  26. outerloop/review_agent.py +7 -7
  27. outerloop/review_agent_cli.py +2 -2
  28. outerloop/review_post_cli.py +2 -2
  29. outerloop/review_summarize_cli.py +8 -6
  30. outerloop/roles.py +27 -0
  31. outerloop/rolespec.py +3 -1
  32. outerloop/steward.py +7 -14
  33. outerloop/syscall.py +261 -47
  34. outerloop/syscall_cli.py +243 -12
  35. outerloop/tick.py +274 -313
  36. outerloop/verify_agent.py +8 -6
  37. outerloop/verify_post_cli.py +2 -2
  38. outerloop/watcher.py +203 -0
  39. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev4.dist-info}/METADATA +4 -1
  40. outerloop_science-0.1.0.dev4.dist-info/RECORD +59 -0
  41. outerloop_science-0.1.0.dev2.dist-info/RECORD +0 -53
  42. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev4.dist-info}/WHEEL +0 -0
  43. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev4.dist-info}/entry_points.txt +0 -0
  44. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev4.dist-info}/licenses/LICENSE +0 -0
  45. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev4.dist-info}/licenses/NOTICE +0 -0
outerloop/init.py CHANGED
@@ -19,6 +19,7 @@ from __future__ import annotations
19
19
  import argparse
20
20
  import getpass
21
21
  import json
22
+ import logging
22
23
  import os
23
24
  import shutil
24
25
  import sys
@@ -33,6 +34,8 @@ from outerloop.cli import ENV_FILE
33
34
  from outerloop.image import ensure_image
34
35
  from outerloop.paths import write_private
35
36
 
37
+ log = logging.getLogger(__name__)
38
+
36
39
  CONFIG_DIR = ENV_FILE.parent
37
40
  DEFAULT_PAT_FILE = CONFIG_DIR / "bot_pat"
38
41
  API = "https://api.github.com"
@@ -69,6 +72,9 @@ def render_env(
69
72
  lines.append(f"OUTERLOOP_ACCOUNT={a.account}")
70
73
  if a.partition: # optional: unset lets Slurm pick its default partition
71
74
  lines.append(f"OUTERLOOP_PARTITION={a.partition}")
75
+ # the checkout stays put until the operator upgrades; `release` follows
76
+ # the release tags, `main` every merge (docs/install.md)
77
+ lines.append("OUTERLOOP_AUTO_UPDATE=off")
72
78
  if a.image:
73
79
  lines.append(f"OUTERLOOP_IMAGE={a.image}")
74
80
  elif a.uncontained:
@@ -356,8 +362,29 @@ def _owner_type(owner: str) -> str:
356
362
  try:
357
363
  with urllib.request.urlopen(req, timeout=15) as resp:
358
364
  return str(json.loads(resp.read()).get("type", ""))
359
- except Exception:
365
+ except Exception as exc:
366
+ log.warning("could not look up the account type of %s: %s", owner, exc)
367
+ return ""
368
+
369
+
370
+ def _app_owner_mismatch_note(conversion: dict[str, Any], expected_owner: str, target: str) -> str:
371
+ """When the App landed under a different account than intended — GitHub's
372
+ fallback when someone who is not an org owner tries to create an org-owned
373
+ App — explain the path that still works, since a personal App does not
374
+ install on an org repo by default. Empty when the App landed as intended."""
375
+ actual = str((conversion.get("owner") or {}).get("login", "")).strip()
376
+ if not actual or actual.casefold() == expected_owner.casefold():
360
377
  return ""
378
+ org = target.split("/")[0]
379
+ return (
380
+ f" note: the App was created under '{actual}', not '{expected_owner}'. Creating an\n"
381
+ f" App owned by an organization needs org-owner rights, so GitHub made a personal\n"
382
+ f" one — which will not install on {target} by default. To use it there anyway:\n"
383
+ f" make the App public (its Settings > 'Make public'), request its installation on\n"
384
+ f" {target}, and have an owner of {org} approve it; then rerun\n"
385
+ f" `outerloop init --force --github-app`. For lab ownership, you can later transfer\n"
386
+ f" the App to {org} from its Advanced settings — the key keeps working."
387
+ )
361
388
 
362
389
 
363
390
  def _app_failure(answers: InitAnswers, slug: str, problem: str) -> int:
@@ -369,7 +396,9 @@ def _app_failure(answers: InitAnswers, slug: str, problem: str) -> int:
369
396
  f" https://github.com/apps/{slug}/installations/new (or the App's page under\n"
370
397
  " Settings > Installations > Configure) install it on that repository and accept\n"
371
398
  " contents, issues and pull-request write access, then run\n"
372
- " `outerloop init --force --github-app` to re-check with these credentials.",
399
+ " `outerloop init --force --github-app` to re-check with these credentials.\n"
400
+ " If the App is under your personal account and the repository belongs to an\n"
401
+ " organization, make the App public first and have an org owner approve the install.",
373
402
  file=sys.stderr,
374
403
  )
375
404
  return 1
@@ -442,6 +471,9 @@ def _github_app_setup(
442
471
  pem_path, app_json = appmanifest.save_app_creds(conversion, CONFIG_DIR)
443
472
  repo = answers.target.split("/", 1)[-1]
444
473
  print(f" created App '{conversion['slug']}'; credentials in {app_json} and {pem_path} (0600)")
474
+ mismatch = _app_owner_mismatch_note(conversion, owner, answers.target)
475
+ if mismatch:
476
+ print(mismatch, file=sys.stderr)
445
477
  print()
446
478
  print(f"Step 2 of 3: install the App on {answers.target}.")
447
479
  print(f" Open {appmanifest.install_url(conversion)}")
@@ -684,6 +716,17 @@ def main(argv: list[str] | None = None) -> int:
684
716
  if login:
685
717
  write_private(env_path, render_env(answers, effective_pat, bot_login=login))
686
718
  print(f" posting as {login} (OUTERLOOP_BOT_LOGIN)")
719
+ else:
720
+ print(
721
+ " the token's login could not be read — set OUTERLOOP_BOT_LOGIN in "
722
+ f"{env_path} before `outerloop start` (the tick skips a target without it)"
723
+ )
724
+ else:
725
+ print(
726
+ " OUTERLOOP_BOT_LOGIN not recorded (the check did not pass) — rerun "
727
+ "`outerloop init --force` with network access, or set it in "
728
+ f"{env_path} (the tick skips a target without it)"
729
+ )
687
730
  else:
688
731
  print(" no PAT set — add OUTERLOOP_PAT_FILE before the agents can open PRs")
689
732
  _author_key_hint(answers)
outerloop/intake.py CHANGED
@@ -13,11 +13,11 @@ from __future__ import annotations
13
13
  import logging
14
14
  from dataclasses import dataclass
15
15
 
16
- from outerloop.brief import MAX_TASK_CHARS, _cap, _fence
16
+ from outerloop.brief import MAX_TASK_CHARS, cap, code_fence
17
17
  from outerloop.contract import Contract
18
18
  from outerloop.followup import QUALIFYING_ASSOCIATIONS
19
19
  from outerloop.github import is_own_login
20
- from outerloop.markers import has_label, has_marker, label_name, marker
20
+ from outerloop.markers import has_label, has_marker, marker
21
21
 
22
22
  log = logging.getLogger(__name__)
23
23
 
@@ -29,9 +29,6 @@ RELEASE_MARKER = marker("claim-released")
29
29
  # failure must not claim/release (and comment) forever. Same idea as the
30
30
  # steward lane's MAX_STEWARD_ATTEMPTS.
31
31
  MAX_INTAKE_ATTEMPTS = 3
32
- # steward work orders carry this label; they are the STEWARD lane's,
33
- # never the solver's (a solver climb cannot touch env paths anyway)
34
- STEWARD_LABEL = label_name("steward")
35
32
 
36
33
 
37
34
  @dataclass(frozen=True)
@@ -119,8 +116,8 @@ def issue_hypothesis(task: IssueTask) -> str:
119
116
  The author passed the standing gate, so the REQUEST is legitimate; the
120
117
  fence marks where quoted text ends and the harness's authority resumes.
121
118
  """
122
- quoted = _cap(f"{task.title}\n\n{task.body}".strip(), MAX_TASK_CHARS - 400)
123
- fence = _fence(quoted)
119
+ quoted = cap(f"{task.title}\n\n{task.body}".strip(), MAX_TASK_CHARS - 400)
120
+ fence = code_fence(quoted)
124
121
  return (
125
122
  f"A maintainer (@{task.author}) opened issue #{task.number} requesting "
126
123
  f"work on the `{task.benchmark}` benchmark. Their request:\n"
outerloop/launchlog.py ADDED
@@ -0,0 +1,239 @@
1
+ """The per-run launch ledger: append-only JSON lines in the run directory, one
2
+ record when a sleep's launches are submitted and one when each job's result
3
+ comes back at the wake. It is what `history` reads and what labels a launch in
4
+ the queue view (docs/design/session-watcher.md, "History"). Kernel-owned: the
5
+ run directory is never the session's to write."""
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ from pathlib import Path
11
+ from typing import Any
12
+
13
+ from outerloop.syscall import Launch, LaunchResult, launch_jobs
14
+
15
+ LEDGER = "launches.jsonl"
16
+ # generous: a run is depth_k launches x sleep_k sleeps x the array width, far below this
17
+ MAX_LEDGER_BYTES = 4_000_000
18
+
19
+
20
+ def _append(run_dir: Path, rows: list[dict[str, Any]]) -> None:
21
+ if not rows:
22
+ return
23
+ run_dir.mkdir(parents=True, exist_ok=True)
24
+ path = run_dir / LEDGER
25
+ # a crash mid-append leaves a torn last line; start on a fresh one so the
26
+ # torn line is the only record lost, never the next one too
27
+ torn = False
28
+ try:
29
+ with path.open("rb") as fh:
30
+ fh.seek(-1, 2)
31
+ torn = fh.read(1) != b"\n"
32
+ except OSError:
33
+ pass
34
+ with path.open("a", encoding="utf-8") as fh:
35
+ if torn:
36
+ fh.write("\n")
37
+ for row in rows:
38
+ fh.write(json.dumps(row, sort_keys=True) + "\n")
39
+
40
+
41
+ def append_submitted(
42
+ run_dir: Path, *, sleep: int, launches: tuple[Launch, ...], job_ids: list[str], at: float
43
+ ) -> None:
44
+ """One record per launch of a sleep, with the job ids it fanned out to. The
45
+ ids are positional over `launch_jobs` order, exactly as the park recorded
46
+ them; a launch whose ids are missing (an older park) gets none. A launch
47
+ already recorded under this (sleep, name) is not written again: a run
48
+ re-parks the same sleep through a multi-stage gate, and the first record
49
+ is the one with the author's words."""
50
+ known = {
51
+ (int(row.get("sleep") or 0), str(row.get("name") or ""))
52
+ for row in read_ledger(run_dir)
53
+ if row.get("event") == "submitted"
54
+ }
55
+ # one id per launch (a sweep is one Slurm job array), or one per task for
56
+ # a park recorded when arrays were separate jobs
57
+ per_launch = len(job_ids) == len(launches)
58
+ rows: list[dict[str, Any]] = []
59
+ k = 0
60
+ for launch in launches:
61
+ n = 1 if per_launch else len(launch_jobs(launch))
62
+ ids = job_ids[k : k + n]
63
+ k += n
64
+ if (sleep, launch.name) in known:
65
+ continue
66
+ rows.append(
67
+ {
68
+ "event": "submitted",
69
+ "sleep": sleep,
70
+ "name": launch.name,
71
+ "why": launch.why,
72
+ "minutes": launch.minutes,
73
+ "array": launch.array,
74
+ "concurrency": launch.concurrency,
75
+ "job_ids": list(ids),
76
+ "at": at,
77
+ }
78
+ )
79
+ _append(run_dir, rows)
80
+
81
+
82
+ def append_ended(
83
+ run_dir: Path,
84
+ *,
85
+ sleep: int,
86
+ results: tuple[LaunchResult, ...],
87
+ at: float,
88
+ elapsed_seconds: list[int | None] | None = None,
89
+ ) -> None:
90
+ """One record per job that came back at the wake, keyed to its sleep, with
91
+ how long it ran when the compute could say (aligned with `results`) and
92
+ the last line it printed — captured now, because a later launch with the
93
+ same name overwrites the job dir. A job already recorded as ended under
94
+ this (sleep, name) is not written again: a submitted park's wake and the
95
+ author's wake may both see the same jobs."""
96
+ known = {
97
+ (int(row.get("sleep") or 0), str(row.get("name") or ""))
98
+ for row in read_ledger(run_dir)
99
+ if row.get("event") == "ended"
100
+ }
101
+ _append(
102
+ run_dir,
103
+ [
104
+ {
105
+ "event": "ended",
106
+ "sleep": sleep,
107
+ "name": r.name,
108
+ "exit_code": r.exit_code,
109
+ "state": r.slurm_state,
110
+ "elapsed": (
111
+ elapsed_seconds[i]
112
+ if elapsed_seconds is not None and i < len(elapsed_seconds)
113
+ else None
114
+ ),
115
+ "last_line": last_line(r.stdout_tail),
116
+ "at": at,
117
+ }
118
+ for i, r in enumerate(results)
119
+ if (sleep, r.name) not in known
120
+ ],
121
+ )
122
+
123
+
124
+ def last_line(text: str, cap: int = 160) -> str:
125
+ """The last non-empty line of a job's output — its result line, as the
126
+ wake shows the author — on one line and bounded. "" when there is none."""
127
+ for line in reversed(text.splitlines()):
128
+ flat = " ".join(line.split())
129
+ if flat:
130
+ return flat[:cap]
131
+ return ""
132
+
133
+
134
+ def read_ledger(run_dir: Path) -> list[dict[str, Any]]:
135
+ """Every record, oldest first; a malformed line is skipped, a missing file
136
+ is an empty history."""
137
+ try:
138
+ with (run_dir / LEDGER).open("rb") as fh:
139
+ raw = fh.read(MAX_LEDGER_BYTES)
140
+ except OSError:
141
+ return []
142
+ rows: list[dict[str, Any]] = []
143
+ for line in raw.decode("utf-8", "replace").splitlines():
144
+ try:
145
+ row = json.loads(line)
146
+ except ValueError:
147
+ continue
148
+ if isinstance(row, dict):
149
+ rows.append(row)
150
+ return rows
151
+
152
+
153
+ def history(run_dir: Path) -> list[dict[str, Any]]:
154
+ """Submitted launches in order, each with the ended records of its jobs.
155
+ The identity is (sleep, name): a name is unique within one sleep only."""
156
+ rows = read_ledger(run_dir)
157
+ entries: dict[tuple[int, str], dict[str, Any]] = {}
158
+ for row in rows:
159
+ if row.get("event") != "submitted":
160
+ continue
161
+ key = (int(row.get("sleep") or 0), str(row.get("name") or ""))
162
+ entries[key] = {
163
+ "sleep": row.get("sleep"),
164
+ "name": row.get("name"),
165
+ "why": row.get("why", ""),
166
+ "minutes": row.get("minutes"),
167
+ "array": row.get("array", 1),
168
+ "concurrency": row.get("concurrency", 0),
169
+ "job_ids": list(row.get("job_ids") or []),
170
+ "submitted_at": row.get("at"),
171
+ "jobs": [],
172
+ }
173
+ for row in rows:
174
+ if row.get("event") != "ended":
175
+ continue
176
+ # an array member is `<name>.<i>`; the dot is outside the name alphabet
177
+ launch_name = str(row.get("name") or "").split(".", 1)[0]
178
+ entry = entries.get((int(row.get("sleep") or 0), launch_name))
179
+ if entry is not None:
180
+ entry["jobs"].append(
181
+ {
182
+ "name": row.get("name"),
183
+ "exit_code": row.get("exit_code"),
184
+ "state": row.get("state", ""),
185
+ "elapsed": row.get("elapsed"),
186
+ "last_line": str(row.get("last_line") or ""),
187
+ "ended_at": row.get("at"),
188
+ }
189
+ )
190
+ return list(entries.values())
191
+
192
+
193
+ def experiments_rows(run_dir: Path) -> list[dict[str, Any]]:
194
+ """The pull request's experiments table: one row per job of every launch
195
+ this run — its sleep, name and why, how it ended and how long it ran, and
196
+ the last line it printed, all from the ledger (the job dir is overwritten
197
+ by a later launch of the same name). A job not back yet says so."""
198
+ rows: list[dict[str, Any]] = []
199
+ for entry in history(run_dir):
200
+ ended = {str(j.get("name")): j for j in entry["jobs"]}
201
+ array = int(entry.get("array") or 1)
202
+ name = str(entry.get("name") or "")
203
+ jobs = [name] if array <= 1 else [f"{name}.{k}" for k in range(array)]
204
+ for job in jobs:
205
+ j = ended.get(job)
206
+ rows.append(
207
+ {
208
+ "sleep": entry.get("sleep"),
209
+ "launch": name,
210
+ "why": str(entry.get("why") or ""),
211
+ "array": array,
212
+ "concurrency": int(entry.get("concurrency") or 0),
213
+ "job": job,
214
+ "back": j is not None,
215
+ "exit_code": j.get("exit_code") if j else None,
216
+ "state": str(j.get("state") or "") if j else "",
217
+ "elapsed": j.get("elapsed") if j else None,
218
+ "result": str(j.get("last_line") or "") if j else "",
219
+ }
220
+ )
221
+ return rows
222
+
223
+
224
+ def why_by_job(run_dir: Path) -> dict[str, dict[str, Any]]:
225
+ """Slurm job id -> the launch it belongs to (name, why, sleep): how the
226
+ queue view labels a launch job for every agent."""
227
+ out: dict[str, dict[str, Any]] = {}
228
+ for row in read_ledger(run_dir):
229
+ if row.get("event") != "submitted":
230
+ continue
231
+ for job_id in row.get("job_ids") or []:
232
+ out[str(job_id)] = {
233
+ "name": row.get("name", ""),
234
+ "why": row.get("why", ""),
235
+ "sleep": row.get("sleep"),
236
+ "array": int(row.get("array") or 1),
237
+ "concurrency": int(row.get("concurrency") or 0),
238
+ }
239
+ return out