outerloop-science 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. outerloop/__init__.py +18 -0
  2. outerloop/__main__.py +3 -0
  3. outerloop/appauth.py +230 -0
  4. outerloop/appmanifest.py +203 -0
  5. outerloop/attempt.py +3784 -0
  6. outerloop/brief.py +528 -0
  7. outerloop/cli.py +621 -0
  8. outerloop/climbboard.py +1395 -0
  9. outerloop/compute.py +654 -0
  10. outerloop/contract.py +492 -0
  11. outerloop/contract_cli.py +63 -0
  12. outerloop/disk.py +164 -0
  13. outerloop/dispatch.py +631 -0
  14. outerloop/evalcache.py +147 -0
  15. outerloop/followup.py +2172 -0
  16. outerloop/github.py +1531 -0
  17. outerloop/harness.py +1435 -0
  18. outerloop/housekeeping.py +151 -0
  19. outerloop/image.py +368 -0
  20. outerloop/init.py +744 -0
  21. outerloop/intake.py +126 -0
  22. outerloop/launchlog.py +239 -0
  23. outerloop/limits.py +80 -0
  24. outerloop/maintain.py +353 -0
  25. outerloop/maintain_agent_cli.py +81 -0
  26. outerloop/maintain_post_cli.py +140 -0
  27. outerloop/markers.py +48 -0
  28. outerloop/measure.py +529 -0
  29. outerloop/orchestrator.py +2011 -0
  30. outerloop/panel.py +188 -0
  31. outerloop/paths.py +40 -0
  32. outerloop/posting.py +160 -0
  33. outerloop/progress.py +170 -0
  34. outerloop/py.typed +0 -0
  35. outerloop/review.py +615 -0
  36. outerloop/review_agent.py +263 -0
  37. outerloop/review_agent_cli.py +209 -0
  38. outerloop/review_post_cli.py +162 -0
  39. outerloop/review_summarize_cli.py +165 -0
  40. outerloop/role_runner.py +229 -0
  41. outerloop/roles.py +274 -0
  42. outerloop/rolespec.py +91 -0
  43. outerloop/runstate.py +385 -0
  44. outerloop/steward.py +845 -0
  45. outerloop/style.py +12 -0
  46. outerloop/syscall.py +1192 -0
  47. outerloop/syscall_cli.py +762 -0
  48. outerloop/tick.py +3422 -0
  49. outerloop/verifier.py +403 -0
  50. outerloop/verify_agent.py +151 -0
  51. outerloop/verify_agent_cli.py +95 -0
  52. outerloop/verify_post_cli.py +116 -0
  53. outerloop/watcher.py +203 -0
  54. outerloop_science-0.1.0.dist-info/METADATA +152 -0
  55. outerloop_science-0.1.0.dist-info/RECORD +59 -0
  56. outerloop_science-0.1.0.dist-info/WHEEL +4 -0
  57. outerloop_science-0.1.0.dist-info/entry_points.txt +2 -0
  58. outerloop_science-0.1.0.dist-info/licenses/LICENSE +202 -0
  59. outerloop_science-0.1.0.dist-info/licenses/NOTICE +5 -0
outerloop/intake.py ADDED
@@ -0,0 +1,126 @@
1
+ """The requested lane: maintainer issues become runs.
2
+
3
+ An open issue on the target repo qualifies when its author carries repo
4
+ standing (same association gate as review comments). The tick claims at most
5
+ one per cycle by commenting a claim marker, then submits a climb job whose
6
+ task carries the issue text data-fenced; the resulting PR references the
7
+ issue, and the run's report lands back on the issue thread — the loop closes
8
+ with whoever asked (docs/design/architecture.md, "The life of a run").
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import logging
14
+ from dataclasses import dataclass
15
+
16
+ from outerloop.brief import MAX_TASK_CHARS, cap, code_fence
17
+ from outerloop.contract import Contract
18
+ from outerloop.followup import QUALIFYING_ASSOCIATIONS
19
+ from outerloop.github import is_own_login
20
+ from outerloop.markers import has_label, has_marker, marker
21
+
22
+ log = logging.getLogger(__name__)
23
+
24
+ CLAIM_MARKER = marker("claimed")
25
+ # Posted (by the bot only) to undo a claim whose run never started — a failed
26
+ # submit must not strand the issue, since the claim scan skips claimed issues.
27
+ RELEASE_MARKER = marker("claim-released")
28
+ # Claim attempts per issue before intake gives up on it: a durable submit
29
+ # failure must not claim/release (and comment) forever. Same idea as the
30
+ # steward lane's MAX_STEWARD_ATTEMPTS.
31
+ MAX_INTAKE_ATTEMPTS = 3
32
+
33
+
34
+ @dataclass(frozen=True)
35
+ class IssueTask:
36
+ number: int
37
+ title: str
38
+ body: str
39
+ author: str
40
+ benchmark: str # inferred from the issue text against the contract
41
+
42
+
43
+ def infer_benchmark(text: str, contract: Contract) -> str:
44
+ """The single contract benchmark the issue names, or "" if not exactly
45
+ one — ambiguity is a human problem, not a guess."""
46
+ lowered = text.casefold()
47
+ named = [b.name for b in contract.benchmarks if b.name.casefold() in lowered]
48
+ return named[0] if len(named) == 1 else ""
49
+
50
+
51
+ def qualifying_issue(issue: dict, bot_login: str) -> bool:
52
+ author = str((issue.get("user") or {}).get("login", ""))
53
+ if is_own_login(author, bot_login):
54
+ return False # the kernel's own issues (research log, alarms) are never orders
55
+ if str(issue.get("author_association", "")) not in QUALIFYING_ASSOCIATIONS:
56
+ return False
57
+ return bool(str(issue.get("title") or "").strip())
58
+
59
+
60
+ def pick_issue(github, repo: str, contract: Contract, bot_login: str) -> IssueTask | None:
61
+ """The oldest qualifying, unclaimed issue that names exactly one
62
+ benchmark. At most one — intake is deliberately slow."""
63
+ if not bot_login.strip():
64
+ # fail closed like the steward picker: with no identity the claim
65
+ # scan below would see NO claims and re-claim every tick — an
66
+ # unbounded paid loop
67
+ log.warning("pick_issue: bot_login is blank; intake lane sits out")
68
+ return None
69
+ issues = sorted(github.list_open_issues(repo), key=lambda i: i.get("number", 0))
70
+ for issue in issues:
71
+ labels = {
72
+ str(label.get("name", "")).casefold()
73
+ for label in issue.get("labels", [])
74
+ if isinstance(label, dict)
75
+ }
76
+ if has_label(labels, "steward"):
77
+ continue # the steward lane's, never the solver's
78
+ if not qualifying_issue(issue, bot_login):
79
+ continue
80
+ number = int(issue["number"])
81
+ claimed = False
82
+ attempts = 0
83
+ for c in github.list_comments(repo, number):
84
+ author = str((c.get("user") or {}).get("login", ""))
85
+ if not is_own_login(author, bot_login):
86
+ continue # only the bot's own markers count — no forged releases
87
+ body = str(c.get("body", ""))
88
+ if has_marker(body, "claimed"):
89
+ claimed = True
90
+ attempts += 1
91
+ if has_marker(body, "claim-released"):
92
+ claimed = False
93
+ if claimed:
94
+ continue # already claimed by a run
95
+ if attempts >= MAX_INTAKE_ATTEMPTS:
96
+ log.info("issue #%s burned %d claim attempts; needs a human look", number, attempts)
97
+ continue
98
+ text = f"{issue.get('title', '')}\n{issue.get('body') or ''}"
99
+ benchmark = infer_benchmark(text, contract)
100
+ if not benchmark:
101
+ log.info("issue #%s names zero or several benchmarks; skipping", number)
102
+ continue
103
+ return IssueTask(
104
+ number=number,
105
+ title=str(issue.get("title") or ""),
106
+ body=str(issue.get("body") or ""),
107
+ author=str((issue.get("user") or {}).get("login", "")),
108
+ benchmark=benchmark,
109
+ )
110
+ return None
111
+
112
+
113
+ def issue_hypothesis(task: IssueTask) -> str:
114
+ """The task text for the brief: the maintainer's ask, data-fenced.
115
+
116
+ The author passed the standing gate, so the REQUEST is legitimate; the
117
+ fence marks where quoted text ends and the harness's authority resumes.
118
+ """
119
+ quoted = cap(f"{task.title}\n\n{task.body}".strip(), MAX_TASK_CHARS - 400)
120
+ fence = code_fence(quoted)
121
+ return (
122
+ f"A maintainer (@{task.author}) opened issue #{task.number} requesting "
123
+ f"work on the `{task.benchmark}` benchmark. Their request:\n"
124
+ f"{fence}\n{quoted}\n{fence}\n"
125
+ "Address the request's substance within the contract's rules."
126
+ )
outerloop/launchlog.py ADDED
@@ -0,0 +1,239 @@
1
+ """The per-run launch ledger: append-only JSON lines in the run directory, one
2
+ record when a sleep's launches are submitted and one when each job's result
3
+ comes back at the wake. It is what `history` reads and what labels a launch in
4
+ the queue view (docs/design/session-watcher.md, "History"). Kernel-owned: the
5
+ run directory is never the session's to write."""
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ from pathlib import Path
11
+ from typing import Any
12
+
13
+ from outerloop.syscall import Launch, LaunchResult, launch_jobs
14
+
15
+ LEDGER = "launches.jsonl"
16
+ # generous: a run is depth_k launches x sleep_k sleeps x the array width, far below this
17
+ MAX_LEDGER_BYTES = 4_000_000
18
+
19
+
20
+ def _append(run_dir: Path, rows: list[dict[str, Any]]) -> None:
21
+ if not rows:
22
+ return
23
+ run_dir.mkdir(parents=True, exist_ok=True)
24
+ path = run_dir / LEDGER
25
+ # a crash mid-append leaves a torn last line; start on a fresh one so the
26
+ # torn line is the only record lost, never the next one too
27
+ torn = False
28
+ try:
29
+ with path.open("rb") as fh:
30
+ fh.seek(-1, 2)
31
+ torn = fh.read(1) != b"\n"
32
+ except OSError:
33
+ pass
34
+ with path.open("a", encoding="utf-8") as fh:
35
+ if torn:
36
+ fh.write("\n")
37
+ for row in rows:
38
+ fh.write(json.dumps(row, sort_keys=True) + "\n")
39
+
40
+
41
+ def append_submitted(
42
+ run_dir: Path, *, sleep: int, launches: tuple[Launch, ...], job_ids: list[str], at: float
43
+ ) -> None:
44
+ """One record per launch of a sleep, with the job ids it fanned out to. The
45
+ ids are positional over `launch_jobs` order, exactly as the park recorded
46
+ them; a launch whose ids are missing (an older park) gets none. A launch
47
+ already recorded under this (sleep, name) is not written again: a run
48
+ re-parks the same sleep through a multi-stage gate, and the first record
49
+ is the one with the author's words."""
50
+ known = {
51
+ (int(row.get("sleep") or 0), str(row.get("name") or ""))
52
+ for row in read_ledger(run_dir)
53
+ if row.get("event") == "submitted"
54
+ }
55
+ # one id per launch (a sweep is one Slurm job array), or one per task for
56
+ # a park recorded when arrays were separate jobs
57
+ per_launch = len(job_ids) == len(launches)
58
+ rows: list[dict[str, Any]] = []
59
+ k = 0
60
+ for launch in launches:
61
+ n = 1 if per_launch else len(launch_jobs(launch))
62
+ ids = job_ids[k : k + n]
63
+ k += n
64
+ if (sleep, launch.name) in known:
65
+ continue
66
+ rows.append(
67
+ {
68
+ "event": "submitted",
69
+ "sleep": sleep,
70
+ "name": launch.name,
71
+ "why": launch.why,
72
+ "minutes": launch.minutes,
73
+ "array": launch.array,
74
+ "concurrency": launch.concurrency,
75
+ "job_ids": list(ids),
76
+ "at": at,
77
+ }
78
+ )
79
+ _append(run_dir, rows)
80
+
81
+
82
+ def append_ended(
83
+ run_dir: Path,
84
+ *,
85
+ sleep: int,
86
+ results: tuple[LaunchResult, ...],
87
+ at: float,
88
+ elapsed_seconds: list[int | None] | None = None,
89
+ ) -> None:
90
+ """One record per job that came back at the wake, keyed to its sleep, with
91
+ how long it ran when the compute could say (aligned with `results`) and
92
+ the last line it printed — captured now, because a later launch with the
93
+ same name overwrites the job dir. A job already recorded as ended under
94
+ this (sleep, name) is not written again: a submitted park's wake and the
95
+ author's wake may both see the same jobs."""
96
+ known = {
97
+ (int(row.get("sleep") or 0), str(row.get("name") or ""))
98
+ for row in read_ledger(run_dir)
99
+ if row.get("event") == "ended"
100
+ }
101
+ _append(
102
+ run_dir,
103
+ [
104
+ {
105
+ "event": "ended",
106
+ "sleep": sleep,
107
+ "name": r.name,
108
+ "exit_code": r.exit_code,
109
+ "state": r.slurm_state,
110
+ "elapsed": (
111
+ elapsed_seconds[i]
112
+ if elapsed_seconds is not None and i < len(elapsed_seconds)
113
+ else None
114
+ ),
115
+ "last_line": last_line(r.stdout_tail),
116
+ "at": at,
117
+ }
118
+ for i, r in enumerate(results)
119
+ if (sleep, r.name) not in known
120
+ ],
121
+ )
122
+
123
+
124
+ def last_line(text: str, cap: int = 160) -> str:
125
+ """The last non-empty line of a job's output — its result line, as the
126
+ wake shows the author — on one line and bounded. "" when there is none."""
127
+ for line in reversed(text.splitlines()):
128
+ flat = " ".join(line.split())
129
+ if flat:
130
+ return flat[:cap]
131
+ return ""
132
+
133
+
134
+ def read_ledger(run_dir: Path) -> list[dict[str, Any]]:
135
+ """Every record, oldest first; a malformed line is skipped, a missing file
136
+ is an empty history."""
137
+ try:
138
+ with (run_dir / LEDGER).open("rb") as fh:
139
+ raw = fh.read(MAX_LEDGER_BYTES)
140
+ except OSError:
141
+ return []
142
+ rows: list[dict[str, Any]] = []
143
+ for line in raw.decode("utf-8", "replace").splitlines():
144
+ try:
145
+ row = json.loads(line)
146
+ except ValueError:
147
+ continue
148
+ if isinstance(row, dict):
149
+ rows.append(row)
150
+ return rows
151
+
152
+
153
+ def history(run_dir: Path) -> list[dict[str, Any]]:
154
+ """Submitted launches in order, each with the ended records of its jobs.
155
+ The identity is (sleep, name): a name is unique within one sleep only."""
156
+ rows = read_ledger(run_dir)
157
+ entries: dict[tuple[int, str], dict[str, Any]] = {}
158
+ for row in rows:
159
+ if row.get("event") != "submitted":
160
+ continue
161
+ key = (int(row.get("sleep") or 0), str(row.get("name") or ""))
162
+ entries[key] = {
163
+ "sleep": row.get("sleep"),
164
+ "name": row.get("name"),
165
+ "why": row.get("why", ""),
166
+ "minutes": row.get("minutes"),
167
+ "array": row.get("array", 1),
168
+ "concurrency": row.get("concurrency", 0),
169
+ "job_ids": list(row.get("job_ids") or []),
170
+ "submitted_at": row.get("at"),
171
+ "jobs": [],
172
+ }
173
+ for row in rows:
174
+ if row.get("event") != "ended":
175
+ continue
176
+ # an array member is `<name>.<i>`; the dot is outside the name alphabet
177
+ launch_name = str(row.get("name") or "").split(".", 1)[0]
178
+ entry = entries.get((int(row.get("sleep") or 0), launch_name))
179
+ if entry is not None:
180
+ entry["jobs"].append(
181
+ {
182
+ "name": row.get("name"),
183
+ "exit_code": row.get("exit_code"),
184
+ "state": row.get("state", ""),
185
+ "elapsed": row.get("elapsed"),
186
+ "last_line": str(row.get("last_line") or ""),
187
+ "ended_at": row.get("at"),
188
+ }
189
+ )
190
+ return list(entries.values())
191
+
192
+
193
+ def experiments_rows(run_dir: Path) -> list[dict[str, Any]]:
194
+ """The pull request's experiments table: one row per job of every launch
195
+ this run — its sleep, name and why, how it ended and how long it ran, and
196
+ the last line it printed, all from the ledger (the job dir is overwritten
197
+ by a later launch of the same name). A job not back yet says so."""
198
+ rows: list[dict[str, Any]] = []
199
+ for entry in history(run_dir):
200
+ ended = {str(j.get("name")): j for j in entry["jobs"]}
201
+ array = int(entry.get("array") or 1)
202
+ name = str(entry.get("name") or "")
203
+ jobs = [name] if array <= 1 else [f"{name}.{k}" for k in range(array)]
204
+ for job in jobs:
205
+ j = ended.get(job)
206
+ rows.append(
207
+ {
208
+ "sleep": entry.get("sleep"),
209
+ "launch": name,
210
+ "why": str(entry.get("why") or ""),
211
+ "array": array,
212
+ "concurrency": int(entry.get("concurrency") or 0),
213
+ "job": job,
214
+ "back": j is not None,
215
+ "exit_code": j.get("exit_code") if j else None,
216
+ "state": str(j.get("state") or "") if j else "",
217
+ "elapsed": j.get("elapsed") if j else None,
218
+ "result": str(j.get("last_line") or "") if j else "",
219
+ }
220
+ )
221
+ return rows
222
+
223
+
224
+ def why_by_job(run_dir: Path) -> dict[str, dict[str, Any]]:
225
+ """Slurm job id -> the launch it belongs to (name, why, sleep): how the
226
+ queue view labels a launch job for every agent."""
227
+ out: dict[str, dict[str, Any]] = {}
228
+ for row in read_ledger(run_dir):
229
+ if row.get("event") != "submitted":
230
+ continue
231
+ for job_id in row.get("job_ids") or []:
232
+ out[str(job_id)] = {
233
+ "name": row.get("name", ""),
234
+ "why": row.get("why", ""),
235
+ "sleep": row.get("sleep"),
236
+ "array": int(row.get("array") or 1),
237
+ "concurrency": int(row.get("concurrency") or 0),
238
+ }
239
+ return out
outerloop/limits.py ADDED
@@ -0,0 +1,80 @@
1
+ """Effective session/job limits: contract wishes clamped by our ceilings.
2
+
3
+ Contracts live in TARGET repos and are untrusted input (contract.py's
4
+ threat model). A target may therefore SHAPE the orchestrator's spend on it
5
+ — shorter sessions, tighter job walltimes — but must never be able to
6
+ raise it: every contract value is clamped into [floor, ceiling], and the
7
+ ceilings are code on the orchestrator side, not configuration a target
8
+ can reach. Absent values fall back to the defaults the pilot has run with
9
+ all along.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from dataclasses import dataclass
15
+ from typing import Any
16
+
17
+ # (default, floor, ceiling) per knob. Floors keep a hostile-or-typo'd
18
+ # contract from starving runs into uselessness (a 1-turn session still
19
+ # spends money and reports nothing). CEILING == DEFAULT, deliberately:
20
+ # contracts are merged by TARGET-repo maintainers, not by us, so any
21
+ # ceiling above the default would let them raise our spend — the knobs
22
+ # shape strictly downward. Raising a target's budget is an
23
+ # orchestrator-side decision (config we control), not a contract edit.
24
+ # Raised from 60/60/90/60 on 2026-08-09 (maintainer decision): the first
25
+ # steward work order to BUILD an env burned its full 60-turn budget mid-
26
+ # work — session budgets sized for solver tweaks starve construction work.
27
+ # floor = session floor + overhead + self-deadline margin: even at the
28
+ # floors, a session must fit inside its job with the ending's runway.
29
+ # Public: the tick's OUTERLOOP_MAX_JOB_MINUTES knob floors here too.
30
+ ATTEMPT_JOB_MINUTES_FLOOR = 40
31
+
32
+ # Public: the tick shrinks a capped job's session with the same floor the
33
+ # contract clamp uses.
34
+ SESSION_MINUTES_FLOOR = 10
35
+
36
+ _BOUNDS: dict[str, tuple[int, int, int]] = {
37
+ "session_max_turns": (120, 10, 120),
38
+ "session_minutes": (90, SESSION_MINUTES_FLOOR, 90),
39
+ "attempt_job_minutes": (120, ATTEMPT_JOB_MINUTES_FLOOR, 120),
40
+ "followup_job_minutes": (90, 20, 90),
41
+ }
42
+
43
+ # A climb job must outlive its session long enough for the orchestrator's
44
+ # own work around it (clone, two evals, publish, ending writes). Public:
45
+ # the tick's cap warning uses it as the no-runway threshold too.
46
+ ATTEMPT_OVERHEAD_MINUTES = 20
47
+
48
+
49
+ @dataclass(frozen=True)
50
+ class EffectiveLimits:
51
+ session_max_turns: int
52
+ session_minutes: int
53
+ attempt_job_minutes: int
54
+ followup_job_minutes: int
55
+
56
+
57
+ def _clamp(name: str, value: int | None) -> int:
58
+ default, floor, ceiling = _BOUNDS[name]
59
+ if value is None:
60
+ return default
61
+ return max(floor, min(int(value), ceiling))
62
+
63
+
64
+ def effective_limits(budgets: Any = None) -> EffectiveLimits:
65
+ """Resolve a contract's optional budget knobs into enforceable limits.
66
+
67
+ `budgets` is the contract's Budgets model (or None for pure defaults);
68
+ unknown/absent attributes read as None. The session is finally shrunk
69
+ to fit inside the climb job with room for the orchestrator's overhead —
70
+ a session that outlives its job ends as a kill, not a report.
71
+ """
72
+ values = {
73
+ name: _clamp(name, getattr(budgets, name, None) if budgets is not None else None)
74
+ for name in _BOUNDS
75
+ }
76
+ max_session = values["attempt_job_minutes"] - ATTEMPT_OVERHEAD_MINUTES
77
+ if values["session_minutes"] > max_session:
78
+ floor = _BOUNDS["session_minutes"][1]
79
+ values["session_minutes"] = max(floor, max_session)
80
+ return EffectiveLimits(**values)