outerloop-science 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. outerloop/__init__.py +18 -0
  2. outerloop/__main__.py +3 -0
  3. outerloop/appauth.py +230 -0
  4. outerloop/appmanifest.py +203 -0
  5. outerloop/attempt.py +3784 -0
  6. outerloop/brief.py +528 -0
  7. outerloop/cli.py +621 -0
  8. outerloop/climbboard.py +1395 -0
  9. outerloop/compute.py +654 -0
  10. outerloop/contract.py +492 -0
  11. outerloop/contract_cli.py +63 -0
  12. outerloop/disk.py +164 -0
  13. outerloop/dispatch.py +631 -0
  14. outerloop/evalcache.py +147 -0
  15. outerloop/followup.py +2172 -0
  16. outerloop/github.py +1531 -0
  17. outerloop/harness.py +1435 -0
  18. outerloop/housekeeping.py +151 -0
  19. outerloop/image.py +368 -0
  20. outerloop/init.py +744 -0
  21. outerloop/intake.py +126 -0
  22. outerloop/launchlog.py +239 -0
  23. outerloop/limits.py +80 -0
  24. outerloop/maintain.py +353 -0
  25. outerloop/maintain_agent_cli.py +81 -0
  26. outerloop/maintain_post_cli.py +140 -0
  27. outerloop/markers.py +48 -0
  28. outerloop/measure.py +529 -0
  29. outerloop/orchestrator.py +2011 -0
  30. outerloop/panel.py +188 -0
  31. outerloop/paths.py +40 -0
  32. outerloop/posting.py +160 -0
  33. outerloop/progress.py +170 -0
  34. outerloop/py.typed +0 -0
  35. outerloop/review.py +615 -0
  36. outerloop/review_agent.py +263 -0
  37. outerloop/review_agent_cli.py +209 -0
  38. outerloop/review_post_cli.py +162 -0
  39. outerloop/review_summarize_cli.py +165 -0
  40. outerloop/role_runner.py +229 -0
  41. outerloop/roles.py +274 -0
  42. outerloop/rolespec.py +91 -0
  43. outerloop/runstate.py +385 -0
  44. outerloop/steward.py +845 -0
  45. outerloop/style.py +12 -0
  46. outerloop/syscall.py +1192 -0
  47. outerloop/syscall_cli.py +762 -0
  48. outerloop/tick.py +3422 -0
  49. outerloop/verifier.py +403 -0
  50. outerloop/verify_agent.py +151 -0
  51. outerloop/verify_agent_cli.py +95 -0
  52. outerloop/verify_post_cli.py +116 -0
  53. outerloop/watcher.py +203 -0
  54. outerloop_science-0.1.0.dist-info/METADATA +152 -0
  55. outerloop_science-0.1.0.dist-info/RECORD +59 -0
  56. outerloop_science-0.1.0.dist-info/WHEEL +4 -0
  57. outerloop_science-0.1.0.dist-info/entry_points.txt +2 -0
  58. outerloop_science-0.1.0.dist-info/licenses/LICENSE +202 -0
  59. outerloop_science-0.1.0.dist-info/licenses/NOTICE +5 -0
outerloop/attempt.py ADDED
@@ -0,0 +1,3784 @@
1
+ """One live climb, end to end: clone → attempt_once → commit/push/PR → report.
2
+
3
+ This is the glue `orchestrator.attempt_once` deliberately does not own: the git
4
+ side (bot-auth clone, veto-checked commit, push, PR) and the run's durable
5
+ record. One invocation = one run = at most one PR.
6
+
7
+ Credential separation holds throughout: the bot PAT is read orchestrator-side
8
+ and used only by Workspace network calls and the PR client, after the session
9
+ has ended; the session sees only its own capped API key inside its container.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import argparse
15
+ import contextlib
16
+ import json
17
+ import logging
18
+ import os
19
+ import re
20
+ import shutil
21
+ import time
22
+ from collections.abc import Callable, Iterable
23
+ from dataclasses import dataclass
24
+ from dataclasses import replace as dc_replace
25
+ from functools import partial
26
+ from pathlib import Path
27
+ from typing import Any, cast
28
+
29
+ from outerloop.appauth import add_credential_args, resolve_bot_auth
30
+ from outerloop.brief import BudgetState, distill_lessons
31
+ from outerloop.compute import LocalCompute
32
+ from outerloop.contract import Benchmark, Contract, contract_text_in_tree, load_contract
33
+ from outerloop.dispatch import (
34
+ Snapshot,
35
+ afterany_ids,
36
+ drop_snapshot,
37
+ should_dispatch,
38
+ snapshot_tree,
39
+ )
40
+ from outerloop.evalcache import seed_dir
41
+ from outerloop.github import (
42
+ GitError,
43
+ GitHubClient,
44
+ TokenProvider,
45
+ Workspace,
46
+ contract_at,
47
+ ensure_regular_git_dir,
48
+ git_identity,
49
+ )
50
+ from outerloop.harness import Harness, SessionResult, default_binary, redact
51
+ from outerloop.launchlog import append_ended, append_submitted, experiments_rows
52
+ from outerloop.markers import has_marker
53
+ from outerloop.measure import DispatchedMeasurer, DispatchSettings
54
+ from outerloop.orchestrator import (
55
+ AttemptResult,
56
+ EvalError,
57
+ Measurer,
58
+ RunConfig,
59
+ RunParked,
60
+ _benchmark,
61
+ attempt_once,
62
+ pr_body,
63
+ resume_attempt,
64
+ )
65
+ from outerloop.panel import PanelLens, PanelVerdict, run_panel
66
+ from outerloop.paths import CONFIG_DIR
67
+ from outerloop.progress import (
68
+ PROGRESS_PATHS,
69
+ load_leader,
70
+ update_leader,
71
+ write_progress,
72
+ )
73
+ from outerloop.review import PullRequest
74
+ from outerloop.role_runner import build_harness, role_key
75
+ from outerloop.roles import author_spec
76
+ from outerloop.rolespec import RoleSpec
77
+ from outerloop.runstate import (
78
+ ABORTED,
79
+ BUDGET_EXHAUSTED,
80
+ ENDED,
81
+ IN_REVIEW,
82
+ NEGATIVE_RESULT,
83
+ STUCK,
84
+ WAITING,
85
+ RunRecord,
86
+ list_runs,
87
+ load_record,
88
+ save_record,
89
+ stamp_outage,
90
+ )
91
+ from outerloop.runstate import (
92
+ run_dir as run_dir_of,
93
+ )
94
+ from outerloop.syscall import (
95
+ CHANNEL_DIR_NAMES,
96
+ MAX_ARTIFACT_BYTES,
97
+ SyscallRequest,
98
+ channel_dir,
99
+ launch_task_ids,
100
+ tool_update_note,
101
+ )
102
+ from outerloop.syscall import ensure_excluded as syscall_excluded
103
+ from outerloop.syscall import install_tool as syscall_install_tool
104
+ from outerloop.syscall import refresh_tool as syscall_refresh_tool
105
+ from outerloop.syscall import write_budget as syscall_write_budget
106
+ from outerloop.syscall import write_siblings as syscall_write_siblings
107
+ from outerloop.verifier import MAX_CLAIM_CHARS
108
+
109
+ log = logging.getLogger(__name__)
110
+
111
+ # Where a climb job reads its keys unless the CLI flags say otherwise. The
112
+ # tick preflights the panel key (and compares it against the author key —
113
+ # role separation) before claiming/submitting.
114
+ PANEL_KEY_DEFAULT = str(CONFIG_DIR / "verifier_key")
115
+ # One rule for author keys: ~/.config/outerloop/<backend>_key. `harness_key` is
116
+ # the pre-rename name of the claude key; it is still read, never written.
117
+ CLAUDE_KEY_DEFAULT = str(CONFIG_DIR / "claude_key")
118
+ HARNESS_KEY_DEFAULT = str(CONFIG_DIR / "harness_key") # legacy claude key
119
+ CODEX_KEY_DEFAULT = str(CONFIG_DIR / "codex_key")
120
+
121
+
122
+ def resolve_author_key_file(backend: str, explicit: str = "") -> str:
123
+ """The author key file for `backend`. Per-backend keys COEXIST (claude's and
124
+ codex's both on disk), selected by backend — so the author backend is a
125
+ config choice, not a key swap, and an in-flight run of either backend can
126
+ still be woken/serviced after a fleet flip. An explicit path always wins;
127
+ otherwise the per-backend env var, then the default path. For claude the
128
+ legacy spellings (`AUTORESEARCH_HARNESS_KEY_FILE`, read through its bridged
129
+ `OUTERLOOP_` twin, and `harness_key`) are still
130
+ honored, so a machine set up before the rename keeps working; the new name
131
+ wins when both exist. The result is always ~-expanded, so every caller gets
132
+ a real path (an env value like "~/.config/..." must not reach the token
133
+ provider verbatim)."""
134
+ if not explicit:
135
+ if backend == "codex":
136
+ explicit = os.environ.get("OUTERLOOP_CODEX_KEY_FILE") or CODEX_KEY_DEFAULT
137
+ else:
138
+ explicit = (
139
+ os.environ.get("OUTERLOOP_CLAUDE_KEY_FILE")
140
+ or os.environ.get("OUTERLOOP_HARNESS_KEY_FILE")
141
+ or ""
142
+ )
143
+ if not explicit:
144
+ explicit = CLAUDE_KEY_DEFAULT
145
+ if not os.path.exists(os.path.expanduser(explicit)) and os.path.exists(
146
+ os.path.expanduser(HARNESS_KEY_DEFAULT)
147
+ ):
148
+ explicit = HARNESS_KEY_DEFAULT
149
+ return os.path.expanduser(explicit)
150
+
151
+
152
+ def codex_author_config_error(backend: str, model: str, image: str) -> str:
153
+ """Why a codex author would die at startup ("" when it won't). Validates the
154
+ EFFECTIVE (backend, model) — the fresh climb passes args; a wake/follow-up
155
+ passes the PARKED RUN's persisted pair — so backend and model are checked as
156
+ a unit and never a fleet backend against a run's model. codex writes+executes,
157
+ so it must be contained (--image) and needs a non-claude model."""
158
+ if backend not in ("claude", "codex"):
159
+ # a typo'd OUTERLOOP_AUTHOR_BACKEND passes the env DEFAULT silently
160
+ # (argparse validates the flag, not its default) and the climb rejects it
161
+ # at build_harness — catch it on the tick host so a claimed intake
162
+ # issue never strands on it
163
+ return f"unknown author backend {backend!r} (expected 'claude' or 'codex')"
164
+ if backend == "claude":
165
+ # symmetric to the codex check: a claude harness 404s on a non-claude
166
+ # model (e.g. OUTERLOOP_AUTHOR_MODEL left on a codex id while the
167
+ # backend is claude) — catch that misconfig before spend
168
+ if model and not model.startswith("claude"):
169
+ return f"author-backend claude needs a claude model (got {model!r})"
170
+ return ""
171
+ if not image:
172
+ return "author-backend codex requires --image (it runs contained)"
173
+ if not model or model.startswith("claude"):
174
+ return (
175
+ "author-backend codex needs a codex/openai model "
176
+ f"(e.g. gpt-5.6-terra), not the claude default (got {model!r})"
177
+ )
178
+ return ""
179
+
180
+
181
+ def resume_author(record: object, fleet_model: str) -> tuple[str, str, str]:
182
+ """The (backend, model, key_file) a wake/follow-up must reproduce for a parked
183
+ run — all from the RECORD, not the current fleet.
184
+
185
+ An empty backend is a legacy record (written before the field) and is
186
+ therefore CLAUDE, never the fleet default; the model pairs with that backend
187
+ (a claude backend falls back to the claude default, a codex backend to the
188
+ fleet model only as a last resort — codex records always carry their model);
189
+ the key file is the exact resolved path the run used (so an explicit
190
+ --key-file survives), falling back to the per-backend resolution for legacy
191
+ records that never recorded it."""
192
+ backend = getattr(record, "author_backend", "") or "claude"
193
+ model = getattr(record, "author_model", "") or (
194
+ "claude-opus-5" if backend == "claude" else fleet_model
195
+ )
196
+ key_file = getattr(record, "author_key_file", "") or resolve_author_key_file(backend)
197
+ return backend, model, key_file
198
+
199
+
200
+ class WorkspaceDrift(RuntimeError):
201
+ """The tree changed between measurement and commit."""
202
+
203
+
204
+ def target_clone_url(target: str) -> str:
205
+ """The canonical HTTPS clone URL for `owner/repo`. The one source of truth
206
+ for where a run's git pushes go — derived from the target, never read from
207
+ the session-writable `remote.origin.url`."""
208
+ return f"https://github.com/{target}.git"
209
+
210
+
211
+ def _blessed_head(ws: Workspace, result: Any, contract: Any) -> str:
212
+ """The pushed PR head the tick may later self-merge — only when this
213
+ publish was under merge:auto with a CLEAN panel (#171's arming
214
+ condition); "" otherwise. Best-effort: an unreadable HEAD blesses
215
+ nothing (never arm on doubt)."""
216
+ if not (
217
+ result.panel_rounds > 0
218
+ and not (result.panel_blocking_open or result.panel_degraded)
219
+ and getattr(contract, "merge", "manual") == "auto"
220
+ ):
221
+ return ""
222
+ try:
223
+ return ws.git("rev-parse", "HEAD").strip()
224
+ except Exception as exc:
225
+ log.warning("could not read HEAD; not arming self-merge: %s", exc)
226
+ return ""
227
+
228
+
229
+ def _arm_unless_base_moved(
230
+ github: GitHubClient,
231
+ ws: Workspace,
232
+ target: str,
233
+ pr_number: str,
234
+ base_branch: str,
235
+ measured_base_sha: str,
236
+ secrets: tuple[str, ...],
237
+ merge_mode: str = "manual",
238
+ panel_ran: bool = False,
239
+ ) -> None:
240
+ """Arm auto-merge only while origin/<base_branch> still equals the base the
241
+ claim was measured against. A moved base still OPENS the PR — review owns
242
+ staleness — but never ARMS it: merging a tree whose gate/suite/panel read
243
+ is stale must be a human's deliberate act, not an armed automation. A
244
+ failed freshness fetch also declines to arm (fail-safe: un-armed is just a
245
+ normal PR). Best-effort throughout, like arming itself."""
246
+
247
+ def _check_and_arm() -> None:
248
+ ws.git_network("fetch", str(ws.url or ws.remote_url()), base_branch)
249
+ fresh = ws.git("rev-parse", "FETCH_HEAD").strip()
250
+ if fresh != measured_base_sha:
251
+ log.info(
252
+ "not arming auto-merge on %s#%s: %s moved since the claim was "
253
+ "measured (%s -> %s); a human merges this one",
254
+ target,
255
+ pr_number,
256
+ base_branch,
257
+ measured_base_sha[:12],
258
+ fresh[:12],
259
+ )
260
+ return
261
+ if merge_mode == "auto" and not panel_ran:
262
+ # the dial's own precondition: auto means GATE+PANEL clean, so a
263
+ # publish that ran no panel must not self-merge — fall back to
264
+ # the manual guard and say so (terra #171: a panel-less
265
+ # deployment could otherwise self-merge on the metric gate alone)
266
+ log.warning(
267
+ "merge mode auto on %s#%s but no panel ran this attempt; "
268
+ "arming manual-mode instead",
269
+ target,
270
+ pr_number,
271
+ )
272
+ if merge_mode == "auto" and panel_ran:
273
+ # the contract's autonomy dial: the owner opted this repo into
274
+ # self-merging gate-clean PRs — arm, or merge directly when
275
+ # nothing is pending to arm against
276
+ github.arm_auto_merge_auto_mode(target, int(pr_number))
277
+ else:
278
+ github.arm_auto_merge_when_review_required(target, int(pr_number))
279
+
280
+ _best_effort("auto-merge arming", _check_and_arm, secrets)
281
+
282
+
283
+ def _title_pair(a: float, b: float) -> str:
284
+ """Compact but never ambiguous: widen precision until the two numbers
285
+ render differently (a title reading '10.00 -> 10.00' looks like no
286
+ change even when the improvement is real)."""
287
+ for precision in range(4, 12):
288
+ fa, fb = f"{a:.{precision}g}", f"{b:.{precision}g}"
289
+ if fa != fb:
290
+ return f"{fa} -> {fb}"
291
+ return f"{a} -> {b}"
292
+
293
+
294
+ RULER = (
295
+ "The metric is computed by the contract's eval command over a frozen "
296
+ "instance pool. Your claim is verified by the orchestrator re-running "
297
+ "that exact command on your tree — and again by CI after the PR opens. "
298
+ "Only changes inside the contract's allowed paths are ever measured."
299
+ )
300
+
301
+ _ENDINGS_BY_OUTCOME = {
302
+ "no-improvement": NEGATIVE_RESULT,
303
+ # the improvement was real but bought by regressing a sibling benchmark —
304
+ # an honest negative with a named cause, not a malfunction
305
+ "suite-regression": NEGATIVE_RESULT,
306
+ "session-error": ABORTED,
307
+ "session-budget": BUDGET_EXHAUSTED,
308
+ "session-outage": STUCK, # infrastructure failure, nothing about the run
309
+ "eval-error": ABORTED,
310
+ "scope-violation": ABORTED,
311
+ }
312
+
313
+
314
+ @dataclass(frozen=True)
315
+ class AttemptOutcome:
316
+ run_id: str
317
+ outcome: str
318
+ pr_url: str = ""
319
+ report_path: str = ""
320
+
321
+
322
+ def _best_effort(what: str, fn: Callable[[], object], secrets: tuple[str, ...] = ()) -> bool:
323
+ """One ending step; a failure is logged, never raised.
324
+
325
+ The terminal sequence (record, report, issue post) must degrade
326
+ independently: a full disk must not block the GitHub post, and a network
327
+ failure must not block the record.
328
+ """
329
+ try:
330
+ fn()
331
+ return True
332
+ except Exception as exc:
333
+ log.warning("%s failed: %s", what, redact(f"{type(exc).__name__}: {exc}", secrets))
334
+ return False
335
+
336
+
337
+ def _clear_stage(record: RunRecord) -> RunRecord:
338
+ """Strip the WAITING-only bookkeeping from a record leaving `waiting` for a
339
+ terminal state. Otherwise a dispatched run's `stage`, `deadline`, and
340
+ especially `wake_attempts` ride into `in-review`, where in-review follow-up
341
+ servicing reuses `wake_attempts` as its OWN retry cap — so a run that woke
342
+ once would reach review with a shrunk follow-up budget."""
343
+ # the run's spend survives the wipe: terminal reporting (the climb
344
+ # board) reads it after the transition
345
+ kept = {k: record.stage[k] for k in ("gpu_hours_used",) if record.stage and k in record.stage}
346
+ return dc_replace(
347
+ record,
348
+ stage=kept,
349
+ experiment_job_id="",
350
+ deadline=0.0,
351
+ terminal_seen=0.0,
352
+ wake_attempts=0,
353
+ )
354
+
355
+
356
+ def _post_issue_finished(
357
+ github: GitHubClient,
358
+ target: str,
359
+ issue_number: int,
360
+ run_id: str,
361
+ outcome_name: str,
362
+ pr_url: str,
363
+ summary: str,
364
+ secrets: tuple[str, ...],
365
+ ) -> None:
366
+ """Post a run's terminal result back to the issue that requested it. When
367
+ the run ends WITHOUT a PR (a negative or an error), include the
368
+ `RELEASE_MARKER` so `intake.pick_issue` can re-select the issue — a comment
369
+ alone does NOT un-claim it. An improved run KEEPS the claim: its PR
370
+ (`Addresses #N`) is the ongoing work, and `followup` releases the claim if
371
+ that PR later closes unmerged."""
372
+ if not issue_number:
373
+ return
374
+ from outerloop.intake import RELEASE_MARKER
375
+
376
+ link = f"\n\nPull request: {pr_url}" if pr_url else ""
377
+ # no PR opened -> nothing will ever resolve this issue, so free the claim
378
+ # (bounded by intake's per-issue attempt cap).
379
+ release = f"{RELEASE_MARKER}\n" if not pr_url else ""
380
+ _best_effort(
381
+ "issue report",
382
+ lambda: github.comment(
383
+ target,
384
+ issue_number,
385
+ f"{release}Run `{run_id}` finished ({outcome_name}).{link}\n\n{summary}",
386
+ ),
387
+ secrets,
388
+ )
389
+
390
+
391
+ # A parked run's deadline is the FLOOR beneath the afterany wake: submit +
392
+ # eval walltime + a generous queue/grace allowance. It must exceed the time a
393
+ # healthy eval can legitimately sit queued-then-running, or `tick._sweep_one`
394
+ # would cancel a still-queued job as "unschedulable".
395
+ PARK_QUEUE_SLACK_MIN = 12 * 60
396
+ # a jobless checkpoint sleep only needs to survive to the next sweep pass:
397
+ # one cadence + coalescing headroom, not queue slack
398
+ CHECKPOINT_SLEEP_SLACK_MIN = 45
399
+
400
+
401
+ def _park_run(
402
+ run_root: Path,
403
+ record: RunRecord,
404
+ parked: RunParked,
405
+ candidate_ref: str,
406
+ eval_minutes: int | None,
407
+ now: float,
408
+ secrets: tuple[str, ...] = (),
409
+ keep_wake_attempts: bool = False,
410
+ base_branch: str = "main",
411
+ panel_reads: int = 0,
412
+ dispatch: DispatchSettings | None = None,
413
+ ) -> None:
414
+ """Persist a dispatched climb's re-entry point as a WAITING record: the
415
+ committed shas, drawn seeds, candidate snapshot ref, and afterany set a
416
+ fresh process reconstructs the measure-and-decide phase from. The caller
417
+ passes the EXACT `candidate_ref` it will keep alive (never re-derive it from
418
+ the commit — two snapshots can share a commit)."""
419
+ from outerloop.dispatch import effective_eval_minutes
420
+
421
+ job_ids = afterany_ids(parked.afterany)
422
+ stage: dict[str, object] = {
423
+ "phase": parked.phase,
424
+ "base_sha": parked.base_sha,
425
+ "candidate_sha": parked.candidate_sha,
426
+ "candidate_ref": candidate_ref,
427
+ "seed": parked.seed,
428
+ "suite_seed": parked.suite_seed,
429
+ "afterany": parked.afterany,
430
+ "launch_afterany": parked.launch_afterany,
431
+ # the branch the run targets, so a wake opens its PR against the SAME
432
+ # branch a non-default `--base-branch` selected — the wake CLI otherwise
433
+ # defaults to main and would mis-target.
434
+ "base_branch": base_branch,
435
+ # verification-panel reads so far — persisted so the next wake
436
+ # continues the count.
437
+ "panel_reads": panel_reads,
438
+ # the session's write-up + spend, saved so a candidate wake can build
439
+ # the PR body / panel claim and report the real cost WITHOUT re-running
440
+ # the session (its edits are already in candidate_sha). REDACTED before
441
+ # it lands in the durable record, like every other persisted final_text
442
+ # — a session that echoed a credential must not leave it in record.json.
443
+ # Empty/zero for a baseline park (the session has not run yet).
444
+ # the author's report at submit, else the session's last words (a park
445
+ # with no submit); either way what the wake's panel and the PR body read
446
+ "report": redact(
447
+ parked.syscall.report
448
+ if parked.syscall is not None and parked.syscall.report
449
+ else (parked.session.final_text if parked.session else ""),
450
+ secrets,
451
+ )[:MAX_CLAIM_CHARS],
452
+ "session_cost_usd": parked.session.cost_usd if parked.session else 0.0,
453
+ "session_turns": parked.session.num_turns if parked.session else 0,
454
+ }
455
+ if parked.submitted:
456
+ # a SUBMITTED candidate park (buildout Phase B): the wake delivers the
457
+ # gate + panel results back to the author instead of deciding by policy
458
+ stage["submitted"] = True
459
+ if parked.syscall is not None:
460
+ # A syscall park — an author-directed sleep (research-loop-buildout.md
461
+ # Phase A) or a submitted candidate carrying sibling launches: the wake
462
+ # gathers each launch's results by NAME from the run dir, delivers the
463
+ # declared artifacts, and resumes the SAME session — so the stage must
464
+ # carry the launch names/artifacts, the author's note, and the budget
465
+ # counts as of this park.
466
+ stage["syscall_launches"] = [
467
+ # minutes ride along so a RE-PARK's deadline floor still covers the
468
+ # longest launch — without them a rebuilt descriptor would
469
+ # undershoot the floor and the sweep could cancel a healthy
470
+ # queued sibling as "pending past deadline"
471
+ {
472
+ "name": launch.name,
473
+ "minutes": launch.minutes,
474
+ "artifacts": list(launch.artifacts),
475
+ **({"array": launch.array} if launch.array > 1 else {}),
476
+ **({"why": redact(launch.why, secrets)} if launch.why else {}),
477
+ **({"concurrency": launch.concurrency} if launch.concurrency else {}),
478
+ }
479
+ for launch in parked.syscall.launches
480
+ ]
481
+ stage["syscall_note"] = redact(parked.syscall.note, secrets)
482
+ if parked.syscall.launches:
483
+ # the run's launch ledger (`history`, and the queue view's labels):
484
+ # ids align with launch_jobs order, as stage_launch_job_ids reads them
485
+ if parked.launch_afterany:
486
+ launch_ids = afterany_ids(parked.launch_afterany)
487
+ elif parked.phase == "author-sleep":
488
+ launch_ids = list(job_ids)
489
+ else:
490
+ launch_ids = []
491
+ ledger_launches = tuple(
492
+ dc_replace(launch, why=redact(launch.why, secrets))
493
+ for launch in parked.syscall.launches
494
+ )
495
+ _best_effort(
496
+ "launch ledger",
497
+ lambda: append_submitted(
498
+ run_dir_of(run_root, record.run_id),
499
+ sleep=parked.sleeps_used,
500
+ launches=ledger_launches,
501
+ job_ids=launch_ids,
502
+ at=now,
503
+ ),
504
+ )
505
+ # (the session id the wake resumes is the record's own
506
+ # resume_session_id, set below for every park — no stage duplicate)
507
+ stage["launches_used"] = parked.launches_used
508
+ stage["sleeps_used"] = parked.sleeps_used
509
+ stage["gpu_hours_used"] = parked.gpu_hours_used
510
+ if parked.judged is not None:
511
+ # the gate's last negative rides the park: a wake that ends on the
512
+ # same tree reuses it instead of measuring again
513
+ judged_sha, verdict = parked.judged
514
+ stage["judged"] = {
515
+ "sha": judged_sha,
516
+ "outcome": verdict.outcome,
517
+ "baseline": verdict.baseline,
518
+ "candidate": verdict.candidate,
519
+ "note": redact(verdict.note, secrets),
520
+ }
521
+ if parked.eval_minutes:
522
+ # the author's declared eval walltime rides the park: the wake's
523
+ # measurer and deadline floor must use it, not the contract's
524
+ stage["eval_minutes"] = parked.eval_minutes
525
+ # A single-job park records its one pollable id; a MULTI-job park records
526
+ # none — the sweep falls back to polling every id in the stage's `afterany`
527
+ # string and wakes only when ALL are done (tick._poll_targets).
528
+ experiment_job_id = job_ids[0] if len(job_ids) == 1 else ""
529
+ # The deadline is a FLOOR: park time (`now` here is the park moment, passed
530
+ # by the caller) + the eval walltime + a generous queue/grace slack, so a
531
+ # healthy queued-then-running eval never trips the sweep's cancel-on-pending.
532
+ floor_minutes = effective_eval_minutes(parked.eval_minutes or eval_minutes)
533
+ if parked.phase == "author-sleep" and parked.syscall is not None:
534
+ # an author launch's walltime is the LAUNCH's ask, not the benchmark's
535
+ # eval hint — the floor must sit past the LONGEST launch, or the sweep
536
+ # cancels still-queued author jobs (a benchmark can be in-job cheap,
537
+ # eval_minutes=None, while its author trains for hours). A checkpoint
538
+ # sleep has no jobs: floor 0 wakes it at the first deadline pass.
539
+ floor_minutes = max((la.minutes for la in parked.syscall.launches), default=0)
540
+ elif parked.syscall is not None:
541
+ # a submitted candidate with sibling launches waits on gate evals AND
542
+ # launches — the floor must sit past the longest of either
543
+ floor_minutes = max(floor_minutes, *(la.minutes for la in parked.syscall.launches), 0)
544
+ checkpoint_sleep = (
545
+ parked.phase == "author-sleep"
546
+ and parked.syscall is not None
547
+ and not parked.syscall.launches
548
+ )
549
+ if checkpoint_sleep:
550
+ # a CHECKPOINT SLEEP has nothing in any queue, so the 12h queue slack
551
+ # (sized to protect queued Slurm jobs from cancel-on-pending) does not
552
+ # apply — the deadline needs only to reach the sweep's next pass.
553
+ # Observed live (yolo heldout_probe, 2026-08-27): a jobless nap
554
+ # inherited the queue slack and became a 12h coma.
555
+ deadline = now + CHECKPOINT_SLEEP_SLACK_MIN * 60
556
+ else:
557
+ deadline = now + (floor_minutes + PARK_QUEUE_SLACK_MIN) * 60
558
+ waiting = RunRecord(
559
+ **{
560
+ **record.__dict__,
561
+ "state": WAITING,
562
+ "experiment_job_id": experiment_job_id,
563
+ "resume_session_id": parked.session.session_id if parked.session else "",
564
+ "deadline": deadline,
565
+ "stage": stage,
566
+ # wake_attempts = "wakes since the run last made progress"; the
567
+ # stuck cap ends a run that keeps waking without advancing. Reset on
568
+ # a PRODUCTIVE park (the IMPLEMENTING->park entry ran a session; a
569
+ # wake that resolved its measures and dispatched NEW ones). A
570
+ # no-progress re-park — results still pending, or a blind re-park
571
+ # (squeue unreachable, nothing new dispatched) — must KEEP the
572
+ # counter (`keep_wake_attempts`), or the loop never reaches the cap.
573
+ "wake_attempts": record.wake_attempts if keep_wake_attempts else 0,
574
+ "terminal_seen": 0.0,
575
+ }
576
+ )
577
+ save_record(run_root, waiting, now)
578
+ if dispatch is not None:
579
+ _arm_park_wake(run_root, record.run_id, now, dispatch)
580
+
581
+
582
+ def _arm_park_wake(run_root: Path, run_id: str, now: float, dispatch: DispatchSettings) -> str:
583
+ """Submit the parked run's wake right away, depending on the jobs it
584
+ waits on (tick.arm_wake), when the tick has published its wake recipe.
585
+ Without the recipe — dispatched wakes not armed, or a local compute —
586
+ the sweep delivers as before."""
587
+ from outerloop.compute import SlurmCompute
588
+ from outerloop.tick import JobWakeDispatcher, arm_wake, dispatch_wake_armed, load_wake_spec
589
+
590
+ if not dispatch_wake_armed(run_root):
591
+ return "" # disarmed: a recipe the tick has not yet removed is not used
592
+ spec = load_wake_spec(run_root)
593
+ if spec is None or not isinstance(dispatch.compute, SlurmCompute):
594
+ return ""
595
+ try:
596
+ record = load_record(run_root, run_id)
597
+ dispatcher = JobWakeDispatcher(dispatch.compute, spec, now)
598
+ return arm_wake(
599
+ run_root, record, dispatcher, now, holder_job_id=os.environ.get("SLURM_JOB_ID", "")
600
+ )
601
+ except Exception as exc:
602
+ log.warning("park-time wake not armed for %s: %s: %s", run_id, type(exc).__name__, exc)
603
+ return ""
604
+
605
+
606
+ def _lease_held_by_another_job(run_root: Path, run_id: str) -> str:
607
+ """The job id of a wake that holds this run's lease and is not us, or "".
608
+ A resume with no job id of its own (a manual run) never counts as the
609
+ holder of a job-held lease."""
610
+ from outerloop.runstate import read_lease
611
+
612
+ lease = read_lease(run_root, run_id)
613
+ mine = os.environ.get("SLURM_JOB_ID", "")
614
+ if lease is not None and lease.holder_job_id and lease.holder_job_id != mine:
615
+ return lease.holder_job_id
616
+ return ""
617
+
618
+
619
+ def _release_own_lease(run_root: Path, run_id: str) -> None:
620
+ """Release the run's lease only while this job still holds it: a park
621
+ hands the lease to the wake it arms, and that wake must keep it."""
622
+ from outerloop.runstate import release_lease
623
+
624
+ if _lease_held_by_another_job(run_root, run_id):
625
+ return
626
+ release_lease(run_root, run_id)
627
+
628
+
629
+ def _dispatch_settings(args: argparse.Namespace) -> DispatchSettings:
630
+ """The cluster coordinates from the CLI, read in ONE place for both the
631
+ fresh climb and the wake (a second constructor drifted once — terra
632
+ #174: the wake dropped the GPU lane)."""
633
+ from outerloop.compute import compute_from_env
634
+
635
+ target = getattr(args, "target", "") or ""
636
+ return DispatchSettings(
637
+ compute=compute_from_env(),
638
+ image=args.image,
639
+ account=args.account,
640
+ partition=args.partition,
641
+ gpu_partition=getattr(args, "gpu_partition", ""),
642
+ gpu_account=getattr(args, "gpu_account", ""),
643
+ seed_cache=seed_dir(Path(args.run_root), target) if target else None,
644
+ )
645
+
646
+
647
+ # Experiments yield to verification. Every kernel job is one Slurm user, so
648
+ # among the kernel's own pending jobs the priority order is ours: launches
649
+ # carry this nice so a gate eval or a follow-up re-measure (nice 0) starts
650
+ # first when the cap frees a slot. Sized above the factors that differ between
651
+ # our jobs on Torch — age tops out at 1000 after a week, job size at 1000, the
652
+ # per-GPU TRES share stays in the hundreds — so the order holds however long a
653
+ # launch has waited. Other users' jobs and the group cap are untouched.
654
+ LAUNCH_NICE = 5000
655
+
656
+
657
+ def with_seed(dispatch: DispatchSettings, run_root: Path, target: str) -> DispatchSettings:
658
+ """These settings with the target's seed cache filled in from the record's
659
+ target when the CLI gave none (wake and follow-up jobs carry the run id,
660
+ not the target)."""
661
+ # tolerant of any settings object: a backend that knows no seed (or a
662
+ # test double) is left exactly as it is
663
+ if not target or getattr(dispatch, "seed_cache", "unknown") is not None:
664
+ return dispatch
665
+ try:
666
+ return dc_replace(dispatch, seed_cache=seed_dir(run_root, target))
667
+ except TypeError:
668
+ return dispatch
669
+
670
+
671
+ def _make_launcher(
672
+ dispatch: DispatchSettings, run_dir: Path, workspace: Path, run_id: str, gpus: int = 0
673
+ ):
674
+ """The launch side of the author syscalls, shared by the first pass
675
+ (live_attempt) and the author-sleep wake: each launch becomes a jailed job on
676
+ the sealed snapshot (write_eval_job's copy-out handles artifacts), and a
677
+ partially-submitted batch is reaped rather than orphaned. `gpus` is the
678
+ benchmark's: an author's experiments run on the same lane as its evals."""
679
+ account, partition = dispatch.placement(gpus)
680
+
681
+ def launcher(sha: str, request: SyscallRequest) -> str:
682
+ from outerloop.dispatch import eval_job_spec, write_eval_job
683
+ from outerloop.syscall import array_spec
684
+
685
+ ids: list[str] = []
686
+ try:
687
+ for launch in request.launches:
688
+ # a sweep is ONE Slurm job array (`--array=0-N%K`): the queue
689
+ # holds one entry, Slurm runs at most K tasks at once, each task
690
+ # derives its job dir and SWEEP_INDEX from its array index, and
691
+ # one afterany on the array id covers every task
692
+ script = write_eval_job(
693
+ run_dir,
694
+ f"launch-{launch.name}",
695
+ repo_root=workspace,
696
+ snapshot_sha=sha,
697
+ command=launch.command,
698
+ image=dispatch.image,
699
+ artifacts=launch.artifacts,
700
+ artifact_max_bytes=MAX_ARTIFACT_BYTES,
701
+ gpus=gpus,
702
+ array=launch.array,
703
+ seed_cache=dispatch.seed_cache,
704
+ )
705
+ spec = eval_job_spec(
706
+ script,
707
+ job_name=f"{run_id}-launch-{launch.name}",
708
+ account=account,
709
+ partition=partition,
710
+ eval_minutes=launch.minutes,
711
+ gpus=gpus,
712
+ nice=LAUNCH_NICE,
713
+ array=array_spec(launch),
714
+ )
715
+ ids.append(dispatch.compute.submit(spec))
716
+ except Exception:
717
+ # a partial batch must not orphan: no park record was written yet,
718
+ # so nothing would ever wake or cancel the jobs that DID submit —
719
+ # reap them here, then let the caller end the run as the error it
720
+ # is (same stance as the failed-_park_run cancel).
721
+ for job_id in ids:
722
+ with contextlib.suppress(Exception):
723
+ dispatch.compute.cancel(job_id)
724
+ raise
725
+ # a checkpoint sleep (no launches) parks with no dependency and wakes
726
+ # on the sweep's deadline floor — slow but correct; a fast requeue wake
727
+ # is a follow-up.
728
+ return "afterany:" + ":".join(ids) if ids else ""
729
+
730
+ return launcher
731
+
732
+
733
+ def _make_watcher(
734
+ dispatch: DispatchSettings, run_root: Path, run_id: str, workspace: Path, config: RunConfig
735
+ ) -> Callable[[], Any]:
736
+ """The session watcher for one run: a thread beside the harness that
737
+ answers `queue` and `history` from the channel (docs/design/session-watcher.md).
738
+ Shared by the first pass and every wake leg."""
739
+ from outerloop.watcher import SessionWatcher, WatcherContext
740
+
741
+ ctx = WatcherContext(
742
+ workspace=workspace,
743
+ run_root=run_root,
744
+ run_id=run_id,
745
+ target=config.target,
746
+ agent_id=config.agent_id,
747
+ compute=dispatch.compute,
748
+ gpu_partition=dispatch.gpu_partition,
749
+ )
750
+ return lambda: SessionWatcher(ctx)
751
+
752
+
753
+ def _wake_author_sleep(
754
+ *,
755
+ run_root: Path,
756
+ run_id: str,
757
+ record: RunRecord,
758
+ ws: Workspace,
759
+ workspace: Path,
760
+ run_dir: Path,
761
+ dispatch: DispatchSettings,
762
+ github: GitHubClient,
763
+ now: float,
764
+ secrets: tuple[str, ...],
765
+ base_branch: str,
766
+ base_sha: str,
767
+ sleep_ref: str,
768
+ contract_text: str,
769
+ contract: Contract,
770
+ bench: Benchmark,
771
+ config: RunConfig,
772
+ measurer: Measurer,
773
+ harness: Harness | None,
774
+ spec: RoleSpec | None,
775
+ panel_lenses: tuple[PanelLens, ...],
776
+ issue_number: int,
777
+ eval_minutes: int | None,
778
+ extra_update: str = "",
779
+ judged: tuple[str, AttemptResult] | None = None,
780
+ ) -> AttemptOutcome:
781
+ """Wake a syscall park and resume the AUTHOR: deliver the launches' results
782
+ into the sandbox, resume the SAME session through the climb's resume-entry
783
+ with them (data-fenced), and let the climb run — it may sleep again
784
+ (re-park), submit, finish into the gate (whose dispatched measures park it
785
+ as a CANDIDATE), or end on a terminal. The session's workspace persisted on
786
+ disk exactly as the author left it (the launches ran on node-local
787
+ checkouts of the sealed sha), so the resumed session continues its own tree
788
+ — cumulative depth. `extra_update` leads the wake text — a SUBMITTED park's
789
+ gate result or panel verdict (buildout Phase B), delivered back to the
790
+ author to act on; `sleep_ref` is whichever snapshot ref this park holds
791
+ (the sleep seal, or the submitted candidate)."""
792
+ from outerloop.syscall import Launch as SyscallLaunch
793
+ from outerloop.syscall import (
794
+ annotate_launch_states,
795
+ gather_results,
796
+ render_wake,
797
+ write_budget,
798
+ )
799
+
800
+ def _end(result: AttemptResult, drop_refs: list[str]) -> AttemptOutcome:
801
+ # a terminal from the resumed climb: report, ending record, issue note —
802
+ # the same ending shape every other terminal takes. The line notebook
803
+ # records it first, while the tree is still the session's final tree.
804
+ _push_line_snapshot(
805
+ ws,
806
+ _line_ref_for(bench, config.agent_id),
807
+ run_id,
808
+ result.outcome,
809
+ secrets,
810
+ bot_login=config.bot_login,
811
+ )
812
+ for ref in drop_refs:
813
+ drop_snapshot(ws, Snapshot(commit="", tree="", ref=ref))
814
+ report_path = run_dir / "report.md"
815
+ _best_effort(
816
+ "run report",
817
+ lambda: report_path.write_text(result.report(config, redact_secrets=secrets)),
818
+ secrets,
819
+ )
820
+ ending = _ENDINGS_BY_OUTCOME.get(result.outcome, ABORTED)
821
+ final = _clear_stage(
822
+ RunRecord(
823
+ **{
824
+ **record.__dict__,
825
+ "state": ENDED,
826
+ "ending": ending,
827
+ "ending_note": redact(result.note, secrets),
828
+ }
829
+ )
830
+ )
831
+ _best_effort("final record", lambda: save_record(run_root, final, now), secrets)
832
+ _post_issue_finished(
833
+ github,
834
+ config.target,
835
+ issue_number,
836
+ run_id,
837
+ result.outcome,
838
+ "",
839
+ redact(result.report(config, redact_secrets=secrets), secrets)[:8000],
840
+ secrets,
841
+ )
842
+ return AttemptOutcome(run_id=run_id, outcome=result.outcome, report_path=str(report_path))
843
+
844
+ # The wake NEEDS the author harness (it resumes the session). Fail as a
845
+ # named ending, not a crash: the run cannot proceed and re-waking will not
846
+ # help without the harness, so leaving it WAITING would just hit the stuck
847
+ # cap slowly.
848
+ if (
849
+ harness is None
850
+ or spec is None
851
+ or not record.resume_session_id
852
+ or not getattr(harness, "supports_resume", True)
853
+ ):
854
+ return _end(
855
+ AttemptResult(
856
+ outcome="session-error",
857
+ note="author-sleep wake needs the author harness/spec and a resumable session",
858
+ ),
859
+ drop_refs=[sleep_ref],
860
+ )
861
+
862
+ # Deliver: read each launch's job output, copy declared artifacts into the
863
+ # excluded channel, and render the data-fenced wake text. The stage stores
864
+ # only what the wake needs (names + artifacts); command/minutes placeholders
865
+ # never reach the author.
866
+ launches = tuple(
867
+ SyscallLaunch(
868
+ name=str(item.get("name", "")),
869
+ command="(ran)",
870
+ minutes=int(item.get("minutes") or 1),
871
+ artifacts=tuple(str(a) for a in item.get("artifacts", [])),
872
+ array=int(item.get("array") or 1),
873
+ why=str(item.get("why") or ""),
874
+ concurrency=int(item.get("concurrency") or 0),
875
+ )
876
+ for item in _stage_launches(record)
877
+ )
878
+ results = gather_results(run_dir, workspace, launches)
879
+ # A launch that left no exit code was SIGKILL'd before its wrapper ran (the
880
+ # cgroup OOM killer, a hard walltime kill, a node failure). The scheduler
881
+ # still knows which; surface it so the author reads "OOM"/"timeout" instead
882
+ # of a blank "job failure". The park's launch job ids align positionally
883
+ # with the results (same launch/array order). Best-effort — the wake never
884
+ # blocks on the scheduler query.
885
+ task_ids = launch_task_ids(launches, stage_launch_job_ids(record))
886
+ status_of = getattr(dispatch.compute, "status", None)
887
+ if status_of is not None:
888
+ results = annotate_launch_states(results, task_ids, status_of)
889
+ launches_used = int(record.stage.get("launches_used", 0)) # type: ignore[call-overload]
890
+ sleeps_used = int(record.stage.get("sleeps_used", 0)) # type: ignore[call-overload]
891
+ elapsed = _launch_elapsed(dispatch, task_ids) if task_ids else None
892
+ _best_effort(
893
+ "launch ledger",
894
+ lambda: append_ended(
895
+ run_dir, sleep=sleeps_used, results=results, at=time.time(), elapsed_seconds=elapsed
896
+ ),
897
+ )
898
+ gpu_hours_used = _reconcile_launch_hours(record, dispatch, bench.gpus, launches, elapsed)
899
+ # the tool the session invokes comes from THIS kernel: a session that
900
+ # started under an older one gets today's verbs and flags at its wake, and
901
+ # is told what is new
902
+ tool_changed = False
903
+ try:
904
+ tool_changed = syscall_refresh_tool(workspace)
905
+ except Exception as exc:
906
+ log.warning("tool refresh failed: %s", redact(f"{type(exc).__name__}: {exc}", secrets))
907
+ wake_text = render_wake(
908
+ results,
909
+ str(record.stage.get("syscall_note", "")),
910
+ launches_used=launches_used,
911
+ launch_budget=bench.depth_k,
912
+ sleeps_used=sleeps_used,
913
+ sleep_budget=bench.sleep_k,
914
+ gpu_hours_remaining=(
915
+ max(0.0, contract.budgets.gpu_hours_per_run - gpu_hours_used) if bench.gpus else None
916
+ ),
917
+ gpus=bench.gpus,
918
+ )
919
+ pacing = [
920
+ f"sweep `{la.name}`: {la.array} tasks, at most {la.concurrency or la.array} at a time"
921
+ for la in launches
922
+ if la.array > 1
923
+ ]
924
+ if pacing:
925
+ wake_text = f"{wake_text}\n\n" + "\n".join(pacing) + " (the contract's ceiling applies)."
926
+ if tool_changed:
927
+ wake_text = f"{wake_text}\n\n{tool_update_note(channel_dir(workspace))}"
928
+ if extra_update:
929
+ # a submitted park's gate/panel feedback leads; launch results follow
930
+ wake_text = f"{extra_update}\n\n{wake_text}"
931
+ # A research line whose base moved while it slept RE-PINS to the fresh base
932
+ # and is told to merge it: the agent does the merge (mirroring the in-review
933
+ # conflict wake, followup.py), the kernel only fetches and re-pins. Re-pinning
934
+ # base_sha to the fresh head is what makes the gate baseline and the scope
935
+ # base the CURRENT base (like followup's base_sha_at_fetch), so a sibling's
936
+ # merged work is never credited to this line and a forbidden conflict
937
+ # resolution (differing from the fresh base) is still scope-checked. Non-line
938
+ # runs and an unmoved base are untouched.
939
+ if _line_ref_for(bench, config.agent_id):
940
+ fresh_base = _line_base_advanced(ws, base_branch, base_sha)
941
+ if fresh_base:
942
+ digest = _reintegration_digest(ws, base_sha, fresh_base)
943
+ base_sha = fresh_base
944
+ wake_text = (
945
+ REINTEGRATE_PROMPT.format(base_branch=base_branch, digest=digest) + wake_text
946
+ )
947
+ _best_effort(
948
+ "budget refresh",
949
+ lambda: write_budget(
950
+ workspace,
951
+ launches_remaining=max(0, bench.depth_k - launches_used),
952
+ sleeps_remaining=max(0, bench.sleep_k - sleeps_used),
953
+ gpu_hours_remaining=(
954
+ max(0.0, contract.budgets.gpu_hours_per_run - gpu_hours_used)
955
+ if bench.gpus
956
+ else None
957
+ ),
958
+ ),
959
+ )
960
+
961
+ # The wake's climb IO: measures go through the DISPATCHED measurer (this is
962
+ # a wake job with bounded walltime — the gate's evals run as their own jobs
963
+ # and park the run as a CANDIDATE); snapshots parent on base (same as the
964
+ # first pass: the clone was at base).
965
+ snapshots: list[Snapshot] = []
966
+ wake_line = _line_ref_for(bench, config.agent_id)
967
+
968
+ def snapshot() -> str:
969
+ snap = snapshot_tree(
970
+ ws, base_sha, exclude=LINE_MEMORY_PATHS if wake_line else (), author=config.bot_login
971
+ )
972
+ snapshots.append(snap)
973
+ return snap.commit
974
+
975
+ def changed_paths() -> list[str]:
976
+ return _paths_changed_from_base(
977
+ ws, f"refs/remotes/origin/{base_branch}", bool(wake_line), fallback=base_sha
978
+ )
979
+
980
+ panel_runner = (
981
+ build_panel_runner(
982
+ ws,
983
+ run_dir,
984
+ base_sha,
985
+ panel_lenses,
986
+ contract_text,
987
+ config.target,
988
+ config.benchmark,
989
+ config.bot_login,
990
+ _utc_date(now),
991
+ exclude=LINE_MEMORY_PATHS if wake_line else (),
992
+ secrets=secrets,
993
+ )
994
+ if panel_lenses
995
+ else None
996
+ )
997
+
998
+ parked: RunParked | None = None
999
+ kept_ref = ""
1000
+ try:
1001
+ result = attempt_once(
1002
+ config,
1003
+ contract_text,
1004
+ workspace,
1005
+ harness,
1006
+ measurer,
1007
+ base_sha,
1008
+ snapshot,
1009
+ ruler=RULER,
1010
+ changed_paths=changed_paths,
1011
+ spec=spec,
1012
+ panel_runner=panel_runner,
1013
+ resume_session_id=record.resume_session_id,
1014
+ improve_prompt=wake_text,
1015
+ launcher=_make_launcher(dispatch, run_dir, workspace, run_id, gpus=bench.gpus),
1016
+ watcher=_make_watcher(dispatch, run_root, run_id, workspace, config),
1017
+ tree_of=lambda sha: ws.git("rev-parse", f"{sha}^{{tree}}").strip(),
1018
+ judged=judged or _stage_judged(record),
1019
+ launches_used=launches_used,
1020
+ sleeps_used=sleeps_used,
1021
+ gpu_hours_used=gpu_hours_used,
1022
+ )
1023
+ except RunParked as p:
1024
+ # slept again, or the gate dispatched its measures (a candidate park the
1025
+ # existing wake path decides). Keep the NEW park's snapshot ref; the OLD
1026
+ # sleep ref is superseded once the new park persists.
1027
+ kept_ref = next((s.ref for s in snapshots if s.commit == p.candidate_sha), "")
1028
+ import time
1029
+
1030
+ try:
1031
+ _park_run(
1032
+ run_root,
1033
+ record,
1034
+ p,
1035
+ kept_ref,
1036
+ eval_minutes,
1037
+ time.time(),
1038
+ secrets,
1039
+ dispatch=dispatch,
1040
+ base_branch=base_branch,
1041
+ )
1042
+ except Exception:
1043
+ for job_id in afterany_ids(p.afterany):
1044
+ dispatch.compute.cancel(job_id)
1045
+ raise
1046
+ parked = p
1047
+ drop_snapshot(ws, Snapshot(commit="", tree="", ref=sleep_ref))
1048
+ return AttemptOutcome(run_id=run_id, outcome="parked")
1049
+ finally:
1050
+ for snap in snapshots:
1051
+ if parked and kept_ref and snap.ref == kept_ref:
1052
+ continue
1053
+ drop_snapshot(ws, snap)
1054
+
1055
+ # a terminal from the resumed session (session-error/-outage/-budget,
1056
+ # scope-violation, eval-error; a dispatched gate never returns improved
1057
+ # inline): end the run and release the sleep snapshot.
1058
+ return _end(result, drop_refs=[sleep_ref])
1059
+
1060
+
1061
+ def _stage_judged(record: RunRecord) -> tuple[str, AttemptResult] | None:
1062
+ """The gate verdict a park carried (written by `_park_run`), or None."""
1063
+ j = (record.stage or {}).get("judged")
1064
+ if not isinstance(j, dict) or not j.get("sha"):
1065
+ return None
1066
+
1067
+ def num(v: object) -> float | None:
1068
+ return float(v) if isinstance(v, int | float) and not isinstance(v, bool) else None
1069
+
1070
+ return (
1071
+ str(j["sha"]),
1072
+ AttemptResult(
1073
+ outcome=str(j.get("outcome") or "no-improvement"),
1074
+ baseline=num(j.get("baseline")),
1075
+ candidate=num(j.get("candidate")),
1076
+ note=str(j.get("note") or ""),
1077
+ ),
1078
+ )
1079
+
1080
+
1081
+ def _ledger_ended(
1082
+ run_dir: Path,
1083
+ record: RunRecord,
1084
+ launches: tuple,
1085
+ task_ids: list[str],
1086
+ dispatch: DispatchSettings,
1087
+ elapsed: list[int | None] | None,
1088
+ ) -> None:
1089
+ """Record a park's finished launches in the ledger from the run dir alone
1090
+ (no delivery into a workspace), with the scheduler's state for jobs that
1091
+ left no exit code."""
1092
+ from outerloop.syscall import annotate_launch_states, read_results
1093
+
1094
+ results = read_results(run_dir, launches)
1095
+ status_of = getattr(dispatch.compute, "status", None)
1096
+ if status_of is not None:
1097
+ results = annotate_launch_states(results, task_ids, status_of)
1098
+ append_ended(
1099
+ run_dir,
1100
+ sleep=int(record.stage.get("sleeps_used", 0)), # type: ignore[call-overload]
1101
+ results=results,
1102
+ at=time.time(),
1103
+ elapsed_seconds=elapsed,
1104
+ )
1105
+
1106
+
1107
+ def _stage_syscall_launches(record: RunRecord) -> tuple:
1108
+ """The park's launches as `Launch` values (command elided: they ran)."""
1109
+ from outerloop.syscall import Launch
1110
+
1111
+ return tuple(
1112
+ Launch(
1113
+ name=str(item.get("name", "")),
1114
+ command="(ran)",
1115
+ minutes=int(item.get("minutes") or 1),
1116
+ artifacts=tuple(str(a) for a in item.get("artifacts", [])),
1117
+ array=int(item.get("array") or 1),
1118
+ why=str(item.get("why") or ""),
1119
+ concurrency=int(item.get("concurrency") or 0),
1120
+ )
1121
+ for item in _stage_launches(record)
1122
+ )
1123
+
1124
+
1125
+ def stage_launch_job_ids(record: RunRecord) -> list[str]:
1126
+ """The park's launch jobs: `launch_afterany` when the park recorded it;
1127
+ for an older author-sleep park every waited job was a launch; for an
1128
+ older candidate park the gate's evals are mixed in, so none."""
1129
+ stage = record.stage or {}
1130
+ if "launch_afterany" in stage:
1131
+ return afterany_ids(str(stage.get("launch_afterany") or ""))
1132
+ if stage.get("phase") == "author-sleep":
1133
+ return afterany_ids(str(stage.get("afterany") or ""))
1134
+ return []
1135
+
1136
+
1137
+ def _reconcile_launch_hours(
1138
+ record: RunRecord,
1139
+ dispatch: DispatchSettings,
1140
+ gpus: int,
1141
+ launches: tuple,
1142
+ elapsed: list[int | None] | None = None,
1143
+ ) -> float:
1144
+ """The run's GPU-hours after handing back the unused walltime of the
1145
+ park's launch jobs — once: the stage remembers the refund, so a wake
1146
+ that follows a gate decision on the same park does not refund twice.
1147
+ Returns the (possibly corrected) `gpu_hours_used`."""
1148
+ stage = record.stage or {}
1149
+ used = float(stage.get("gpu_hours_used", 0.0)) # type: ignore[arg-type]
1150
+ if not gpus or stage.get("launch_hours_refunded"):
1151
+ return used
1152
+ refund = _launch_refund(
1153
+ dispatch, launches, launch_task_ids(launches, stage_launch_job_ids(record)), gpus, elapsed
1154
+ )
1155
+ if refund > 0:
1156
+ log.info("%s: refunding %.2f GPU-hours of unused launch walltime", record.run_id, refund)
1157
+ used = max(0.0, used - refund)
1158
+ stage["gpu_hours_used"] = used
1159
+ stage["launch_hours_refunded"] = True
1160
+ return used
1161
+
1162
+
1163
+ def _launch_elapsed(dispatch: DispatchSettings, job_ids: list[str]) -> list[int | None] | None:
1164
+ """How long each launch job ran, from the compute, aligned with `job_ids`;
1165
+ None when the compute cannot say (nothing is refunded or recorded on a
1166
+ guess)."""
1167
+ query = getattr(dispatch.compute, "elapsed_seconds", None)
1168
+ if query is None or not job_ids:
1169
+ return None
1170
+ try:
1171
+ return [query(jid) for jid in job_ids]
1172
+ except Exception as exc:
1173
+ log.warning("launch walltime unknown (%s: %s)", type(exc).__name__, exc)
1174
+ return None
1175
+
1176
+
1177
+ def _launch_refund(
1178
+ dispatch: DispatchSettings,
1179
+ launches: tuple,
1180
+ job_ids: list[str],
1181
+ gpus: int,
1182
+ elapsed: list[int | None] | None = None,
1183
+ ) -> float:
1184
+ """The unused walltime of a park's launch jobs, in GPU-hours, or 0 when
1185
+ the compute cannot say how long they ran (nothing is refunded on a
1186
+ guess). `elapsed` may be handed in when the caller already asked."""
1187
+ from outerloop.syscall import launch_hours_refund
1188
+
1189
+ if elapsed is None:
1190
+ elapsed = _launch_elapsed(dispatch, job_ids)
1191
+ if elapsed is None:
1192
+ return 0.0
1193
+ return launch_hours_refund(launches, elapsed, gpus=gpus)
1194
+
1195
+
1196
+ RESEARCH_LOG_BRANCH = "research-log"
1197
+ MAX_ARCHIVED_REPORTS = 30 # materialized for the session to read; newest first
1198
+ MAX_ARCHIVED_REPORT_CHARS = 100_000 # per report; branch content is remote-controlled
1199
+
1200
+
1201
+ def _fetch_research_reports(ws: Workspace, count: int) -> list[tuple[str, str]]:
1202
+ """The newest `count` reports from the target's research-log branch, as
1203
+ (name, text), newest first — the shared memory of every attempt on this
1204
+ target, wherever it ran. Fail-soft: a target with no research log yet
1205
+ (or an unreachable remote) is an empty memory, never a dead attempt."""
1206
+ try:
1207
+ ws.fetch_branch(RESEARCH_LOG_BRANCH)
1208
+ listing = ws.git("ls-tree", "-r", "--name-only", "FETCH_HEAD", "reports/")
1209
+ # only direct children (the publisher's layout): a nested path would
1210
+ # flatten to a basename that overwrites another archived report
1211
+ names = [
1212
+ line.strip()
1213
+ for line in listing.splitlines()
1214
+ if line.strip().endswith(".md") and line.strip().count("/") == 1
1215
+ ]
1216
+ # report files are dated (reports/<YYYY-MM-DD>-<run_id>.md): the name
1217
+ # sorts by day; same-day order is arbitrary and does not matter
1218
+ out: list[tuple[str, str]] = []
1219
+ for name in sorted(names, reverse=True):
1220
+ if len(out) >= count:
1221
+ break
1222
+ # size BEFORE content: `git show` would load the whole blob, and
1223
+ # the branch's content is remote-controlled
1224
+ if int(ws.git("cat-file", "-s", f"FETCH_HEAD:{name}").strip()) > (
1225
+ MAX_ARCHIVED_REPORT_CHARS
1226
+ ):
1227
+ log.info("research report %s exceeds the size cap; skipped", name)
1228
+ continue
1229
+ out.append((Path(name).name, ws.git("show", f"FETCH_HEAD:{name}")))
1230
+ return out
1231
+ except Exception as exc:
1232
+ log.info("research log unavailable (%s: %s); starting without it", type(exc).__name__, exc)
1233
+ return []
1234
+
1235
+
1236
+ def _exclude_merge_artifacts(workspace: Path) -> None:
1237
+ """Ignore *.orig and *.rej — git's merge/patch conflict backups, which
1238
+ should never be committed — via .git/info/exclude (repo-local, never a
1239
+ tracked edit). A line run merges main at start and the agent resolves
1240
+ conflicts as its first task; a leftover train.py.orig would otherwise
1241
+ read as an out-of-scope edit and abort the run at launch. Idempotent,
1242
+ and effective for the `git add -A` behind changed-paths and every seal."""
1243
+ exclude = workspace / ".git" / "info" / "exclude"
1244
+ wanted = ["*.orig", "*.rej"]
1245
+ try:
1246
+ existing = exclude.read_text()
1247
+ except OSError:
1248
+ existing = ""
1249
+ lines = existing.splitlines()
1250
+ missing = [p for p in wanted if p not in lines]
1251
+ if missing:
1252
+ exclude.parent.mkdir(parents=True, exist_ok=True)
1253
+ sep = "" if existing.endswith("\n") or not existing else "\n"
1254
+ exclude.write_text(existing + sep + "\n".join(missing) + "\n")
1255
+
1256
+
1257
+ def _reset_instruction_files(ws: Workspace, workspace: Path, base_ref: str) -> None:
1258
+ """Make the checkout's instruction-bearing files EQUAL the base branch's
1259
+ reviewed versions (review_agent.INSTRUCTION_FILES owns the list): a line
1260
+ is an author-written tree, and must never instruct its own successor
1261
+ sessions — or a sibling's (docs/design/research-lines.md)."""
1262
+ from outerloop.review_agent import INSTRUCTION_FILES
1263
+
1264
+ # bottom-up so a removed directory doesn't orphan paths found beneath it
1265
+ for path in sorted(workspace.rglob("*"), key=lambda p: len(p.parts), reverse=True):
1266
+ if ".git" in path.parts or path.name not in INSTRUCTION_FILES:
1267
+ continue
1268
+ if path.is_dir() and not path.is_symlink():
1269
+ shutil.rmtree(path, ignore_errors=True)
1270
+ else:
1271
+ path.unlink(missing_ok=True)
1272
+ base_paths = [
1273
+ p
1274
+ for p in ws.git("ls-tree", "-r", "--name-only", base_ref).splitlines()
1275
+ if any(part in INSTRUCTION_FILES for part in Path(p).parts)
1276
+ ]
1277
+ if base_paths:
1278
+ ws.git("checkout", base_ref, "--", *base_paths)
1279
+
1280
+
1281
+ # The agent's memory on its line (docs/design/research-lines.md): the bounded
1282
+ # index at the branch root plus the topic-file folder. They ride the NOTEBOOK
1283
+ # seal (the line branch is exactly where they live) and are excluded from
1284
+ # every MEASURABLE seal and from changed-path accounting — never a scope
1285
+ # violation, never claimable work, never part of a main-PR candidate.
1286
+ LINE_MEMORY_PATHS = ("AGENT_MEMORY.md", "agent_memory")
1287
+
1288
+
1289
+ def _is_line_memory(path: str) -> bool:
1290
+ return path in LINE_MEMORY_PATHS or path.startswith("agent_memory/")
1291
+
1292
+
1293
+ def _without_line_memory(paths: Iterable[str], line: str) -> list[str]:
1294
+ """The run's OWN changes: with a line active, its memory paths are dropped.
1295
+ Every sealed tree excludes them by construction, and the run's base is the
1296
+ line tip that carries them — so a base..candidate diff lists them as
1297
+ deletions that are not the run's change (live: a dispatched wake on
1298
+ gpt-speedrun read eight memory files as out of scope and aborted a run
1299
+ whose paired eval had already finished)."""
1300
+ return [p for p in paths if not (line and _is_line_memory(p))]
1301
+
1302
+
1303
+ def _line_ref_for(bench: Benchmark | None, agent_id: str) -> str:
1304
+ """The agent's line branch when the benchmark opted in, or the empty
1305
+ string when the feature is off. Wake paths recompute this from the
1306
+ record — a malformed agent id could never have created a line at run
1307
+ start, so empty (not an error) is right there too."""
1308
+ if bench is None or not bench.lines or not agent_id:
1309
+ return ""
1310
+ if not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9_-]{0,63}", agent_id):
1311
+ return ""
1312
+ return f"agents/{agent_id}"
1313
+
1314
+
1315
+ # The base moved under a line while it slept: rather than the kernel doing a
1316
+ # git merge (which mishandles the agent's in-flight tree, its conflicts, and
1317
+ # the scope gate), the wake mirrors the in-review conflict wake — fetch the
1318
+ # fresh base into the workspace and TELL THE AGENT to merge it. The agent has
1319
+ # git and already resolves the run-start merge as its first task; only the
1320
+ # credential-bearing fetch/push are the kernel's. No commit text goes in the
1321
+ # prompt (no cross-agent prompt-injection surface); the agent reads what
1322
+ # landed from git itself.
1323
+ REINTEGRATE_PROMPT = (
1324
+ "# The base moved while you were asleep\n"
1325
+ "`origin/{base_branch}` advanced since your last run and is fetched into "
1326
+ "your workspace. What landed:\n{digest}\n"
1327
+ "Merge it into your line and resolve any conflicts honestly, then decide "
1328
+ "what to re-run given what landed — if a sibling took your direction "
1329
+ "further, pivot or say so plainly rather than pushing on. Your change is "
1330
+ "measured against the current base.\n\n"
1331
+ )
1332
+
1333
+
1334
+ def _reintegration_digest(ws: Workspace, base_sha: str, fresh_head: str) -> str:
1335
+ """A short 'what landed' list: the subjects of the commits merged into the
1336
+ base since this line's base, newest first, capped. These are MERGED commits
1337
+ — vetted by the human-merge gate — so they are context, not untrusted input;
1338
+ the agent also has them in git to read in full."""
1339
+ try:
1340
+ out = ws.git("log", "--no-merges", "--format=%s", f"{base_sha}..{fresh_head}")
1341
+ except Exception:
1342
+ return " (recent changes on the base; see `git log`)"
1343
+ subjects = [ln.strip() for ln in out.splitlines() if ln.strip()][:12]
1344
+ return "\n".join(f" - {s}" for s in subjects) or " (a merge on the base; see `git log`)"
1345
+
1346
+
1347
+ def _line_base_advanced(ws: Workspace, base_branch: str, base_sha: str) -> str:
1348
+ """Fetch `origin/<base_branch>` and return its head when it has advanced
1349
+ past the line's pinned base, else "". The fetch doubles as making the
1350
+ fresh base available for the agent to merge and refreshes the origin refs
1351
+ the scope/measure path pairs against. Best-effort: any git failure returns
1352
+ "" and the wake proceeds exactly as today."""
1353
+ try:
1354
+ ws.fetch_origin()
1355
+ new = ws.git("rev-parse", f"refs/remotes/origin/{base_branch}").strip()
1356
+ merge_base = ws.git("merge-base", new, base_sha).strip()
1357
+ return new if new and merge_base != new else ""
1358
+ except Exception as exc:
1359
+ log.warning("base-moved check failed (%s); wake proceeds unchanged", type(exc).__name__)
1360
+ return ""
1361
+
1362
+
1363
+ def _push_line_snapshot(
1364
+ ws: Workspace,
1365
+ line_ref: str,
1366
+ run_id: str,
1367
+ outcome: str,
1368
+ secrets: tuple[str, ...] = (),
1369
+ bot_login: str = "",
1370
+ ) -> None:
1371
+ """Publish the session's final tree to the agent's line as a sealed
1372
+ snapshot commit — every terminal path, any outcome
1373
+ (docs/design/research-lines.md). The seal parents on the LOCAL line ref
1374
+ and advances it, so sequential terminals within one run chain as
1375
+ fast-forwards; one slot never runs twice concurrently, so the remote
1376
+ cannot have moved under us. Best-effort throughout: the notebook never
1377
+ changes a run's outcome. An unchanged tree is not pushed (the run-start
1378
+ push already holds it)."""
1379
+ if not line_ref:
1380
+ return
1381
+
1382
+ def _seal_and_push() -> None:
1383
+ # raises if the ref is absent (e.g. a park that predates the line
1384
+ # feature) or the session altered .git (every ws.git call checks) —
1385
+ # _best_effort turns either into a logged skip
1386
+ local = ws.git("rev-parse", f"refs/heads/{line_ref}").strip()
1387
+ memory = tuple(p for p in LINE_MEMORY_PATHS if (Path(ws.root) / p).exists())
1388
+ last_exc: Exception | None = None
1389
+ # the line commit this workspace's untouched files currently match:
1390
+ # the local ref at first, then each remote head reconciled into it
1391
+ # (so a retry does not mistake copied-in files for this run's edits)
1392
+ fork = local
1393
+ for _ in range(3):
1394
+ parent = fork
1395
+ # A park frees the slot, so a newer run on the same line can end
1396
+ # (and push) while this one is parked: the remote line then sits
1397
+ # past our local ref, and a seal parented on the stale ref would be
1398
+ # refused as a non-fast-forward (gpt-speedrun, 2026-09-03: agent-01's
1399
+ # winning run left no snapshot this way, and its line fell behind
1400
+ # main). Parent on the remote head whenever our ref is an ancestor
1401
+ # of it, keeping the files that head added since; offline, seal on
1402
+ # the local ref as before.
1403
+ try:
1404
+ ws.fetch_origin()
1405
+ remote = ws.git("rev-parse", f"refs/remotes/origin/{line_ref}").strip()
1406
+ if remote != fork:
1407
+ ws.git("merge-base", "--is-ancestor", fork, remote) # raises when not
1408
+ _reconcile_with_remote(ws, fork, remote)
1409
+ fork = parent = remote
1410
+ except Exception as exc:
1411
+ log.info("line %s: sealing on the local ref (%s)", line_ref, type(exc).__name__)
1412
+ snap = snapshot_tree(ws, parent, force=memory, author=bot_login)
1413
+ try:
1414
+ # seal only when the tree moved past the parent; the PUSH runs
1415
+ # either way — a session that COMMITTED its work advanced the
1416
+ # local ref without dirtying the tree, and that commit must
1417
+ # still reach the remote (an already-current ref push is a no-op)
1418
+ sealed = parent
1419
+ if snap.tree != ws.git("rev-parse", f"{parent}^{{tree}}").strip():
1420
+ sealed = ws.git(
1421
+ *git_identity(bot_login),
1422
+ "commit-tree",
1423
+ snap.tree,
1424
+ "-p",
1425
+ parent,
1426
+ "-m",
1427
+ f"line snapshot: {run_id} ({outcome})",
1428
+ ).strip()
1429
+ ws.git("update-ref", f"refs/heads/{line_ref}", sealed)
1430
+ try:
1431
+ ws.push(line_ref)
1432
+ return
1433
+ except Exception as exc:
1434
+ # another run pushed between our fetch and this push:
1435
+ # re-read the line and seal again on its new head
1436
+ last_exc = exc
1437
+ log.info(
1438
+ "line %s: push refused, re-sealing on the moved line (%s)",
1439
+ line_ref,
1440
+ type(exc).__name__,
1441
+ )
1442
+ finally:
1443
+ drop_snapshot(ws, snap)
1444
+ assert last_exc is not None
1445
+ raise last_exc
1446
+
1447
+ _best_effort(f"line push ({outcome})", _seal_and_push, secrets)
1448
+
1449
+
1450
+ def _reconcile_with_remote(ws: Workspace, old: str, new: str) -> None:
1451
+ """The seal is built from THIS workspace, which forked the line at `old`;
1452
+ another run has since moved the line to `new`. For every path that run
1453
+ changed, take its state when this workspace left the path untouched
1454
+ since `old` (added: materialize, modified: update, deleted: remove); a
1455
+ path this run touched keeps this run's version, as the line always did.
1456
+ Without this, a seal would silently undo the other run's work on files
1457
+ this run never looked at."""
1458
+ entries = [
1459
+ e for e in ws.git("diff", "--name-status", "--no-renames", "-z", old, new).split("\0") if e
1460
+ ]
1461
+ remote_changes = list(zip(entries[0::2], entries[1::2], strict=True))
1462
+ if not remote_changes:
1463
+ return
1464
+ touched = {
1465
+ p for p in ws.git("diff", "--name-only", "-z", old).split("\0") if p
1466
+ } # tracked paths this run changed or deleted
1467
+ touched |= {
1468
+ p for p in ws.git("ls-files", "--others", "--exclude-standard", "-z").split("\0") if p
1469
+ } # and the files it created
1470
+ for status, path in remote_changes:
1471
+ if path in touched:
1472
+ continue
1473
+ if status.startswith("D"):
1474
+ target = Path(ws.root) / path
1475
+ if target.is_file() or target.is_symlink():
1476
+ target.unlink()
1477
+ else:
1478
+ ws.git("checkout", new, "--", path)
1479
+
1480
+
1481
+ def _checkout_line(
1482
+ ws: Workspace, workspace: Path, agent_id: str, base_branch: str, bot_login: str = ""
1483
+ ) -> str:
1484
+ """Check out the agent's research line: the persistent branch
1485
+ `agents/<agent-id>`, created from the base branch when absent, with the
1486
+ base branch merged in when it exists — a conflicted merge is left in the
1487
+ tree as the session's first task. Instruction-bearing files are reset to
1488
+ the base branch's reviewed versions and the hygiene is committed (never
1489
+ left to masquerade as agent edits). Returns the line ref name; raises on
1490
+ anything unrecoverable (the caller falls back to the base branch)."""
1491
+ if not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9_-]{0,63}", agent_id):
1492
+ raise ValueError(f"agent id {agent_id!r} cannot shape a line ref")
1493
+ line = f"agents/{agent_id}"
1494
+ base_ref = f"origin/{base_branch}"
1495
+ if not ws.git("branch", "--list", "-r", f"origin/{line}").strip():
1496
+ ws.git("checkout", "-q", "-B", line, base_ref)
1497
+ ws.push(line) # the line is durable from its first run
1498
+ return line
1499
+ ws.git("checkout", "-q", "-B", line, f"origin/{line}")
1500
+ conflicted = False
1501
+ try:
1502
+ ws.git(
1503
+ *git_identity(bot_login),
1504
+ "merge",
1505
+ "--no-edit",
1506
+ base_ref,
1507
+ )
1508
+ except Exception:
1509
+ conflicted = True
1510
+ log.info(
1511
+ "line %s: merging %s conflicts; left as the session's first task", line, base_branch
1512
+ )
1513
+ _reset_instruction_files(ws, workspace, base_ref)
1514
+ if not conflicted:
1515
+ ws.git("add", "-A")
1516
+ if ws.git("status", "--porcelain").strip():
1517
+ ws.git(
1518
+ *git_identity(bot_login),
1519
+ "commit",
1520
+ "-q",
1521
+ "-m",
1522
+ f"line hygiene: instruction files reset to {base_branch}",
1523
+ )
1524
+ # Run-START persistence: the branch exists on the remote from its
1525
+ # first run, and the merge-main + hygiene state survives a crashed
1526
+ # run. One slot never runs twice concurrently, so this is a fast-
1527
+ # forward. The run-END push of the sealed session tree is the next
1528
+ # phase PR (it requires all-terminal sealing; the push must publish
1529
+ # a sealed sha, never invent a commit). A conflicted merge is not
1530
+ # pushed: the conflict is session work, not line state.
1531
+ ws.push(line)
1532
+ return line
1533
+
1534
+
1535
+ def _paths_changed_from_base(
1536
+ ws: Workspace, base: str, exclude_memory: bool, fallback: str = "HEAD"
1537
+ ) -> list[str]:
1538
+ """The paths the session changed, measured against the BASE BRANCH head
1539
+ (`base`, a commit-ish such as refs/remotes/origin/main), not HEAD. On a
1540
+ research line HEAD can be behind main: a run starts by merging main in,
1541
+ and when that merge conflicts it stays uncommitted, so the files git
1542
+ auto-merged (the project's BENCHMARKS.md and results/leader.json) sit
1543
+ staged against the stale HEAD while being identical to main. Those are
1544
+ main's edits, not the agent's, and must not read as scope violations
1545
+ (gpt-speedrun, 2026-09-03: agent-01's first run after its own win ended
1546
+ scope-violation on exactly those two files). A path counts only when it
1547
+ moved against HEAD AND differs from the base; new and deleted files
1548
+ count. `fallback` is used when `base` does not resolve (no remote).
1549
+ `exclude_memory` drops the line's memory files whenever lines are active
1550
+ for the benchmark (a failed line checkout still keeps them out)."""
1551
+ try:
1552
+ base_commit = ws.git("rev-parse", "--verify", f"{base}^{{commit}}").strip()
1553
+ except Exception:
1554
+ base_commit = fallback
1555
+ ws.git("add", "-A")
1556
+ try:
1557
+ staged = ws.staged_paths()
1558
+ if not staged:
1559
+ return []
1560
+ # index vs base, restricted to what moved against HEAD: a path identical
1561
+ # to base drops out, a new or deleted file still counts
1562
+ differs = {
1563
+ entry
1564
+ for entry in ws.git(
1565
+ "diff", "--cached", "--name-only", "-z", base_commit, "--", *staged
1566
+ ).split("\0")
1567
+ if entry
1568
+ }
1569
+ finally:
1570
+ ws.git("reset")
1571
+ kept = [p for p in staged if p in differs]
1572
+ return [p for p in kept if not (exclude_memory and _is_line_memory(p))]
1573
+
1574
+
1575
+ def _sibling_entries(ws: Workspace, self_agent: str) -> list[dict]:
1576
+ """The other agents' live directions from the research-log's
1577
+ status.json (already fetched: the SAME FETCH_HEAD the reports came
1578
+ from). Size-checked BEFORE show like the report blobs, entries and
1579
+ fields bounded — the branch is bot-written but never trusted with
1580
+ unbounded memory. Any failure means no siblings known, never a crash."""
1581
+ try:
1582
+ blob = "FETCH_HEAD:climb/status.json"
1583
+ if int(ws.git("cat-file", "-s", blob).strip()) > 1_000_000:
1584
+ raise ValueError("status snapshot oversized; skipped")
1585
+ fleet = json.loads(ws.git("show", blob))
1586
+ return [
1587
+ {
1588
+ "agent": str(r.get("agent", ""))[:64],
1589
+ "state": str(r.get("state", ""))[:32],
1590
+ "phase": str(r.get("phase", ""))[:32],
1591
+ "direction": str(r.get("direction", ""))[:160],
1592
+ }
1593
+ for r in fleet.get("runs", [])[:64]
1594
+ if isinstance(r, dict) and r.get("agent") != self_agent
1595
+ ]
1596
+ except Exception as exc:
1597
+ log.info("no sibling snapshot for this session (%s)", exc)
1598
+ return []
1599
+
1600
+
1601
+ def _install_report_archive(workspace: Path, reports: list[tuple[str, str]]) -> None:
1602
+ """Materialize the fetched reports under the kernel-owned channel
1603
+ (`.outerloop/reports/`) so the session can read and search the full
1604
+ texts with its own tools; the brief inlines only the newest few."""
1605
+ dest = workspace / channel_dir(workspace) / "reports"
1606
+ dest.mkdir(parents=True, exist_ok=True)
1607
+ for name, text in reports:
1608
+ if Path(name).name != name: # branch content is remote-controlled
1609
+ continue
1610
+ (dest / name).write_text(text)
1611
+
1612
+
1613
+ def _stage_launches(record: RunRecord) -> list[dict]:
1614
+ """The persisted launch descriptors of an author-sleep stage (name +
1615
+ artifacts), tolerating a malformed entry by skipping it (the job dirs are
1616
+ keyed by name; a nameless entry has nothing to gather)."""
1617
+ raw = record.stage.get("syscall_launches", [])
1618
+ if not isinstance(raw, list):
1619
+ return []
1620
+ return [item for item in raw if isinstance(item, dict) and item.get("name")]
1621
+
1622
+
1623
+ def _utc_date(now: float) -> str:
1624
+ from datetime import UTC, datetime
1625
+
1626
+ return datetime.fromtimestamp(now, UTC).strftime("%Y-%m-%d")
1627
+
1628
+
1629
+ def _is_git_tamper(exc: GitError) -> bool:
1630
+ """True when a GitError means the workspace's object store or refs are
1631
+ damaged or session-altered — as opposed to an ordinary git failure (a push
1632
+ conflict, a fetch outage). The guard raises its own `_altered(...)` for the
1633
+ states it models; a raw object-store error ("could not parse object", "bad
1634
+ tree object", a corrupt or unreadable pack) is the same damage reaching a
1635
+ git command in the gap between the guard's check and the command. Either
1636
+ way the wake cannot be trusted and must END, not crash."""
1637
+ msg = str(exc).lower()
1638
+ return (
1639
+ "altered by the session" in msg # the guard's own _altered signal
1640
+ or "could not parse object" in msg
1641
+ or "bad tree object" in msg
1642
+ or "bad object" in msg
1643
+ or "unable to read tree" in msg
1644
+ or "data stream error" in msg # inflate: a corrupt pack
1645
+ or "is corrupt" in msg # "packed object ... is corrupt", loose too
1646
+ or ("object file" in msg and "empty" in msg)
1647
+ )
1648
+
1649
+
1650
+ def _end_refused_wake(
1651
+ run_root: Path, record: RunRecord, exc: Exception, now: float, secrets: tuple[str, ...]
1652
+ ) -> AttemptOutcome:
1653
+ """End a parked run whose workspace the wake refused (a session altered
1654
+ .git): ABORTED with the tampering as the note. The candidate snapshot's
1655
+ retaining ref lives inside that same, now untrusted, repository and is
1656
+ deliberately not touched: deleting it would mean writing through the
1657
+ very structure the guard refused (a symlinked refs dir carries the write
1658
+ elsewhere), and an ENDED workspace is inert — the ref keeps a commit
1659
+ alive only within a repository nothing will read again."""
1660
+ note = redact(str(exc), secrets)[:480]
1661
+ log.warning("wake refused for %s: %s", record.run_id, note)
1662
+ failed = _clear_stage(
1663
+ RunRecord(**{**record.__dict__, "state": ENDED, "ending": ABORTED, "ending_note": note})
1664
+ )
1665
+ _best_effort("ending record", lambda: save_record(run_root, failed, now), secrets)
1666
+ return AttemptOutcome(run_id=record.run_id, outcome="attempt-error")
1667
+
1668
+
1669
+ def resume_run(
1670
+ run_root: Path,
1671
+ run_id: str,
1672
+ *,
1673
+ dispatch: DispatchSettings,
1674
+ github: GitHubClient,
1675
+ bot_auth: TokenProvider,
1676
+ now: float,
1677
+ secrets: tuple[str, ...] = (),
1678
+ base_branch: str = "main",
1679
+ panel_lenses: tuple[PanelLens, ...] = (),
1680
+ harness: Harness | None = None,
1681
+ spec: RoleSpec | None = None,
1682
+ ) -> AttemptOutcome:
1683
+ """Wake a parked dispatched climb and re-enter its decision WITHOUT the
1684
+ session (`orchestrator.resume_attempt`), from the record `_park_run` wrote.
1685
+ The three exits:
1686
+
1687
+ * **re-park** — the wake dispatched a measure that is not done yet (the
1688
+ suite pairs an improving candidate fans out, "another round of
1689
+ experiments"): `resume_attempt` raises `RunParked`, and this re-persists
1690
+ the WAITING stage on the new afterany, keeping the same candidate
1691
+ snapshot;
1692
+ * **a negative terminal** (no-improvement / suite-regression / eval-error):
1693
+ drop the candidate snapshot and end the record;
1694
+ * **improved** — branch the SEALED `candidate_sha` (never the live tree,
1695
+ which may have drifted since the park; the diff was scope-checked so it
1696
+ carries only in-scope changes), layer the ledger update on top, push, and
1697
+ open the PR. A moved base is NOT merged and re-measured here
1698
+ (docs/design/research-loop.md): a stale PR is a re-wake, not an
1699
+ orchestrator auto-merge.
1700
+ """
1701
+ run_dir = run_root / "runs" / run_id
1702
+ workspace = run_dir / "ws"
1703
+ record = load_record(run_root, run_id)
1704
+ dispatch = with_seed(dispatch, run_root, record.target)
1705
+ stage = record.stage
1706
+ # Push to the CANONICAL target URL, never the workspace's remote.origin.url:
1707
+ # the session could have rewritten that config to exfil the bot token / code
1708
+ # to another remote. Passing `url` here means `Workspace.push` uses it
1709
+ # instead of reading `remote.origin.url`.
1710
+ ws = Workspace(root=workspace, auth=bot_auth, url=target_clone_url(record.target))
1711
+ # A session reshaped .git (symlinked object store, gitdir file, FIFO) is
1712
+ # refused BEFORE anything writes through it: the exclude below opens
1713
+ # .git/info/exclude, and every ws.git call re-checks. The refusal ENDS
1714
+ # the parked run with the tampering as its note — the tree cannot be
1715
+ # trusted, and a record left waiting would only be re-woken into the
1716
+ # same refusal.
1717
+ try:
1718
+ ensure_regular_git_dir(workspace)
1719
+ except GitError as exc:
1720
+ return _end_refused_wake(run_root, record, exc, now, secrets)
1721
+ # Re-establish the merge-artifact exclude on the wake too: the workspace
1722
+ # persisted across the park, but a session could have removed the exclude,
1723
+ # and this wake's changed_paths / seal run `git add -A`. Idempotent.
1724
+ _exclude_merge_artifacts(workspace)
1725
+ # Refresh origin refs on EVERY wake: the clone's refs froze at run
1726
+ # start, and this is the one credential-free freshness point — the
1727
+ # kernel fetches (from the canonical URL, never the session-writable
1728
+ # remote config), the session only ever reads local refs. `sleep`
1729
+ # thereby doubles as the author's sync primitive. Best-effort: a fetch
1730
+ # outage must not cost the wake.
1731
+ try:
1732
+ ws.fetch_origin()
1733
+ except Exception as exc:
1734
+ log.warning("wake fetch failed for %s: %s", run_id, exc)
1735
+
1736
+ # Two park kinds reach the wake: a CANDIDATE park (the gate's measures were
1737
+ # dispatched) and an AUTHOR-SLEEP park (the author launched work and slept —
1738
+ # research-loop-buildout.md Phase A). Both carry a sealed sha. Anything else
1739
+ # is a stray record: guard rather than crash on `git diff base ""`.
1740
+ if stage.get("phase") not in ("candidate", "author-sleep") or not stage.get("candidate_sha"):
1741
+ raise EvalError(f"resume_run: run {run_id} is not a wakeable park (stage={stage!r})")
1742
+
1743
+ base_sha = str(stage["base_sha"])
1744
+ candidate_sha = str(stage["candidate_sha"])
1745
+ candidate_ref = str(stage["candidate_ref"])
1746
+ issue_number = record.issue_number
1747
+ # the run's target branch rides the stage, so a wake opens its PR against
1748
+ # the branch the ORIGINAL climb selected — not the CLI's default (the wake
1749
+ # job carries no --base-branch).
1750
+ base_branch = str(stage.get("base_branch") or base_branch)
1751
+
1752
+ # The contract gates scope and names the eval command, so read it from the
1753
+ # BASE commit (the tree the run started on), NOT the working tree the
1754
+ # session left dirty — a session that widened its own scope in
1755
+ # its contract must not have the wake gate on the doctored rules.
1756
+ contract_text = contract_at(ws, base_sha)
1757
+ contract = load_contract(contract_text, record.target)
1758
+ bench = _benchmark(contract, record.benchmark)
1759
+ config = RunConfig(target=record.target, benchmark=record.benchmark, agent_id=record.agent_id)
1760
+ eval_minutes = next(
1761
+ (b.eval_minutes for b in contract.benchmarks if b.name == record.benchmark), None
1762
+ )
1763
+ # a submitted park carries the author's declared eval walltime: the
1764
+ # wake's measurer (a re-dispatch) and deadline floor honor it
1765
+ declared = int(record.stage.get("eval_minutes", 0) or 0) # type: ignore[call-overload]
1766
+ if declared:
1767
+ eval_minutes = declared
1768
+ measurer = dispatch.measurer(
1769
+ run_dir, repo_root=workspace, eval_minutes=int(eval_minutes or 0), run_tag=run_id
1770
+ )
1771
+ # measured_paths from the COMMITTED base..candidate diff — the sealed
1772
+ # candidate, never `changed_paths()` on a live tree that may have drifted.
1773
+ # NUL-delimited (like Workspace.staged_paths) so a path with a space is one
1774
+ # entry, not two that could each slip past the scope check. The same
1775
+ # line-memory rule as the climb's changed_paths(): the base is the line
1776
+ # tip, the seal excluded the memory, the diff must not read it as a change.
1777
+ measured_paths = tuple(
1778
+ _without_line_memory(
1779
+ (
1780
+ p
1781
+ for p in ws.git("diff", "--name-only", "-z", base_sha, candidate_sha).split("\0")
1782
+ if p
1783
+ ),
1784
+ _line_ref_for(bench, config.agent_id),
1785
+ )
1786
+ )
1787
+ seed = int(stage["seed"]) # type: ignore[call-overload]
1788
+ suite_seed = int(stage["suite_seed"]) # type: ignore[call-overload]
1789
+ panel_reads = int(stage.get("panel_reads", 0)) # type: ignore[call-overload]
1790
+
1791
+ if stage.get("phase") == "author-sleep":
1792
+ # The author slept on launches: deliver their results and RESUME the
1793
+ # same session through the climb's resume-entry. Every exit is a park
1794
+ # (slept again, or the gate dispatched its measures -> a candidate park
1795
+ # the NEXT wake decides through the path below) or a terminal ending —
1796
+ # never a publish, so the publish tail stays candidate-only.
1797
+ return _wake_author_sleep(
1798
+ run_root=run_root,
1799
+ run_id=run_id,
1800
+ record=record,
1801
+ ws=ws,
1802
+ workspace=workspace,
1803
+ run_dir=run_dir,
1804
+ dispatch=dispatch,
1805
+ github=github,
1806
+ now=now,
1807
+ secrets=secrets,
1808
+ base_branch=base_branch,
1809
+ base_sha=base_sha,
1810
+ sleep_ref=candidate_ref,
1811
+ contract_text=contract_text,
1812
+ contract=contract,
1813
+ bench=bench,
1814
+ config=config,
1815
+ measurer=measurer,
1816
+ harness=harness,
1817
+ spec=spec,
1818
+ panel_lenses=panel_lenses,
1819
+ issue_number=issue_number,
1820
+ eval_minutes=eval_minutes,
1821
+ )
1822
+
1823
+ # rebuild the session from what the park saved: the (redacted) write-up and
1824
+ # its real spend, so the report shows true cost/turns. It is never re-run.
1825
+ session = SessionResult(
1826
+ stop_reason="resumed",
1827
+ is_error=False,
1828
+ cost_usd=float(stage.get("session_cost_usd", 0.0)), # type: ignore[arg-type]
1829
+ num_turns=int(stage.get("session_turns", 0)), # type: ignore[call-overload]
1830
+ session_id=record.resume_session_id,
1831
+ final_text=str(stage.get("report", "")),
1832
+ transcript_path="",
1833
+ )
1834
+ try:
1835
+ result = resume_attempt(
1836
+ contract,
1837
+ bench,
1838
+ base_sha=base_sha,
1839
+ candidate_sha=candidate_sha,
1840
+ seed=seed,
1841
+ suite_seed=suite_seed,
1842
+ measured_paths=measured_paths,
1843
+ session=session,
1844
+ measurer=measurer,
1845
+ min_relative_improvement=config.min_relative_improvement,
1846
+ )
1847
+ except RunParked as parked:
1848
+ # another measure this wake dispatched is not done — re-park on the new
1849
+ # afterany, keeping the SAME candidate snapshot the next wake reads.
1850
+ # PROGRESS only if this wake dispatched a NEW job set (e.g. the candidate
1851
+ # resolved and the suite pairs fanned out); a blind re-park (empty
1852
+ # afterany) or the same jobs still pending is NO progress, so the stuck
1853
+ # cap must keep counting.
1854
+ if stage.get("submitted"):
1855
+ # No author was woken yet, so a SUBMITTED park's re-park (the suite
1856
+ # fanned out) still owes the author the gate+panel results and its
1857
+ # sibling launches' results — carry the submit context forward, or
1858
+ # the next wake drafts instead of waking the author and the launch
1859
+ # descriptors are lost. Commands/minutes are spent
1860
+ # history; the wake needs only names + artifacts (as persisted).
1861
+ from outerloop.syscall import Launch as _Launch
1862
+ from outerloop.syscall import SyscallRequest as _SyscallRequest
1863
+
1864
+ parked.submitted = True
1865
+ parked.launches_used = int(stage.get("launches_used", 0)) # type: ignore[call-overload]
1866
+ parked.sleeps_used = int(stage.get("sleeps_used", 0)) # type: ignore[call-overload]
1867
+ parked.gpu_hours_used = float(stage.get("gpu_hours_used", 0.0)) # type: ignore[arg-type]
1868
+ parked.eval_minutes = int(stage.get("eval_minutes", 0) or 0) or None # type: ignore[call-overload]
1869
+ parked.judged = parked.judged or _stage_judged(record)
1870
+ parked.launch_afterany = parked.launch_afterany or str(stage.get("launch_afterany", ""))
1871
+ if parked.syscall is None:
1872
+ parked.syscall = _SyscallRequest(
1873
+ launches=tuple(
1874
+ _Launch(
1875
+ name=str(item.get("name", "")),
1876
+ command="(ran)",
1877
+ # the persisted walltime, so the re-park's deadline
1878
+ # floor still covers the longest launch
1879
+ minutes=int(item.get("minutes") or 1),
1880
+ artifacts=tuple(str(a) for a in item.get("artifacts", [])),
1881
+ array=int(item.get("array") or 1),
1882
+ why=str(item.get("why") or ""),
1883
+ concurrency=int(item.get("concurrency") or 0),
1884
+ )
1885
+ for item in _stage_launches(record)
1886
+ ),
1887
+ note=str(stage.get("syscall_note", "")),
1888
+ submit=True,
1889
+ # the author's report rides every re-park: a suite fan-out
1890
+ # must not drop what the panel and the PR read
1891
+ report=str(stage.get("report", "")),
1892
+ )
1893
+ old_afterany = str(record.stage.get("afterany", ""))
1894
+ made_progress = bool(parked.afterany) and parked.afterany != old_afterany
1895
+ _park_run(
1896
+ run_root,
1897
+ record,
1898
+ parked,
1899
+ candidate_ref,
1900
+ eval_minutes,
1901
+ now,
1902
+ secrets,
1903
+ dispatch=dispatch,
1904
+ keep_wake_attempts=not made_progress,
1905
+ base_branch=base_branch,
1906
+ panel_reads=panel_reads,
1907
+ )
1908
+ return AttemptOutcome(run_id=run_id, outcome="parked")
1909
+
1910
+ # A SUBMITTED park (the author's `submit` syscall, buildout Phase B): gate
1911
+ # and panel results go back to the AUTHOR — it revises and resubmits, runs
1912
+ # more experiments, or concludes — instead of being decided by policy here.
1913
+ # Falls back to the plain-finish behavior (negative terminal / draft PR)
1914
+ # when the session cannot be resumed.
1915
+ submitted_park = bool(stage.get("submitted"))
1916
+ author_resumable = (
1917
+ harness is not None
1918
+ and spec is not None
1919
+ and bool(record.resume_session_id)
1920
+ and getattr(harness, "supports_resume", True)
1921
+ )
1922
+
1923
+ # the park's sibling launches are done too: settle their charge before
1924
+ # any path — publish or hand back to the author — reads the budget
1925
+ if _stage_launches(record):
1926
+ sibling_launches = _stage_syscall_launches(record)
1927
+ sibling_ids = launch_task_ids(sibling_launches, stage_launch_job_ids(record))
1928
+ sibling_elapsed = _launch_elapsed(dispatch, sibling_ids) if sibling_ids else None
1929
+ _reconcile_launch_hours(record, dispatch, bench.gpus, sibling_launches, sibling_elapsed)
1930
+ # the ledger's ended records for the sibling launches, whether or not
1931
+ # the author is woken: the PR's experiments table reads them (an
1932
+ # author wake that follows records nothing twice)
1933
+ _best_effort(
1934
+ "launch ledger",
1935
+ lambda: _ledger_ended(
1936
+ run_dir, record, sibling_launches, sibling_ids, dispatch, sibling_elapsed
1937
+ ),
1938
+ )
1939
+
1940
+ def _wake_author(
1941
+ extra_update: str, judged: tuple[str, AttemptResult] | None = None
1942
+ ) -> AttemptOutcome:
1943
+ # resume the submitted park's author with the gate/panel feedback
1944
+ # leading its wake text; the candidate ref is this park's held snapshot
1945
+ return _wake_author_sleep(
1946
+ run_root=run_root,
1947
+ run_id=run_id,
1948
+ record=record,
1949
+ ws=ws,
1950
+ workspace=workspace,
1951
+ run_dir=run_dir,
1952
+ dispatch=dispatch,
1953
+ github=github,
1954
+ now=now,
1955
+ secrets=secrets,
1956
+ base_branch=base_branch,
1957
+ base_sha=base_sha,
1958
+ sleep_ref=candidate_ref,
1959
+ contract_text=contract_text,
1960
+ contract=contract,
1961
+ bench=bench,
1962
+ config=config,
1963
+ measurer=measurer,
1964
+ harness=harness,
1965
+ spec=spec,
1966
+ panel_lenses=panel_lenses,
1967
+ issue_number=issue_number,
1968
+ eval_minutes=eval_minutes,
1969
+ extra_update=extra_update,
1970
+ judged=judged,
1971
+ )
1972
+
1973
+ if (
1974
+ submitted_park
1975
+ and author_resumable
1976
+ and result.outcome in ("no-improvement", "suite-regression", "eval-error")
1977
+ ):
1978
+ # the submitted candidate failed the gate — including an eval that
1979
+ # errored: feedback, never a silent terminal — the author decides
1980
+ # what happens next (rounds stay bounded by sleep_k)
1981
+ return _wake_author(
1982
+ "Your `submit` did NOT clear the gate: "
1983
+ f"{result.note or result.outcome} "
1984
+ f"(baseline {result.baseline}, candidate {result.candidate}). "
1985
+ "Revise and submit again, run more experiments, or finish with an "
1986
+ "honest negative report.",
1987
+ # the verdict rides the resume: the same tree, sealed again after
1988
+ # the author concludes, is not measured twice; only an explicit
1989
+ # resubmit runs an errored eval again
1990
+ judged=(candidate_sha, result),
1991
+ )
1992
+
1993
+ def _notebook(outcome: str) -> None:
1994
+ # Research lines: record the tree AS OF THIS DECIDED TERMINAL — never
1995
+ # earlier, because a blocking panel verdict can still resume the
1996
+ # author (a continuation, not a terminal).
1997
+ _push_line_snapshot(
1998
+ ws,
1999
+ _line_ref_for(bench, config.agent_id),
2000
+ run_id,
2001
+ outcome,
2002
+ secrets,
2003
+ bot_login=config.bot_login,
2004
+ )
2005
+
2006
+ if result.outcome == "improved":
2007
+ # Publish: branch the SEALED candidate sha, fold in the ledger, push,
2008
+ # open the PR. No moved-base merge (research-loop.md) — a stale PR is a
2009
+ # re-wake, not an auto-merge.
2010
+ from datetime import UTC, datetime
2011
+
2012
+ assert result.baseline is not None and result.candidate is not None
2013
+ baseline, candidate = result.baseline, result.candidate
2014
+ # a zero-change "improvement" is metric noise, not progress — never a PR
2015
+ # (defense in depth; measure_and_decide already requires a real delta,
2016
+ # and an empty base..candidate diff implies baseline == candidate).
2017
+ if not measured_paths:
2018
+ result = dc_replace(
2019
+ result, outcome="no-improvement", note="no code change; metric noise"
2020
+ )
2021
+ drop_snapshot(ws, Snapshot(commit=candidate_sha, tree="", ref=candidate_ref))
2022
+ final = _clear_stage(
2023
+ RunRecord(**{**record.__dict__, "state": ENDED, "ending": NEGATIVE_RESULT})
2024
+ )
2025
+ _best_effort("final record", lambda: save_record(run_root, final, now), secrets)
2026
+ _notebook("no-improvement")
2027
+ return AttemptOutcome(run_id=run_id, outcome="no-improvement")
2028
+
2029
+ from datetime import UTC as _UTC
2030
+ from datetime import datetime as _dt
2031
+
2032
+ # seal the notebook before any checkout mutates the persisted session
2033
+ # tree (the memory files are excluded from the sealed candidate and
2034
+ # would not survive the force-checkout + clean below)
2035
+ _notebook("improved")
2036
+
2037
+ branch = f"{config.branch_prefix}/{run_id}"
2038
+
2039
+ # IDEMPOTENCY: a prior wake may have opened the PR but died before
2040
+ # recording it (leaving the run WAITING). On re-entry, if a PR is already
2041
+ # open for this head->base, reconcile to it — do NOT re-push
2042
+ # (non-fast-forward) or open a duplicate. The reconcile does the FULL
2043
+ # terminal (branch checkout, arm, report, issue, in-review record); it
2044
+ # only SKIPS the push + create_pull the prior wake already did. A lookup
2045
+ # failure just falls through to the normal publish.
2046
+ existing: dict[str, object] | None = None
2047
+ try:
2048
+ existing = github.find_open_pull_for_head(config.target, branch, base_branch)
2049
+ except Exception as exc:
2050
+ log.warning(
2051
+ "idempotency PR lookup failed for %s: %s",
2052
+ run_id,
2053
+ redact(f"{type(exc).__name__}: {exc}", secrets),
2054
+ )
2055
+ if existing:
2056
+ pr_url = str(existing.get("html_url", ""))
2057
+ log.info("run %s: PR %s already open; reconciling the record", run_id, pr_url)
2058
+ # put the workspace on the branch (a later follow-up expects it) and
2059
+ # finish the steps the prior wake may have died before completing.
2060
+ _best_effort(
2061
+ "reconcile checkout", lambda: ws.git("checkout", "-f", "-B", branch, candidate_sha)
2062
+ )
2063
+ pr_number = pr_url.rstrip("/").rsplit("/", 1)[-1]
2064
+ if pr_number.isdigit() and not existing.get("draft"):
2065
+ _arm_unless_base_moved(
2066
+ github,
2067
+ ws,
2068
+ config.target,
2069
+ pr_number,
2070
+ base_branch,
2071
+ base_sha,
2072
+ secrets,
2073
+ merge_mode=getattr(contract, "merge", "manual"),
2074
+ panel_ran=result.panel_rounds > 0,
2075
+ )
2076
+ report_path = run_dir / "report.md"
2077
+ _best_effort(
2078
+ "run report",
2079
+ lambda: report_path.write_text(result.report(config, redact_secrets=secrets)),
2080
+ secrets,
2081
+ )
2082
+ final = _clear_stage(
2083
+ RunRecord(
2084
+ **{
2085
+ **record.__dict__,
2086
+ "state": IN_REVIEW,
2087
+ "pr_url": pr_url,
2088
+ "auto_blessed_head": _blessed_head(ws, result, contract),
2089
+ "resume_session_id": result.session.session_id if result.session else "",
2090
+ "ending_note": pr_url,
2091
+ }
2092
+ )
2093
+ )
2094
+ if _best_effort("final record", lambda: save_record(run_root, final, now), secrets):
2095
+ drop_snapshot(ws, Snapshot(commit=candidate_sha, tree="", ref=candidate_ref))
2096
+ _post_issue_finished(
2097
+ github,
2098
+ config.target,
2099
+ issue_number,
2100
+ run_id,
2101
+ "improved",
2102
+ pr_url,
2103
+ redact(result.report(config, redact_secrets=secrets), secrets)[:8000],
2104
+ secrets,
2105
+ )
2106
+ return AttemptOutcome(
2107
+ run_id=run_id, outcome="improved", pr_url=pr_url, report_path=str(report_path)
2108
+ )
2109
+
2110
+ try:
2111
+ # FORCE-checkout the sealed candidate: at wake the workspace still
2112
+ # holds the session's dirty tree (HEAD is pre_session_sha), so a
2113
+ # plain checkout could be blocked; the sha already captured exactly
2114
+ # the measured content.
2115
+ ws.git("checkout", "-f", "-B", branch, candidate_sha)
2116
+ # `checkout -f` does NOT remove untracked files, and the panel's
2117
+ # `git add -A` (in build_panel_runner) would sweep any post-snapshot
2118
+ # cruft into the tree it judges. Clean untracked (non-ignored) files
2119
+ # so the panel reads EXACTLY candidate_sha. The ledger commit stages
2120
+ # only PROGRESS_PATHS, so it was never affected.
2121
+ ws.git("clean", "-fd")
2122
+
2123
+ # Verification panel on the credited claim — the SAME gate the
2124
+ # inline path runs (docs/design/orchestrator-verify.md), so a
2125
+ # dispatched improvement is not published unverified. It reads the
2126
+ # workspace tree, now checked out to the SEALED candidate_sha (the
2127
+ # dispatched evals ran on node-local scratch, so the tree is exactly
2128
+ # what was measured), over base_sha. A blocking or degraded
2129
+ # verdict opens a DRAFT PR carrying the findings and never arms
2130
+ # auto-merge; a clean verdict (or no panel) arms.
2131
+ if panel_lenses:
2132
+ # A panel ERROR (a git op in build_panel_runner, not a finding)
2133
+ # must NOT abort the publish and drop the candidate snapshot —
2134
+ # the improvement is real and measured. Fail closed to DEGRADED:
2135
+ # open a DRAFT for a human, keep the candidate. (run_panel itself
2136
+ # already fails closed per-lens; this catches the git setup.)
2137
+ try:
2138
+ verdict = build_panel_runner(
2139
+ ws,
2140
+ run_dir,
2141
+ base_sha,
2142
+ panel_lenses,
2143
+ contract_text,
2144
+ config.target,
2145
+ config.benchmark,
2146
+ config.bot_login,
2147
+ _dt.fromtimestamp(now, _UTC).strftime("%Y-%m-%d"),
2148
+ exclude=(
2149
+ LINE_MEMORY_PATHS if _line_ref_for(bench, config.agent_id) else ()
2150
+ ),
2151
+ secrets=secrets,
2152
+ )(baseline, candidate, str(stage.get("report", "")))
2153
+ except Exception as exc:
2154
+ if isinstance(exc, GitError) and _is_git_tamper(exc):
2155
+ # tamper during the panel is not a panel error to draft
2156
+ # around — end as a refused wake.
2157
+ return _end_refused_wake(run_root, record, exc, now, secrets)
2158
+ log.warning(
2159
+ "wake panel errored for %s (%s); opening a DRAFT",
2160
+ run_id,
2161
+ redact(f"{type(exc).__name__}: {exc}", secrets),
2162
+ )
2163
+ verdict = PanelVerdict(
2164
+ blocking=(),
2165
+ transcript="panel setup failed — NOT a clean read",
2166
+ wake_text="",
2167
+ degraded=True,
2168
+ )
2169
+ reads = panel_reads + 1
2170
+ # DEPTH AXIS (docs/design/research-loop.md): blocking findings
2171
+ # on a SUBMITTED claim go back to the AUTHOR (buildout Phase B)
2172
+ # — it revises and resubmits (a fresh seal + gate + panel), or
2173
+ # concludes. A plain finish (or an unresumable session) DRAFTs
2174
+ # the PR with the findings open for a human to triage.
2175
+ if bool(verdict.blocking) and submitted_park and author_resumable:
2176
+ return _wake_author(verdict.wake_text)
2177
+ result = dc_replace(
2178
+ result,
2179
+ panel_transcript=verdict.transcript,
2180
+ panel_rounds=reads,
2181
+ panel_blocking_open=bool(verdict.blocking),
2182
+ panel_degraded=verdict.degraded,
2183
+ )
2184
+
2185
+ entries = update_leader(
2186
+ load_leader(workspace),
2187
+ benchmark=bench.name,
2188
+ metric=bench.metric,
2189
+ direction=bench.direction,
2190
+ baseline=baseline,
2191
+ candidate=candidate,
2192
+ run_id=run_id,
2193
+ date=datetime.fromtimestamp(now, UTC).strftime("%Y-%m-%d"),
2194
+ run_seed=result.run_seed,
2195
+ )
2196
+ write_progress(
2197
+ workspace,
2198
+ entries,
2199
+ config.target,
2200
+ digits={b.name: b.display_digits for b in contract.benchmarks if b.display_digits},
2201
+ )
2202
+ # Stage ONLY the ledger files on top of the sealed candidate — never
2203
+ # `git add -A`, which would sweep in untracked cruft the session left
2204
+ # (eval caches) that was neither measured nor scope-checked. The
2205
+ # candidate content is already vetted (measure_and_decide's scope
2206
+ # check on measured_paths); assert nothing but the ledger is staged.
2207
+ ws.git("add", "--", *PROGRESS_PATHS)
2208
+ staged = ws.staged_paths()
2209
+ extra = [p for p in staged if p not in PROGRESS_PATHS]
2210
+ if extra:
2211
+ raise WorkspaceDrift(f"wake commit would stage non-ledger paths: {extra[:10]}")
2212
+ # Commit the ledger update on top of the sealed candidate ONLY when
2213
+ # it actually moved. When the candidate beat its baseline but not the
2214
+ # recorded best, update_leader is a no-op (the ledger's `best` does
2215
+ # not advance) — a valid composable win with no leaderboard change,
2216
+ # so push the candidate as-is rather than an empty commit.
2217
+ if staged:
2218
+ ws.git(
2219
+ *git_identity(config.bot_login),
2220
+ "commit",
2221
+ "-m",
2222
+ f"agent: improve {config.benchmark} ({_title_pair(baseline, candidate)})"
2223
+ f"\n\nAgent: {config.agent_id}",
2224
+ )
2225
+ ws.push(branch)
2226
+ if record.stage.get("submitted"):
2227
+ # the author's report at submit rides the stage: the PR shows it
2228
+ # as the research report, over the ledger's experiments
2229
+ result = dc_replace(result, submit_report=str(record.stage.get("report") or ""))
2230
+ body = pr_body(
2231
+ result,
2232
+ config,
2233
+ redact_secrets=secrets,
2234
+ display_digits=bench.display_digits,
2235
+ experiments=experiments_rows(run_dir),
2236
+ )
2237
+ if issue_number:
2238
+ body = f"Addresses #{issue_number}.\n\n{body}"
2239
+ # blocking findings still open at the panel, or a degraded final
2240
+ # read, mean a human must look: open a DRAFT and never arm. A clean
2241
+ # verdict (or no panel configured) opens non-draft and arms
2242
+ # auto-merge only where branch protection requires a review — same
2243
+ # policy as the inline path.
2244
+ draft = result.panel_blocking_open or result.panel_degraded
2245
+ pr_url = github.create_pull(
2246
+ config.target,
2247
+ title=f"[agent] {config.benchmark}: {_title_pair(baseline, candidate)}",
2248
+ head=branch,
2249
+ base=base_branch,
2250
+ body=body,
2251
+ draft=draft,
2252
+ )
2253
+ pr_number = pr_url.rstrip("/").rsplit("/", 1)[-1]
2254
+ if pr_number.isdigit() and not draft:
2255
+ _arm_unless_base_moved(
2256
+ github,
2257
+ ws,
2258
+ config.target,
2259
+ pr_number,
2260
+ base_branch,
2261
+ base_sha,
2262
+ secrets,
2263
+ merge_mode=getattr(contract, "merge", "manual"),
2264
+ panel_ran=result.panel_rounds > 0,
2265
+ )
2266
+ except Exception as exc:
2267
+ if isinstance(exc, GitError) and _is_git_tamper(exc):
2268
+ # a concurrent object-store removal during the publish block is
2269
+ # tamper, not a publish failure: end as a refused wake (the
2270
+ # clean tamper terminal) rather than mask it as a publish-error.
2271
+ return _end_refused_wake(run_root, record, exc, now, secrets)
2272
+ # push / PR / commit failed — end as an error. Save the ENDED record
2273
+ # BEFORE dropping the snapshot (same ordering as the other terminals):
2274
+ # a failed save then leaves the run WAITING with its snapshot intact
2275
+ # (recoverable), never WAITING with the candidate already gone. On a
2276
+ # successful end, drop the snapshot — ENDED runs are never swept, so
2277
+ # keeping it would only leak the ref (a retry is a fresh climb, not a
2278
+ # re-wake). Never delete a remote branch (a push may have
2279
+ # half-succeeded).
2280
+ note = redact(f"{type(exc).__name__}: {exc}", secrets)[:480]
2281
+ log.warning("wake publish failed for %s: %s", run_id, note)
2282
+ failed = _clear_stage(
2283
+ RunRecord(
2284
+ **{**record.__dict__, "state": ENDED, "ending": ABORTED, "ending_note": note}
2285
+ )
2286
+ )
2287
+ if _best_effort("ending record", lambda: save_record(run_root, failed, now), secrets):
2288
+ drop_snapshot(ws, Snapshot(commit=candidate_sha, tree="", ref=candidate_ref))
2289
+ _notebook("publish-error")
2290
+ return AttemptOutcome(run_id=run_id, outcome="publish-error")
2291
+ # PR opened. Record IN_REVIEW *before* dropping the snapshot: if the save
2292
+ # fails, the record stays `waiting` with the snapshot intact, so the run
2293
+ # is recoverable rather than an ABORTED record over a live PR.
2294
+ report_path = run_dir / "report.md"
2295
+ _best_effort(
2296
+ "run report",
2297
+ lambda: report_path.write_text(result.report(config, redact_secrets=secrets)),
2298
+ secrets,
2299
+ )
2300
+ final = _clear_stage(
2301
+ RunRecord(
2302
+ **{
2303
+ **record.__dict__,
2304
+ "state": IN_REVIEW,
2305
+ "pr_url": pr_url,
2306
+ "auto_blessed_head": _blessed_head(ws, result, contract),
2307
+ "resume_session_id": result.session.session_id if result.session else "",
2308
+ "ending_note": pr_url,
2309
+ }
2310
+ )
2311
+ )
2312
+ if _best_effort("final record", lambda: save_record(run_root, final, now), secrets):
2313
+ drop_snapshot(ws, Snapshot(commit=candidate_sha, tree="", ref=candidate_ref))
2314
+ else:
2315
+ log.warning(
2316
+ "run %s: PR %s opened but in-review record unsaved; snapshot kept", run_id, pr_url
2317
+ )
2318
+ _post_issue_finished(
2319
+ github,
2320
+ config.target,
2321
+ issue_number,
2322
+ run_id,
2323
+ "improved",
2324
+ pr_url,
2325
+ redact(result.report(config, redact_secrets=secrets), secrets)[:8000],
2326
+ secrets,
2327
+ )
2328
+ return AttemptOutcome(
2329
+ run_id=run_id, outcome="improved", pr_url=pr_url, report_path=str(report_path)
2330
+ )
2331
+
2332
+ # a negative terminal: end the record, THEN release the snapshot. Save
2333
+ # BEFORE dropping (same ordering as the improved path): if the save fails,
2334
+ # the run stays WAITING with its snapshot intact, so a re-wake can still
2335
+ # reconstruct — never WAITING with the snapshot already gone.
2336
+ _notebook(result.outcome)
2337
+ report_path = run_dir / "report.md"
2338
+ _best_effort(
2339
+ "run report",
2340
+ lambda: report_path.write_text(result.report(config, redact_secrets=secrets)),
2341
+ secrets,
2342
+ )
2343
+ final = _clear_stage(
2344
+ RunRecord(
2345
+ **{
2346
+ **record.__dict__,
2347
+ "state": ENDED,
2348
+ "ending": _ENDINGS_BY_OUTCOME[result.outcome],
2349
+ "ending_note": redact(result.note, secrets),
2350
+ }
2351
+ )
2352
+ )
2353
+ if _best_effort("final record", lambda: save_record(run_root, final, now), secrets):
2354
+ drop_snapshot(ws, Snapshot(commit=candidate_sha, tree="", ref=candidate_ref))
2355
+ else:
2356
+ log.warning(
2357
+ "run %s: ended negative but record unsaved; snapshot kept for a re-wake", run_id
2358
+ )
2359
+ _post_issue_finished(
2360
+ github,
2361
+ config.target,
2362
+ issue_number,
2363
+ run_id,
2364
+ result.outcome,
2365
+ "",
2366
+ redact(result.report(config, redact_secrets=secrets), secrets)[:8000],
2367
+ secrets,
2368
+ )
2369
+ return AttemptOutcome(run_id=run_id, outcome=result.outcome, report_path=str(report_path))
2370
+
2371
+
2372
+ def _judge_lens_key(
2373
+ *,
2374
+ backend: str,
2375
+ key_file_env: str,
2376
+ author_backend: str,
2377
+ claude_panel_path: Path,
2378
+ image: str,
2379
+ ) -> str:
2380
+ """Resolve a non-claude judge lens's OWN key file, enforcing the three
2381
+ separations every shelled judge shares (codex, hermes, any future
2382
+ backend): the image is required (a judge never runs uncontained next to
2383
+ key files), the key must be named explicitly, and it must differ from BOTH
2384
+ the author's key of the same provider AND the claude panel key (a
2385
+ cross-provider send would leak an anthropic credential to another login).
2386
+ Returns the redacted key (or "" under an ADC-covered deployment)."""
2387
+ if not image:
2388
+ raise ValueError(
2389
+ f"a {backend} panel lens requires --image (a shelled judge only "
2390
+ "ever runs inside the container)"
2391
+ )
2392
+ raw = os.environ.get(key_file_env, "").strip()
2393
+ if not raw:
2394
+ raise ValueError(
2395
+ f"a {backend} panel lens needs {key_file_env} "
2396
+ "(role separation: the judge's own key, never the author's)"
2397
+ )
2398
+ path = Path(raw).expanduser()
2399
+ author_path = Path(resolve_author_key_file(author_backend)).expanduser()
2400
+ if path.resolve() == author_path.resolve():
2401
+ raise ValueError(
2402
+ f"{backend} panel key file {path} is the {author_backend} author key "
2403
+ "(role separation: the judge needs its own key)"
2404
+ )
2405
+ if path.resolve() == claude_panel_path.resolve():
2406
+ raise ValueError(
2407
+ f"{backend} panel key file {path} is the claude panel key file "
2408
+ "(an anthropic key must never reach another provider's login)"
2409
+ )
2410
+ return role_key(raw, author_backend)
2411
+
2412
+
2413
+ def _panel_lenses_from_args(args: Any) -> tuple[tuple[PanelLens, ...], tuple[str, ...]]:
2414
+ """Build the verification-panel lenses from the CLI args (empty `--panel`
2415
+ disables it), returning `(lenses, panel_secrets)` — the ONE owner of
2416
+ panel credentials: each backend's judge key is read only when a lens uses
2417
+ it, role separation is enforced HERE (a manual climb gets the same rule
2418
+ as the tick preflight), and every key a judge holds joins the caller's
2419
+ redaction set via `panel_secrets`. Shared by the fresh-climb and the
2420
+ `--resume` wake paths so a dispatched improvement runs the SAME panel as
2421
+ an inline one. Raises ValueError on a bad panel/backend config — a
2422
+ configured gate must never silently vanish."""
2423
+ import os
2424
+
2425
+ if not args.panel.strip():
2426
+ return (), ()
2427
+ from outerloop.panel import parse_lenses
2428
+ from outerloop.roles import reviewer_spec
2429
+
2430
+ parsed = parse_lenses(args.panel)
2431
+ # the anthropic panel key is read only when a claude lens will use it —
2432
+ # a codex-only panel must not demand an unrelated credential
2433
+ panel_key = role_key(args.panel_key_file) if any(b == "claude" for _, b, _ in parsed) else ""
2434
+ lenses = []
2435
+ secrets: list[str] = [panel_key] if panel_key else []
2436
+ for kind, backend, model in parsed:
2437
+ hermes_repo_env = os.environ.get("REVIEW_HERMES_REPO", "").strip()
2438
+ # per-backend judge keys coexist — a codex lens is never handed the
2439
+ # anthropic panel key, and role separation forbids defaulting to the
2440
+ # AUTHOR's codex key: the judge key is its own, named explicitly
2441
+ claude_panel_path = Path(args.panel_key_file or PANEL_KEY_DEFAULT).expanduser()
2442
+ if backend == "codex":
2443
+ lens_key = _judge_lens_key(
2444
+ backend="codex",
2445
+ key_file_env="OUTERLOOP_PANEL_CODEX_KEY_FILE",
2446
+ author_backend="codex",
2447
+ claude_panel_path=claude_panel_path,
2448
+ image=args.image,
2449
+ )
2450
+ if lens_key:
2451
+ secrets.append(lens_key)
2452
+ elif backend == "hermes":
2453
+ # hermes reads its key from its provider's env var, but the FILE
2454
+ # is resolved and separated exactly like codex's (the key still
2455
+ # lands next to the session). The author's OpenAI key coexists, so
2456
+ # separate against the codex author key.
2457
+ lens_key = _judge_lens_key(
2458
+ backend="hermes",
2459
+ key_file_env="OUTERLOOP_PANEL_HERMES_KEY_FILE",
2460
+ author_backend="codex",
2461
+ claude_panel_path=claude_panel_path,
2462
+ image=args.image,
2463
+ )
2464
+ if lens_key:
2465
+ secrets.append(lens_key)
2466
+ else:
2467
+ lens_key = panel_key
2468
+ try:
2469
+ if not args.image:
2470
+ log.warning(
2471
+ "panel lens %s runs uncontained (local mode, no image): the judge "
2472
+ "shares this machine with the operator's keys",
2473
+ backend,
2474
+ )
2475
+ judge = build_harness(
2476
+ lens_key,
2477
+ reviewer_spec(),
2478
+ backend=backend,
2479
+ binary=args.claude_bin if backend == "claude" else args.codex_bin,
2480
+ model=model or None,
2481
+ # ALWAYS contained: the panel runs on the climb host next to key
2482
+ # files, and a judge now holds a shell (codex `danger-full-access`),
2483
+ # so it must run inside the image. `parse_lenses` gates panel
2484
+ # backends to those containable here (claude today); passing the
2485
+ # image unconditionally means codex is safe the moment it is
2486
+ # enabled, never accidentally uncontained.
2487
+ container_image=args.image,
2488
+ hermes_repo=Path(hermes_repo_env) if hermes_repo_env else None,
2489
+ hermes_provider=os.environ.get("REVIEW_HERMES_PROVIDER", "openrouter"),
2490
+ )
2491
+ except ValueError as exc:
2492
+ raise ValueError(f"panel entry {kind}:{backend}: {exc}") from exc
2493
+ lenses.append(PanelLens(kind=kind, harness=judge))
2494
+ return tuple(lenses), tuple(dict.fromkeys(secrets))
2495
+
2496
+
2497
+ def _panel_claim_body(
2498
+ benchmark: str, baseline: float, candidate: float, report: str, *, lines: bool
2499
+ ) -> str:
2500
+ """The synthetic claim the panel judges. On a research-lines target the
2501
+ one-contribution mandate is part of the claim itself: the panel is the
2502
+ backstop against a line's accumulated tweaks reaching main as one PR
2503
+ (docs/design/research-lines.md)."""
2504
+ mandate = (
2505
+ "\n\nThis target runs research lines: a PR to main must be ONE "
2506
+ "clean contribution, extracted onto the base branch. A diff that "
2507
+ "bundles unrelated or unablated changes is a BLOCKING finding — "
2508
+ "name the pieces that should be separated."
2509
+ if lines
2510
+ else ""
2511
+ )
2512
+ return (
2513
+ f"Automated improvement claim (pre-PR): {benchmark} "
2514
+ f"{baseline} -> {candidate}, measured by the orchestrator.{mandate}\n\n"
2515
+ f"## Research report\n\n*Session prose, written before "
2516
+ f"the orchestrator measured.*\n\n{report[:MAX_CLAIM_CHARS]}"
2517
+ )
2518
+
2519
+
2520
+ def build_panel_runner(
2521
+ ws: Workspace,
2522
+ run_dir: Path,
2523
+ base_sha: str,
2524
+ lenses: tuple[PanelLens, ...],
2525
+ contract_text: str,
2526
+ target: str,
2527
+ benchmark: str,
2528
+ bot_login: str,
2529
+ today: str,
2530
+ start_round: int = 0,
2531
+ exclude: tuple[str, ...] = (),
2532
+ claim_body: Callable[[float, float, str], str] | None = None,
2533
+ secrets: tuple[str, ...] = (),
2534
+ ) -> Callable[[float, float, str], PanelVerdict]:
2535
+ """The git half of the pre-PR panel: prepare the two read-only checkouts
2536
+ and the synthetic claim, then hand off to `run_panel` (which owns no git).
2537
+
2538
+ Each call snapshots the CURRENT workspace tree as a detached commit and
2539
+ checks it out as `pr-head/` (sanitized — the candidate is an untrusted
2540
+ tree), next to `base/` (the trusted pre-session commit: contract and
2541
+ ruler). Worktrees are removed after the read; a fresh pair is built per
2542
+ round because the tree changes with every revision.
2543
+
2544
+ `claim_body` renders the claim the panel judges from (baseline,
2545
+ candidate, report); the default is the pre-PR improvement claim, a
2546
+ follow-up re-read passes its own wording."""
2547
+ from outerloop.review_agent import sanitize_checkout
2548
+
2549
+ reads = {"n": start_round}
2550
+ render_claim = claim_body or (
2551
+ lambda baseline, candidate, report: _panel_claim_body(
2552
+ benchmark, baseline, candidate, report, lines=bool(exclude)
2553
+ )
2554
+ )
2555
+
2556
+ def runner(baseline: float, candidate: float, report: str) -> PanelVerdict:
2557
+ # the claim is author text (the report at submit, or the session's
2558
+ # last words): redacted before any lens sees it, like the record and
2559
+ # the PR body
2560
+ report = redact(report, secrets)
2561
+ reads["n"] += 1
2562
+ panel_ws = run_dir / "panel"
2563
+ shutil.rmtree(panel_ws, ignore_errors=True)
2564
+ panel_ws.mkdir(parents=True, exist_ok=True)
2565
+ ws.git("add", "-A")
2566
+ if exclude:
2567
+ # the panel judges the CLAIM — the same tree the gate measured,
2568
+ # which excludes line memory (docs/design/research-lines.md)
2569
+ ws.git("rm", "--cached", "-r", "-q", "--ignore-unmatch", "--", *exclude)
2570
+ tree = ws.git("write-tree").strip()
2571
+ ws.git("reset")
2572
+ snapshot = ws.git(
2573
+ *git_identity(bot_login),
2574
+ "commit-tree",
2575
+ tree,
2576
+ "-p",
2577
+ base_sha,
2578
+ "-m",
2579
+ "panel snapshot (never pushed)",
2580
+ ).strip()
2581
+ try:
2582
+ ws.git("worktree", "add", "--detach", str(panel_ws / "base"), base_sha)
2583
+ ws.git("worktree", "add", "--detach", str(panel_ws / "pr-head"), snapshot)
2584
+ _renamed, failed = sanitize_checkout(panel_ws / "pr-head")
2585
+ if failed:
2586
+ # fail closed for the read, loudly in the transcript: an
2587
+ # unsanitizable tree is never judged, and never certified
2588
+ return PanelVerdict(
2589
+ blocking=(),
2590
+ transcript=(
2591
+ f"**Verification round {reads['n']}**\n- panel skipped: "
2592
+ f"the candidate tree could not be sanitized "
2593
+ f"({failed} instruction file(s) left) — NOT a clean read"
2594
+ ),
2595
+ wake_text="",
2596
+ degraded=True,
2597
+ )
2598
+ claim = PullRequest(
2599
+ repo=target,
2600
+ number=0,
2601
+ title=f"[agent] {benchmark}: {_title_pair(baseline, candidate)}",
2602
+ body=render_claim(baseline, candidate, report),
2603
+ # base..snapshot, never base..worktree: the snapshot commit
2604
+ # includes newly ADDED files, which a working-tree diff omits.
2605
+ # Excluded (line-memory) paths are excluded from the diff too:
2606
+ # the snapshot dropped them, so against a line-tip base they
2607
+ # would read as deletions the author never made
2608
+ diff=ws.git(
2609
+ "diff",
2610
+ f"{base_sha}..{snapshot}",
2611
+ *(["--", ".", *(f":(exclude){p}" for p in exclude)] if exclude else []),
2612
+ ),
2613
+ author=bot_login,
2614
+ )
2615
+ return run_panel(lenses, panel_ws, claim, contract_text, today, reads["n"])
2616
+ finally:
2617
+ for name in ("base", "pr-head"):
2618
+ _best_effort(
2619
+ "panel worktree cleanup",
2620
+ partial(ws.git, "worktree", "remove", "--force", str(panel_ws / name)),
2621
+ )
2622
+ _best_effort("panel dir removal", lambda: shutil.rmtree(panel_ws, ignore_errors=True))
2623
+ _best_effort("panel worktree prune", lambda: ws.git("worktree", "prune"))
2624
+
2625
+ return runner
2626
+
2627
+
2628
+ def live_attempt(
2629
+ config: RunConfig,
2630
+ run_root: Path,
2631
+ run_id: str,
2632
+ harness: Harness,
2633
+ github: GitHubClient,
2634
+ bot_auth: TokenProvider,
2635
+ now: float,
2636
+ created: str,
2637
+ secrets: tuple[str, ...] = (),
2638
+ base_branch: str = "main",
2639
+ issue_number: int = 0,
2640
+ author_backend: str = "claude",
2641
+ author_model: str = "",
2642
+ author_key_file: str = "",
2643
+ task_hypothesis: str = "",
2644
+ spec: RoleSpec | None = None,
2645
+ panel_lenses: tuple[PanelLens, ...] = (),
2646
+ dispatch: DispatchSettings | None = None,
2647
+ eval_image: str = "",
2648
+ ) -> AttemptOutcome:
2649
+ """Run one climb against the real target repo. With `panel_lenses`, the
2650
+ pre-PR verification panel gates the claim before any PR exists
2651
+ (docs/design/orchestrator-verify.md); blocking findings still open at
2652
+ the cap open a DRAFT PR carrying them."""
2653
+ run_dir = run_root / "runs" / run_id
2654
+ run_dir.mkdir(parents=True, exist_ok=True)
2655
+ workspace = run_dir / "ws"
2656
+
2657
+ # The record exists before any network or clone work: every crash from
2658
+ # here on has a record to end.
2659
+ import os as _os
2660
+
2661
+ record = RunRecord(
2662
+ run_id=run_id,
2663
+ target=config.target,
2664
+ task_title=f"improve {config.benchmark}",
2665
+ benchmark=config.benchmark,
2666
+ state="implementing",
2667
+ agent_id=config.agent_id,
2668
+ deadline=now + 24 * 3600,
2669
+ issue_number=issue_number,
2670
+ author_backend=author_backend,
2671
+ author_model=author_model,
2672
+ author_key_file=author_key_file,
2673
+ run_job_id=_os.environ.get("SLURM_JOB_ID", ""),
2674
+ )
2675
+ try:
2676
+ save_record(run_root, record, now)
2677
+ except Exception as exc:
2678
+ # No record could be written, so the run must not proceed invisibly:
2679
+ # nothing would ever end it. Submit-time evidence (the claim comment
2680
+ # or the pending marker) plus this post keep the failure visible.
2681
+ exc_name = type(exc).__name__
2682
+ log.warning(
2683
+ "could not create run record for %s: %s",
2684
+ run_id,
2685
+ redact(f"{exc_name}: {exc}", secrets),
2686
+ )
2687
+ if issue_number:
2688
+ _best_effort(
2689
+ "issue report",
2690
+ lambda: github.comment(
2691
+ config.target,
2692
+ issue_number,
2693
+ f"Run `{run_id}` could not start ({exc_name} while writing its run record).",
2694
+ ),
2695
+ secrets,
2696
+ )
2697
+ return AttemptOutcome(run_id=run_id, outcome="attempt-error")
2698
+
2699
+ # what the attempt-error handler needs to salvage the line notebook: the
2700
+ # exception path cannot rely on names bound inside the try
2701
+ salvage: dict[str, object] = {}
2702
+ try:
2703
+ ws = Workspace.clone(target_clone_url(config.target), workspace, auth=bot_auth)
2704
+ # Build ON the requested PR base: the clone checks out the remote
2705
+ # DEFAULT branch, which need not be `base_branch` — the session must
2706
+ # edit, and the gate must measure, the tree the PR will land on.
2707
+ # A missing base branch fails loudly as attempt-error.
2708
+ ws.git("checkout", "-q", "-B", base_branch, f"origin/{base_branch}")
2709
+ _exclude_merge_artifacts(workspace)
2710
+ contract_text = contract_text_in_tree(workspace)
2711
+ contract = load_contract(contract_text, config.target)
2712
+ # Load the brief budget from the contract and run state: callers do
2713
+ # not supply it (the dataclass default rendered "0.0 GPU-hours" and
2714
+ # honest agents refused to launch). Same weekly counting rule as the
2715
+ # tick's cap: records plus live pending markers, minus this run's own.
2716
+ from outerloop.tick import list_pendings
2717
+
2718
+ week_ago = now - 7 * 24 * 3600
2719
+ recent = [
2720
+ r for r in list_runs(run_root) if r.target == config.target and r.created >= week_ago
2721
+ ]
2722
+ # a marker whose job already has a record (this run's included) is
2723
+ # the same attempt, not a second one — count each job once
2724
+ recorded_jobs = {r.run_job_id for r in recent if r.run_job_id} | {record.run_job_id}
2725
+ # every unrecorded week-fresh marker counts: the tick reaps dead
2726
+ # markers on its own cadence (with the squeue liveness reads a brief
2727
+ # must not make), so an unreaped marker is either a live queued run
2728
+ # the weekly cap WILL count, or dead for at most a sweep — the brief
2729
+ # stays on the cap's conservative side either way
2730
+ used_week = len(recent) + sum(
2731
+ 1
2732
+ for _agent, marker in list_pendings(run_root, config.target)
2733
+ if float(marker.get("submitted_at", 0) or 0) >= week_ago
2734
+ and str(marker.get("job_id", "")) not in recorded_jobs
2735
+ )
2736
+ _budget_bench = next((b for b in contract.benchmarks if b.name == config.benchmark), None)
2737
+ config = dc_replace(
2738
+ config,
2739
+ budget=BudgetState(
2740
+ gpu_hours_remaining=(
2741
+ float(contract.budgets.gpu_hours_per_run or 0.0)
2742
+ if _budget_bench is not None and _budget_bench.gpus
2743
+ else 0.0
2744
+ ),
2745
+ runs_remaining_this_week=max(0, int(contract.budgets.runs_per_week) - used_week),
2746
+ ),
2747
+ )
2748
+ # Author syscalls (research-loop.md, "one syscall") are CONTRACT-DRIVEN:
2749
+ # armed whenever the deployment can deliver them — dispatch coords (the
2750
+ # launches and the gate run as Slurm jobs) and a resumable backend (the
2751
+ # wake resumes the SAME session) — and the benchmark has not opted out
2752
+ # (`depth_k: 0`). With the channel (`.outerloop/`) armed it never
2753
+ # enters diffs or scope — repo-local exclude. With the feature off, an
2754
+ # untracked `.outerloop/` file must be staged and judged like any
2755
+ # other agent edit, not silently hidden by a magic dir name (the off
2756
+ # state stays byte-identical).
2757
+ _bench = next((b for b in contract.benchmarks if b.name == config.benchmark), None)
2758
+ # Research lines: move HEAD to the agent's own branch BEFORE anything
2759
+ # reads the tree — the contract above came from the base branch (a
2760
+ # line must not shape its own budgets), and the syscall-channel check
2761
+ # below must see the line's tree. A failed checkout falls back to the
2762
+ # base branch: a run is never lost to its notebook.
2763
+ lines_active = _bench is not None and _bench.lines and bool(config.agent_id)
2764
+ line_ref = ""
2765
+ if lines_active:
2766
+ try:
2767
+ line_ref = _checkout_line(
2768
+ ws, workspace, config.agent_id, base_branch, config.bot_login
2769
+ )
2770
+ except Exception as exc:
2771
+ log.warning(
2772
+ "line checkout failed (%s); running on %s",
2773
+ redact(f"{type(exc).__name__}: {exc}", secrets),
2774
+ base_branch,
2775
+ )
2776
+ _best_effort("line merge abort", lambda: ws.git("merge", "--abort"))
2777
+ ws.git("checkout", "-q", "-B", base_branch, f"origin/{base_branch}")
2778
+ line_memory = ""
2779
+ line_divergence = ""
2780
+ if line_ref:
2781
+ salvage.update(ws=ws, line_ref=line_ref)
2782
+ try:
2783
+ # the line's own memory index, rendered into the brief
2784
+ # (data-fenced there); topic files are read on demand from
2785
+ # the checkout, never rendered
2786
+ memory_path = workspace / "AGENT_MEMORY.md"
2787
+ if memory_path.is_file() and not memory_path.is_symlink():
2788
+ # byte-mode bounded read: never load an oversized file
2789
+ with memory_path.open("rb") as fh:
2790
+ line_memory = fh.read(65_536).decode("utf-8", errors="replace")
2791
+ except OSError as exc:
2792
+ log.warning("could not read AGENT_MEMORY.md: %s", exc)
2793
+ try:
2794
+ # divergence debt, made visible each session (a conflicted
2795
+ # merge skips it — the diff is not meaningful mid-merge)
2796
+ if not ws.git("diff", "--name-only", "--diff-filter=U").strip():
2797
+ line_divergence = ws.git(
2798
+ "diff", "--shortstat", f"origin/{base_branch}", "HEAD"
2799
+ ).strip()
2800
+ except Exception:
2801
+ line_divergence = ""
2802
+ author_syscalls = (
2803
+ dispatch is not None
2804
+ and getattr(harness, "supports_resume", True)
2805
+ and _bench is not None
2806
+ and _bench.depth_k > 0
2807
+ )
2808
+ # The `.outerloop/` channel must be KERNEL-OWNED. In a fresh clone,
2809
+ # anything already at that path was committed by the TARGET — a symlink
2810
+ # (install would write through it to a host path with our permissions),
2811
+ # a tracked request (free cluster compute), or
2812
+ # any other booby trap. If the path pre-exists in ANY form, disable the
2813
+ # feature for the run, loudly; otherwise we create a dir we own.
2814
+ # A target must not ship EITHER channel name (both are booby-trap risks:
2815
+ # a symlink install writes through, a planted request steals compute).
2816
+ shipped = next(
2817
+ (
2818
+ n
2819
+ for n in CHANNEL_DIR_NAMES
2820
+ if (workspace / n).is_symlink() or (workspace / n).exists()
2821
+ ),
2822
+ "",
2823
+ )
2824
+ if author_syscalls and shipped:
2825
+ log.warning(
2826
+ "target ships a %s path (symlink=%s); author syscalls disabled for this run",
2827
+ shipped,
2828
+ (workspace / shipped).is_symlink(),
2829
+ )
2830
+ author_syscalls = False
2831
+ reports = _fetch_research_reports(ws, MAX_ARCHIVED_REPORTS)
2832
+ if author_syscalls:
2833
+ assert _bench is not None
2834
+ syscall_excluded(workspace)
2835
+ # the author's interface is the TOOL (`python .outerloop/syscall
2836
+ # launch ... -- <cmd>`; `... sleep`), never the raw ABI file —
2837
+ # install it plus the informational budget its `status` shows.
2838
+ syscall_install_tool(workspace)
2839
+ syscall_write_budget(
2840
+ workspace,
2841
+ launches_remaining=_bench.depth_k,
2842
+ sleeps_remaining=_bench.sleep_k,
2843
+ gpu_hours_remaining=(
2844
+ float(contract.budgets.gpu_hours_per_run) if _bench.gpus else None
2845
+ ),
2846
+ )
2847
+ # AFTER install_tool: installing the tool recreates the channel
2848
+ # dir it owns, which would delete an archive written earlier
2849
+ _install_report_archive(workspace, reports)
2850
+ # the fleet snapshot the `siblings` command shows — read from the
2851
+ # research-log's status.json (the SAME branch the reports came
2852
+ # from, so it works across clusters) and best-effort throughout:
2853
+ # a missing or malformed snapshot just means no siblings known
2854
+ syscall_write_siblings(workspace, _sibling_entries(ws, config.agent_id))
2855
+
2856
+ def changed_paths() -> list[str]:
2857
+ # against the base branch head, never the line tip a conflicted
2858
+ # merge can leave HEAD on (see _paths_changed_from_base)
2859
+ return _paths_changed_from_base(ws, f"refs/remotes/origin/{base_branch}", lines_active)
2860
+
2861
+ if issue_number:
2862
+ from outerloop.intake import CLAIM_MARKER
2863
+
2864
+ already = any(
2865
+ has_marker(str(c.get("body", "")), "claimed")
2866
+ for c in github.list_comments(config.target, issue_number)
2867
+ )
2868
+ if not already: # manual CLI runs claim here; tick runs claimed at submit
2869
+ github.comment(
2870
+ config.target,
2871
+ issue_number,
2872
+ f"{CLAIM_MARKER}\nPicked up as run `{run_id}` "
2873
+ f"(benchmark `{config.benchmark}`). A report will follow here.",
2874
+ )
2875
+
2876
+ # Expensive benchmarks measure as dispatched cluster jobs; cheap ones
2877
+ # (and any run with no cluster coordinates) measure inline. The choice
2878
+ # is the benchmark's eval-time hint against the in-job runway, decided
2879
+ # ONCE here so the baseline setup, the measurer, and the park deadline
2880
+ # all agree on it.
2881
+ eval_minutes = next(
2882
+ (b.eval_minutes for b in contract.benchmarks if b.name == config.benchmark), None
2883
+ )
2884
+ wants_dispatch = should_dispatch(eval_minutes)
2885
+ dispatched = dispatch is not None and wants_dispatch
2886
+ if wants_dispatch and dispatch is None:
2887
+ # a benchmark asked to be dispatched but no cluster coordinates
2888
+ # reached us — never silently: name it, then measure inline
2889
+ log.warning(
2890
+ "benchmark %s wants dispatched eval (eval_minutes=%s) but no cluster "
2891
+ "coordinates (image/account/partition) are set; measuring inline",
2892
+ config.benchmark,
2893
+ eval_minutes,
2894
+ )
2895
+
2896
+ # the panel's base is the PRE-SESSION commit — the exact tree the
2897
+ # baseline was measured on — never origin/<base_branch>, which can
2898
+ # name a different branch than the clone's checkout
2899
+ pre_session_sha = ws.git("rev-parse", "HEAD").strip()
2900
+ panel_runner = (
2901
+ build_panel_runner(
2902
+ ws,
2903
+ run_dir,
2904
+ pre_session_sha,
2905
+ panel_lenses,
2906
+ contract_text,
2907
+ config.target,
2908
+ config.benchmark,
2909
+ config.bot_login,
2910
+ created[:10],
2911
+ exclude=LINE_MEMORY_PATHS if lines_active else (),
2912
+ secrets=secrets,
2913
+ )
2914
+ if panel_lenses
2915
+ else None
2916
+ )
2917
+ # ONE measurer either way: every measure is a job that checks its
2918
+ # tree sha out fresh from this workspace's `refs/dispatch/*` and
2919
+ # writes its result to the run dir. DISPATCHED: the jobs go to the
2920
+ # cluster and a not-yet-done measure PARKS the climb. LOCAL: the SAME
2921
+ # jobs run synchronously in this allocation (LocalCompute), so every
2922
+ # measure is done when checked and nothing parks. `snapshot` commits
2923
+ # the workspace's current content to a candidate sha and we own the
2924
+ # ref lifecycle, keeping the one candidate ref a park needs and
2925
+ # dropping the rest when the climb ends.
2926
+ measurer: Measurer
2927
+ if dispatched:
2928
+ assert dispatch is not None and eval_minutes is not None # should_dispatch(None) False
2929
+ measurer = dispatch.measurer(
2930
+ run_dir, repo_root=workspace, eval_minutes=eval_minutes, run_tag=run_id
2931
+ )
2932
+ else:
2933
+ measurer = DispatchedMeasurer(
2934
+ compute=LocalCompute(),
2935
+ run_dir=run_dir,
2936
+ repo_root=workspace,
2937
+ # a configured image contains LOCAL evals too — the cluster
2938
+ # triple being incomplete must not silently drop the jail
2939
+ image=dispatch.image if dispatch is not None else eval_image,
2940
+ account="",
2941
+ partition="",
2942
+ eval_minutes=int(eval_minutes or 0),
2943
+ run_tag=run_id,
2944
+ # an inline gate shares the same target-wide baseline cache
2945
+ baseline_cache=run_dir.parent / "baselines",
2946
+ seed_cache=dispatch.seed_cache if dispatch is not None else None,
2947
+ )
2948
+ snapshots: list[Snapshot] = []
2949
+
2950
+ def snapshot() -> str:
2951
+ snap = snapshot_tree(
2952
+ ws,
2953
+ pre_session_sha,
2954
+ exclude=LINE_MEMORY_PATHS if lines_active else (),
2955
+ author=config.bot_login,
2956
+ )
2957
+ snapshots.append(snap)
2958
+ return snap.commit
2959
+
2960
+ # `author_syscalls` already folds every enablement condition — dispatch
2961
+ # coords, a resumable backend, the benchmark's opt-out, and the
2962
+ # channel-ownership guard above (a target-shipped `.autoresearch` —
2963
+ # symlink, tracked request, or any other pre-existing form — has
2964
+ # disabled the feature for this run).
2965
+ launcher = None
2966
+ if author_syscalls:
2967
+ assert dispatch is not None # folded into author_syscalls above
2968
+ launcher = _make_launcher(
2969
+ dispatch, run_dir, workspace, run_id, gpus=_bench.gpus if _bench else 0
2970
+ )
2971
+
2972
+ parked: RunParked | None = None
2973
+ kept_ref = "" # the ONE candidate snapshot ref that must outlive a park
2974
+ try:
2975
+ # the last-known score orients the brief only; the gate re-measures
2976
+ # both sides after the session, so None (a first run) is fine.
2977
+ prior_best = load_leader(workspace).get(config.benchmark)
2978
+ result = attempt_once(
2979
+ config,
2980
+ contract_text,
2981
+ workspace,
2982
+ harness,
2983
+ measurer,
2984
+ pre_session_sha,
2985
+ snapshot,
2986
+ ruler=RULER,
2987
+ changed_paths=changed_paths,
2988
+ created=created,
2989
+ task_hypothesis=task_hypothesis,
2990
+ recent_reports=tuple(text for _name, text in reports),
2991
+ lessons=distill_lessons(reports),
2992
+ report_archive=author_syscalls,
2993
+ spec=spec,
2994
+ panel_runner=panel_runner,
2995
+ brief_baseline=prior_best.best if prior_best else None,
2996
+ line_ref=line_ref,
2997
+ line_memory=line_memory,
2998
+ line_divergence=line_divergence,
2999
+ launcher=launcher,
3000
+ watcher=(
3001
+ _make_watcher(dispatch, run_root, run_id, workspace, config)
3002
+ if dispatch is not None
3003
+ else None
3004
+ ),
3005
+ tree_of=lambda sha: ws.git("rev-parse", f"{sha}^{{tree}}").strip(),
3006
+ )
3007
+ except RunParked as p:
3008
+ # The climb dispatched its measures and hibernated. Persist the
3009
+ # re-entry stage as a WAITING record (not an error), keep the
3010
+ # candidate snapshot alive for the wake, and end. The wake re-enters
3011
+ # from the record. `parked` is set only
3012
+ # AFTER a successful write: if _park_run raises, it stays None so the
3013
+ # finally drops every snapshot (no leak) and the outer handler ends
3014
+ # the run as an error rather than a half-written hibernation.
3015
+ if p.phase in ("candidate", "author-sleep"):
3016
+ # keep exactly ONE snapshot for that sha (two can share a
3017
+ # commit); record and keep that same ref, drop the rest. An
3018
+ # author-sleep's snapshot is the tree the wake re-delivers.
3019
+ kept_ref = next((s.ref for s in snapshots if s.commit == p.candidate_sha), "")
3020
+ import time
3021
+
3022
+ # anchor the deadline to the PARK (when the evals were submitted),
3023
+ # not the run's start `now` — a session lasting hours would otherwise
3024
+ # eat the queue budget and let the sweep cancel a still-queued eval.
3025
+ try:
3026
+ _park_run(
3027
+ run_root,
3028
+ record,
3029
+ p,
3030
+ kept_ref,
3031
+ eval_minutes,
3032
+ time.time(),
3033
+ secrets,
3034
+ dispatch=dispatch,
3035
+ base_branch=base_branch,
3036
+ )
3037
+ except Exception:
3038
+ # The WAITING record did not persist, so nothing will ever wake
3039
+ # the eval jobs this park already submitted. Cancel them so they
3040
+ # don't sit in the queue as orphans (best-effort, self-logging),
3041
+ # then fall through to the error handler — `parked` stays None,
3042
+ # so the finally still drops every snapshot. A park only happens
3043
+ # on the dispatched path, so `dispatch` is set here.
3044
+ assert dispatch is not None
3045
+ for job_id in afterany_ids(p.afterany):
3046
+ dispatch.compute.cancel(job_id)
3047
+ raise
3048
+ parked = p
3049
+ return AttemptOutcome(run_id=run_id, outcome="parked")
3050
+ finally:
3051
+ for snap in snapshots:
3052
+ # a candidate park must OUTLIVE the wake — keep the ONE recorded
3053
+ # snapshot (matched by ref, not commit); drop every other one.
3054
+ if parked and kept_ref and snap.ref == kept_ref:
3055
+ continue
3056
+ drop_snapshot(ws, snap) # best-effort + self-logging; never raises
3057
+ except Exception as exc:
3058
+ exc_name = type(exc).__name__
3059
+ note = redact(f"{exc_name}: {exc}", secrets)[:500]
3060
+ log.warning("climb failed for %s: %s", run_id, note)
3061
+ if salvage:
3062
+ # a crashed attempt's tree is still notebook-worthy (best-effort)
3063
+ _push_line_snapshot(
3064
+ cast(Workspace, salvage["ws"]),
3065
+ str(salvage["line_ref"]),
3066
+ run_id,
3067
+ "attempt-error",
3068
+ secrets,
3069
+ bot_login=config.bot_login,
3070
+ )
3071
+ failed = RunRecord(
3072
+ **{
3073
+ **record.__dict__,
3074
+ "state": ENDED,
3075
+ "ending": ABORTED,
3076
+ "ending_note": note,
3077
+ }
3078
+ )
3079
+ report_path = run_dir / "report.md"
3080
+ _best_effort("ending record", lambda: save_record(run_root, failed, now), secrets)
3081
+ wrote = _best_effort(
3082
+ "error report",
3083
+ lambda: report_path.write_text(
3084
+ f"# Run report — {config.target} / {config.benchmark}\n"
3085
+ f"Outcome: **attempt-error**\n"
3086
+ f"Note: {note}\n"
3087
+ ),
3088
+ secrets,
3089
+ )
3090
+ if issue_number:
3091
+ # Exception detail stays in the local record and report: redact()
3092
+ # only knows the secrets it was handed, and raw messages can carry
3093
+ # paths or tokens the tuple does not cover. The issue gets the
3094
+ # exception TYPE only.
3095
+ _best_effort(
3096
+ "issue report",
3097
+ lambda: github.comment(
3098
+ config.target,
3099
+ issue_number,
3100
+ f"Run `{run_id}` finished (attempt-error): {exc_name}. "
3101
+ f"Details are in the run's record and report on the orchestrator.",
3102
+ ),
3103
+ secrets,
3104
+ )
3105
+ return AttemptOutcome(
3106
+ run_id=run_id,
3107
+ outcome="attempt-error",
3108
+ # an outcome must never point at a report that was not written
3109
+ report_path=str(report_path) if wrote else "",
3110
+ )
3111
+
3112
+ if result.outcome == "improved" and not result.measured_paths:
3113
+ # a zero-change "improvement" is metric noise, not progress — never a
3114
+ # PR (same rule as the wake publish)
3115
+ result = dc_replace(result, outcome="no-improvement", note="no code change; metric noise")
3116
+
3117
+ report = result.report(config, redact_secrets=secrets)
3118
+ report_path = run_dir / "report.md"
3119
+ wrote_report = _best_effort("run report", lambda: report_path.write_text(report), secrets)
3120
+
3121
+ # Research lines: seal the notebook NOW, while the tree is still the
3122
+ # session's final tree — the publish below force-checkouts the sealed
3123
+ # candidate and cleans untracked files, which would drop the agent's
3124
+ # memory (it is excluded from measurable seals by design). The label is
3125
+ # the GATE outcome, correct at this moment; a publish failure appends a
3126
+ # publish-error snapshot at the tail.
3127
+ _push_line_snapshot(ws, line_ref, run_id, result.outcome, secrets, bot_login=config.bot_login)
3128
+
3129
+ pr_url = ""
3130
+ outcome_name = result.outcome
3131
+ branch = ""
3132
+ pushed = False
3133
+ if result.outcome == "improved":
3134
+ try:
3135
+ # Publish the SEALED candidate sha — never the live tree, which
3136
+ # may have drifted since the snapshot (eval caches, stray writes);
3137
+ # the sha is exactly the measured, scope-checked content. A base
3138
+ # branch that moved during the climb is NOT merged and re-measured
3139
+ # here: a stale PR is review's to handle (research-loop.md).
3140
+ branch = f"{config.branch_prefix}/{run_id}"
3141
+ if result.baseline is None or result.candidate is None or not result.candidate_sha:
3142
+ raise EvalError("improved result missing measurements or the sealed sha")
3143
+ bench = next(b for b in contract.benchmarks if b.name == config.benchmark)
3144
+ baseline, candidate = result.baseline, result.candidate
3145
+ # FORCE-checkout: the workspace still holds the session's dirty
3146
+ # tree. The snapshot commit is anchored by the new branch (the
3147
+ # dropped dispatch ref left it unreferenced; nothing pruned it in
3148
+ # this process). clean -fd drops post-snapshot cruft so the
3149
+ # pushed tree is exactly candidate_sha plus the ledger commit.
3150
+ ws.git("checkout", "-f", "-B", branch, result.candidate_sha)
3151
+ ws.git("clean", "-fd")
3152
+ entries = update_leader(
3153
+ load_leader(workspace),
3154
+ benchmark=bench.name,
3155
+ metric=bench.metric,
3156
+ direction=bench.direction,
3157
+ baseline=baseline,
3158
+ candidate=candidate,
3159
+ run_id=run_id,
3160
+ date=created[:10],
3161
+ run_seed=result.run_seed,
3162
+ )
3163
+ write_progress(
3164
+ workspace,
3165
+ entries,
3166
+ config.target,
3167
+ digits={b.name: b.display_digits for b in contract.benchmarks if b.display_digits},
3168
+ )
3169
+ # Stage ONLY the ledger files on top of the sealed candidate —
3170
+ # never `git add -A`, which would sweep in anything a session or
3171
+ # eval left behind (same rule as the wake publish).
3172
+ ws.git("add", "--", *PROGRESS_PATHS)
3173
+ staged = ws.staged_paths()
3174
+ extra = [p for p in staged if p not in PROGRESS_PATHS]
3175
+ if extra:
3176
+ raise WorkspaceDrift(f"publish would stage non-ledger paths: {extra[:10]}")
3177
+ if staged:
3178
+ ws.git(
3179
+ *git_identity(config.bot_login),
3180
+ "commit",
3181
+ "-m",
3182
+ f"agent: improve {config.benchmark} ({_title_pair(baseline, candidate)})"
3183
+ f"\n\nAgent: {config.agent_id}",
3184
+ )
3185
+ ws.push(branch)
3186
+ pushed = True
3187
+ body = pr_body(
3188
+ result,
3189
+ config,
3190
+ redact_secrets=secrets,
3191
+ display_digits=bench.display_digits,
3192
+ experiments=experiments_rows(run_dir),
3193
+ )
3194
+ if issue_number:
3195
+ body = f"Addresses #{issue_number}.\n\n{body}"
3196
+ pr_url = github.create_pull(
3197
+ config.target,
3198
+ # short precision in the title; full precision lives in the
3199
+ # PR body table and the ledger
3200
+ title=f"[agent] {config.benchmark}: {_title_pair(baseline, candidate)}",
3201
+ head=branch,
3202
+ base=base_branch,
3203
+ body=body,
3204
+ # blocking findings open at the panel, or a degraded final
3205
+ # read: visible, plainly not merge-ready
3206
+ draft=result.panel_blocking_open or result.panel_degraded,
3207
+ )
3208
+ # Arm auto-merge, best-effort, and ONLY when branch protection
3209
+ # requires a human review — the guard keeps bot-never-merges
3210
+ # enforced in code, not in per-repo config. Never arm a draft,
3211
+ # and never arm a claim whose base has moved (_arm_unless_base_moved).
3212
+ pr_number = pr_url.rstrip("/").rsplit("/", 1)[-1]
3213
+ if pr_number.isdigit() and not (result.panel_blocking_open or result.panel_degraded):
3214
+ _arm_unless_base_moved(
3215
+ github,
3216
+ ws,
3217
+ config.target,
3218
+ pr_number,
3219
+ base_branch,
3220
+ pre_session_sha,
3221
+ secrets,
3222
+ merge_mode=getattr(contract, "merge", "manual"),
3223
+ panel_ran=result.panel_rounds > 0,
3224
+ )
3225
+ final = RunRecord(
3226
+ **{
3227
+ **record.__dict__,
3228
+ "state": IN_REVIEW,
3229
+ "pr_url": pr_url,
3230
+ "auto_blessed_head": _blessed_head(ws, result, contract),
3231
+ "resume_session_id": result.session.session_id if result.session else "",
3232
+ "ending_note": pr_url,
3233
+ }
3234
+ )
3235
+ except Exception as exc:
3236
+ log.warning(
3237
+ "publish failed for %s: %s",
3238
+ run_id,
3239
+ redact(f"{type(exc).__name__}: {exc}", secrets),
3240
+ )
3241
+ # Never delete the remote branch: an exception from create_pull
3242
+ # does not prove no PR exists (a 422-already-exists or a timeout
3243
+ # after a successful POST both land here), and deleting the ref
3244
+ # would close such a PR and discard the only pushed copy. Leave
3245
+ # it and record it; a sweeper can reap confirmed orphans later.
3246
+ outcome_name = "publish-error"
3247
+ final = RunRecord(
3248
+ **{
3249
+ **record.__dict__,
3250
+ "state": ENDED,
3251
+ "ending": ABORTED,
3252
+ "ending_note": (
3253
+ (f"branch left on remote: {branch}; " if pushed else "")
3254
+ + redact(f"{type(exc).__name__}: {exc}", secrets)[:480]
3255
+ ),
3256
+ }
3257
+ )
3258
+ else:
3259
+ if result.outcome == "session-outage":
3260
+ _best_effort(
3261
+ "outage stamp",
3262
+ lambda: stamp_outage(run_root, redact(result.note, secrets)[:300], now),
3263
+ secrets,
3264
+ )
3265
+ final = RunRecord(
3266
+ **{
3267
+ **record.__dict__,
3268
+ "state": ENDED,
3269
+ "ending": _ENDINGS_BY_OUTCOME[result.outcome],
3270
+ "ending_note": redact(result.note, secrets),
3271
+ }
3272
+ )
3273
+ if not _best_effort("final record", lambda: save_record(run_root, final, now), secrets):
3274
+ # The on-disk record still says `implementing`, so automated
3275
+ # follow-up servicing will not track this run — and if a PR was
3276
+ # opened, its humans are the only ones who can act. Say so WHERE
3277
+ # they are looking: GitHub is the one store still writable when the
3278
+ # local disk is gone.
3279
+ pr_number = pr_url.rstrip("/").rsplit("/", 1)[-1] if pr_url else ""
3280
+ if pr_number.isdigit():
3281
+ _best_effort(
3282
+ "pr state warning",
3283
+ lambda: github.comment(
3284
+ config.target,
3285
+ int(pr_number),
3286
+ f"State record for run `{run_id}` could not be saved; "
3287
+ f"automated follow-up servicing is offline for this run. "
3288
+ f"A maintainer owns any follow-ups on this PR.",
3289
+ ),
3290
+ secrets,
3291
+ )
3292
+ if issue_number:
3293
+ _post_issue_finished(
3294
+ github,
3295
+ config.target,
3296
+ issue_number,
3297
+ run_id,
3298
+ outcome_name,
3299
+ pr_url,
3300
+ redact(result.report(config, redact_secrets=secrets), secrets)[:8000],
3301
+ secrets,
3302
+ )
3303
+ if outcome_name != result.outcome:
3304
+ # the publish failed after the gate credited the tree: the improved
3305
+ # snapshot above stands (the measurement was real); append the
3306
+ # publish-error marker so the notebook records how the run ended
3307
+ _push_line_snapshot(ws, line_ref, run_id, outcome_name, secrets, bot_login=config.bot_login)
3308
+ log.info("run %s: %s %s", run_id, outcome_name, pr_url)
3309
+ return AttemptOutcome(
3310
+ run_id=run_id,
3311
+ outcome=outcome_name,
3312
+ pr_url=pr_url,
3313
+ report_path=str(report_path) if wrote_report else "",
3314
+ )
3315
+
3316
+
3317
+ class Terminated(Exception):
3318
+ """Slurm sent SIGTERM (walltime, preemption, scancel): raised into the
3319
+ main thread so the ordinary exception containment ends the run inside
3320
+ the KillWait grace window before SIGKILL arrives."""
3321
+
3322
+
3323
+ # Below this, arming is pointless: the alarm would fire during setup,
3324
+ # outside containment, and a job this short cannot finish a climb anyway.
3325
+ MIN_ARM_S = 180
3326
+
3327
+
3328
+ def arm_self_deadline(job_minutes: int, margin_s: float = 120.0) -> int:
3329
+ """Arm our own end-of-walltime alarm; returns the armed seconds (0 = off).
3330
+
3331
+ Slurm delivers NO signal to our process on Torch before SIGKILL
3332
+ (scancel and walltime timeout both signal the
3333
+ batch shell only) — so the only way to end a run richly before the
3334
+ wall is our own clock. SIGALRM fires `margin_s` before the job's
3335
+ walltime and raises Terminated into the ordinary containment; the
3336
+ margin floor covers the containment's own tail (GitHub calls are 30s
3337
+ timeout x retries). The walltime clock starts at JOB start, not
3338
+ process start — SLURM_JOB_START_TIME anchors the deadline when
3339
+ present so startup latency erodes the runway, never the margin.
3340
+ """
3341
+ if job_minutes <= 0:
3342
+ return 0
3343
+ import signal
3344
+ import time as _time
3345
+
3346
+ margin = max(60.0, margin_s)
3347
+ now = _time.time()
3348
+ start_raw = os.environ.get("SLURM_JOB_START_TIME", "")
3349
+ # Sanity-bounded: the env can carry a STALE value inherited from the
3350
+ # submitting job (tick jobs sbatch climb jobs). A start time outside
3351
+ # [now - walltime, now] is not this job's — fall back to the process
3352
+ # clock rather than silently disarm (past) or overshoot the wall
3353
+ # (future).
3354
+ if start_raw.isdigit() and now - job_minutes * 60 <= int(start_raw) <= now:
3355
+ remaining = int(int(start_raw) + job_minutes * 60 - margin - now)
3356
+ else:
3357
+ remaining = int(job_minutes * 60 - margin)
3358
+ if remaining < MIN_ARM_S:
3359
+ log.warning(
3360
+ "self-deadline NOT armed: %ds runway is below the %ds floor", remaining, MIN_ARM_S
3361
+ )
3362
+ return 0
3363
+
3364
+ def _on_alarm(signum: int, frame: object) -> None:
3365
+ raise Terminated(
3366
+ f"self-deadline: {margin:.0f}s before the job's {job_minutes}-minute walltime"
3367
+ )
3368
+
3369
+ signal.signal(signal.SIGALRM, _on_alarm)
3370
+ signal.alarm(remaining)
3371
+ return remaining
3372
+
3373
+
3374
+ def arm_sigterm_containment() -> None:
3375
+ """Convert the FIRST SIGTERM into a Terminated exception, one-shot.
3376
+
3377
+ Repeats are absorbed by a flag rather than SIG_IGN: a second SIGTERM
3378
+ (repeated scancel, site KillWait re-sends) must not abort the very
3379
+ containment the first one enabled — and SIG_IGN would be inherited
3380
+ across exec by children spawned during containment, leaving them
3381
+ unkillable by TERM. A Python-level handler is reset on exec, so
3382
+ children keep default signal behavior.
3383
+ """
3384
+ import signal
3385
+
3386
+ fired = {"done": False}
3387
+
3388
+ def _on_sigterm(signum: int, frame: object) -> None:
3389
+ if fired["done"]:
3390
+ return # containment already unwinding; absorb the repeat
3391
+ fired["done"] = True
3392
+ raise Terminated("SIGTERM from Slurm (walltime, preemption, or scancel)")
3393
+
3394
+ signal.signal(signal.SIGTERM, _on_sigterm)
3395
+
3396
+
3397
+ def main() -> int:
3398
+ import argparse
3399
+ import os
3400
+ from datetime import UTC, datetime
3401
+
3402
+ arm_sigterm_containment()
3403
+
3404
+ parser = argparse.ArgumentParser(description="One live attempt on one benchmark.")
3405
+ # --target/--benchmark drive a fresh climb; they are read from the record
3406
+ # on a --resume wake instead, so they are optional (validated below).
3407
+ parser.add_argument("--target", default="")
3408
+ parser.add_argument("--benchmark", default="")
3409
+
3410
+ # WIDTH: the tick assigns each concurrent slot its own agent identity;
3411
+ # branches, ledger rows, and reports key on it. Resumed runs inherit
3412
+ # the identity from their record instead.
3413
+ def _agent_id(value: str) -> str:
3414
+ # the id shapes refs (feat/auto/<id>/<run>, agents/<id>): slug only —
3415
+ # same rule the line checkout enforces
3416
+ if not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9_-]{0,63}", value):
3417
+ raise argparse.ArgumentTypeError(
3418
+ f"agent id {value!r} cannot shape a git ref (want [A-Za-z0-9][A-Za-z0-9_-]*)"
3419
+ )
3420
+ return value
3421
+
3422
+ parser.add_argument("--agent-id", default="agent-01", type=_agent_id)
3423
+ parser.add_argument("--run-root", required=True, type=Path)
3424
+ parser.add_argument(
3425
+ "--resume",
3426
+ default="",
3427
+ metavar="RUN_ID",
3428
+ help="wake a parked dispatched run instead of starting a fresh climb",
3429
+ )
3430
+ parser.add_argument("--base-branch", default="main")
3431
+ # All three default from the chain env the tick sets on the climb job, so
3432
+ # a contained run with OUTERLOOP_{IMAGE,ACCOUNT,PARTITION} set selects
3433
+ # dispatched measurement without extra flags. Only the image is required
3434
+ # for dispatch (it also contains the session + inline eval); without it,
3435
+ # measurement stays inline regardless of the benchmark's eval hint. Empty
3436
+ # account/partition use Slurm's defaults.
3437
+ parser.add_argument(
3438
+ "--image",
3439
+ default=os.environ.get("OUTERLOOP_IMAGE", ""),
3440
+ help="apptainer image for session+eval",
3441
+ )
3442
+ parser.add_argument("--account", default=os.environ.get("OUTERLOOP_ACCOUNT", ""))
3443
+ parser.add_argument("--partition", default=os.environ.get("OUTERLOOP_PARTITION", ""))
3444
+ # the GPU lane for benchmarks with `gpus > 0` (evals + author launches);
3445
+ # empty = this deployment cannot place GPU jobs
3446
+ parser.add_argument("--gpu-partition", default=os.environ.get("OUTERLOOP_GPU_PARTITION", ""))
3447
+ parser.add_argument("--gpu-account", default=os.environ.get("OUTERLOOP_GPU_ACCOUNT", ""))
3448
+ parser.add_argument(
3449
+ "--uncontained",
3450
+ action="store_true",
3451
+ help="run WITHOUT a container (dev only: sessions can then read "
3452
+ "same-user files, including credential files)",
3453
+ )
3454
+ parser.add_argument(
3455
+ "--claude-bin",
3456
+ default=default_binary("claude"),
3457
+ help="host claude binary for the claude author (OUTERLOOP_CLAUDE_BIN, which init records)",
3458
+ )
3459
+ parser.add_argument(
3460
+ "--codex-bin",
3461
+ default=default_binary("codex"),
3462
+ help="host codex binary for the codex author; bind-mounted into apptainer "
3463
+ "(must be an absolute path).",
3464
+ )
3465
+ parser.add_argument(
3466
+ "--model", default=os.environ.get("OUTERLOOP_AUTHOR_MODEL") or "claude-opus-5"
3467
+ )
3468
+ parser.add_argument(
3469
+ "--author-backend",
3470
+ choices=("claude", "codex"),
3471
+ default=os.environ.get("OUTERLOOP_AUTHOR_BACKEND") or "claude",
3472
+ help="agent backend for the author/editor role (config-driven: default "
3473
+ "from OUTERLOOP_AUTHOR_BACKEND). codex runs contained (apptainer + "
3474
+ "--sandbox danger-full-access) and REQUIRES --image and a codex/openai "
3475
+ "--model (e.g. gpt-5.6-terra).",
3476
+ )
3477
+ parser.add_argument(
3478
+ "--codex-config",
3479
+ action="append",
3480
+ default=[],
3481
+ metavar="KEY=VALUE",
3482
+ help="codex `-c KEY=VALUE` config for the codex author (repeatable), "
3483
+ "e.g. --codex-config use_legacy_landlock=true for a host that needs it.",
3484
+ )
3485
+ parser.add_argument("--max-turns", type=int, default=60)
3486
+ parser.add_argument("--session-minutes", type=int, default=60)
3487
+ parser.add_argument(
3488
+ "--panel",
3489
+ default="",
3490
+ help=(
3491
+ "pre-PR verification lenses, comma-separated kind[:backend[:model]] "
3492
+ "entries (e.g. 'verify,review' or 'verify:claude:MODEL'); only the "
3493
+ "claude backend is contained on this host so far; empty disables "
3494
+ "the panel"
3495
+ ),
3496
+ )
3497
+ parser.add_argument(
3498
+ "--panel-key-file",
3499
+ default=PANEL_KEY_DEFAULT,
3500
+ help="key file for panel judge sessions (the verifier's own key, never the author's)",
3501
+ )
3502
+ parser.add_argument(
3503
+ "--job-minutes",
3504
+ type=int,
3505
+ default=0,
3506
+ help="this job's Slurm walltime; arms the self-deadline (0 = off)",
3507
+ )
3508
+ parser.add_argument(
3509
+ "--deadline-margin-s",
3510
+ type=float,
3511
+ default=120.0,
3512
+ help="how long before the walltime the self-deadline fires (floor 60)",
3513
+ )
3514
+ add_credential_args(parser)
3515
+ parser.add_argument(
3516
+ "--key-file",
3517
+ default="",
3518
+ help="author key file; default resolves per backend (config-driven): "
3519
+ "OUTERLOOP_CLAUDE_KEY_FILE for claude, OUTERLOOP_CODEX_KEY_FILE for codex",
3520
+ )
3521
+ parser.add_argument("--issue", type=int, default=0)
3522
+ parser.add_argument(
3523
+ "--min-free-gb",
3524
+ type=float,
3525
+ default=10.0,
3526
+ help="refuse to start when the run root has less free space",
3527
+ )
3528
+ parser.add_argument(
3529
+ "--hypothesis-b64", default="", help="base64 task hypothesis (issue text, fenced)"
3530
+ )
3531
+ args = parser.parse_args()
3532
+ logging.basicConfig(level=logging.INFO, format="%(asctime)s %(message)s")
3533
+ if not args.image and not args.uncontained:
3534
+ parser.error("--image is required (or pass --uncontained explicitly, dev only)")
3535
+ # NOTE: the codex author is validated on the EFFECTIVE author per path — the
3536
+ # fresh climb on args (below), a wake on the parked run's persisted pair — not
3537
+ # here, where args.author_backend is the FLEET default and would misjudge a
3538
+ # resume after a fleet flip.
3539
+ # each --codex-config KEY=VALUE becomes a `-c KEY=VALUE` pair for codex
3540
+ codex_extra = tuple(a for c in args.codex_config for a in ("-c", c))
3541
+
3542
+ bot_auth = resolve_bot_auth(args.pat_file, args.github_app_file)
3543
+
3544
+ # --resume WAKES a parked dispatched run: rebuild the dispatched measurer
3545
+ # and re-enter the decision. The wake job the WakeDispatcher submits runs
3546
+ # exactly this.
3547
+ if args.resume:
3548
+ if not (args.image and Path(args.image).is_file()):
3549
+ parser.error(
3550
+ "--resume needs --image to rebuild the dispatched measurer "
3551
+ "(account and partition are optional: empty ones use Slurm's "
3552
+ "defaults, and local compute has no placement)"
3553
+ )
3554
+ from outerloop.runstate import load_record
3555
+
3556
+ # a wake that is not the lease holder is a straggler (a replacement was
3557
+ # dispatched after it was cancelled, or it was armed and then lost):
3558
+ # it must not touch the run beside the holder
3559
+ other = _lease_held_by_another_job(args.run_root, args.resume)
3560
+ if other:
3561
+ print(f"run {args.resume}: wake job {other} holds the lease; this one exits")
3562
+ return 0
3563
+
3564
+ # Reproduce the PARKED run's author, not the current fleet default: the
3565
+ # (backend, model) PAIR is persisted on the record — a fleet flip must not
3566
+ # wake a codex run as claude (or with the new fleet's model). A legacy or
3567
+ # unreadable record is treated as claude (resume_author). The key then
3568
+ # resolves for THAT backend (keys coexist).
3569
+ try:
3570
+ _wake_record: object | None = load_record(args.run_root, args.resume)
3571
+ except Exception:
3572
+ # a wake must never crash on an unreadable/odd record — fall back to
3573
+ # the claude author (resume_author), same fail-safe as the sweep
3574
+ _wake_record = None
3575
+ wake_backend, wake_model, wake_key_file = resume_author(_wake_record, args.model)
3576
+ # an explicit --key-file still overrides (a manual re-run pinning a key)
3577
+ if args.key_file:
3578
+ wake_key_file = os.path.expanduser(args.key_file)
3579
+ _err = codex_author_config_error(wake_backend, wake_model, args.image)
3580
+ if _err:
3581
+ # this wake job HOLDS the run's lease (transferred on dispatch); release
3582
+ # it before exiting so a misconfig doesn't strand the run until the TTL
3583
+ # reap (the resume_run finally below only runs once we reach it)
3584
+ _release_own_lease(args.run_root, args.resume)
3585
+ parser.error(f"parked run {args.resume}: {_err}")
3586
+ # the wake runs the SAME verification panel as a fresh climb, so a
3587
+ # dispatched improvement is not published unverified.
3588
+ try:
3589
+ wake_lenses, wake_panel_secrets = _panel_lenses_from_args(args)
3590
+ except ValueError as exc:
3591
+ parser.error(str(exc))
3592
+ wake_api_key = ""
3593
+ wake_harness = None
3594
+ wake_spec = None
3595
+ # The editor harness is built when the wake may RESUME the session: a
3596
+ # panel is configured (a blocking finding wakes the author to revise),
3597
+ # or the park is an AUTHOR-SLEEP (that wake always resumes the session
3598
+ # with its launches' results). A panel-less candidate wake stays a pure
3599
+ # read-decide-publish job and must not require the author key.
3600
+ _wake_stage = getattr(_wake_record, "stage", None) or {}
3601
+ if (
3602
+ wake_lenses
3603
+ or _wake_stage.get("phase") == "author-sleep"
3604
+ or _wake_stage.get("submitted")
3605
+ ):
3606
+ wake_api_key = role_key(wake_key_file, wake_backend)
3607
+ wake_spec = author_spec(max_turns=args.max_turns, walltime_s=args.session_minutes * 60)
3608
+ wake_harness = build_harness(
3609
+ wake_api_key,
3610
+ wake_spec,
3611
+ backend=wake_backend,
3612
+ binary=args.claude_bin if wake_backend == "claude" else args.codex_bin,
3613
+ model=wake_model,
3614
+ container_image=args.image,
3615
+ codex_extra_args=codex_extra,
3616
+ )
3617
+ wake_secrets = tuple(k for k in (bot_auth.token(), *wake_panel_secrets, wake_api_key) if k)
3618
+ try:
3619
+ resumed = resume_run(
3620
+ args.run_root,
3621
+ args.resume,
3622
+ dispatch=_dispatch_settings(args),
3623
+ github=GitHubClient(auth=bot_auth),
3624
+ bot_auth=bot_auth,
3625
+ now=time.time(),
3626
+ secrets=wake_secrets,
3627
+ base_branch=args.base_branch,
3628
+ panel_lenses=wake_lenses,
3629
+ harness=wake_harness,
3630
+ spec=wake_spec,
3631
+ )
3632
+ except GitError as exc:
3633
+ # A tamper-class GitError that escaped the wake body (a TOCTOU
3634
+ # object removal past the entry guard, or a git op that reached a
3635
+ # damaged object) ends the run cleanly as a refused wake — "a
3636
+ # refused wake ends the run" — instead of crashing the job into a
3637
+ # re-park that only wakes into the same failure. A non-tamper
3638
+ # GitError (push conflict, fetch outage) propagates unchanged.
3639
+ if not _is_git_tamper(exc):
3640
+ raise
3641
+ resumed = _end_refused_wake(
3642
+ args.run_root,
3643
+ load_record(args.run_root, args.resume),
3644
+ exc,
3645
+ time.time(),
3646
+ wake_secrets,
3647
+ )
3648
+ finally:
3649
+ # This wake job HOLDS the run's lease (the sweep transferred it on
3650
+ # dispatch); release it on every exit so a re-parked run is
3651
+ # immediately eligible for the next sweep instead of waiting out the
3652
+ # TTL reap. Idempotent (no-op if no lease file).
3653
+ _release_own_lease(args.run_root, args.resume)
3654
+ print(f"outcome={resumed.outcome} pr={resumed.pr_url or '-'} report={resumed.report_path}")
3655
+ return 0
3656
+
3657
+ if not (args.target and args.benchmark):
3658
+ parser.error("--target and --benchmark are required for a fresh climb")
3659
+
3660
+ # a fresh climb authors on the FLEET's configured backend; validate it (codex
3661
+ # writes+executes, so --image + a non-claude model) before any spend.
3662
+ _err = codex_author_config_error(args.author_backend, args.model, args.image)
3663
+ if _err:
3664
+ parser.error(_err)
3665
+ # config-driven: the author key defaults per backend (claude vs codex) so the
3666
+ # tick never threads it — see resolve_author_key_file (result is ~-expanded).
3667
+ args.key_file = resolve_author_key_file(args.author_backend, args.key_file)
3668
+ # same 0600 discipline as the PAT: this key spends real money. A missing
3669
+ # file is tolerated only when Vertex (ADC) covers the claude backend.
3670
+ api_key = role_key(args.key_file, args.author_backend)
3671
+ stamp = datetime.now(UTC).strftime("%Y%m%d-%H%M%S")
3672
+ # the agent id keeps concurrent same-benchmark slots (the width dial's
3673
+ # portfolio case) from minting one run directory in the same second
3674
+ run_id = f"{args.benchmark}-{stamp}-{args.agent_id}"
3675
+
3676
+ # Disk preflight BEFORE any run state exists: a session started on a
3677
+ # full filesystem dies mid-flight in ways that lose its own evidence
3678
+ # (quota errors are invisible until a write fails on some clusters).
3679
+ from outerloop.disk import check_mount
3680
+
3681
+ health = check_mount(args.run_root, min_free_bytes=int(args.min_free_gb * 1024**3))
3682
+ if not health.ok():
3683
+ log.error("disk preflight failed: %s — refusing to start a run", health.describe())
3684
+ if args.issue:
3685
+ _best_effort(
3686
+ "issue report",
3687
+ lambda: GitHubClient(auth=bot_auth).comment(
3688
+ args.target,
3689
+ args.issue,
3690
+ "A run for this issue could not start: the orchestrator's "
3691
+ "storage failed its disk preflight. The claim on this issue "
3692
+ "stays until a maintainer removes the claim comment "
3693
+ "(automated claim release is on the roadmap).",
3694
+ ),
3695
+ )
3696
+ return 3
3697
+
3698
+ # Armed LAST, immediately before the contained region — and DISARMED
3699
+ # right after it: a run finishing inside the margin must not have the
3700
+ # alarm fire during the uncontained epilogue (print/exit).
3701
+ import signal as _signal
3702
+
3703
+ armed = arm_self_deadline(args.job_minutes, args.deadline_margin_s)
3704
+ if armed:
3705
+ log.info("self-deadline armed: Terminated in %ds", armed)
3706
+
3707
+ # the manifest first, the harness from it: budget has one source (the args)
3708
+ spec = author_spec(max_turns=args.max_turns, walltime_s=args.session_minutes * 60)
3709
+
3710
+ # Pre-PR panel lenses: judge sessions on the verifier's own key (separate
3711
+ # identity from the author). kind[:backend[:model]]; claude by default.
3712
+ try:
3713
+ panel_lenses, panel_secrets = _panel_lenses_from_args(args)
3714
+ except ValueError as exc:
3715
+ parser.error(str(exc))
3716
+
3717
+ # Dispatched measurement needs a real image file to bind against; without
3718
+ # one the climb measures inline. Account and partition are optional: empty
3719
+ # ones use Slurm's defaults (the tick sets them on the climb job's env from
3720
+ # the chain's), and local compute has no placement.
3721
+ dispatch: DispatchSettings | None = None
3722
+ if args.image and Path(args.image).is_file():
3723
+ dispatch = _dispatch_settings(args)
3724
+ try:
3725
+ try:
3726
+ outcome = live_attempt(
3727
+ config=RunConfig(
3728
+ target=args.target, benchmark=args.benchmark, agent_id=args.agent_id
3729
+ ),
3730
+ base_branch=args.base_branch,
3731
+ run_root=args.run_root,
3732
+ run_id=run_id,
3733
+ harness=build_harness(
3734
+ api_key,
3735
+ spec,
3736
+ backend=args.author_backend,
3737
+ binary=args.claude_bin if args.author_backend == "claude" else args.codex_bin,
3738
+ model=args.model,
3739
+ container_image=args.image,
3740
+ codex_extra_args=codex_extra,
3741
+ ),
3742
+ spec=spec,
3743
+ panel_lenses=panel_lenses,
3744
+ dispatch=dispatch,
3745
+ eval_image=args.image,
3746
+ github=GitHubClient(auth=bot_auth),
3747
+ bot_auth=bot_auth,
3748
+ now=time.time(),
3749
+ created=datetime.now(UTC).isoformat(),
3750
+ # the panel key joins the redaction set: judge error text can
3751
+ # echo request material like any other model error
3752
+ secrets=tuple(
3753
+ k
3754
+ for k in (
3755
+ api_key,
3756
+ bot_auth.token(),
3757
+ *panel_secrets,
3758
+ )
3759
+ if k
3760
+ ),
3761
+ issue_number=args.issue,
3762
+ author_backend=args.author_backend,
3763
+ author_model=args.model,
3764
+ author_key_file=args.key_file,
3765
+ task_hypothesis=(
3766
+ __import__("base64").b64decode(args.hypothesis_b64).decode()
3767
+ if args.hypothesis_b64
3768
+ else ""
3769
+ ),
3770
+ )
3771
+ except Terminated as exc:
3772
+ # Fired in live_attempt's microseconds-wide pre-containment window:
3773
+ # any record it saved strands and the sweep ends it from Slurm
3774
+ # truth; here we only avoid dying as an unexplained traceback.
3775
+ log.error("self-deadline fired before containment: %s", exc)
3776
+ return 3
3777
+ finally:
3778
+ _signal.alarm(0)
3779
+ print(f"outcome={outcome.outcome} pr={outcome.pr_url or '-'} report={outcome.report_path}")
3780
+ return 0
3781
+
3782
+
3783
+ if __name__ == "__main__":
3784
+ raise SystemExit(main())