outerloop-science 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. outerloop/__init__.py +18 -0
  2. outerloop/__main__.py +3 -0
  3. outerloop/appauth.py +230 -0
  4. outerloop/appmanifest.py +203 -0
  5. outerloop/attempt.py +3784 -0
  6. outerloop/brief.py +528 -0
  7. outerloop/cli.py +621 -0
  8. outerloop/climbboard.py +1395 -0
  9. outerloop/compute.py +654 -0
  10. outerloop/contract.py +492 -0
  11. outerloop/contract_cli.py +63 -0
  12. outerloop/disk.py +164 -0
  13. outerloop/dispatch.py +631 -0
  14. outerloop/evalcache.py +147 -0
  15. outerloop/followup.py +2172 -0
  16. outerloop/github.py +1531 -0
  17. outerloop/harness.py +1435 -0
  18. outerloop/housekeeping.py +151 -0
  19. outerloop/image.py +368 -0
  20. outerloop/init.py +744 -0
  21. outerloop/intake.py +126 -0
  22. outerloop/launchlog.py +239 -0
  23. outerloop/limits.py +80 -0
  24. outerloop/maintain.py +353 -0
  25. outerloop/maintain_agent_cli.py +81 -0
  26. outerloop/maintain_post_cli.py +140 -0
  27. outerloop/markers.py +48 -0
  28. outerloop/measure.py +529 -0
  29. outerloop/orchestrator.py +2011 -0
  30. outerloop/panel.py +188 -0
  31. outerloop/paths.py +40 -0
  32. outerloop/posting.py +160 -0
  33. outerloop/progress.py +170 -0
  34. outerloop/py.typed +0 -0
  35. outerloop/review.py +615 -0
  36. outerloop/review_agent.py +263 -0
  37. outerloop/review_agent_cli.py +209 -0
  38. outerloop/review_post_cli.py +162 -0
  39. outerloop/review_summarize_cli.py +165 -0
  40. outerloop/role_runner.py +229 -0
  41. outerloop/roles.py +274 -0
  42. outerloop/rolespec.py +91 -0
  43. outerloop/runstate.py +385 -0
  44. outerloop/steward.py +845 -0
  45. outerloop/style.py +12 -0
  46. outerloop/syscall.py +1192 -0
  47. outerloop/syscall_cli.py +762 -0
  48. outerloop/tick.py +3422 -0
  49. outerloop/verifier.py +403 -0
  50. outerloop/verify_agent.py +151 -0
  51. outerloop/verify_agent_cli.py +95 -0
  52. outerloop/verify_post_cli.py +116 -0
  53. outerloop/watcher.py +203 -0
  54. outerloop_science-0.1.0.dist-info/METADATA +152 -0
  55. outerloop_science-0.1.0.dist-info/RECORD +59 -0
  56. outerloop_science-0.1.0.dist-info/WHEEL +4 -0
  57. outerloop_science-0.1.0.dist-info/entry_points.txt +2 -0
  58. outerloop_science-0.1.0.dist-info/licenses/LICENSE +202 -0
  59. outerloop_science-0.1.0.dist-info/licenses/NOTICE +5 -0
outerloop/tick.py ADDED
@@ -0,0 +1,3422 @@
1
+ """One tick of the loop: sentinel, heartbeat, and the fail-safe sweep.
2
+
3
+ The tick is stateless and bounded — everything durable lives in run-state
4
+ files (`runstate`) and Slurm. It implements the backup layers of the wake
5
+ design (docs/design/architecture.md, "Wake delivery and fail-safety"); the
6
+ primary layer (the afterany dependency job) is submitted by whoever launches
7
+ an experiment and needs no help from here.
8
+
9
+ Wake *delivery* is behind a seam (`WakeDispatcher`) so this module stays
10
+ testable and the actual session dispatch (harness + brief) can evolve
11
+ independently.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import contextlib
17
+ import json
18
+ import logging
19
+ import math
20
+ import os
21
+ import re
22
+ import socket
23
+ import subprocess
24
+ import sys
25
+ from collections.abc import Sequence
26
+ from dataclasses import asdict, dataclass, field, replace
27
+ from pathlib import Path
28
+ from typing import Any, Protocol
29
+ from uuid import uuid4
30
+
31
+ from outerloop.compute import (
32
+ GONE,
33
+ Compute,
34
+ JobSpec,
35
+ LocalCompute,
36
+ SlurmError,
37
+ SlurmQueryError,
38
+ compute_from_env,
39
+ is_pending,
40
+ is_terminal,
41
+ local_mode,
42
+ quote_command,
43
+ )
44
+ from outerloop.disk import DEFAULT_MIN_FREE_BYTES, check_disk
45
+ from outerloop.harness import DEFAULT_MAX_TURNS, redact
46
+ from outerloop.housekeeping import shed_ended_workspaces
47
+ from outerloop.limits import EffectiveLimits, effective_limits
48
+ from outerloop.markers import has_marker, marker
49
+ from outerloop.runstate import (
50
+ ABORTED,
51
+ ENDED,
52
+ IMPLEMENTING,
53
+ IN_REVIEW,
54
+ MAX_WAKE_ATTEMPTS,
55
+ STUCK,
56
+ WAITING,
57
+ Lease,
58
+ RunRecord,
59
+ acquire_lease,
60
+ lease_is_stale,
61
+ list_runs,
62
+ load_record,
63
+ outage_active,
64
+ read_lease,
65
+ reap_lease,
66
+ release_lease,
67
+ run_dir,
68
+ save_record,
69
+ update_lease_holder,
70
+ )
71
+
72
+ log = logging.getLogger(__name__)
73
+
74
+ PAUSE_SENTINEL = "PAUSE"
75
+ # Operator on-switch for dispatched-wake, mirroring PAUSE: a root-relative
76
+ # sentinel an operator arms/disarms with a touch/rm — no chain restart, no
77
+ # env-var surgery on a live tick. The OUTERLOOP_DISPATCH_WAKE env var still
78
+ # works too (either arms it); the sentinel is the reversible, restart-free path.
79
+ DISPATCH_WAKE_SENTINEL = "DISPATCH_WAKE"
80
+ HEARTBEAT_NAME = "heartbeat.json"
81
+ # Written at a full tick's END (not its start) — the coalesce guard's signal, so
82
+ # a tick that crashes mid-work cannot suppress the next (recovery) tick.
83
+ WORK_MARKER_NAME = "last_worked.json"
84
+
85
+ # Grace between "experiment terminal" and the sweep stepping in: the afterany
86
+ # job gets this long to deliver before the backup assumes it lost.
87
+ DEFAULT_GRACE_S = 15 * 60
88
+ # a blind park (no job ids to poll) waits its eval walltime plus this queue
89
+ # slack before a follow-up is sent to look for the result
90
+ BLIND_PARK_SLACK_MIN = 12 * 60
91
+
92
+ # Slurm pending reasons that mean "the job will run when its turn comes": the
93
+ # queue is busy, a reservation or dependency is ahead of it, or the account's
94
+ # OWN jobs hold a per-user/per-account/group cap that clears as they finish. A
95
+ # job in this state keeps accruing priority age; cancelling and resubmitting
96
+ # would put it at the back of the queue behind itself. Per-job limits
97
+ # (`...PerJob...`), a dependency that can never be satisfied, an invalid
98
+ # account/QOS or a held job are NOT waits: those never clear on their own.
99
+ # JobHeldUser: an operator's hold is theirs to lift; the job is waiting, not stuck.
100
+ QUEUE_WAIT_REASONS = frozenset(
101
+ {"Priority", "Resources", "Reservation", "Dependency", "ReqNodeNotAvail", "JobHeldUser"}
102
+ )
103
+
104
+
105
+ def is_queue_wait(reason: str) -> bool:
106
+ """Whether a squeue pending reason describes a healthy wait (see
107
+ QUEUE_WAIT_REASONS). Reasons arrive as `Name` or `Name, detail...`."""
108
+ head = reason.strip().split(",")[0].split(" ")[0].strip()
109
+ if not head:
110
+ return False
111
+ if head in QUEUE_WAIT_REASONS:
112
+ return True
113
+ if "PerJob" in head:
114
+ return False
115
+ return any(tag in head for tag in ("PerUser", "PerAccount", "Grp"))
116
+
117
+
118
+ # A held lease is stale after the session timeout plus slack.
119
+ DEFAULT_LEASE_TTL_S = 3600 + 15 * 60
120
+ # Coalesce guard: skip a tick's work if another ran within this window. Under
121
+ # partition congestion, queued ticks bunch up and become eligible together
122
+ # (serialized by the singleton dependency), so they would run back-to-back and
123
+ # redundantly re-sweep. The chain schedules ticks a full cadence apart by
124
+ # begin-time, so only late-bunched pile-ups fall inside this window; keep it
125
+ # well BELOW the cadence (default 30 min). 0 disables. Env: OUTERLOOP_MIN_TICK_MINUTES.
126
+ DEFAULT_MIN_TICK_S = 10 * 60
127
+ # Ceiling for the coalesce window: a value above this is almost certainly a typo
128
+ # (a window near/over the cadence would coalesce every on-cadence tick and stall
129
+ # the loop). Clamp + warn rather than silently freeze. The operator is still
130
+ # responsible for keeping it below their configured cadence.
131
+ MAX_MIN_TICK_S = 60 * 60
132
+
133
+
134
+ class WakeDispatcher(Protocol):
135
+ """Delivers one wake, called with the lease already held.
136
+
137
+ Returns "" when delivery completed synchronously (the caller releases the
138
+ lease), or the Slurm job id of an asynchronous wake job that now owns the
139
+ lease (released by that job on completion; reaped by TTL if it dies).
140
+
141
+ Contract for real dispatchers: a wake that RESULTS IN PROGRESS
142
+ must either move the run out of `waiting` or reset `wake_attempts` —
143
+ the counter means "wakes since the run last made progress", and layer 5
144
+ ends the run as stuck when it reaches MAX_WAKE_ATTEMPTS."""
145
+
146
+ def dispatch(self, record: RunRecord, reason: str) -> str: ...
147
+
148
+
149
+ @dataclass(frozen=True)
150
+ class TickReport:
151
+ paused: bool = False
152
+ coalesced: bool = False # skipped as a redundant pile-up (a tick ran too recently)
153
+ swept: int = 0
154
+ woken: tuple[tuple[str, str], ...] = () # (run_id, reason)
155
+ deferred: tuple[str, ...] = () # runs skipped on "Slurm unknown"
156
+ reaped_leases: tuple[str, ...] = ()
157
+ stuck: tuple[str, ...] = ()
158
+ implementing_ended: tuple[str, ...] = () # killed climbs the sweep closed out
159
+ review_ended: tuple[tuple[str, str], ...] = () # (run_id, ending)
160
+ followups_submitted: tuple[tuple[str, str], ...] = () # (run_id, job_id)
161
+ intake: tuple[str, str] = ("", "") # (issue tag, job_id) when one was claimed
162
+ self_initiated: tuple[str, str] = ("", "") # (benchmark, job_id) when one launched
163
+ steward: tuple[str, str] = ("", "") # (issue tag, job_id) when a stewardship launched
164
+ disk: tuple[str, ...] = () # preflight warnings (home entries are warn-only)
165
+ launch_blocked: bool = False # True when the preflight turned launch lanes off
166
+ shed: tuple[str, ...] = () # ended runs whose workspaces housekeeping removed
167
+
168
+
169
+ # The submitted walltime must never exceed the job partition's MaxTime —
170
+ # sbatch REJECTS a longer request outright, which would ground every climb.
171
+ # The DEFAULT matches cpu_short (6 h); an operator moving work jobs to a
172
+ # longer partition (OUTERLOOP_JOB_PARTITION=cpu48) raises the cap with
173
+ # OUTERLOOP_MAX_JOB_MINUTES. Code-side ceiling: the cap must stay under
174
+ # STRANDED_IMPLEMENTING_S or the picker declares live runs stranded — jobs
175
+ # longer than 10 h need that window made spec-aware first (named gap). The
176
+ # self-deadline arms at the CLAMPED value, so a job that wanted more time
177
+ # fails safe mid-panel instead of never starting.
178
+ MAX_ATTEMPT_JOB_MINUTES = 6 * 60
179
+ MAX_JOB_MINUTES_CEILING = 10 * 60
180
+
181
+
182
+ def _bot_login_default() -> str:
183
+ """FollowupSpec's login default, resolved at construction (the tick reads
184
+ the chain's env, jobs inherit it); github is imported here on purpose —
185
+ the tick module stays importable without it."""
186
+ from outerloop.github import bot_login_from_env
187
+
188
+ return bot_login_from_env()
189
+
190
+
191
+ @dataclass(frozen=True)
192
+ class FollowupSpec:
193
+ """How the tick launches follow-up jobs for in-review runs."""
194
+
195
+ account: str
196
+ partition: str
197
+ run_root: Path
198
+ image: str
199
+ home: Path # OUTERLOOP_HOME: cwd for the submitted job
200
+ bot_login: str = field(default_factory=_bot_login_default)
201
+ time_minutes: int = 90 # min()'d with the contract's followup_job_minutes
202
+ max_turns: int = DEFAULT_MAX_TURNS # session turn budget for follow-up jobs
203
+ pat_file: str = "" # forwarded to the job; "" = the followup CLI default
204
+ # GitHub App config path; jobs inherit OUTERLOOP_GITHUB_APP_FILE from
205
+ # the tick environment, so it is never threaded through argv
206
+ github_app_file: str = ""
207
+ target: str = "" # the repo the intake pass scans for requested-lane issues
208
+ # the STEWARD'S OWN key (role separation): the steward lane stays off
209
+ # until the operator provisions it
210
+ steward_key_file: str = ""
211
+ # Pre-PR verification panel for climb jobs (docs/design/orchestrator-verify.md).
212
+ # DEFAULT ON — the flip is code, the off-switch is OUTERLOOP_PANEL="".
213
+ # The climb CLI fails LOUDLY on a bad panel config (a configured gate must
214
+ # never silently vanish); the tick preflights the same rules — lens
215
+ # grammar AND key file — before claiming or submitting so nothing is
216
+ # stranded. The panel's walltime is the orchestrator's own overhead: the
217
+ # tick ADDS a panel allowance to the contract-clamped job budget
218
+ # (_panel_job_minutes) rather than eating the author's time; a residual
219
+ # overrun still fails safe through the self-deadline.
220
+ panel: str = "verify,review"
221
+ panel_key_file: str = "" # "" = the climb CLI's default verifier-key path
222
+ # The GPU lane for benchmarks whose contract sets `gpus > 0` (their evals
223
+ # and author launches); empty = this deployment cannot place GPU jobs,
224
+ # and the launch lanes refuse such benchmarks (a queue that can never
225
+ # run is worse than a loud refusal). gpu_account "" = same as `account`.
226
+ gpu_partition: str = ""
227
+ gpu_account: str = ""
228
+ # Where submitted WORK jobs (climb/steward/followup) run; empty = same as
229
+ # `partition`. The tick chain itself always stays on `partition` — ticks
230
+ # are minutes, work jobs can be hours, and Slurm prices walltime into
231
+ # scheduling priority, so the two deserve independent placement.
232
+ job_partition: str = ""
233
+ # Partition MaxTime for work jobs — the panel-augmented walltime clamps
234
+ # here (see MAX_ATTEMPT_JOB_MINUTES). Raise together with job_partition.
235
+ max_job_minutes: int = MAX_ATTEMPT_JOB_MINUTES
236
+
237
+
238
+ # Generous vs the ~2 h job walltimes plus queue wait, tight enough that
239
+ # full trees + per-flight venvs cannot pile up for days; same-day
240
+ # forensics is the norm, and the disk preflight is the backstop.
241
+ FLIGHT_TTL_S = 24 * 3600
242
+
243
+
244
+ def flight_checkout(home: Path, name: str, now: float) -> Path:
245
+ """A detached git worktree of the checkout's HEAD commit, for one
246
+ submitted job to run from. The shared checkout is reset --hard at
247
+ every tick's deploy, so a queued job that cd's into it can have its
248
+ code swapped mid-flight; a flight pins the deployed commit, and the
249
+ tree survives for forensics after a crash. HEAD, deliberately:
250
+ uncommitted hand-edits in the shared checkout do not fly — only
251
+ deployed code does. Failures fall back to the shared checkout — a
252
+ snapshot must never ground the fleet."""
253
+ if not (home / ".git").exists():
254
+ # not a checkout (the local loop on an installed package): nothing to
255
+ # pin, the job runs from the home directory itself
256
+ return home
257
+ flights = home.parent / "flights"
258
+ try:
259
+ flights.mkdir(parents=True, exist_ok=True)
260
+ # same name in the same tick (e.g. two orders on one benchmark, or
261
+ # truncation collisions) must get its own tree, not a silent
262
+ # fallback: suffix until free, with a unique tail as the backstop
263
+ target = flights / f"{name}-{int(now)}"
264
+ for attempt in range(2, 6):
265
+ if not target.exists():
266
+ break
267
+ target = flights / f"{name}-{int(now)}-{attempt}"
268
+ if target.exists():
269
+ target = flights / f"{name}-{int(now)}-{uuid4().hex[:8]}"
270
+ subprocess.run(
271
+ ["git", "-C", str(home), "worktree", "add", "--detach", str(target)],
272
+ check=True,
273
+ capture_output=True,
274
+ text=True,
275
+ timeout=60,
276
+ )
277
+ return target
278
+ except (OSError, subprocess.SubprocessError) as exc:
279
+ log.warning("flight snapshot failed for %s (%s); using the shared checkout", name, exc)
280
+ return home
281
+
282
+
283
+ def _benchmark_gpus(contract: Any, benchmark: str) -> int:
284
+ bench = next((b for b in getattr(contract, "benchmarks", []) if b.name == benchmark), None)
285
+ return int(getattr(bench, "gpus", 0) or 0)
286
+
287
+
288
+ def _gpu_lane_error(contract: Any, benchmark: str, spec: FollowupSpec) -> str:
289
+ """Why an attempt on `benchmark` cannot launch here, or "": a contract
290
+ with GPU benchmarks needs this deployment to name a GPU lane — otherwise
291
+ evals would queue into jobs that can never run (the climb would then
292
+ park forever on a phantom eval). ANY GPU benchmark in the contract
293
+ counts, not just the climbed one: the suite gate measures siblings.
294
+ Local compute has no lanes — jobs run on whatever GPUs the machine
295
+ has — so the check is waived there."""
296
+ if spec.gpu_partition or local_mode():
297
+ return ""
298
+ gpu_benches = [
299
+ b.name for b in getattr(contract, "benchmarks", []) if int(getattr(b, "gpus", 0) or 0)
300
+ ]
301
+ if gpu_benches:
302
+ return (
303
+ f"contract has GPU benchmarks ({', '.join(gpu_benches)}) but no GPU lane is "
304
+ "configured (set OUTERLOOP_GPU_PARTITION)"
305
+ )
306
+ return ""
307
+
308
+
309
+ _UNCONTAINED_WARNED = False # the local-mode warning is said once per process
310
+
311
+
312
+ def _containment(image: str) -> list[str]:
313
+ """The job's containment flags: the image when there is one, else the
314
+ explicit uncontained flag every entry point requires in its absence."""
315
+ return ["--image", image] if image else ["--uncontained"]
316
+
317
+
318
+ def _interpreter(home: Path) -> list[str]:
319
+ """How a job runs Python. From a source checkout, `uv run python` resolves
320
+ the project's own environment, as the cluster's flights do. From a plain
321
+ home directory (the local loop on an installed package) the job runs the
322
+ interpreter this tick runs under, which is where the package is."""
323
+ if (home / "pyproject.toml").is_file():
324
+ return ["uv", "run", "python"]
325
+ return [sys.executable]
326
+
327
+
328
+ def _flight_command(home: Path, job_name: str, now: float, argv: list[str]) -> str:
329
+ """The job's shell command, cd'ing into a fresh flight snapshot.
330
+
331
+ The flight is named FROM the job name (one truncation rule, here) so
332
+ the reaper's pending-job immunity — live job name prefixes flight
333
+ name — holds by construction at every submit site. argv must contain
334
+ absolute paths only; every spec path is absolute by construction, and
335
+ a relative path would resolve inside a tree that is reaped later."""
336
+ flight = flight_checkout(home, job_name[:40], now)
337
+ return f"cd {quote_command([str(flight)])} && {quote_command(argv)}"
338
+
339
+
340
+ def reap_flights(
341
+ home: Path,
342
+ now: float,
343
+ ttl_s: float = FLIGHT_TTL_S,
344
+ live_job_names: Sequence[str] = (),
345
+ ) -> int:
346
+ """Remove flight worktrees older than the TTL — age by directory
347
+ mtime, no name parsing — UNLESS a pending or running job's NAME
348
+ prefixes the flight's (flights are named after their job). Queue wait
349
+ is unbounded (GPU partitions can pend for days), so age alone must
350
+ never delete a tree a job will cd into; the TTL is purely the
351
+ forensics-retention window for flights whose job is gone. Name
352
+ matching is conservative: one live job name protects every flight it
353
+ prefixes. Best-effort: a stubborn flight is logged, not fatal."""
354
+ flights = home.parent / "flights"
355
+ if not flights.is_dir():
356
+ return 0
357
+ reaped = 0
358
+ for entry in flights.iterdir():
359
+ if any(name and entry.name.startswith(name[:40]) for name in live_job_names):
360
+ continue # a queued or running job still needs this tree
361
+ try:
362
+ age = now - entry.stat().st_mtime
363
+ except OSError:
364
+ continue
365
+ if age < ttl_s:
366
+ continue
367
+ try:
368
+ subprocess.run(
369
+ ["git", "-C", str(home), "worktree", "remove", "--force", str(entry)],
370
+ check=True,
371
+ capture_output=True,
372
+ text=True,
373
+ timeout=60,
374
+ )
375
+ reaped += 1
376
+ except (OSError, subprocess.SubprocessError):
377
+ # not a registered worktree (a half-created flight, or debris):
378
+ # remove the directory itself and prune the registry, or the
379
+ # entry warns forever without ever going away
380
+ import shutil
381
+
382
+ shutil.rmtree(entry, ignore_errors=True)
383
+ with contextlib.suppress(OSError, subprocess.SubprocessError):
384
+ subprocess.run(
385
+ ["git", "-C", str(home), "worktree", "prune"],
386
+ check=True,
387
+ capture_output=True,
388
+ text=True,
389
+ timeout=60,
390
+ )
391
+ if not entry.exists():
392
+ reaped += 1
393
+ else:
394
+ log.warning("could not reap flight %s", entry.name)
395
+ return reaped
396
+
397
+
398
+ CONTRACT_ALARM_MARKER = marker("contract-alarm")
399
+ CONTRACT_ALARM_AFTER = 3 # consecutive failing ticks (~1.5 h) before alarming
400
+
401
+
402
+ def contract_alarm(
403
+ root: Path,
404
+ github: Any,
405
+ target: str,
406
+ error: str | None,
407
+ now: float,
408
+ bot_login: str = "",
409
+ ) -> None:
410
+ """Persistent contract failure must surface where humans look.
411
+
412
+ A rejected or unfetchable contract silently idles every launch lane.
413
+ After CONTRACT_ALARM_AFTER consecutive failing ticks this
414
+ opens ONE issue on the target repo; the next successful load closes
415
+ it and says so. Alarm plumbing is best-effort by construction: it
416
+ must never break the tick it reports for."""
417
+ bot_login = bot_login or _bot_login_default()
418
+ state_path = root / "contract-alarm.json"
419
+ try:
420
+ state = json.loads(state_path.read_text())
421
+ except (OSError, ValueError):
422
+ state = {}
423
+ if error is None:
424
+ open_alarm = int(state.get("issue", 0))
425
+ if not open_alarm and state:
426
+ # creation may have landed without a recorded number (dry-run,
427
+ # odd response, lost state file): search so recovery can still
428
+ # close it. Only ticks that follow SOME failure signal pay the
429
+ # search; total state loss + instant recovery leaves the issue
430
+ # for the next alarm cycle's search to adopt and close.
431
+ with contextlib.suppress(Exception):
432
+ open_alarm = _find_alarm_issue(github, target, bot_login)
433
+ if open_alarm:
434
+ try:
435
+ github.close_issue(target, open_alarm)
436
+ except Exception as exc:
437
+ # keep the state so the NEXT healthy tick retries the close;
438
+ # unlinking here would orphan the open alarm forever — and
439
+ # no comment yet, or every retry would repeat it
440
+ log.warning("could not close contract alarm #%s: %s", open_alarm, exc)
441
+ return
442
+ with contextlib.suppress(Exception):
443
+ github.comment(
444
+ target, open_alarm, "The tick loads again cleanly; launch lanes resume."
445
+ )
446
+ if state:
447
+ with contextlib.suppress(OSError):
448
+ state_path.unlink()
449
+ return
450
+ count = int(state.get("count", 0)) + 1
451
+ state["count"] = count
452
+ # redacted (the client's own token is the one secret this process
453
+ # holds) and fenced with a run longer than any backtick run inside —
454
+ # transport errors can echo request material, loader errors can echo
455
+ # contract content, and both are untrusted for a public issue body
456
+ safe_error = redact(error, _client_secrets(github)).replace(str(Path.home()), "~")[:600]
457
+ # A recorded issue a human closed by hand stays closed: closing the
458
+ # alarm is the maintainer's "I know" — re-opening or re-creating it
459
+ # every threshold would be alarm spam, and recovery still clears state.
460
+ if count >= CONTRACT_ALARM_AFTER and not state.get("issue"):
461
+ # search open issues first: state loss must not spawn duplicates
462
+ try:
463
+ number = _find_alarm_issue(github, target, bot_login) or github.create_issue(
464
+ target,
465
+ "outerloop: launch lanes are paused",
466
+ f"{CONTRACT_ALARM_MARKER}\nThe orchestrator's launch lanes "
467
+ f"(intake, steward, self-initiated) have sat out {count} "
468
+ f"consecutive ticks. The error below names the cause — a "
469
+ f"contract that failed to load, or a panel config the climb "
470
+ f"would reject.\n\n"
471
+ f"{_fence(safe_error)}\n{safe_error}\n{_fence(safe_error)}\n\n"
472
+ f"This issue closes itself when a tick passes cleanly.",
473
+ )
474
+ if number:
475
+ state["issue"] = number
476
+ except Exception as exc:
477
+ log.warning("contract alarm could not post to %s: %s", target, exc)
478
+ with contextlib.suppress(OSError):
479
+ tmp = state_path.with_suffix(f".{os.getpid()}.tmp")
480
+ tmp.write_text(json.dumps(state))
481
+ os.replace(tmp, state_path)
482
+
483
+
484
+ def _client_secrets(github: Any) -> tuple[str, ...]:
485
+ try:
486
+ token = github.auth.token()
487
+ if token:
488
+ return (token,)
489
+ except Exception as exc:
490
+ # degraded redaction must not be silent: the error text goes to a
491
+ # public issue, and a renamed auth surface would no-op the redact
492
+ log.warning("alarm redaction has no client token (%s)", exc)
493
+ return ()
494
+
495
+
496
+ def _contract_text(github: Any, target: str, ref: str) -> str | None:
497
+ """The target's contract at `ref` — `.outerloop.yaml`, else the legacy
498
+ `.autoresearch.yaml` — or None when it has neither."""
499
+ from outerloop.contract import find_contract
500
+
501
+ found = find_contract(lambda name: github.get_file_content(target, name, ref))
502
+ return found[1] if found else None
503
+
504
+
505
+ def _fence(content: str) -> str:
506
+ longest = max((len(run) for run in re.findall(r"`+", content)), default=0)
507
+ return "`" * max(3, longest + 1)
508
+
509
+
510
+ def _find_alarm_issue(github: Any, target: str, bot_login: str) -> int:
511
+ """Only the BOT'S own marker'd issue counts: the marker is a public
512
+ string, and adopting a stranger's issue would let anyone suppress the
513
+ real alarm or get their issue closed by the bot."""
514
+ from outerloop.github import is_own_login
515
+
516
+ return next(
517
+ (
518
+ int(issue.get("number", 0))
519
+ for issue in github.list_open_issues(target, max_pages=10)
520
+ if has_marker(str(issue.get("body", "")), "contract-alarm")
521
+ and is_own_login(str((issue.get("user") or {}).get("login", "")), bot_login)
522
+ ),
523
+ 0,
524
+ )
525
+
526
+
527
+ def shape_followup_spec(spec: FollowupSpec, limits: EffectiveLimits, contract: Any) -> FollowupSpec:
528
+ """Clamp the operator's follow-up spec by the contract's effective
529
+ limits. Both knobs clamp only when the contract EXPLICITLY sets them:
530
+ a contract shapes spend downward, but an operator's deliberate config
531
+ is never silently reduced by defaults (raising budgets is operator
532
+ territory)."""
533
+ if contract is None:
534
+ return spec
535
+ # direct attribute access: Budgets is our typed model, and a rename
536
+ # must fail loudly here, not silently stop shaping spend downward
537
+ if contract.budgets.session_max_turns is not None:
538
+ spec = replace(spec, max_turns=min(spec.max_turns, limits.session_max_turns))
539
+ if contract.budgets.followup_job_minutes is not None:
540
+ spec = replace(spec, time_minutes=min(spec.time_minutes, limits.followup_job_minutes))
541
+ return spec
542
+
543
+
544
+ def _base_dial(
545
+ github: Any, target: str, pr: dict, main_contract: Any, main_target: str = ""
546
+ ) -> str:
547
+ """The merge dial that governs THIS PR: its own target's base-branch
548
+ contract. The tick's contract is read from ITS configured target's main;
549
+ it applies only to a main-based PR of that same target — any other
550
+ target or base is fetched from the PR's own coordinates, and unreadable
551
+ or unparsable means "manual" (never arm on doubt)."""
552
+ base_ref = str((pr.get("base") or {}).get("ref", "")) or "main"
553
+ if base_ref == "main" and target == main_target:
554
+ return str(getattr(main_contract, "merge", "manual"))
555
+ try:
556
+ from outerloop.contract import load_contract
557
+
558
+ raw = _contract_text(github, target, base_ref)
559
+ if raw is None:
560
+ return "manual"
561
+ return str(getattr(load_contract(raw, target), "merge", "manual"))
562
+ except Exception as exc:
563
+ log.warning("base-contract read failed for %s@%s: %s", target, base_ref, exc)
564
+ return "manual"
565
+
566
+
567
+ def service_in_review(
568
+ root: Path,
569
+ github: Any, # GitHubClient (Any keeps tick importable without github deps)
570
+ compute: Compute,
571
+ spec: FollowupSpec,
572
+ now: float,
573
+ dry_run: bool = False,
574
+ allow_submit: bool = True,
575
+ contract: Any = None,
576
+ records: list[RunRecord] | None = None,
577
+ ) -> tuple[list[tuple[str, str]], list[tuple[str, str]]]:
578
+ """PR-state transitions + follow-up job submission for in-review runs.
579
+
580
+ allow_submit=False (disk preflight failed) keeps the cheap state
581
+ transitions — ending merged/closed runs still matters — but submits no
582
+ new session jobs.
583
+
584
+ The tick only READS GitHub here (cheap, every cycle); the session-running
585
+ work happens in a submitted job, which takes the run lease itself — a
586
+ duplicate submission no-ops on the lease, and `followup_job_id` keeps the
587
+ tick from queueing duplicates in the first place.
588
+ """
589
+ from outerloop.followup import (
590
+ _pr_number,
591
+ close_if_done,
592
+ conflict_wake_action,
593
+ has_new_comments,
594
+ panel_wake_pending,
595
+ )
596
+
597
+ if records is None:
598
+ records = list_runs(root)
599
+ ended: list[tuple[str, str]] = []
600
+ submitted: list[tuple[str, str]] = []
601
+ for record in records:
602
+ if record.state != IN_REVIEW or not record.pr_url:
603
+ continue
604
+ try:
605
+ ending = close_if_done(root, load_record(root, record.run_id), github, now)
606
+ if ending:
607
+ ended.append((record.run_id, ending))
608
+ continue
609
+ # Steward records are serviced with the STEWARD'S key and the
610
+ # steward scope check (respond_once derives the mode from the
611
+ # record's agent id); without a provisioned steward key the
612
+ # lane stays human-answered.
613
+ is_steward = record.agent_id.startswith("steward")
614
+ if is_steward and not spec.steward_key_file:
615
+ continue
616
+ # per-ROLE outage latch: state transitions above still ran,
617
+ # only this record's session spawn sits the cooldown out
618
+ paused = outage_active(root, now, role="steward" if is_steward else "solver")
619
+ if paused:
620
+ log.info("follow-up for %s paused (api outage: %s)", record.run_id, paused)
621
+ continue
622
+ try:
623
+ pr = github.get_pull_request(record.target, _pr_number(record.pr_url))
624
+ except Exception:
625
+ continue # unreadable PR: nothing to decide this tick
626
+ # Idempotent auto-arm: once GitHub reports the PR CLEAN (green
627
+ # checks AND up-to-date with the CURRENT base — GitHub's own
628
+ # freshness proof), the kernel-read contract STILL says auto, and
629
+ # A dispatched re-measure in flight: nothing else is serviced (the
630
+ # sealed change lands first, so the next comment is answered on
631
+ # the tree it will actually see) and nothing is armed. Once every
632
+ # eval job is terminal, a follow-up is submitted to finish it.
633
+ measure_ready = False
634
+ if record.followup_stage:
635
+ raw_ids = record.followup_stage.get("job_ids")
636
+ job_ids = [str(j) for j in raw_ids] if isinstance(raw_ids, list) else []
637
+ if job_ids:
638
+ try:
639
+ states = [compute.status(j) for j in job_ids]
640
+ except SlurmQueryError:
641
+ continue # unknown: neither service nor arm
642
+ if not all(is_terminal(s) or s == GONE for s in states):
643
+ continue
644
+ measure_ready = True
645
+ else:
646
+ # a BLIND park (the measurer could not read the queue at
647
+ # dispatch): no ids to poll, so the eval walltime plus the
648
+ # climb's queue slack is the floor before a follow-up is
649
+ # sent to look — never one per tick (terra #241 r1)
650
+ from outerloop.dispatch import effective_eval_minutes
651
+
652
+ parked_at = float(record.followup_stage.get("parked_at", 0.0) or 0.0) # type: ignore[arg-type]
653
+ floor_min = int(record.followup_stage.get("eval_minutes", 0) or 0) # type: ignore[call-overload]
654
+ floor_s = (effective_eval_minutes(floor_min) + BLIND_PARK_SLACK_MIN) * 60
655
+ if now - parked_at < floor_s:
656
+ continue
657
+ measure_ready = True
658
+ wake_action = conflict_wake_action(record, pr)
659
+ if wake_action == "clear":
660
+ # the PR is clean again: re-arm the wake for this head — the
661
+ # base can move and conflict the SAME head a second time
662
+ try:
663
+ save_record(root, replace(record, dirty_wake_head=""), now)
664
+ except OSError as exc:
665
+ log.warning("conflict cursor clear failed for %s: %s", record.run_id, exc)
666
+ if (
667
+ not measure_ready
668
+ and not has_new_comments(record, github, spec.bot_login)
669
+ and wake_action != "wake"
670
+ and not panel_wake_pending(record, pr)
671
+ ):
672
+ # NOTHING awaits servicing — only a fully quiet PR may
673
+ # self-merge (pending reviewer feedback always wins over
674
+ # arming: a followup must service it first, and a pushed
675
+ # change would kill the blessing anyway). The RECORD says the
676
+ # publish was auto-eligible (published under merge:auto with
677
+ # a clean panel — #171's exact arming condition; a manual
678
+ # publish never consented, and contracts alone cannot prove
679
+ # either fact after a dial flip); GitHub's own CLEAN state is
680
+ # the freshness proof; the PR's base-branch contract is the
681
+ # governing dial. Running every tick survives any crash
682
+ # between a sync push and this step; the helper direct-merges
683
+ # when nothing is pending to arm against.
684
+ # ...and no follow-up job may be LIVE: a running responder can
685
+ # have pushed a code change whose record write (clearing the
686
+ # blessing) has not landed yet — arming on that head would
687
+ # merge code the panel never saw (terra #228 r7)
688
+ followup_live = False
689
+ if record.followup_job_id:
690
+ try:
691
+ state = compute.status(record.followup_job_id)
692
+ followup_live = not (is_terminal(state) or state == GONE)
693
+ except SlurmQueryError:
694
+ followup_live = True # unknown = assume live, never arm
695
+ if (
696
+ not dry_run
697
+ and not is_steward
698
+ and not followup_live
699
+ and record.auto_blessed_head
700
+ and str((pr.get("head") or {}).get("sha", "")) == record.auto_blessed_head
701
+ and contract is not None
702
+ and pr.get("state") == "open"
703
+ and not pr.get("merged")
704
+ and not pr.get("draft")
705
+ and pr.get("mergeable_state") == "clean"
706
+ and _base_dial(github, record.target, pr, contract, spec.target) == "auto"
707
+ ):
708
+ try:
709
+ # the mutation itself is bound to the blessed head: a
710
+ # push racing this check is refused by GitHub, not
711
+ # merged (terra #228 r9)
712
+ github.arm_auto_merge_auto_mode(
713
+ record.target,
714
+ _pr_number(record.pr_url),
715
+ expected_head=record.auto_blessed_head,
716
+ )
717
+ except Exception as exc:
718
+ log.warning("auto-arm failed for %s: %s", record.run_id, exc)
719
+ continue
720
+ if record.followup_job_id:
721
+ try:
722
+ state = compute.status(record.followup_job_id)
723
+ if not (is_terminal(state) or state == GONE):
724
+ continue # a follow-up job is already queued/running
725
+ except SlurmQueryError:
726
+ continue # unknown — do not stack another job
727
+ # the wake-attempt counter caps follow-up retries too: a responder
728
+ # that cannot advance its cursors must not burn a session per tick.
729
+ # A LANDED re-measure gets its own allowance: the sessions that
730
+ # synced the base and dispatched it were progress, and capping the
731
+ # follow-up that pushes the sealed number parks the result forever
732
+ # (speedrun agent-03, 2026-09-04: three syncs spent the cap, the
733
+ # measure completed, and the run idled for two days on this
734
+ # branch). The allowance is counted on the stage itself, so a
735
+ # finishing session that reverts and leaves the stage intact is
736
+ # retried MAX_WAKE_ATTEMPTS times, not once per tick; a stage that
737
+ # finishes is cleared, and with it the count.
738
+ finish_attempts = (
739
+ int(record.followup_stage.get("finish_attempts", 0) or 0) # type: ignore[call-overload]
740
+ if measure_ready
741
+ else 0
742
+ )
743
+ if measure_ready and finish_attempts >= MAX_WAKE_ATTEMPTS:
744
+ log.warning(
745
+ "run %s: %d follow-ups failed to finish the landed re-measure; parked",
746
+ record.run_id,
747
+ finish_attempts,
748
+ )
749
+ continue
750
+ if record.wake_attempts >= MAX_WAKE_ATTEMPTS and not measure_ready:
751
+ log.warning(
752
+ "run %s: %d follow-up attempts without progress; not resubmitting",
753
+ record.run_id,
754
+ record.wake_attempts,
755
+ )
756
+ continue
757
+ if not allow_submit:
758
+ log.warning("run %s has new comments but disk preflight failed", record.run_id)
759
+ continue
760
+ if dry_run:
761
+ submitted.append((record.run_id, "dry-run"))
762
+ continue
763
+ # The author follow-up carries the climb's panel so a pushed code
764
+ # change is RE-READ before the tick may arm it (followup.py). The
765
+ # panel brings its own walltime, like the climb's allowance — the
766
+ # contract's followup budget caps the author, not the gate. A
767
+ # panel that would die at startup is left off: the reply still
768
+ # goes out, the PR simply stays human-merged (the tick's contract
769
+ # alarm already names the misconfig).
770
+ panel_argv: list[str] = []
771
+ panel_minutes = 0
772
+ if not is_steward and spec.panel.strip():
773
+ panel_error = _panel_preflight_error(spec)
774
+ if panel_error:
775
+ log.warning(
776
+ "follow-up for %s runs without the panel: %s", record.run_id, panel_error
777
+ )
778
+ else:
779
+ from outerloop.panel import panel_read_minutes
780
+
781
+ panel_argv = _climb_panel_argv(spec)
782
+ panel_minutes = panel_read_minutes(spec.panel)
783
+ # the author's budget first, the read on top, both under the
784
+ # partition cap — and the follow-up is told how many minutes the
785
+ # read actually got (--panel-minutes), so a cap that eats the
786
+ # allowance costs the READ (skipped, said so), never the author
787
+ author_minutes = min(spec.time_minutes, spec.max_job_minutes)
788
+ job_minutes = min(author_minutes + panel_minutes, spec.max_job_minutes)
789
+ if panel_argv:
790
+ panel_argv = [*panel_argv, "--panel-minutes", str(job_minutes - author_minutes)]
791
+ argv = [
792
+ *_interpreter(spec.home),
793
+ "-m",
794
+ "outerloop.followup",
795
+ "--run-root",
796
+ str(spec.run_root),
797
+ "--run-id",
798
+ record.run_id,
799
+ *_containment(spec.image),
800
+ "--bot-login",
801
+ spec.bot_login,
802
+ "--job-minutes",
803
+ # the SAME clamped value Slurm gets: a deadline armed past
804
+ # the real walltime is a Slurm kill before a clean ending
805
+ str(job_minutes),
806
+ "--max-turns",
807
+ str(spec.max_turns),
808
+ # the cluster coordinates the climb gets: a GPU benchmark's
809
+ # re-measure is dispatched to the GPU lane, never run here
810
+ "--account",
811
+ spec.account,
812
+ "--partition",
813
+ spec.partition,
814
+ "--gpu-partition",
815
+ spec.gpu_partition,
816
+ "--gpu-account",
817
+ spec.gpu_account,
818
+ *panel_argv,
819
+ ]
820
+ if spec.pat_file:
821
+ argv += ["--pat-file", spec.pat_file]
822
+ # config-driven author: the author follow-up resolves its key per the
823
+ # RUN's backend (from the record) inside followup.main — the tick does
824
+ # not thread it. The steward is a distinct role with its own key.
825
+ if is_steward and spec.steward_key_file:
826
+ argv += ["--key-file", spec.steward_key_file]
827
+ job_id = compute.submit(
828
+ JobSpec(
829
+ job_name=f"followup-{record.run_id}"[:60],
830
+ account=spec.account,
831
+ partition=spec.job_partition or spec.partition,
832
+ time_minutes=job_minutes,
833
+ command=_flight_command(spec.home, f"followup-{record.run_id}"[:60], now, argv),
834
+ cpus=4,
835
+ mem="8G",
836
+ )
837
+ )
838
+ # read-modify-write on the FRESH record: the submitted job may
839
+ # already be saving its own fields
840
+ latest = load_record(root, record.run_id)
841
+ stage = latest.followup_stage
842
+ if measure_ready and stage:
843
+ # one finishing attempt billed against this landed measure
844
+ stage = {**stage, "finish_attempts": finish_attempts + 1}
845
+ save_record(
846
+ root,
847
+ replace(
848
+ latest,
849
+ followup_job_id=job_id,
850
+ followup_stage=stage,
851
+ wake_attempts=latest.wake_attempts + 1,
852
+ ),
853
+ now,
854
+ )
855
+ submitted.append((record.run_id, job_id))
856
+ except (SlurmError, Exception) as exc:
857
+ log.warning("in-review service failed for %s: %s", record.run_id, exc)
858
+ return ended, submitted
859
+
860
+
861
+ def _last_worked_ts(root: Path) -> float | None:
862
+ """The `ts` of the last tick that COMPLETED its work, or None if there is no
863
+ readable marker (first tick, or a corrupt/missing file). The coalesce guard
864
+ keys on this, NOT the heartbeat: the heartbeat is stamped at tick START (for
865
+ the watchdog), so a tick that crashes mid-work still leaves a fresh
866
+ heartbeat — coalescing on that would suppress the very recovery tick. The
867
+ work marker is written only at a full tick's END, so a failed tick never
868
+ hides behind it."""
869
+ # The marker is a best-effort optimization we write ourselves; ANY failure
870
+ # reading/parsing/converting a corrupt file (OSError, ValueError,
871
+ # OverflowError on a huge int, RecursionError on deep nesting, ...) must
872
+ # degrade to "no marker" so coalesce simply proceeds — it can never crash the
873
+ # tick before its heartbeat. bool is an int subclass, so exclude it; inf/nan
874
+ # are not usable elapsed anchors.
875
+ try:
876
+ payload = json.loads((root / WORK_MARKER_NAME).read_text())
877
+ ts = payload.get("ts") if isinstance(payload, dict) else None
878
+ if not isinstance(ts, int | float) or isinstance(ts, bool):
879
+ return None
880
+ val = float(ts)
881
+ return val if math.isfinite(val) else None
882
+ except Exception as exc:
883
+ log.debug("work marker unreadable, coalescing as if absent: %s", exc)
884
+ return None
885
+
886
+
887
+ def _mark_worked(root: Path, now: float) -> None:
888
+ """Record that a tick completed its work at `now` — the coalesce signal.
889
+ Best-effort: a marker that cannot be written must not fail the tick."""
890
+ try:
891
+ tmp = root / f".{WORK_MARKER_NAME}.tmp"
892
+ tmp.write_text(json.dumps({"ts": now}))
893
+ os.replace(tmp, root / WORK_MARKER_NAME)
894
+ except OSError as exc:
895
+ log.warning("work-marker write failed: %s", exc)
896
+
897
+
898
+ def mark_tick_complete(root: Path, report: TickReport, now: float) -> None:
899
+ """Stamp the coalesce marker iff the tick actually did work, at real
900
+ COMPLETION time (the caller passes time.time() AFTER tick() returns). A
901
+ paused/coalesced tick leaves it untouched; a tick that raised never reaches
902
+ here — so only a genuinely completed tick can coalesce the next one, and a
903
+ long tick's marker reflects when it finished, not when it started."""
904
+ if not report.paused and not report.coalesced:
905
+ _mark_worked(root, now)
906
+
907
+
908
+ def write_heartbeat(root: Path, now: float, disk: dict[str, object] | None = None) -> None:
909
+ """Best-effort: a heartbeat that cannot be written (full disk) must not
910
+ kill the tick — the tick can still end runs and post to GitHub."""
911
+ payload: dict[str, object] = {"ts": now, "host": socket.gethostname(), "pid": os.getpid()}
912
+ if disk is not None:
913
+ payload["disk"] = disk
914
+ try:
915
+ tmp = root / f".{HEARTBEAT_NAME}.tmp"
916
+ tmp.write_text(json.dumps(payload))
917
+ os.replace(tmp, root / HEARTBEAT_NAME)
918
+ except OSError as exc:
919
+ log.warning("heartbeat write failed: %s", exc)
920
+
921
+
922
+ def _holder_alive(compute: Compute, lease_job_id: str) -> bool | None:
923
+ """True/False when Slurm answered; None when it could not (an outage
924
+ must not look like a dead holder)."""
925
+ if not lease_job_id:
926
+ return None
927
+ try:
928
+ state = compute.status(lease_job_id)
929
+ except SlurmQueryError:
930
+ return None
931
+ return not (is_terminal(state) or state == GONE)
932
+
933
+
934
+ def _wake(
935
+ root: Path,
936
+ record: RunRecord,
937
+ reason: str,
938
+ dispatcher: WakeDispatcher,
939
+ now: float,
940
+ holder: str,
941
+ ) -> bool:
942
+ """Lease-guarded wake. True when this tick delivered (or handed off) it.
943
+
944
+ The attempt counter is bumped BEFORE dispatch, so a dispatcher that dies
945
+ mid-delivery still counts toward the stuck threshold.
946
+ """
947
+ if not acquire_lease(root, record.run_id, holder, holder_job_id="", now=now):
948
+ return False
949
+ bumped = replace(
950
+ record,
951
+ wake_attempts=record.wake_attempts + 1,
952
+ # repair legacy records as we touch them: save_record (rightly)
953
+ # refuses to write a waiting run without a deadline
954
+ deadline=record.deadline if record.deadline > 0 else now,
955
+ )
956
+ save_record(root, bumped, now)
957
+ try:
958
+ holder_job = dispatcher.dispatch(bumped, reason)
959
+ except Exception as exc:
960
+ log.warning("wake dispatch failed for %s: %s: %s", record.run_id, type(exc).__name__, exc)
961
+ release_lease(root, record.run_id)
962
+ return False
963
+ if holder_job and not local_mode():
964
+ # An async wake job now owns the lease; it releases on completion,
965
+ # and the TTL/holder-dead check reaps it if it dies.
966
+ update_lease_holder(root, record.run_id, f"wake-job:{holder_job}", holder_job, now)
967
+ else:
968
+ # No job to hand the lease to — or a LOCAL dispatch, which ran the
969
+ # whole wake synchronously: the attempt already finished and released
970
+ # its own lease, and recreating one under a terminal job id would
971
+ # make the next sweep reap a corpse instead of delivering.
972
+ release_lease(root, record.run_id)
973
+ return True
974
+
975
+
976
+ WAKE_SPEC_NAME = "wake-spec.json"
977
+
978
+
979
+ def dispatch_wake_armed(root: Path) -> bool:
980
+ """The operator's on-switch for dispatched wakes: the env var, or the
981
+ sentinel file (touch/rm, no chain restart). Read by the tick and by every
982
+ park, so a disarm takes effect at once."""
983
+ return (
984
+ bool(os.environ.get("OUTERLOOP_DISPATCH_WAKE", "").strip())
985
+ or (root / DISPATCH_WAKE_SENTINEL).exists()
986
+ )
987
+
988
+
989
+ def write_wake_spec(root: Path, spec: FollowupSpec) -> None:
990
+ """Publish the tick's wake recipe for the jobs that park runs: a park
991
+ submits its own wake (`arm_wake`) with exactly the tick's settings, so
992
+ dispatched wakes stay one recipe with one owner."""
993
+ data = {k: (str(v) if isinstance(v, Path) else v) for k, v in asdict(spec).items()}
994
+ tmp = root / f".{WAKE_SPEC_NAME}.{os.getpid()}.tmp"
995
+ tmp.write_text(json.dumps(data))
996
+ os.replace(tmp, root / WAKE_SPEC_NAME)
997
+
998
+
999
+ def remove_wake_spec(root: Path) -> None:
1000
+ with contextlib.suppress(FileNotFoundError):
1001
+ (root / WAKE_SPEC_NAME).unlink()
1002
+
1003
+
1004
+ def load_wake_spec(root: Path) -> FollowupSpec | None:
1005
+ """The published wake recipe, or None when dispatched wakes are not armed
1006
+ (or the file is unreadable — the sweep still delivers)."""
1007
+ try:
1008
+ data = json.loads((root / WAKE_SPEC_NAME).read_text())
1009
+ except (OSError, ValueError):
1010
+ return None
1011
+ if not isinstance(data, dict):
1012
+ return None
1013
+ names = {f.name for f in FollowupSpec.__dataclass_fields__.values()}
1014
+ kwargs: dict[str, Any] = {
1015
+ k: (Path(v) if k in ("run_root", "home") else v) for k, v in data.items() if k in names
1016
+ }
1017
+ try:
1018
+ return FollowupSpec(**kwargs)
1019
+ except TypeError:
1020
+ return None
1021
+
1022
+
1023
+ def arm_wake(
1024
+ root: Path, record: RunRecord, dispatcher: WakeDispatcher, now: float, *, holder_job_id: str
1025
+ ) -> str:
1026
+ """Submit a parked run's wake NOW, depending on the jobs it waits on, so
1027
+ it fires the moment they finish instead of a sweep cadence (plus grace)
1028
+ later. The same job and lease as a sweep-delivered wake: a wake job that
1029
+ parks again hands its own lease to the wake it arms; any other holder (a
1030
+ tick mid-delivery) keeps it and the sweep delivers as before. Arming is
1031
+ not a redelivery, so it leaves `wake_attempts` — the sweep's stuck
1032
+ counter — alone: an eval requeued by preemption fires the afterany
1033
+ early, the wake re-parks, and that must not count toward STUCK. Returns
1034
+ the wake job id, or "" when nothing was armed. A park with nothing to
1035
+ depend on (a checkpoint sleep, a blind park) is not armed: it rides the
1036
+ deadline floor, as before."""
1037
+ if not _poll_targets(record):
1038
+ return ""
1039
+ lease = read_lease(root, record.run_id)
1040
+ if lease is not None:
1041
+ if not holder_job_id or lease.holder_job_id != holder_job_id:
1042
+ return ""
1043
+ elif not acquire_lease(
1044
+ root, record.run_id, f"park:{holder_job_id or os.getpid()}", holder_job_id="", now=now
1045
+ ):
1046
+ return ""
1047
+ if record.deadline <= 0: # a waiting record always carries a deadline
1048
+ record = replace(record, deadline=now)
1049
+ save_record(root, record, now)
1050
+ try:
1051
+ job = dispatcher.dispatch(record, "parked")
1052
+ except Exception as exc:
1053
+ log.warning("arming the wake failed for %s: %s: %s", record.run_id, type(exc).__name__, exc)
1054
+ job = ""
1055
+ if job:
1056
+ update_lease_holder(root, record.run_id, f"wake-job:{job}", job, now)
1057
+ elif lease is None:
1058
+ release_lease(root, record.run_id)
1059
+ return job
1060
+
1061
+
1062
+ def _armed_wake_lost(
1063
+ root: Path,
1064
+ compute: Compute,
1065
+ record: RunRecord,
1066
+ lease: Lease,
1067
+ now: float,
1068
+ grace_s: float,
1069
+ dry_run: bool,
1070
+ ) -> bool:
1071
+ """An armed wake that is still PENDING on its dependency after every job
1072
+ it waits on has been terminal for a full grace window is not coming
1073
+ (Slurm reports the dependency as never satisfiable, or the afterany was
1074
+ lost). Cancel it so the sweep redelivers; the lease is then reaped.
1075
+
1076
+ A wake the SITE moved off the partition it was submitted to may be
1077
+ starving there — on Torch a pending job can be shifted to a lower-tier
1078
+ partition (2026-09-02, wake 16787511 sat on `all` for hours) — but
1079
+ relocation alone is routine (jobs move to `cs` while still waiting on
1080
+ their dependencies and start on time). So a relocated holder counts as
1081
+ lost only once every job it waited on is terminal AND the grace window
1082
+ has run out without it starting; then it is cancelled and redelivered
1083
+ onto the requested partition."""
1084
+ if not lease.holder_job_id:
1085
+ return False
1086
+ job_ids = _poll_targets(record)
1087
+ if not job_ids:
1088
+ return False
1089
+ try:
1090
+ holder_state = compute.status(lease.holder_job_id)
1091
+ if not is_pending(holder_state):
1092
+ return False
1093
+ reason = compute.pending_reason(lease.holder_job_id)
1094
+ states = [compute.status(jid) for jid in job_ids]
1095
+ except SlurmQueryError:
1096
+ return False
1097
+ moved = _moved_off_partition(root, compute, lease.holder_job_id)
1098
+ if reason == "DependencyNeverSatisfied":
1099
+ pass
1100
+ elif not all(is_terminal(s) for s in states):
1101
+ return False # its dependencies are still running: nothing to redeliver yet
1102
+ elif reason != "Dependency" and not moved:
1103
+ return False
1104
+ else:
1105
+ # Dependency-pending past its dependencies, or RELOCATED and eligible:
1106
+ # both get the grace window. Relocation alone is normal (the site
1107
+ # moves pending jobs routinely); a relocated wake counts as lost only
1108
+ # when every job it waited on is terminal and it still has not
1109
+ # started once the grace has run out — cancelling earlier would only
1110
+ # reset its queue age and burn a wake attempt.
1111
+ if moved:
1112
+ log.info(
1113
+ "armed wake %s for %s sits on partition %s (asked for %s) past its dependencies",
1114
+ lease.holder_job_id,
1115
+ record.run_id,
1116
+ moved[0],
1117
+ moved[1],
1118
+ )
1119
+ if record.terminal_seen <= 0:
1120
+ if not dry_run:
1121
+ save_record(root, replace(record, terminal_seen=now), now)
1122
+ return False
1123
+ if now - record.terminal_seen < grace_s:
1124
+ return False
1125
+ if dry_run:
1126
+ return True
1127
+ try:
1128
+ compute.cancel(lease.holder_job_id)
1129
+ # only a cancellation Slurm confirms lets the sweep redeliver: a
1130
+ # still-pending wake would otherwise run beside its replacement
1131
+ return not is_pending(compute.status(lease.holder_job_id))
1132
+ except Exception:
1133
+ return False
1134
+
1135
+
1136
+ def sweep(
1137
+ root: Path,
1138
+ compute: Compute,
1139
+ dispatcher: WakeDispatcher,
1140
+ now: float,
1141
+ grace_s: float = DEFAULT_GRACE_S,
1142
+ lease_ttl_s: float = DEFAULT_LEASE_TTL_S,
1143
+ dry_run: bool = False,
1144
+ ) -> TickReport:
1145
+ """The backup wake layers, applied to every waiting run.
1146
+
1147
+ dry_run reports what WOULD happen with zero writes — no leases, no
1148
+ attempt counters, no dispatch.
1149
+ """
1150
+ woken: list[tuple[str, str]] = []
1151
+ deferred: list[str] = []
1152
+ reaped: list[str] = []
1153
+ stuck: list[str] = []
1154
+ holder = f"tick:{socket.gethostname()}:{os.getpid()}"
1155
+ records = [r for r in list_runs(root) if r.state == WAITING]
1156
+
1157
+ def wake(record: RunRecord, reason: str, tag: str) -> None:
1158
+ if dry_run or _wake(root, record, reason, dispatcher, now, holder):
1159
+ woken.append((record.run_id, tag))
1160
+
1161
+ for record in records:
1162
+ try:
1163
+ _sweep_one(
1164
+ root,
1165
+ compute,
1166
+ dispatcher,
1167
+ now,
1168
+ grace_s,
1169
+ lease_ttl_s,
1170
+ dry_run,
1171
+ record,
1172
+ holder,
1173
+ wake,
1174
+ deferred,
1175
+ reaped,
1176
+ stuck,
1177
+ )
1178
+ except Exception as exc:
1179
+ log.warning("sweep failed on %s: %s: %s", record.run_id, type(exc).__name__, exc)
1180
+
1181
+ try:
1182
+ cancel_ended_launches(root, compute, now, dry_run)
1183
+ except Exception as exc:
1184
+ log.warning("cancel-on-end failed: %s: %s", type(exc).__name__, exc)
1185
+
1186
+ return TickReport(
1187
+ swept=len(records),
1188
+ woken=tuple(woken),
1189
+ deferred=tuple(deferred),
1190
+ reaped_leases=tuple(reaped),
1191
+ stuck=tuple(stuck),
1192
+ # NOT the global dry_run: that flag only dries WAKE delivery;
1193
+ # ending killed climbs' records dispatches nothing and must run
1194
+ # live even while wakes stay dry.
1195
+ implementing_ended=tuple(_sweep_implementing(root, compute, now, grace_s)),
1196
+ )
1197
+
1198
+
1199
+ CANCEL_ON_END_WINDOW_S = 7 * 86400
1200
+
1201
+
1202
+ def cancel_ended_launches(
1203
+ root: Path, compute: Compute, now: float, dry_run: bool = False
1204
+ ) -> list[str]:
1205
+ """Cancel on end: a run that ended while its author's launches are still
1206
+ queued or running has nothing left to read their results, so the jobs are
1207
+ cancelled — a running one included, since that is the one holding GPUs for
1208
+ a result nobody will read. The stage is stamped once every cancel
1209
+ succeeded, so an ended run costs no query afterwards; a failed scancel
1210
+ leaves the run unstamped for the next tick, and runs ended longer ago than
1211
+ the window are stamped without a query (their jobs have long left the
1212
+ queue). Returns the cancelled ids."""
1213
+ from outerloop.attempt import stage_launch_job_ids
1214
+
1215
+ cancelled: list[str] = []
1216
+ for record in list_runs(root):
1217
+ if record.state != ENDED:
1218
+ continue
1219
+ stage = dict(record.stage or {})
1220
+ if stage.get("launches_cancelled"):
1221
+ continue
1222
+ job_ids = stage_launch_job_ids(record)
1223
+ recent = now - float(getattr(record, "updated", 0.0) or 0.0) <= CANCEL_ON_END_WINDOW_S
1224
+ live: list[str] = []
1225
+ if job_ids and recent:
1226
+ try:
1227
+ for jid in job_ids:
1228
+ state = compute.status(jid)
1229
+ if state != GONE and not is_terminal(state):
1230
+ live.append(jid)
1231
+ except SlurmQueryError as exc:
1232
+ log.warning("cancel-on-end: %s: query failed (%s); next tick", record.run_id, exc)
1233
+ continue
1234
+ if dry_run:
1235
+ cancelled.extend(live)
1236
+ continue
1237
+ failed = False
1238
+ for jid in live:
1239
+ try:
1240
+ ok = compute.cancel(jid)
1241
+ except Exception as exc:
1242
+ ok = False
1243
+ log.warning("cancel-on-end: %s: scancel %s failed: %s", record.run_id, jid, exc)
1244
+ if ok:
1245
+ cancelled.append(jid)
1246
+ else:
1247
+ failed = True
1248
+ if failed:
1249
+ continue # unstamped: the next tick tries again, until the window closes
1250
+ stage["launches_cancelled"] = True
1251
+ save_record(root, replace(record, stage=stage), now)
1252
+ if cancelled:
1253
+ log.info("cancel-on-end: cancelled %d launch job(s) of ended runs", len(cancelled))
1254
+ return cancelled
1255
+
1256
+
1257
+ RESEARCH_LOG_BRANCH = "research-log"
1258
+ RESEARCH_LOG_MARKER = marker("research-log")
1259
+ RESEARCH_LOG_PER_TICK = 3
1260
+
1261
+
1262
+ def _ledger_marker(root: Path, run_id: str) -> Path:
1263
+ return run_dir(root, run_id) / "ledger-published"
1264
+
1265
+
1266
+ def _ledger_since(root: Path, target: str) -> Path:
1267
+ return root / ("research-log-since-" + target.replace("/", "__"))
1268
+
1269
+
1270
+ def _ledger_issue_cache(root: Path, target: str) -> Path:
1271
+ return root / ("research-log-issue-" + target.replace("/", "__"))
1272
+
1273
+
1274
+ def service_research_log(
1275
+ root: Path,
1276
+ github: Any,
1277
+ spec: FollowupSpec,
1278
+ now: float,
1279
+ records: list[RunRecord] | None = None,
1280
+ ) -> int:
1281
+ """STATE-driven ledger (terra #170 r1: wiring publisher calls at terminal
1282
+ sites missed four terminal paths — attempt error, zero-change resume,
1283
+ the direct live terminal, steward): any run of this target whose
1284
+ terminal report exists but carries no published marker gets archived on
1285
+ the `research-log` branch and a two-line pointer routed — to the run's
1286
+ claimed issue never (it already got the full finish post), else to an
1287
+ open order issue naming the benchmark, else to the rolling log issue
1288
+ whose number is CACHED in the state dir (no 300-issue pagination scan,
1289
+ no duplicate creation). Running in the single tick also removes the
1290
+ concurrent-first-archive races by construction. The pointer posts only
1291
+ AFTER a successful archive (no dead links); the marker is written only
1292
+ after full success, so a crashed publish retries next tick. Runs ended
1293
+ before the feature's first pass are marker-stamped silently (no
1294
+ backfill spam).
1295
+ """
1296
+ if records is None:
1297
+ records = list_runs(root)
1298
+ since_path = _ledger_since(root, spec.target)
1299
+ first_pass = not since_path.exists()
1300
+ if first_pass:
1301
+ with contextlib.suppress(OSError):
1302
+ since_path.write_text(str(now))
1303
+ published = 0
1304
+ for record in records:
1305
+ if record.target != spec.target or record.state not in (ENDED, IN_REVIEW):
1306
+ continue
1307
+ marker = _ledger_marker(root, record.run_id)
1308
+ report_path = run_dir(root, record.run_id) / "report.md"
1309
+ state = ""
1310
+ with contextlib.suppress(OSError):
1311
+ state = marker.read_text()
1312
+ if state.startswith(("done", "adopted")) or not report_path.exists():
1313
+ continue
1314
+ if first_pass and (record.updated or record.created) < now:
1315
+ # adopt pre-feature history silently: browsable going forward,
1316
+ # no retroactive issue spam. Only records OLDER than the since
1317
+ # marker — a run that goes terminal during this very pass is new
1318
+ # work and publishes normally (terra #170 r2).
1319
+ with contextlib.suppress(OSError):
1320
+ marker.write_text("adopted-unpublished")
1321
+ continue
1322
+ if published >= RESEARCH_LOG_PER_TICK:
1323
+ break
1324
+ try:
1325
+ report = report_path.read_text()
1326
+ except OSError:
1327
+ continue
1328
+ outcome = record.ending or ("improved" if record.state == IN_REVIEW else "ended")
1329
+ if _publish_ledger_entry(github, spec.target, root, record, outcome, report, marker, state):
1330
+ published += 1
1331
+ return published
1332
+
1333
+
1334
+ def _publish_ledger_entry(
1335
+ github: Any,
1336
+ target: str,
1337
+ root: Path,
1338
+ record: RunRecord,
1339
+ outcome: str,
1340
+ report: str,
1341
+ marker: Path,
1342
+ state: str,
1343
+ ) -> bool:
1344
+ """Staged publish whose marker doubles as a WRITABILITY PROBE: stages
1345
+ are "archived" -> "pointer-pending" -> "done", and no pointer is ever
1346
+ posted in a pass where a marker write is failing (terra #170 r4: a
1347
+ persistently unwritable marker must stall the publish, not stream a
1348
+ duplicate pointer every tick). A pointer failure retries pointer-only;
1349
+ a marker lost after full success re-posts at most once; the residual
1350
+ crash-between-probe-and-post window costs at most one duplicate per
1351
+ incident, never an unbounded stream."""
1352
+ from datetime import UTC, datetime
1353
+
1354
+ date = datetime.fromtimestamp(record.updated or record.created, tz=UTC).strftime("%Y-%m-%d")
1355
+ path = f"reports/{date}-{record.run_id}.md"
1356
+ # an earlier pass may have archived under an earlier date (an in-review
1357
+ # archive whose record re-stamped `updated` at ENDED): the marker's own
1358
+ # second line is the authoritative path for retries and pointers
1359
+ prior = state.splitlines()
1360
+ if len(prior) > 1 and prior[1].startswith("reports/") and prior[1].endswith(".md"):
1361
+ path = prior[1]
1362
+
1363
+ def _mark(value: str) -> bool:
1364
+ try:
1365
+ # the path rides the marker so readers (the board) never have to
1366
+ # re-derive the date from a timestamp that may have moved on
1367
+ marker.write_text(value + "\n" + path)
1368
+ return True
1369
+ except OSError as exc:
1370
+ log.warning("ledger marker write failed for %s: %s", record.run_id, exc)
1371
+ return False
1372
+
1373
+ try:
1374
+ if not state.startswith(("archived", "pointer-pending")):
1375
+ if not github.ensure_branch(target, RESEARCH_LOG_BRANCH):
1376
+ return False
1377
+ if not github.put_file(
1378
+ target,
1379
+ path,
1380
+ report,
1381
+ RESEARCH_LOG_BRANCH,
1382
+ f"research log: {record.run_id} ({outcome})",
1383
+ ):
1384
+ return False # retry the whole publish next tick
1385
+ if not _mark("archived"):
1386
+ return False # unwritable state: stall BEFORE any pointer
1387
+ # the probe: a fresh successful write is the license to post
1388
+ if not _mark("pointer-pending"):
1389
+ return False
1390
+ url = f"https://github.com/{target}/blob/{RESEARCH_LOG_BRANCH}/{path}"
1391
+ line = f"**{outcome}** `{record.benchmark}` — [report]({url})"
1392
+ if record.pr_url:
1393
+ line += f" · {record.pr_url}"
1394
+ if record.issue_number:
1395
+ _mark("done") # the claimed issue already received the full finish
1396
+ return True
1397
+ bench = record.benchmark.casefold()
1398
+ posted = False
1399
+ if bench:
1400
+ for issue in github.list_open_issues(target):
1401
+ text = f"{issue.get('title', '')}\n{issue.get('body') or ''}"
1402
+ if has_marker(text, "research-log") or issue.get("pull_request"):
1403
+ continue
1404
+ if bench in text.casefold():
1405
+ github.comment(target, int(issue["number"]), line)
1406
+ posted = True
1407
+ break
1408
+ if not posted:
1409
+ cache = _ledger_issue_cache(root, target)
1410
+ log_issue = 0
1411
+ with contextlib.suppress(OSError, ValueError):
1412
+ log_issue = int(cache.read_text().strip())
1413
+ if not log_issue:
1414
+ # cache miss (first use, or a lost/failed cache write): find
1415
+ # the rolling issue by its marker BEFORE creating another —
1416
+ # the cache is a fast path, never the source of truth
1417
+ # (terra #170 r5: a failed cache write must not duplicate
1418
+ # the rolling issue)
1419
+ for issue in github.list_open_issues(target):
1420
+ if has_marker(str(issue.get("body") or ""), "research-log"):
1421
+ log_issue = int(issue.get("number", 0))
1422
+ break
1423
+ if not log_issue:
1424
+ log_issue = github.create_issue(
1425
+ target,
1426
+ "Research log",
1427
+ f"{RESEARCH_LOG_MARKER}\nOne two-line comment per finished "
1428
+ f"autoresearch run — full reports live on the [`{RESEARCH_LOG_BRANCH}`]"
1429
+ f"(https://github.com/{target}/tree/{RESEARCH_LOG_BRANCH}/reports) "
1430
+ "branch. Results relevant to an open order issue are posted "
1431
+ "there instead.",
1432
+ )
1433
+ if log_issue:
1434
+ with contextlib.suppress(OSError):
1435
+ cache.write_text(str(log_issue))
1436
+ try:
1437
+ github.comment(target, log_issue, line)
1438
+ except Exception:
1439
+ # a stale cached number (locked/deleted/transferred issue)
1440
+ # must not stall delivery forever: drop the cache so the
1441
+ # next pass re-discovers or re-creates (terra #170 r5)
1442
+ with contextlib.suppress(OSError):
1443
+ cache.unlink()
1444
+ raise
1445
+ _mark("done") # write just proved out via the probe; failure = freak
1446
+ return True
1447
+ except Exception as exc: # advisory ledger: never fail the tick
1448
+ log.warning("research-log publish failed for %s: %s", record.run_id, exc)
1449
+ return False
1450
+
1451
+
1452
+ def _kill_stamp(root: Path, run_id: str) -> Path:
1453
+ return run_dir(root, run_id) / "attempt-terminal-seen"
1454
+
1455
+
1456
+ def _sweep_implementing(root: Path, compute: Compute, now: float, grace_s: float) -> list[str]:
1457
+ """End `implementing` records whose climb job died without a verdict.
1458
+
1459
+ A climb that CRASHES contains its own ending (attempt.py); a climb that is
1460
+ KILLED — walltime, preemption, scancel after the SIGTERM grace, node
1461
+ death — leaves no exception to contain, so this pass records the ending
1462
+ (the picker's stranded guard only frees the lane). Slurm truth decides:
1463
+ job terminal or GONE, plus a grace so a just-finished healthy climb can
1464
+ write its own final state first. Outage never reads as dead. Legacy
1465
+ records without a job id age out on the stranded window instead.
1466
+ """
1467
+ ended: list[str] = []
1468
+ for record in list_runs(root):
1469
+ if record.state != IMPLEMENTING:
1470
+ continue
1471
+ try:
1472
+ if record.run_job_id:
1473
+ try:
1474
+ state = compute.status(record.run_job_id)
1475
+ except SlurmQueryError:
1476
+ continue # Slurm outage must not read as a dead job
1477
+ if not (is_terminal(state) or state == GONE):
1478
+ continue # alive; the climb owns its own record
1479
+ # Grace runs from FIRST OBSERVED terminal: during Slurm's
1480
+ # KillWait the job already reports terminal while the
1481
+ # SIGTERM containment is still writing its honest ending —
1482
+ # record age would be hours and protect nothing. The stamp
1483
+ # is a write-once SIDECAR file, never a record write: while
1484
+ # the climb may still be alive the sweep must not touch the
1485
+ # record at all (a load-modify-replace here could revert a
1486
+ # concurrently written ending), and the waiting sweep's
1487
+ # terminal_seen field stays reserved for the EXPERIMENT job.
1488
+ stamp = _kill_stamp(root, record.run_id)
1489
+ if not stamp.exists():
1490
+ try:
1491
+ fd = os.open(stamp, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o644)
1492
+ try:
1493
+ os.write(fd, f"{now}".encode())
1494
+ finally:
1495
+ os.close(fd)
1496
+ except FileExistsError:
1497
+ pass # a concurrent tick stamped it; its clock stands
1498
+ except OSError as exc:
1499
+ log.warning("kill-stamp write failed for %s: %s", record.run_id, exc)
1500
+ continue
1501
+ # An empty stamp (write failed after create — the disk-full
1502
+ # case — or a concurrent tick mid-write) must fall back to
1503
+ # mtime, NOT to epoch 0, which would skip the grace outright.
1504
+ try:
1505
+ raw = stamp.read_text().strip()
1506
+ seen = float(raw) if raw else stamp.stat().st_mtime
1507
+ except (OSError, ValueError):
1508
+ try:
1509
+ seen = stamp.stat().st_mtime
1510
+ except OSError:
1511
+ continue # stamp vanished mid-read; next tick decides
1512
+ if now - seen < grace_s:
1513
+ continue
1514
+ note = f"climb job {record.run_job_id} ended {state} without a verdict"
1515
+ else:
1516
+ # No Slurm evidence at all (legacy record, or a manual dev
1517
+ # invocation without SLURM_JOB_ID): only the run DEADLINE —
1518
+ # past which nothing legitimately lives — justifies a
1519
+ # terminal verdict; the shorter stranded window merely
1520
+ # frees the picker lane and must not author endings.
1521
+ deadline = record.deadline if record.deadline > 0 else (record.created + 24 * 3600)
1522
+ if now < deadline:
1523
+ continue
1524
+ note = "implementing with no recorded climb job, past its run deadline"
1525
+ fresh = load_record(root, record.run_id)
1526
+ if fresh.state != IMPLEMENTING:
1527
+ continue # the climb landed its own ending meanwhile
1528
+ for jid in _poll_targets(fresh):
1529
+ # defensive: no current path records an experiment while
1530
+ # still implementing, but an orphan GPU job burning budget
1531
+ # after its run is declared dead must never survive one
1532
+ with contextlib.suppress(Exception):
1533
+ compute.cancel(jid)
1534
+ save_record(
1535
+ root,
1536
+ replace(
1537
+ fresh,
1538
+ state=ENDED,
1539
+ ending=ABORTED,
1540
+ ending_note=(
1541
+ f"{note} — ended by the sweep (a killed climb "
1542
+ f"leaves no exception to contain)"
1543
+ ),
1544
+ ),
1545
+ now,
1546
+ )
1547
+ # every ending produces a report — but never clobber one the
1548
+ # climb already wrote before it was killed
1549
+ report_path = run_dir(root, record.run_id) / "report.md"
1550
+ if not report_path.exists():
1551
+ try:
1552
+ report_path.write_text(
1553
+ f"# Run report — {record.target} / {record.benchmark}\n"
1554
+ f"Outcome: **aborted** (climb job killed)\n"
1555
+ f"Note: {note}\n"
1556
+ )
1557
+ except OSError as exc:
1558
+ log.warning("sweep report write failed for %s: %s", record.run_id, exc)
1559
+ log.warning("sweep ended implementing run %s: %s", record.run_id, note)
1560
+ ended.append(record.run_id)
1561
+ except Exception as exc: # per-record isolation, like the waiting sweep
1562
+ log.warning("implementing-sweep failed on %s: %s", record.run_id, exc)
1563
+ return ended
1564
+
1565
+
1566
+ def _moved_off_partition(root: Path, compute: Compute, job_id: str) -> tuple[str, str] | None:
1567
+ """(actual, wanted) when a queued kernel job no longer sits in the
1568
+ partition the wake recipe asks for, else None. Unknown either way —
1569
+ no recipe, a compute without partitions, a failed query — is None:
1570
+ never cancel on doubt."""
1571
+ spec = load_wake_spec(root)
1572
+ if spec is None:
1573
+ return None
1574
+ wanted = spec.job_partition or spec.partition
1575
+ if not wanted:
1576
+ return None
1577
+ try:
1578
+ actual = compute.job_partition(job_id)
1579
+ except (SlurmQueryError, ValueError, AttributeError):
1580
+ return None
1581
+ # both sides may be comma-separated lists: moved only when the job holds
1582
+ # NONE of the partitions the recipe asked for; empty is unknown, not moved
1583
+ have = {p.strip() for p in actual.split(",") if p.strip()}
1584
+ want = {p.strip() for p in wanted.split(",") if p.strip()}
1585
+ if not have or have & want:
1586
+ return None
1587
+ return actual, wanted
1588
+
1589
+
1590
+ def _poll_targets(record: RunRecord) -> list[str]:
1591
+ """Every Slurm job this run waits on. The single `experiment_job_id` is
1592
+ the common case; a MULTI-job park (candidate + siblings, or several author
1593
+ launches) records no single id — its jobs live in the stage's `afterany`
1594
+ dependency string, the one source that always names them all. Without
1595
+ this fallback a multi-job park is blind and rides the deadline floor."""
1596
+ if record.experiment_job_id:
1597
+ return [record.experiment_job_id]
1598
+ afterany = str((record.stage or {}).get("afterany", ""))
1599
+ return [t for t in afterany.split(":")[1:] if t]
1600
+
1601
+
1602
+ def _sweep_one(
1603
+ root: Path,
1604
+ compute: Compute,
1605
+ dispatcher: WakeDispatcher,
1606
+ now: float,
1607
+ grace_s: float,
1608
+ lease_ttl_s: float,
1609
+ dry_run: bool,
1610
+ record: RunRecord,
1611
+ holder: str,
1612
+ wake,
1613
+ deferred: list[str],
1614
+ reaped: list[str],
1615
+ stuck: list[str],
1616
+ ) -> None:
1617
+ # Leases first: a LIVE wake in flight owns this run — even the stuck
1618
+ # verdict must wait for it (its session may be the one that succeeds).
1619
+ lease = read_lease(root, record.run_id)
1620
+ if lease is not None:
1621
+ alive = _holder_alive(compute, lease.holder_job_id)
1622
+ if not lease_is_stale(lease, now, lease_ttl_s, alive):
1623
+ if not _armed_wake_lost(root, compute, record, lease, now, grace_s, dry_run):
1624
+ return
1625
+ record = load_record(root, record.run_id)
1626
+ if dry_run:
1627
+ reaped.append(record.run_id)
1628
+ return
1629
+ if not reap_lease(root, record.run_id, reaper=f"{os.getpid()}-{now}", expected=lease):
1630
+ return # a concurrent tick reaped it first; it owns redelivery
1631
+ reaped.append(record.run_id)
1632
+
1633
+ # Layer 5: too many failed attempts is a terminal, reported state.
1634
+ if record.wake_attempts >= MAX_WAKE_ATTEMPTS:
1635
+ if not dry_run:
1636
+ ended = replace(
1637
+ record,
1638
+ state=ENDED,
1639
+ ending=STUCK,
1640
+ ending_note=(
1641
+ f"{record.wake_attempts} wake attempts without the run leaving 'waiting'"
1642
+ ),
1643
+ )
1644
+ save_record(root, ended, now)
1645
+ stuck.append(record.run_id)
1646
+ return
1647
+
1648
+ job_ids = _poll_targets(record)
1649
+ if not job_ids:
1650
+ # No job ids to poll. A BLIND PARK (the measurer could not read Slurm,
1651
+ # so `MeasurementPending` carried no ids) still hibernated with a
1652
+ # deadline — the deadline floor is its ONLY wake, so fire on it. A
1653
+ # genuinely mid-write record has no deadline and is left alone.
1654
+ # (A jobless CHECKPOINT SLEEP arrives here too — its deadline is
1655
+ # near-term by construction, attempt.py sizes it to the next sweep
1656
+ # pass, not the 12h queue slack that protects queued jobs.)
1657
+ if record.deadline > 0 and now > record.deadline:
1658
+ wake(record, "blind park past deadline", "deadline")
1659
+ return
1660
+
1661
+ try:
1662
+ states = [compute.status(jid) for jid in job_ids]
1663
+ except SlurmQueryError:
1664
+ # Layer 4's rule: query failure is "Slurm unknown", never "gone".
1665
+ deferred.append(record.run_id)
1666
+ return
1667
+
1668
+ # deadline <= 0 cannot be written by save_record for waiting runs;
1669
+ # if one exists anyway (legacy/hand-edited), treat it as already past
1670
+ # for GONE — a vanished-experiment wake is safe — but never for
1671
+ # PENDING, where the consequence would be cancelling a healthy job.
1672
+ past_deadline = record.deadline <= 0 or now > record.deadline
1673
+
1674
+ if all(is_terminal(s) for s in states):
1675
+ state = ",".join(sorted(set(states)))
1676
+ # Layer 3, with real grace: time runs from when the sweep FIRST
1677
+ # saw every job terminal, not from submission — the afterany
1678
+ # job gets the full window to deliver before the backup steps in.
1679
+ # Local compute has no afterany jobs to wait for (jobs are
1680
+ # terminal at submit): the sweep IS the delivery, so grace would
1681
+ # only cost a whole extra loop iteration.
1682
+ if local_mode():
1683
+ wake(record, f"experiment {state}", state)
1684
+ return
1685
+ if record.terminal_seen <= 0:
1686
+ if dry_run:
1687
+ # no writes in dry-run: report the would-wake now so the
1688
+ # terminal path is visible to live plumbing checks
1689
+ wake(record, f"experiment {state}", state)
1690
+ else:
1691
+ save_record(
1692
+ root,
1693
+ replace(
1694
+ record,
1695
+ terminal_seen=now,
1696
+ # repair legacy records as we touch them (see _wake)
1697
+ deadline=record.deadline if record.deadline > 0 else now,
1698
+ ),
1699
+ now,
1700
+ )
1701
+ return
1702
+ if now - record.terminal_seen >= grace_s:
1703
+ wake(record, f"experiment {state}", state)
1704
+ elif any(is_pending(s) for s in states) and record.deadline > 0 and now > record.deadline:
1705
+ # Past the deadline with a job still queued. Ask Slurm WHY before
1706
+ # calling it unschedulable: a busy queue or the account's own cap is
1707
+ # a wait a scientist would sit out, so the deadline moves out by one
1708
+ # slack window instead (Torch 2026-09-06: four launches pending on
1709
+ # QOSMaxGRESPerUser were about to be cancelled and re-launched into
1710
+ # the same cap). Local compute has no queue and no reasons.
1711
+ pending = [j for j, s in zip(job_ids, states, strict=True) if is_pending(s)]
1712
+ reason_of = getattr(compute, "pending_reason", None)
1713
+ reasons: dict[str, str] = {}
1714
+ if reason_of is not None:
1715
+ try:
1716
+ reasons = {j: str(reason_of(j)) for j in pending}
1717
+ except SlurmQueryError:
1718
+ deferred.append(record.run_id) # unknown is never "cancel"
1719
+ return
1720
+ if reasons and all(is_queue_wait(r) for r in reasons.values()):
1721
+ extended = now + BLIND_PARK_SLACK_MIN * 60
1722
+ if not dry_run:
1723
+ save_record(root, replace(record, deadline=extended), now)
1724
+ log.info(
1725
+ "sweep: %s waits in the queue (%s); deadline extended by %d min",
1726
+ record.run_id,
1727
+ ", ".join(f"{j}={r}" for j, r in reasons.items()),
1728
+ BLIND_PARK_SLACK_MIN,
1729
+ )
1730
+ return
1731
+ # Unschedulable in practice: cancel every non-terminal job
1732
+ # (best-effort — scancel trouble must not abort the sweep), then
1733
+ # wake with that fact.
1734
+ if not dry_run:
1735
+ for jid, s in zip(job_ids, states, strict=True):
1736
+ if is_terminal(s) or s == GONE:
1737
+ continue
1738
+ try:
1739
+ compute.cancel(jid)
1740
+ except Exception as exc: # scancel trouble is never fatal here
1741
+ log.warning("cancel %s failed: %s", jid, exc)
1742
+ wake(record, "experiment unschedulable (pending past deadline)", "unschedulable")
1743
+ elif all(is_terminal(s) or s == GONE for s in states):
1744
+ # done-or-vanished, at least one GONE (all-terminal handled above)
1745
+ if past_deadline:
1746
+ wake(record, "experiment vanished from Slurm", "vanished")
1747
+ # else: sacct lag right after submission is normal; wait.
1748
+ # something RUNNING (or recently pending): nothing to do yet.
1749
+
1750
+
1751
+ def tick(
1752
+ root: Path,
1753
+ compute: Compute,
1754
+ dispatcher: WakeDispatcher,
1755
+ now: float,
1756
+ grace_s: float = DEFAULT_GRACE_S,
1757
+ lease_ttl_s: float = DEFAULT_LEASE_TTL_S,
1758
+ dry_run: bool = False,
1759
+ github: Any = None,
1760
+ followup_spec: FollowupSpec | None = None,
1761
+ followup_dry_run: bool = False,
1762
+ min_free_bytes: int = DEFAULT_MIN_FREE_BYTES,
1763
+ min_tick_s: float = DEFAULT_MIN_TICK_S,
1764
+ ) -> TickReport:
1765
+ """One full tick. Pause sentinel wins over everything: a paused loop
1766
+ heartbeats (so the watchdog stays quiet) but touches nothing.
1767
+
1768
+ Disk preflight gates every lane that LAUNCHES new work (follow-up jobs,
1769
+ intake claims, self-initiated climbs): a session started on a full or
1770
+ nearly-full filesystem dies mid-flight in ways that lose data. The sweep
1771
+ still runs — its writes are small, per-record contained, and ending runs
1772
+ matters more when storage is failing, not less.
1773
+ """
1774
+ # Heartbeat FIRST, before any probe: check_disk touches $HOME (a
1775
+ # different filesystem), and a hung mount there must not starve the
1776
+ # watchdog signal. The disk-annotated heartbeat follows once known. The
1777
+ # coalesce guard reads the last COMPLETED tick's marker (not the heartbeat).
1778
+ prior_worked = _last_worked_ts(root)
1779
+ write_heartbeat(root, now)
1780
+ disk_health = check_disk(root, min_free_bytes=min_free_bytes)
1781
+ write_heartbeat(root, now, disk=disk_health.as_dict())
1782
+ for warning in disk_health.warnings():
1783
+ log.warning("disk: %s", warning)
1784
+ if (root / PAUSE_SENTINEL).exists():
1785
+ log.info("pause sentinel present; tick is a no-op")
1786
+ return TickReport(paused=True)
1787
+ # Coalesce a congestion pile-up: if a tick COMPLETED its work within
1788
+ # min_tick_s, this one is redundant (that recent tick already swept and
1789
+ # launched). Keyed on the work marker (stamped by the CALLER at real
1790
+ # completion time), not the heartbeat, so a tick that crashed mid-work does
1791
+ # not suppress this recovery tick. Heartbeat still written above, so the
1792
+ # watchdog stays fed and the chain stays alive.
1793
+ if min_tick_s > 0 and prior_worked is not None:
1794
+ elapsed = now - prior_worked
1795
+ if elapsed < 0:
1796
+ # marker dated in the future -> the clock jumped back; never coalesce
1797
+ # on it (that could stall the loop), and surface it rather than fail
1798
+ # silently.
1799
+ log.warning(
1800
+ "work marker is %.0fs in the future (clock skew?); not coalescing", -elapsed
1801
+ )
1802
+ elif elapsed < min_tick_s:
1803
+ log.info(
1804
+ "coalescing: last completed tick %.0fs ago (< %.0fs); tick is a no-op",
1805
+ elapsed,
1806
+ min_tick_s,
1807
+ )
1808
+ return TickReport(coalesced=True)
1809
+ report = sweep(root, compute, dispatcher, now, grace_s, lease_ttl_s, dry_run=dry_run)
1810
+ # Housekeeping: ended runs shed ws/ and ws-home/ after a grace period;
1811
+ # when the state filesystem's write probe failed, the grace is waived and
1812
+ # the sweep frees oldest-first until the probe passes, then the preflight
1813
+ # is taken again so launch lanes can come back this very tick.
1814
+ # Force-shed (waive the grace) only when the state root cannot be WRITTEN,
1815
+ # not merely when it is below the free-space threshold: a writable disk
1816
+ # that is just low keeps the 24 h grace so a post-mortem is not deleted
1817
+ # under someone. A dry-run tick sheds nothing (the destructive lane obeys
1818
+ # the zero-writes contract, like sweep()).
1819
+ shed: list[str] = []
1820
+ if not dry_run:
1821
+ cannot_write = not disk_health.state_root.writable
1822
+ # Bounded per tick so shedding (rm -rf of tens of thousands of files
1823
+ # per workspace on a networked FS) never blows the tick's own timeout:
1824
+ # a few runs on a healthy disk, more but still time-boxed when it is
1825
+ # failing. The backlog drains over several ticks.
1826
+ shed = shed_ended_workspaces(
1827
+ root,
1828
+ now,
1829
+ force=cannot_write,
1830
+ limit=25 if cannot_write else 3,
1831
+ time_budget_s=300.0 if cannot_write else 120.0,
1832
+ until_ok=(lambda: check_disk(root, min_free_bytes=min_free_bytes).state_root.writable)
1833
+ if cannot_write
1834
+ else None,
1835
+ )
1836
+ if shed and cannot_write:
1837
+ disk_health = check_disk(root, min_free_bytes=min_free_bytes)
1838
+ write_heartbeat(root, now, disk=disk_health.as_dict())
1839
+ if shed:
1840
+ from dataclasses import replace as _dc_replace
1841
+
1842
+ report = _dc_replace(report, shed=tuple(shed))
1843
+ launch_ok = disk_health.launch_ok()
1844
+ if not launch_ok:
1845
+ log.warning("disk preflight failed; launch lanes are OFF this tick")
1846
+ # Mid-leg sync is serviced regardless of follow-up/board servicing: it
1847
+ # only needs the workspace and the PAT (a git fetch, no GitHub REST and
1848
+ # no contract), and a live session waiting on `sync` must not depend on
1849
+ # whether github/contract loaded this tick.
1850
+ if followup_spec is not None:
1851
+ # ONE run-record snapshot for every READ-heavy service that follows the
1852
+ # mutation phase (sweep/park/reap and housekeeping have already run and
1853
+ # written what they will). Sharing it means walking runs/ once, not once
1854
+ # per service. The launch lanes read a stale-but-conservative picture on
1855
+ # purpose: the only record-writer between here and them is
1856
+ # service_in_review, and it only ENDS runs (frees slots), so the
1857
+ # snapshot can only OVER-count active runs — never launch a duplicate.
1858
+ # The one exception is service_research_log, which PUBLISHES terminal
1859
+ # outcomes: it reads fresh (below), never the snapshot. Any service
1860
+ # that acts on a single record's CURRENT state re-reads that one
1861
+ # record fresh via load_record (the freshness guard).
1862
+ tick_records = list_runs(root)
1863
+ service_syncs(root, followup_spec, now, tick_records)
1864
+ # a dry run reports and writes nothing: no download, no seed
1865
+ if github is not None and followup_spec is not None and followup_spec.target and not dry_run:
1866
+ try:
1867
+ service_eval_cache(root, github, followup_spec.target)
1868
+ except Exception as exc: # advisory: a cold cache costs a download, never a tick
1869
+ log.warning("eval cache warm failed: %s: %s", type(exc).__name__, exc)
1870
+ if github is not None and followup_spec is not None:
1871
+ # expired flight snapshots die with their TTL, not with a human.
1872
+ # One home suffices: every lane's spec derives from followup_spec
1873
+ # via replace(), so all flights share this checkout's flights/ dir.
1874
+ # Blind means delete nothing — but only QUERY failures count as
1875
+ # blindness; a compute backend missing the method is a programming
1876
+ # error and propagates.
1877
+ try:
1878
+ live_names = compute.active_job_names()
1879
+ except SlurmQueryError as exc:
1880
+ log.warning("cannot list live jobs (%s); reaping no flights this tick", exc)
1881
+ live_names = None
1882
+ if live_names is not None:
1883
+ with contextlib.suppress(Exception):
1884
+ reaped = reap_flights(followup_spec.home, now, live_job_names=live_names)
1885
+ if reaped:
1886
+ log.info("reaped %d expired flight snapshot(s)", reaped)
1887
+ # ONE contract fetch per tick feeds every lane: the requested and
1888
+ # self-initiated lanes need its benchmarks, and all three lanes now
1889
+ # take their session/job limits from its budgets — clamped by our
1890
+ # ceilings (limits.py), so a target shapes spend, never raises it.
1891
+ # A failed fetch leaves in-review servicing running on defaults;
1892
+ # the launch lanes need the contract and sit out this tick.
1893
+ contract = None
1894
+ if followup_spec.target:
1895
+ contract_error: str | None = "contract file missing on main"
1896
+ try:
1897
+ from outerloop.contract import load_contract
1898
+
1899
+ raw = _contract_text(github, followup_spec.target, "main")
1900
+ if raw is not None:
1901
+ contract = load_contract(raw, followup_spec.target)
1902
+ contract_error = None
1903
+ except Exception as exc:
1904
+ log.warning("contract fetch failed for %s: %s", followup_spec.target, exc)
1905
+ contract_error = f"{type(exc).__name__}: {exc}"
1906
+ if contract_error is None:
1907
+ # a bad panel config idles the same launch lanes a bad
1908
+ # contract does — same silent-idle class, so it rides the
1909
+ # same alarm
1910
+ panel_error = _panel_preflight_error(followup_spec)
1911
+ if panel_error:
1912
+ contract_error = f"panel preflight: {panel_error}"
1913
+ try:
1914
+ contract_alarm(
1915
+ root,
1916
+ github,
1917
+ followup_spec.target,
1918
+ contract_error,
1919
+ now,
1920
+ bot_login=followup_spec.bot_login,
1921
+ )
1922
+ except Exception as exc:
1923
+ log.warning("contract alarm failed: %s", exc)
1924
+ limits = effective_limits(contract.budgets if contract is not None else None)
1925
+ # The contract's followup walltime only overrides when EXPLICITLY
1926
+ # set — and only DOWNWARD from the operator's spec value: strictly-
1927
+ # downward shaping must hold against operator config too, not just
1928
+ # against the module defaults.
1929
+ spec = shape_followup_spec(followup_spec, limits, contract)
1930
+ ended, submitted = service_in_review(
1931
+ root,
1932
+ github,
1933
+ compute,
1934
+ spec,
1935
+ now,
1936
+ dry_run=followup_dry_run,
1937
+ allow_submit=launch_ok,
1938
+ contract=contract,
1939
+ records=tick_records,
1940
+ )
1941
+ try:
1942
+ # research_log reads FRESH, not the shared snapshot: it publishes
1943
+ # terminal outcomes and writes done markers, and service_in_review
1944
+ # just above may have ended a run this tick — a stale in-review
1945
+ # record would be published as 'improved' and locked, wrong.
1946
+ service_research_log(root, github, spec, now)
1947
+ except Exception as exc: # the ledger is advisory; the tick continues
1948
+ log.warning("research-log service failed: %s", exc)
1949
+ intake_job = (
1950
+ service_intake(
1951
+ root, github, compute, spec, now, contract, limits, dry_run=followup_dry_run
1952
+ )
1953
+ if launch_ok and contract is not None
1954
+ else None
1955
+ )
1956
+ steward_job = (
1957
+ service_steward(
1958
+ root,
1959
+ github,
1960
+ compute,
1961
+ spec,
1962
+ now,
1963
+ contract,
1964
+ limits,
1965
+ dry_run=followup_dry_run,
1966
+ records=tick_records,
1967
+ )
1968
+ if launch_ok and intake_job is None and contract is not None
1969
+ else None
1970
+ )
1971
+ self_job = None
1972
+ if launch_ok and intake_job is None and steward_job is None and contract is not None:
1973
+ try:
1974
+ self_job = service_self_initiated(
1975
+ root,
1976
+ compute,
1977
+ spec,
1978
+ contract,
1979
+ now,
1980
+ limits=limits,
1981
+ dry_run=followup_dry_run,
1982
+ records=tick_records,
1983
+ )
1984
+ except Exception as exc:
1985
+ log.warning("self-initiated selection failed: %s", exc)
1986
+ # AFTER the launch block: a run started this tick is on the strip
1987
+ # this tick, not the next one
1988
+ service_boards(root, github, spec.target, contract, now, compute)
1989
+ report = replace_report(
1990
+ report,
1991
+ ended,
1992
+ submitted,
1993
+ intake_job,
1994
+ self_job,
1995
+ disk_health.warnings(),
1996
+ not launch_ok,
1997
+ steward_job,
1998
+ )
1999
+ # The coalesce marker is stamped by the CALLER at real completion time (see
2000
+ # main / mark_tick_complete) — not here with the start-of-tick `now`, which
2001
+ # a tick longer than the window would leave stale.
2002
+ return report
2003
+
2004
+
2005
+ def service_eval_cache(root: Path, github: Any, target: str) -> str:
2006
+ """Warm the target's seed cache from its default branch's lockfile
2007
+ (docs/design/eval-cache.md). Cheap when nothing changed: two file reads
2008
+ and a hash. Returns the warmer's one-word status."""
2009
+ from outerloop.evalcache import warm
2010
+
2011
+ status = warm(root, target, github, github.default_branch(target))
2012
+ if status != "unchanged":
2013
+ log.info("eval cache for %s: %s", target, status)
2014
+ return status
2015
+
2016
+
2017
+ def service_syncs(
2018
+ root: Path, spec: Any, now: float, records: list[RunRecord] | None = None
2019
+ ) -> None:
2020
+ """Honor mid-leg sync requests: a LIVE session asked for fresh origin/*
2021
+ refs and is waiting inside its own clock. The fetch pins the canonical
2022
+ URL (never the workspace's mutable remote config) and only refs/remotes
2023
+ are written — safe next to the session's local git use. Best-effort per
2024
+ run; a failure leaves the request standing for the next cycle."""
2025
+ from outerloop.appauth import resolve_bot_auth
2026
+ from outerloop.attempt import target_clone_url
2027
+ from outerloop.github import Workspace
2028
+ from outerloop.syscall import mark_synced, sync_requested
2029
+
2030
+ if records is None:
2031
+ records = list_runs(root)
2032
+ for record in records:
2033
+ if record.state != IMPLEMENTING:
2034
+ continue
2035
+ workspace = run_dir(root, record.run_id) / "ws"
2036
+ if not workspace.is_dir():
2037
+ continue
2038
+ requested_at = sync_requested(workspace)
2039
+ if requested_at is None:
2040
+ continue
2041
+ try:
2042
+ ws = Workspace(
2043
+ root=workspace,
2044
+ auth=(
2045
+ resolve_bot_auth(spec.pat_file, spec.github_app_file)
2046
+ if (spec.pat_file or spec.github_app_file)
2047
+ else None
2048
+ ),
2049
+ url=target_clone_url(record.target),
2050
+ )
2051
+ ws.fetch_origin()
2052
+ mark_synced(workspace, requested_at)
2053
+ log.info("synced origin refs for %s", record.run_id)
2054
+ except Exception as exc:
2055
+ log.warning("sync failed for %s: %s", record.run_id, exc)
2056
+
2057
+
2058
+ def service_boards(
2059
+ root: Path, github: Any, target: str, contract: Any, now: float, compute: Any = None
2060
+ ) -> None:
2061
+ """The climb board and the live strip, together and advisory: the views
2062
+ publish from the first tick (before any run ends), and a failure never
2063
+ stops the tick. With a compute backend, the strip also carries the
2064
+ kernel's queue (its own Slurm jobs, attributed to agents)."""
2065
+ # ONE snapshot for the whole board pass: the climb rows, the live strip,
2066
+ # and the queue's owner maps all read the SAME records, so they share it
2067
+ # rather than each re-walking runs/. Taken here (end of tick) so a run
2068
+ # ended earlier this tick already shows.
2069
+ records = list_runs(root)
2070
+ try:
2071
+ from outerloop.climbboard import contract_directions, service_climb_board
2072
+
2073
+ service_climb_board(root, github, target, contract_directions(contract), records)
2074
+ except Exception as exc:
2075
+ log.warning("climb board service failed: %s", exc)
2076
+ try:
2077
+ from outerloop.climbboard import service_status
2078
+
2079
+ queue = None
2080
+ if compute is not None:
2081
+ try:
2082
+ queue = compute.queue_snapshot()
2083
+ except Exception as exc: # blind this tick: the strip publishes without a queue
2084
+ log.warning("queue snapshot failed: %s", exc)
2085
+ service_status(root, github, target, now, contract, queue=queue, records=records)
2086
+ except Exception as exc: # each is advisory ALONE: one failing never mutes the other
2087
+ log.warning("status strip service failed: %s", exc)
2088
+
2089
+
2090
+ def replace_report(
2091
+ report: TickReport,
2092
+ ended: list[tuple[str, str]],
2093
+ submitted: list[tuple[str, str]],
2094
+ intake_job: tuple[str, str] | None = None,
2095
+ self_job: tuple[str, str] | None = None,
2096
+ disk_warnings: list[str] | None = None,
2097
+ launch_blocked: bool = False,
2098
+ steward_job: tuple[str, str] | None = None,
2099
+ ) -> TickReport:
2100
+ from dataclasses import replace as dc_replace
2101
+
2102
+ return dc_replace(
2103
+ report,
2104
+ review_ended=tuple(ended),
2105
+ followups_submitted=tuple(submitted),
2106
+ intake=intake_job or ("", ""),
2107
+ self_initiated=self_job or ("", ""),
2108
+ disk=tuple(disk_warnings or ()),
2109
+ launch_blocked=launch_blocked,
2110
+ steward=steward_job or ("", ""),
2111
+ )
2112
+
2113
+
2114
+ MAX_ACTIVE_RUNS_PER_TARGET = 1
2115
+ SELF_INITIATED_COOLDOWN_S = 6 * 3600
2116
+ # the crash-loop floor: a launch that died pre-record backs off at least
2117
+ # this long regardless of the contract's cooldown dial
2118
+ DEAD_LAUNCH_BACKOFF_S = 30 * 60
2119
+ # An implementing run untouched for this long is a crashed climb job; it must
2120
+ # not block the lane forever, but the window must exceed the LONGEST honest
2121
+ # job — the 120-min contract ceiling plus the panel allowance the tick adds
2122
+ # (~4.5 h at the defaults) plus queue-start slack — or the picker declares a
2123
+ # live run stranded and starts a second one on the same target, breaking the
2124
+ # one-active-run serialization. Its cooldown entry still applies, so a
2125
+ # crashed benchmark isn't immediately retried.
2126
+ STRANDED_IMPLEMENTING_S = 12 * 3600
2127
+ # A pending marker older than this is dead even if squeue can't be read.
2128
+ PENDING_TTL_S = 4 * 3600
2129
+
2130
+
2131
+ def pick_self_initiated(
2132
+ records: list[RunRecord],
2133
+ contract: Any,
2134
+ target: str,
2135
+ now: float,
2136
+ dead_attempts: dict[str, float] | None = None,
2137
+ live_pendings: list[tuple[str, float]] | None = None,
2138
+ ) -> str | None:
2139
+ """The benchmark to climb next on `target`, or None.
2140
+
2141
+ Deliberately boring (the planning agent upgrades this later): serialize
2142
+ to one active run per target, respect the contract's weekly budget and a
2143
+ per-benchmark cooldown, then choose the benchmark least recently
2144
+ attempted — untouched ones first. Only this target's runs count toward
2145
+ any of it. `dead_attempts` maps benchmark -> submitted_at for launches
2146
+ that died BEFORE writing a run record (per-benchmark tombstones) — each
2147
+ counts toward cooldown with a crash-loop floor, so alternating
2148
+ pre-record failures can't ping-pong every tick (terra #172 r2/r3).
2149
+ """
2150
+ mine = [r for r in records if r.target == target]
2151
+
2152
+ def stranded(r: RunRecord) -> bool:
2153
+ return r.state == IMPLEMENTING and now - max(r.updated, r.created) > STRANDED_IMPLEMENTING_S
2154
+
2155
+ active = [r for r in mine if r.state != ENDED and not stranded(r)]
2156
+ if len(active) >= _attempt_width(contract):
2157
+ return None
2158
+ week_ago = now - 7 * 24 * 3600
2159
+ # queued slots count toward the weekly budget BEFORE their records
2160
+ # exist, or a width-N target with one run left could submit N (terra
2161
+ # #173 r2); when a marker lands, the service clears it before calling
2162
+ # here, so a run is never counted twice
2163
+ queued = sum(1 for _, submitted_at in live_pendings or [] if submitted_at >= week_ago)
2164
+ if sum(1 for r in mine if r.created >= week_ago) + queued >= contract.budgets.runs_per_week:
2165
+ return None
2166
+ last_attempt: dict[str, float] = {}
2167
+ for r in mine:
2168
+ if r.benchmark:
2169
+ last_attempt[r.benchmark] = max(last_attempt.get(r.benchmark, 0.0), r.created)
2170
+ for bench_name, submitted_at in (dead_attempts or {}).items():
2171
+ last_attempt[bench_name] = max(last_attempt.get(bench_name, 0.0), submitted_at)
2172
+ for bench_name, submitted_at in live_pendings or []:
2173
+ # a queued sibling starts its benchmark's cooldown clock too:
2174
+ # width spreads across benchmarks first, and re-picking the same
2175
+ # one needs the contract to have set cooldown to 0 (portfolio)
2176
+ if bench_name:
2177
+ last_attempt[bench_name] = max(last_attempt.get(bench_name, 0.0), submitted_at)
2178
+ cooldown_min = getattr(contract.budgets, "attempt_cooldown_minutes", None)
2179
+ cooldown_s = SELF_INITIATED_COOLDOWN_S if cooldown_min is None else cooldown_min * 60
2180
+ # A launch that died BEFORE writing a run record is invisible to the
2181
+ # runs_per_week cap, so its cooldown attribution is the ONLY crash-loop
2182
+ # guard — it keeps a floor even when the contract dials cooldown to 0
2183
+ # (terra #172: zero cooldown otherwise resubmits a crashing launch
2184
+ # every tick, uncapped).
2185
+ dead_benches = set(dead_attempts or ())
2186
+ candidates = sorted(
2187
+ contract.benchmarks,
2188
+ key=lambda b: (last_attempt.get(b.name, 0.0), b.name),
2189
+ )
2190
+ for bench in candidates:
2191
+ floor_s = cooldown_s
2192
+ if bench.name in dead_benches:
2193
+ floor_s = max(cooldown_s, DEAD_LAUNCH_BACKOFF_S)
2194
+ if now - last_attempt.get(bench.name, 0.0) >= floor_s:
2195
+ return str(bench.name)
2196
+ return None
2197
+
2198
+
2199
+ def _tombstone_path(root: Path, target: str, benchmark: str) -> Path:
2200
+ safe = f"{target.replace('/', '__')}__{benchmark}"
2201
+ return root / "pending-dead" / (safe + ".json")
2202
+
2203
+
2204
+ def write_tombstone(root: Path, target: str, benchmark: str, submitted_at: float) -> None:
2205
+ """Per-benchmark crash memory: a launch died before writing a record, so
2206
+ nothing else (runs_per_week, cooldown-by-records) can see it. The
2207
+ tombstone persists independently of the live pending marker — a second
2208
+ benchmark's launch must not erase it (terra #172 r3)."""
2209
+ path = _tombstone_path(root, target, benchmark)
2210
+ try:
2211
+ path.parent.mkdir(parents=True, exist_ok=True)
2212
+ path.write_text(json.dumps({"submitted_at": submitted_at}))
2213
+ except OSError as exc:
2214
+ log.warning("tombstone write failed for %s/%s: %s", target, benchmark, exc)
2215
+
2216
+
2217
+ def read_tombstones(root: Path, target: str, contract: Any, now: float) -> dict[str, float]:
2218
+ """benchmark -> submitted_at for unserved crash tombstones; entries past
2219
+ their window (the larger of the crash floor and the contract cooldown)
2220
+ are pruned on read."""
2221
+ cooldown_min = getattr(contract.budgets, "attempt_cooldown_minutes", None)
2222
+ cooldown_s = SELF_INITIATED_COOLDOWN_S if cooldown_min is None else cooldown_min * 60
2223
+ window = max(DEAD_LAUNCH_BACKOFF_S, cooldown_s)
2224
+ out: dict[str, float] = {}
2225
+ prefix = target.replace("/", "__") + "__"
2226
+ dead_dir = root / "pending-dead"
2227
+ if not dead_dir.is_dir():
2228
+ return out
2229
+ for path in dead_dir.glob(prefix + "*.json"):
2230
+ bench = path.stem[len(prefix) :]
2231
+ try:
2232
+ submitted_at = float(json.loads(path.read_text())["submitted_at"])
2233
+ except (OSError, ValueError, KeyError, TypeError):
2234
+ with contextlib.suppress(OSError):
2235
+ path.unlink()
2236
+ continue
2237
+ if now - submitted_at > window:
2238
+ with contextlib.suppress(OSError):
2239
+ path.unlink() # backoff served
2240
+ continue
2241
+ out[bench] = submitted_at
2242
+ return out
2243
+
2244
+
2245
+ # WIDTH slots are the only suffixed marker names; the pattern also fences
2246
+ # list_pendings against a longer target that shares this one's file-name
2247
+ # prefix (org/foo vs org/foobar — "/" encodes as "__", so glob alone is
2248
+ # ambiguous)
2249
+ _SLOT_AGENT_RE = re.compile(r"agent-\d+")
2250
+
2251
+
2252
+ def _pending_path(root: Path, target: str, agent: str = "") -> Path:
2253
+ # agent "" is the legacy single-slot name, still read for back-compat
2254
+ # with a marker written before the width dial deployed. The slot
2255
+ # separator is "@" because it CANNOT appear in a GitHub owner/repo
2256
+ # name — any character legal in repo names ("_", ".", "-") would make
2257
+ # org/pilot's slot file collide with some other target's legacy file
2258
+ # (org/pilot__agent-01 is a valid repo).
2259
+ suffix = f"@{agent}" if agent else ""
2260
+ return root / "pending" / (target.replace("/", "__") + suffix + ".json")
2261
+
2262
+
2263
+ def list_pendings(root: Path, target: str) -> list[tuple[str, dict[str, Any]]]:
2264
+ """(agent, marker) for every live pending marker of `target` — one per
2265
+ WIDTH slot, plus the legacy un-suffixed marker from a pre-width deploy
2266
+ (attributed to agent-01)."""
2267
+ out: list[tuple[str, dict[str, Any]]] = []
2268
+ stem = target.replace("/", "__")
2269
+ pending_dir = root / "pending"
2270
+ if not pending_dir.is_dir():
2271
+ return out
2272
+ for path in sorted(pending_dir.glob(stem + "*.json")):
2273
+ name = path.stem
2274
+ if name == stem:
2275
+ agent = ""
2276
+ elif name.startswith(stem + "@") and _SLOT_AGENT_RE.fullmatch(name[len(stem) + 1 :]):
2277
+ agent = name[len(stem) + 1 :]
2278
+ else:
2279
+ continue # a longer target sharing this prefix (org/foo vs org/foobar)
2280
+ try:
2281
+ data = json.loads(path.read_text())
2282
+ except (OSError, ValueError):
2283
+ continue
2284
+ if isinstance(data, dict) and "submitted_at" in data:
2285
+ out.append((agent or str(data.get("agent_id") or "agent-01"), data))
2286
+ return out
2287
+
2288
+
2289
+ def write_pending(
2290
+ root: Path, target: str, benchmark: str, job_id: str, now: float, agent: str = ""
2291
+ ) -> None:
2292
+ path = _pending_path(root, target, agent)
2293
+ path.parent.mkdir(parents=True, exist_ok=True)
2294
+ tmp = path.with_suffix(".tmp")
2295
+ tmp.write_text(
2296
+ json.dumps(
2297
+ {"benchmark": benchmark, "job_id": job_id, "submitted_at": now, "agent_id": agent}
2298
+ )
2299
+ )
2300
+ os.replace(tmp, path)
2301
+
2302
+
2303
+ def clear_pending(root: Path, target: str, agent: str = "") -> None:
2304
+ _pending_path(root, target, agent).unlink(missing_ok=True)
2305
+
2306
+
2307
+ def _attempt_width(contract: Any) -> int:
2308
+ width = getattr(getattr(contract, "budgets", None), "max_active_attempts", None)
2309
+ return int(width) if width else MAX_ACTIVE_RUNS_PER_TARGET
2310
+
2311
+
2312
+ def _free_agent_slot(occupied: set[str], width: int) -> str | None:
2313
+ for i in range(1, width + 1):
2314
+ agent = f"agent-{i:02d}"
2315
+ if agent not in occupied:
2316
+ return agent
2317
+ return None
2318
+
2319
+
2320
+ def _climb_panel_argv(spec: FollowupSpec) -> list[str]:
2321
+ """Panel args for a climb job; empty when the operator disabled the panel."""
2322
+ if not spec.panel.strip():
2323
+ return []
2324
+ argv = ["--panel", spec.panel]
2325
+ if spec.panel_key_file:
2326
+ argv += ["--panel-key-file", spec.panel_key_file]
2327
+ return argv
2328
+
2329
+
2330
+ def _author_config_error(spec: FollowupSpec) -> str:
2331
+ """Why the config-driven author would die at the climb's startup ("" when it
2332
+ won't), checked on the tick host BEFORE a claim/submit so a codex misconfig
2333
+ (e.g. OUTERLOOP_AUTHOR_BACKEND=codex with no non-claude model) never
2334
+ strands a claimed intake issue. Reads the fleet author config from env — the
2335
+ same source the climb defaults from — and the image the tick already knows."""
2336
+ from outerloop.attempt import codex_author_config_error
2337
+
2338
+ backend = os.environ.get("OUTERLOOP_AUTHOR_BACKEND") or "claude"
2339
+ model = os.environ.get("OUTERLOOP_AUTHOR_MODEL") or "claude-opus-5"
2340
+ return codex_author_config_error(backend, model, spec.image)
2341
+
2342
+
2343
+ def _panel_preflight_error(spec: FollowupSpec) -> str:
2344
+ """Why the climb would die at startup on this panel config ("" when it
2345
+ won't): the lens spec, then the key file — each checked with the climb's
2346
+ OWN rules (parse_lenses for the grammar and claude-only backend;
2347
+ FileTokenProvider for exists/mode-600/non-empty), so preflight and climb
2348
+ cannot disagree.
2349
+
2350
+ Preflighted BEFORE claiming or submitting: the climb CLI fails loudly,
2351
+ but by then an intake issue is already claimed — and pick_issue never
2352
+ reclaims — so the strand must be caught on the tick host, which shares
2353
+ the home filesystem the climb will read."""
2354
+ if not spec.panel.strip():
2355
+ return ""
2356
+ try:
2357
+ from outerloop.attempt import PANEL_KEY_DEFAULT, resolve_author_key_file
2358
+ from outerloop.github import FileTokenProvider
2359
+ from outerloop.panel import parse_lenses
2360
+
2361
+ try:
2362
+ lenses = parse_lenses(spec.panel)
2363
+ except ValueError as exc:
2364
+ return str(exc)
2365
+ # non-claude (shelled) lenses: mirror the climb's rules exactly, per
2366
+ # backend — image required, the judge's OWN key (set + absolute +
2367
+ # neither the author's nor the claude panel key + readable), and for
2368
+ # hermes its pinned clone.
2369
+ shelled = {
2370
+ "codex": "OUTERLOOP_PANEL_CODEX_KEY_FILE",
2371
+ "hermes": "OUTERLOOP_PANEL_HERMES_KEY_FILE",
2372
+ }
2373
+ for lens_backend, key_env in shelled.items():
2374
+ if not any(backend == lens_backend for _, backend, _ in lenses):
2375
+ continue
2376
+ if not spec.image or not Path(spec.image).is_file():
2377
+ return (
2378
+ f"a {lens_backend} panel lens requires a real container image "
2379
+ f"(OUTERLOOP_IMAGE={spec.image!r})"
2380
+ )
2381
+ key_raw = os.environ.get(key_env, "").strip()
2382
+ if not key_raw:
2383
+ return (
2384
+ f"a {lens_backend} panel lens needs {key_env} "
2385
+ "(role separation: the judge's own key, never the author's)"
2386
+ )
2387
+ key_path = Path(key_raw).expanduser()
2388
+ if not key_path.is_absolute():
2389
+ return (
2390
+ f"{lens_backend} panel key path {key_path} is relative; only absolute paths fly"
2391
+ )
2392
+ author = Path(resolve_author_key_file("codex")).expanduser()
2393
+ if key_path.resolve() == author.resolve():
2394
+ return (
2395
+ f"{lens_backend} panel key file {key_path} is the codex author "
2396
+ "key (role separation: the judge needs its own key)"
2397
+ )
2398
+ claude_panel = Path(spec.panel_key_file or PANEL_KEY_DEFAULT).expanduser()
2399
+ if key_path.resolve() == claude_panel.resolve():
2400
+ return (
2401
+ f"{lens_backend} panel key file {key_path} is the claude panel "
2402
+ "key file (an anthropic key must never reach another "
2403
+ "provider's login)"
2404
+ )
2405
+ FileTokenProvider(key_path).token()
2406
+ if lens_backend == "hermes":
2407
+ repo = os.environ.get("REVIEW_HERMES_REPO", "").strip()
2408
+ # a REAL clone, not merely a directory: the harness executes
2409
+ # run_agent.py from it with the panel key, so an arbitrary or
2410
+ # empty path must fail here, never after a run is claimed
2411
+ if not repo or not (Path(repo).expanduser() / "run_agent.py").is_file():
2412
+ return (
2413
+ f"a hermes panel lens needs REVIEW_HERMES_REPO pointing at "
2414
+ f"the pinned clone (run_agent.py not found under {repo!r})"
2415
+ )
2416
+ from outerloop.role_runner import _HERMES_PROVIDERS
2417
+
2418
+ provider = os.environ.get("REVIEW_HERMES_PROVIDER", "").lower() or "openrouter"
2419
+ if provider not in _HERMES_PROVIDERS:
2420
+ return (
2421
+ f"unknown REVIEW_HERMES_PROVIDER {provider!r} "
2422
+ f"(have: {sorted(_HERMES_PROVIDERS)})"
2423
+ )
2424
+ if not any(backend == "claude" for _, backend, _ in lenses):
2425
+ return "" # codex-only panel: the claude key checks below don't apply
2426
+ path = Path(spec.panel_key_file or PANEL_KEY_DEFAULT).expanduser()
2427
+ if not path.is_absolute():
2428
+ # the climb runs from a flight directory, not the tick's cwd — a
2429
+ # relative path that resolves here could still miss there
2430
+ return f"panel key path {path} is relative; only absolute paths fly"
2431
+ # the AUTHOR key the climb will actually use resolves per the fleet backend
2432
+ # (claude vs codex keys coexist), config-driven like the climb itself — so
2433
+ # the role-separation check compares the panel key against the RIGHT author
2434
+ # key, and a codex run is never judged by a stray Claude key.
2435
+ fleet_backend = os.environ.get("OUTERLOOP_AUTHOR_BACKEND") or "claude"
2436
+ author = Path(resolve_author_key_file(fleet_backend))
2437
+ if not author.is_absolute():
2438
+ # same rule as the panel key: the climb resolves paths from a
2439
+ # flight directory, so a relative author path both misconfigures
2440
+ # the author AND defeats the role-separation comparison below
2441
+ return f"author key path {author} is relative; only absolute paths fly"
2442
+ if path.resolve() == author.resolve():
2443
+ return (
2444
+ f"panel key file {path} is the author key file "
2445
+ "(role separation: the verifier needs its own key)"
2446
+ )
2447
+ # ADC-only deployments (Vertex covering the claude panel) hold no
2448
+ # Anthropic key at all — the same tolerance role_key applies at run
2449
+ # time, so the preflight and the climb agree.
2450
+ from outerloop.role_runner import role_key
2451
+
2452
+ role_key(path)
2453
+ return ""
2454
+ except Exception as exc:
2455
+ # never raises: an unexpected failure (partial deploy, ELOOP, unset
2456
+ # HOME) must fail closed WITH the alarm, not abort the tick that
2457
+ # would have written it
2458
+ return f"{type(exc).__name__}: {exc}"
2459
+
2460
+
2461
+ def _attempt_job_minutes(spec: FollowupSpec, limits: EffectiveLimits) -> int:
2462
+ """The submitted climb walltime: contract budget + panel allowance,
2463
+ clamped at the partition cap. Warns when the cap cuts below the session
2464
+ budget — the self-deadline would then fire before the author's own
2465
+ clock, and that must be a visible operator choice, never a silent
2466
+ surprise."""
2467
+ from outerloop.limits import ATTEMPT_OVERHEAD_MINUTES
2468
+
2469
+ wanted = limits.attempt_job_minutes + _panel_job_minutes(spec, limits)
2470
+ job = min(wanted, spec.max_job_minutes)
2471
+ if job < wanted:
2472
+ log.info(
2473
+ "climb job clamped to %d min by the partition cap (worst case "
2474
+ "wanted %d); slow panel rounds fail safe via the self-deadline",
2475
+ job,
2476
+ wanted,
2477
+ )
2478
+ if job < limits.session_minutes + ATTEMPT_OVERHEAD_MINUTES:
2479
+ log.warning(
2480
+ "work-job cap %d min leaves no runway around the %d-min session "
2481
+ "(the orchestrator needs ~%d min); sessions or endings will be "
2482
+ "cut short by the self-deadline",
2483
+ job,
2484
+ limits.session_minutes,
2485
+ ATTEMPT_OVERHEAD_MINUTES,
2486
+ )
2487
+ return job
2488
+
2489
+
2490
+ def _panel_job_minutes(spec: FollowupSpec, limits: EffectiveLimits) -> int:
2491
+ """Extra walltime the panel needs, ADDED to the contract-clamped job
2492
+ budget: the contract's knobs cap the AUTHOR's spend and their ceilings
2493
+ deliberately cannot raise ours (limits.py), so the panel — the
2494
+ orchestrator's own gate, flipped on by the tick — brings its own time.
2495
+ Worst case: three sequential reads of every lens (initial, post-revision,
2496
+ merged-tree) on the judge budget, plus one revision wake on the session
2497
+ budget. The revision's re-measure rides the margin the self-deadline
2498
+ already fails safe on."""
2499
+ lenses = [entry for entry in spec.panel.split(",") if entry.strip()]
2500
+ if not lenses:
2501
+ return 0
2502
+ from outerloop.roles import reviewer_spec, verifier_spec
2503
+
2504
+ judge_minutes = max(reviewer_spec().budget.walltime_s, verifier_spec().budget.walltime_s) // 60
2505
+ return 3 * len(lenses) * judge_minutes + limits.session_minutes
2506
+
2507
+
2508
+ def _climb_limit_argv(limits: EffectiveLimits, job_minutes: int) -> list[str]:
2509
+ """Climb-CLI flags carrying the tick-resolved limits: the job's ACTUAL
2510
+ walltime rides along (contract budget + any panel allowance, clamped at
2511
+ the partition cap — exactly what the JobSpec gets) so the climb arms its
2512
+ self-deadline against the real clock (Slurm delivers no signals to our
2513
+ processes on Torch). The session shrinks to fit a
2514
+ CAPPED job with the same rule limits.effective_limits applies to
2515
+ contract values — better a short session that ends cleanly than a full
2516
+ one the self-deadline kills mid-flight."""
2517
+ from outerloop.limits import ATTEMPT_OVERHEAD_MINUTES, SESSION_MINUTES_FLOOR
2518
+
2519
+ session = min(limits.session_minutes, job_minutes - ATTEMPT_OVERHEAD_MINUTES)
2520
+ return [
2521
+ "--max-turns",
2522
+ str(limits.session_max_turns),
2523
+ "--session-minutes",
2524
+ str(max(SESSION_MINUTES_FLOOR, session)),
2525
+ "--job-minutes",
2526
+ str(job_minutes),
2527
+ ]
2528
+
2529
+
2530
+ def service_self_initiated(
2531
+ root: Path,
2532
+ compute: Compute,
2533
+ spec: FollowupSpec,
2534
+ contract: Any,
2535
+ now: float,
2536
+ limits: EffectiveLimits | None = None,
2537
+ dry_run: bool = False,
2538
+ records: list[RunRecord] | None = None,
2539
+ ) -> tuple[str, str] | None:
2540
+ """The default background mode: when nothing else needs doing, climb the
2541
+ least-recently-attempted benchmark.
2542
+
2543
+ A pending marker written at submit time bridges the gap between
2544
+ `compute.submit` and the climb job writing its run record — without it,
2545
+ every tick during Slurm queue latency would launch a duplicate climb.
2546
+ """
2547
+ limits = limits if limits is not None else effective_limits(getattr(contract, "budgets", None))
2548
+ paused = outage_active(root, now, role="solver")
2549
+ if paused:
2550
+ log.info("self-initiated lane paused (api outage: %s)", paused)
2551
+ return None
2552
+ try:
2553
+ if records is None:
2554
+ records = list_runs(root)
2555
+ width = _attempt_width(contract)
2556
+ # WIDTH: every live pending marker occupies a slot; landed ones
2557
+ # clear; dead ones become per-benchmark tombstones and free theirs.
2558
+ occupied: set[str] = set()
2559
+ live_pendings: list[tuple[str, float]] = []
2560
+ nonslot_busy = False
2561
+ for agent, pending in list_pendings(root, spec.target):
2562
+ marker_agent = "" if not pending.get("agent_id") else agent
2563
+ submitted_at = float(pending["submitted_at"])
2564
+ # A slotted marker lands only when ITS OWN record appears —
2565
+ # matching on target+time alone would let a sibling slot's
2566
+ # record clear a still-live marker (terra #173). A legacy
2567
+ # marker names no slot, so it keeps the lax match.
2568
+ landed = any(
2569
+ r.target == spec.target
2570
+ and r.created >= submitted_at - 60
2571
+ and (not marker_agent or r.agent_id == marker_agent)
2572
+ for r in records
2573
+ )
2574
+ expired = now - submitted_at > PENDING_TTL_S
2575
+ if landed:
2576
+ clear_pending(root, spec.target, marker_agent)
2577
+ elif (alive := _holder_alive(compute, str(pending.get("job_id", "")))) is True or (
2578
+ not expired and alive is not False
2579
+ ):
2580
+ # climb queued or starting; its record isn't written yet. A
2581
+ # provably-alive job holds its SLOT regardless of the
2582
+ # marker's TTL — queue wait can exceed it — the TTL only
2583
+ # breaks ties when Slurm can't say.
2584
+ occupied.add(agent)
2585
+ live_pendings.append((str(pending.get("benchmark", "")), submitted_at))
2586
+ if not marker_agent:
2587
+ # a live un-slotted marker is another lane's submit
2588
+ # (steward/intake, or a pre-width deploy): serial
2589
+ nonslot_busy = True
2590
+ else:
2591
+ # Died before writing a record: persist the crash memory as a
2592
+ # PER-BENCHMARK tombstone (a sibling launch must not erase
2593
+ # this — terra #172 r3), then free the slot.
2594
+ write_tombstone(root, spec.target, str(pending.get("benchmark", "")), submitted_at)
2595
+ clear_pending(root, spec.target, marker_agent)
2596
+ stranded_cutoff = now - STRANDED_IMPLEMENTING_S
2597
+ for r in records:
2598
+ if r.target == spec.target and r.state != ENDED:
2599
+ if r.state == IMPLEMENTING and max(r.updated, r.created) <= stranded_cutoff:
2600
+ continue # stranded: pick ignores it, so must occupancy
2601
+ occupied.add(r.agent_id)
2602
+ if not _SLOT_AGENT_RE.fullmatch(r.agent_id):
2603
+ nonslot_busy = True
2604
+ if nonslot_busy:
2605
+ # steward and intake keep their pre-width one-run-per-target
2606
+ # exclusivity: width applies AMONG self-initiated slots, it
2607
+ # does not license launching beside another lane (terra #173)
2608
+ return None
2609
+ if len(occupied) >= width:
2610
+ return None
2611
+ slot_agent = _free_agent_slot(occupied, width)
2612
+ if slot_agent is None:
2613
+ return None
2614
+ dead_attempts = read_tombstones(root, spec.target, contract, now)
2615
+ benchmark = pick_self_initiated(
2616
+ records, contract, spec.target, now, dead_attempts, live_pendings
2617
+ )
2618
+ if benchmark is None:
2619
+ return None
2620
+ if getattr(contract, "merge", "manual") == "auto" and not spec.panel:
2621
+ # auto merge mode means gate+PANEL clean self-merges; a
2622
+ # deployment with no panel configured must not launch attempts
2623
+ # that would publish panel-less self-merging PRs (terra #171)
2624
+ log.error(
2625
+ "attempt on %s not launched: contract sets merge:auto but "
2626
+ "no panel is configured (set OUTERLOOP_PANEL, or the "
2627
+ "contract back to merge:manual)",
2628
+ benchmark,
2629
+ )
2630
+ return None
2631
+ if lane_error := _gpu_lane_error(contract, benchmark, spec):
2632
+ log.error("attempt on %s not launched: %s", benchmark, lane_error)
2633
+ return None
2634
+ author_error = _author_config_error(spec)
2635
+ if author_error:
2636
+ log.error(
2637
+ "climb on %s not launched: author misconfigured — %s "
2638
+ "(fix OUTERLOOP_AUTHOR_BACKEND/_MODEL)",
2639
+ benchmark,
2640
+ author_error,
2641
+ )
2642
+ return None
2643
+ panel_error = _panel_preflight_error(spec)
2644
+ if panel_error:
2645
+ log.error(
2646
+ "climb on %s not launched: panel misconfigured — %s "
2647
+ "(fix it, or set OUTERLOOP_PANEL='' to disable the panel)",
2648
+ benchmark,
2649
+ panel_error,
2650
+ )
2651
+ return None
2652
+ if dry_run:
2653
+ return (benchmark, "dry-run")
2654
+ job_minutes = _attempt_job_minutes(spec, limits)
2655
+ argv = [
2656
+ *_interpreter(spec.home),
2657
+ "-m",
2658
+ "outerloop.attempt",
2659
+ "--target",
2660
+ spec.target,
2661
+ "--benchmark",
2662
+ benchmark,
2663
+ "--run-root",
2664
+ str(spec.run_root),
2665
+ *_containment(spec.image),
2666
+ "--agent-id",
2667
+ slot_agent,
2668
+ *_climb_limit_argv(limits, job_minutes),
2669
+ *_climb_panel_argv(spec),
2670
+ ]
2671
+ if spec.pat_file:
2672
+ argv += ["--pat-file", spec.pat_file]
2673
+ # config-driven author: climb resolves the author backend/model/key from
2674
+ # OUTERLOOP_AUTHOR_* env (inherited by the job), so the tick threads
2675
+ # neither the backend nor its key — a new backend needs zero tick change.
2676
+ job_id = compute.submit(
2677
+ JobSpec(
2678
+ job_name=f"climb-{benchmark}-{slot_agent}"[:60],
2679
+ account=spec.account,
2680
+ partition=spec.job_partition or spec.partition,
2681
+ time_minutes=job_minutes,
2682
+ command=_flight_command(
2683
+ spec.home, f"climb-{benchmark}-{slot_agent}"[:60], now, argv
2684
+ ),
2685
+ cpus=4,
2686
+ mem="8G",
2687
+ )
2688
+ )
2689
+ write_pending(root, spec.target, benchmark, job_id, now, agent=slot_agent)
2690
+ log.info("self-initiated climb on %s: job %s", benchmark, job_id)
2691
+ return (benchmark, job_id)
2692
+ except Exception as exc: # one bad pass must not break the tick
2693
+ log.warning("self-initiated pass failed: %s", exc)
2694
+ return None
2695
+
2696
+
2697
+ def service_steward(
2698
+ root: Path,
2699
+ github: Any,
2700
+ compute: Compute,
2701
+ spec: FollowupSpec,
2702
+ now: float,
2703
+ contract: Any,
2704
+ limits: EffectiveLimits,
2705
+ dry_run: bool = False,
2706
+ records: list[RunRecord] | None = None,
2707
+ ) -> tuple[str, str] | None:
2708
+ """The steward lane: claim at most ONE labeled work-order issue per tick
2709
+ and submit a stewardship job. Off until the operator provisions the
2710
+ steward's own key (role separation) and the contract declares a steward
2711
+ scope."""
2712
+ from outerloop.steward import pick_steward_issue
2713
+
2714
+ target = spec.target
2715
+ if not target or not spec.steward_key_file:
2716
+ return None
2717
+ if getattr(contract, "steward", None) is None:
2718
+ return None
2719
+ try:
2720
+ from outerloop.steward import release_orphaned_claims
2721
+
2722
+ # ONE active run per target covers stewardships too: an env rewrite
2723
+ # must not fly alongside a solver climb or another stewardship.
2724
+ if records is None:
2725
+ records = list_runs(root)
2726
+ # reconcile first: killed jobs never post their own release — and
2727
+ # BEFORE the outage pause below, because a claim orphaned by the
2728
+ # very session the outage killed must not stay held all cooldown
2729
+ # (reconciliation is model-free bookkeeping; only spawning pauses)
2730
+ release_orphaned_claims(github, target, records, now, bot_login=spec.bot_login)
2731
+ paused = outage_active(root, now, role="steward")
2732
+ if paused:
2733
+ log.info("steward lane paused (api outage: %s)", paused)
2734
+ return None
2735
+ if any(r.target == target and r.state != ENDED for r in records):
2736
+ return None
2737
+ # The queue window (submit -> job writes its record) is bridged by
2738
+ # the SAME per-target pending markers the self-initiated lane uses
2739
+ # — ALL of them, slotted included: a width slot queued without a
2740
+ # record yet must block a stewardship the same way an active run
2741
+ # does. Liveness first, TTL only breaks unknown ties (queue wait
2742
+ # can outlive the TTL).
2743
+ for slot, pending in list_pendings(root, target):
2744
+ marker_agent = "" if not pending.get("agent_id") else slot
2745
+ submitted_at = float(pending.get("submitted_at", 0.0))
2746
+ landed = any(
2747
+ r.target == target
2748
+ and r.created >= submitted_at - 60
2749
+ and (not marker_agent or r.agent_id == marker_agent)
2750
+ for r in records
2751
+ )
2752
+ expired = now - submitted_at > PENDING_TTL_S
2753
+ alive = _holder_alive(compute, str(pending.get("job_id", "")))
2754
+ if not landed and (alive is True or (not expired and alive is not False)):
2755
+ return None
2756
+ task = pick_steward_issue(github, target, contract, spec.bot_login)
2757
+ if task is None:
2758
+ return None
2759
+ if _benchmark_gpus(contract, task.benchmark) > 0:
2760
+ # the stewardship validates its rewrite IN-JOB (SubprocessEvaluator
2761
+ # inside the CPU work job — no GPUs, no --nv), so a GPU benchmark
2762
+ # cannot be stewarded yet; refuse rather than launch a validation
2763
+ # that can only fail (terra #174 r2)
2764
+ log.error(
2765
+ "stewardship on %s not launched: GPU benchmarks validate in-job "
2766
+ "and the steward job has no GPU allocation",
2767
+ task.benchmark,
2768
+ )
2769
+ return None
2770
+ if dry_run:
2771
+ return (f"steward-issue-{task.number}", "dry-run")
2772
+ from outerloop.intake import CLAIM_MARKER, issue_hypothesis
2773
+
2774
+ github.comment(
2775
+ target,
2776
+ task.number,
2777
+ f"{CLAIM_MARKER}\nClaimed by the steward for benchmark "
2778
+ f"`{task.benchmark}`; a run is queued and a report will follow here.",
2779
+ )
2780
+ import base64 as _b64
2781
+
2782
+ work_order_b64 = _b64.b64encode(issue_hypothesis(task).encode()).decode()
2783
+ argv = [
2784
+ *_interpreter(spec.home),
2785
+ "-m",
2786
+ "outerloop.steward",
2787
+ "--target",
2788
+ target,
2789
+ "--benchmark",
2790
+ task.benchmark,
2791
+ "--run-root",
2792
+ str(spec.run_root),
2793
+ *_containment(spec.image),
2794
+ "--issue",
2795
+ str(task.number),
2796
+ "--work-order-b64",
2797
+ work_order_b64,
2798
+ "--key-file",
2799
+ spec.steward_key_file,
2800
+ # the SAME clamped walltime the JobSpec requests, so the
2801
+ # self-deadline arms against the real clock
2802
+ *_climb_limit_argv(limits, min(limits.attempt_job_minutes, spec.max_job_minutes)),
2803
+ ]
2804
+ if spec.pat_file:
2805
+ argv += ["--pat-file", spec.pat_file]
2806
+ try:
2807
+ job_id = compute.submit(
2808
+ JobSpec(
2809
+ job_name=f"steward-issue-{task.number}",
2810
+ account=spec.account,
2811
+ partition=spec.job_partition or spec.partition,
2812
+ time_minutes=min(limits.attempt_job_minutes, spec.max_job_minutes),
2813
+ command=_flight_command(spec.home, f"steward-issue-{task.number}", now, argv),
2814
+ cpus=4,
2815
+ mem="8G",
2816
+ )
2817
+ )
2818
+ except Exception:
2819
+ # release the claim: a claim with no job behind it would orphan
2820
+ # the work order forever (pick skips claimed issues)
2821
+ from outerloop.steward import RELEASE_MARKER
2822
+
2823
+ with contextlib.suppress(Exception):
2824
+ github.comment(
2825
+ target,
2826
+ task.number,
2827
+ f"{RELEASE_MARKER}\nSubmission failed; claim released — "
2828
+ f"a later tick will retry this work order.",
2829
+ )
2830
+ raise
2831
+ write_pending(root, target, f"steward:{task.benchmark}", job_id, now)
2832
+ log.info("steward issue #%s claimed for job %s", task.number, job_id)
2833
+ return (f"steward-issue-{task.number}", job_id)
2834
+ except Exception as exc: # the steward lane must not break the tick
2835
+ log.warning("steward pass failed: %s", exc)
2836
+ return None
2837
+
2838
+
2839
+ def service_intake(
2840
+ root: Path,
2841
+ github: Any,
2842
+ compute: Compute,
2843
+ spec: FollowupSpec,
2844
+ now: float,
2845
+ contract: Any = None,
2846
+ limits: EffectiveLimits | None = None,
2847
+ dry_run: bool = False,
2848
+ ) -> tuple[str, str] | None:
2849
+ """The requested lane: claim at most ONE qualifying issue per tick and
2850
+ submit a climb job for it. The claim comment (posted by the climb job
2851
+ before its session) marks an issue taken; one-per-tick keeps a burst of
2852
+ issues from bursting the budget. The contract arrives from the tick's
2853
+ single per-target fetch; None (fetch failed) sits the lane out."""
2854
+ from outerloop.contract import load_contract
2855
+ from outerloop.intake import issue_hypothesis, pick_issue
2856
+
2857
+ target = spec.target
2858
+ if not target:
2859
+ return None
2860
+ paused = outage_active(root, now, role="solver")
2861
+ if paused:
2862
+ log.info("intake lane paused (api outage: %s)", paused)
2863
+ return None
2864
+ try:
2865
+ if contract is None:
2866
+ contract_raw = _contract_text(github, target, "main")
2867
+ if contract_raw is None:
2868
+ return None
2869
+ contract = load_contract(contract_raw, target)
2870
+ limits = limits if limits is not None else effective_limits(contract.budgets)
2871
+ task = pick_issue(github, target, contract, spec.bot_login)
2872
+ if task is None:
2873
+ return None
2874
+ if getattr(contract, "merge", "manual") == "auto" and not spec.panel:
2875
+ # auto merge mode means gate+PANEL clean self-merges; a
2876
+ # deployment with no panel configured must not launch attempts
2877
+ # that would publish panel-less self-merging PRs (terra #171)
2878
+ log.error(
2879
+ "attempt on %s not launched: contract sets merge:auto but "
2880
+ "no panel is configured (set OUTERLOOP_PANEL, or the "
2881
+ "contract back to merge:manual)",
2882
+ task.benchmark,
2883
+ )
2884
+ return None
2885
+ if lane_error := _gpu_lane_error(contract, task.benchmark, spec):
2886
+ log.error("attempt on %s not launched: %s", task.benchmark, lane_error)
2887
+ return None
2888
+ author_error = _author_config_error(spec)
2889
+ if author_error:
2890
+ log.error(
2891
+ "issue #%d not claimed: author misconfigured — %s "
2892
+ "(fix OUTERLOOP_AUTHOR_BACKEND/_MODEL)",
2893
+ task.number,
2894
+ author_error,
2895
+ )
2896
+ return None
2897
+ panel_error = _panel_preflight_error(spec)
2898
+ if panel_error:
2899
+ log.error(
2900
+ "issue #%d not claimed: panel misconfigured — %s "
2901
+ "(fix it, or set OUTERLOOP_PANEL='' to disable the panel)",
2902
+ task.number,
2903
+ panel_error,
2904
+ )
2905
+ return None
2906
+ if dry_run:
2907
+ return (f"issue-{task.number}", "dry-run")
2908
+ job_minutes = _attempt_job_minutes(spec, limits)
2909
+ # claim BEFORE submit: Slurm queueing can take minutes, and the next
2910
+ # tick must not re-claim the same issue in that window
2911
+ from outerloop.intake import CLAIM_MARKER, MAX_INTAKE_ATTEMPTS, RELEASE_MARKER
2912
+
2913
+ github.comment(
2914
+ target,
2915
+ task.number,
2916
+ f"{CLAIM_MARKER}\nClaimed for benchmark `{task.benchmark}`; a run "
2917
+ "is queued and a report will follow here.",
2918
+ )
2919
+ import base64 as _b64
2920
+
2921
+ hypothesis_b64 = _b64.b64encode(issue_hypothesis(task).encode()).decode()
2922
+ argv = [
2923
+ *_interpreter(spec.home),
2924
+ "-m",
2925
+ "outerloop.attempt",
2926
+ "--target",
2927
+ target,
2928
+ "--benchmark",
2929
+ task.benchmark,
2930
+ "--run-root",
2931
+ str(spec.run_root),
2932
+ *_containment(spec.image),
2933
+ "--issue",
2934
+ str(task.number),
2935
+ "--hypothesis-b64",
2936
+ hypothesis_b64,
2937
+ *_climb_limit_argv(limits, job_minutes),
2938
+ *_climb_panel_argv(spec),
2939
+ ]
2940
+ if spec.pat_file:
2941
+ argv += ["--pat-file", spec.pat_file]
2942
+ # config-driven author: climb resolves the author key from the
2943
+ # OUTERLOOP_AUTHOR_* env by backend; the tick does not thread it.
2944
+ try:
2945
+ job_id = compute.submit(
2946
+ JobSpec(
2947
+ job_name=f"climb-issue-{task.number}",
2948
+ account=spec.account,
2949
+ partition=spec.job_partition or spec.partition,
2950
+ time_minutes=job_minutes,
2951
+ command=_flight_command(spec.home, f"climb-issue-{task.number}", now, argv),
2952
+ cpus=4,
2953
+ mem="8G",
2954
+ )
2955
+ )
2956
+ except Exception:
2957
+ # the claim is already posted and pick_issue skips claimed
2958
+ # issues, so a failed submit must release it (same pattern as
2959
+ # the steward lane) or the issue is stranded forever
2960
+ with contextlib.suppress(Exception):
2961
+ github.comment(
2962
+ target,
2963
+ task.number,
2964
+ f"{RELEASE_MARKER}\nSubmission failed; claim released — "
2965
+ f"a later tick will retry this issue (intake gives up "
2966
+ f"after {MAX_INTAKE_ATTEMPTS} claim attempts and leaves "
2967
+ f"it for a human).",
2968
+ )
2969
+ raise
2970
+ log.info("issue #%s claimed for climb job %s", task.number, job_id)
2971
+ return (f"issue-{task.number}", job_id)
2972
+ except Exception as exc: # intake must not break the tick
2973
+ log.warning("intake pass failed: %s", exc)
2974
+ return None
2975
+
2976
+
2977
+ @dataclass
2978
+ class LoggingDispatcher:
2979
+ """Never dispatched in production: main() runs the sweep dry unless
2980
+ dispatched wake is armed, so no lease is taken and no attempt is
2981
+ counted. This exists for the seam."""
2982
+
2983
+ def dispatch(self, record: RunRecord, reason: str) -> str:
2984
+ log.info("WOULD WAKE %s (%s) — session dispatch lands in phase 5", record.run_id, reason)
2985
+ return ""
2986
+
2987
+
2988
+ @dataclass
2989
+ class JobWakeDispatcher:
2990
+ """Delivers a wake by submitting a Slurm job that runs the wake CLI
2991
+ (`climb --resume <run_id>`), depending on the run's eval jobs (the record's
2992
+ `afterany`) so it fires when they finish — or immediately if they already
2993
+ have. CPU-only and short: a wake reads cached results and opens a PR, it
2994
+ never holds a GPU. Returns the wake job id (async: it owns the lease until
2995
+ it completes)."""
2996
+
2997
+ compute: Compute
2998
+ spec: FollowupSpec
2999
+ now: float
3000
+ wake_minutes: int = 20
3001
+
3002
+ def dispatch(self, record: RunRecord, reason: str) -> str:
3003
+ argv = [
3004
+ *_interpreter(self.spec.home),
3005
+ "-m",
3006
+ "outerloop.attempt",
3007
+ "--resume",
3008
+ record.run_id,
3009
+ "--run-root",
3010
+ str(self.spec.run_root),
3011
+ *_containment(self.spec.image),
3012
+ "--account",
3013
+ self.spec.account,
3014
+ "--partition",
3015
+ self.spec.partition,
3016
+ "--gpu-partition",
3017
+ self.spec.gpu_partition,
3018
+ "--gpu-account",
3019
+ self.spec.gpu_account,
3020
+ # the wake runs the SAME verification panel as the fresh climb, so a
3021
+ # dispatched improvement is verified before it is published.
3022
+ *_climb_panel_argv(self.spec),
3023
+ # session budget for the depth-axis REVISION (a blocking panel
3024
+ # finding wakes the author to revise).
3025
+ "--max-turns",
3026
+ str(self.spec.max_turns),
3027
+ ]
3028
+ # An AUTHOR-SLEEP wake resumes a FULL author session (not the short
3029
+ # read-decide a candidate wake runs), so the Slurm job must fit that
3030
+ # session or walltime kills the resumed session mid-run and the run just
3031
+ # waits for another wake. Size the job to the session
3032
+ # budget + overhead and pass --session-minutes so the in-job
3033
+ # self-deadline fires BEFORE Slurm's walltime. A candidate wake keeps
3034
+ # the short budget (read results + panel).
3035
+ from outerloop.limits import ATTEMPT_OVERHEAD_MINUTES
3036
+ from outerloop.roles import author_spec
3037
+
3038
+ if record.stage.get("phase") == "author-sleep":
3039
+ session_minutes = author_spec().budget.walltime_s // 60
3040
+ argv += ["--session-minutes", str(session_minutes)]
3041
+ job_minutes = min(session_minutes + ATTEMPT_OVERHEAD_MINUTES, self.spec.max_job_minutes)
3042
+ else:
3043
+ job_minutes = self.wake_minutes + _wake_panel_minutes(self.spec)
3044
+ if self.spec.pat_file:
3045
+ argv += ["--pat-file", self.spec.pat_file]
3046
+ # config-driven author: `climb --resume` resolves the author key from the
3047
+ # PARKED RUN's backend (persisted on its record) inside climb.main — the
3048
+ # tick does not thread the key, so a fleet flip picks the right one.
3049
+ name = f"wake-{record.run_id}"[:60]
3050
+ afterany = str(record.stage.get("afterany", ""))
3051
+ return self.compute.submit(
3052
+ JobSpec(
3053
+ job_name=name,
3054
+ account=self.spec.account,
3055
+ partition=self.spec.job_partition or self.spec.partition,
3056
+ time_minutes=job_minutes,
3057
+ command=_flight_command(self.spec.home, name, self.now, argv),
3058
+ dependency=afterany,
3059
+ cpus=2,
3060
+ mem="4G",
3061
+ )
3062
+ )
3063
+
3064
+
3065
+ def _wake_panel_minutes(spec: FollowupSpec) -> int:
3066
+ """Extra wake walltime for the verification panel it now runs — the base
3067
+ `wake_minutes` covers only reading results + opening the PR. Budgeted for
3068
+ the worst case a single wake reaches: one read per lens PLUS one revision
3069
+ author session (the depth-axis wake-to-revise). The revision's re-measure
3070
+ only DISPATCHES (then the job ends, parked), so it needs no extra time.
3071
+ Grounded in the same judge/author budgets the climb job uses."""
3072
+ from outerloop.panel import panel_read_minutes
3073
+ from outerloop.roles import author_spec
3074
+
3075
+ read_minutes = panel_read_minutes(spec.panel)
3076
+ if not read_minutes:
3077
+ return 0
3078
+ return read_minutes + author_spec().budget.walltime_s // 60
3079
+
3080
+
3081
+ def _wake_dispatcher_from_env(
3082
+ compute: Compute, followup_spec: FollowupSpec | None, now: float, root: Path
3083
+ ) -> tuple[WakeDispatcher, bool]:
3084
+ """The wake delivery for this tick, behind an EXPLICIT on-switch so the
3085
+ dispatched-wake path lands DARK. Returns `(dispatcher, live)`:
3086
+
3087
+ * armed (the `OUTERLOOP_DISPATCH_WAKE` env var OR a `<root>/DISPATCH_WAKE`
3088
+ sentinel file) AND the chain env carries what a wake job needs -> the real
3089
+ `JobWakeDispatcher` and a LIVE sweep;
3090
+ * otherwise -> the `LoggingDispatcher` and a DRY sweep.
3091
+
3092
+ The sentinel mirrors PAUSE: an operator arms/disarms with a touch/rm, no
3093
+ chain restart. So dispatched climbing is turned on deliberately, and a
3094
+ half-configured environment fails safe to dry rather than to a wake job
3095
+ that cannot run."""
3096
+ if not dispatch_wake_armed(root):
3097
+ return LoggingDispatcher(), False
3098
+ if followup_spec is None:
3099
+ log.warning("dispatch-wake armed but the chain env is incomplete; wake stays dry")
3100
+ return LoggingDispatcher(), False
3101
+ log.info("dispatched-wake ON: the waiting-run sweep delivers real wakes this tick")
3102
+ return JobWakeDispatcher(compute, followup_spec, now), True
3103
+
3104
+
3105
+ def _max_job_minutes_from_env() -> int:
3106
+ """OUTERLOOP_MAX_JOB_MINUTES, clamped into what the code can honor:
3107
+ at least the climb-job floor (an operator on a short-MaxTime partition
3108
+ must be able to LOWER the cap below cpu_short's 6h, or every submit is
3109
+ rejected), at most the ceiling the stranded window allows. A clamped
3110
+ value logs — a silently-changed cap would read as the partition
3111
+ rejecting jobs for no reason."""
3112
+ from outerloop.limits import ATTEMPT_JOB_MINUTES_FLOOR
3113
+
3114
+ raw = os.environ.get("OUTERLOOP_MAX_JOB_MINUTES", "").strip()
3115
+ if not raw:
3116
+ return MAX_ATTEMPT_JOB_MINUTES
3117
+ try:
3118
+ value = int(raw)
3119
+ except ValueError:
3120
+ log.warning("OUTERLOOP_MAX_JOB_MINUTES=%r is not an integer; using default", raw)
3121
+ return MAX_ATTEMPT_JOB_MINUTES
3122
+ clamped = max(ATTEMPT_JOB_MINUTES_FLOOR, min(value, MAX_JOB_MINUTES_CEILING))
3123
+ if clamped != value:
3124
+ log.warning("OUTERLOOP_MAX_JOB_MINUTES=%d clamped to %d", value, clamped)
3125
+ return clamped
3126
+
3127
+
3128
+ def _cadence_s() -> float:
3129
+ """The chain's tick cadence in seconds (OUTERLOOP_CADENCE_MIN, the same
3130
+ knob tick_chain.sbatch uses), defaulting to 30 min when unset/invalid."""
3131
+ raw = os.environ.get("OUTERLOOP_CADENCE_MIN", "").strip()
3132
+ try:
3133
+ cadence_s = float(raw) * 60 if raw else 30 * 60
3134
+ except ValueError:
3135
+ cadence_s = 30 * 60
3136
+ return cadence_s if (math.isfinite(cadence_s) and cadence_s > 0) else 30 * 60
3137
+
3138
+
3139
+ def _coalesce_ceiling_s() -> float:
3140
+ """The largest SAFE coalesce window, bounding both the default and an
3141
+ explicit OUTERLOOP_MIN_TICK_MINUTES: half the cadence (so an on-cadence
3142
+ tick is never coalesced even when the previous one ran a little late), and
3143
+ never above the absolute MAX_MIN_TICK_S. A window at/above the cadence would
3144
+ swallow every normal tick and stall the loop — this is what forbids it."""
3145
+ return min(float(MAX_MIN_TICK_S), _cadence_s() / 2)
3146
+
3147
+
3148
+ def _default_min_tick_s() -> float:
3149
+ """The coalesce window when none is set: the safe ceiling, further capped at
3150
+ the 10-min DEFAULT_MIN_TICK_S — small enough to only catch pile-ups, and
3151
+ cadence-aware so a short cadence scales it down instead of swallowing every
3152
+ tick."""
3153
+ return min(DEFAULT_MIN_TICK_S, _coalesce_ceiling_s())
3154
+
3155
+
3156
+ def _min_tick_s_from_env() -> float:
3157
+ """OUTERLOOP_MIN_TICK_MINUTES -> the coalesce window in seconds. Unset
3158
+ derives a cadence-aware default; non-numeric/non-finite also fall back to it;
3159
+ negative clamps to 0 (coalesce disabled); a value at/above the safe ceiling
3160
+ (half the cadence, capped at MAX_MIN_TICK_S) clamps down so it cannot stall
3161
+ the loop."""
3162
+ raw = os.environ.get("OUTERLOOP_MIN_TICK_MINUTES", "").strip()
3163
+ if not raw:
3164
+ return _default_min_tick_s()
3165
+ try:
3166
+ minutes = float(raw)
3167
+ except ValueError:
3168
+ log.warning("OUTERLOOP_MIN_TICK_MINUTES=%r is not a number; using default", raw)
3169
+ return _default_min_tick_s()
3170
+ # reject inf/nan: an infinite window would coalesce every future tick and
3171
+ # freeze the loop (a finite elapsed time is always < inf)
3172
+ if not math.isfinite(minutes):
3173
+ log.warning("OUTERLOOP_MIN_TICK_MINUTES=%r is not finite; using default", raw)
3174
+ return _default_min_tick_s()
3175
+ seconds = max(0.0, minutes * 60)
3176
+ ceiling = _coalesce_ceiling_s()
3177
+ if seconds > ceiling:
3178
+ log.warning(
3179
+ "OUTERLOOP_MIN_TICK_MINUTES=%s exceeds the safe ceiling "
3180
+ "(%.0f min, ~half the cadence); clamping so normal ticks are not coalesced",
3181
+ raw,
3182
+ ceiling / 60,
3183
+ )
3184
+ return ceiling
3185
+ return seconds
3186
+
3187
+
3188
+ def _default_image() -> str:
3189
+ """~/outerloop-images/agent-py312.sif, or the pre-rename ~/autoresearch-images
3190
+ path when only that one exists. The fallback is dropped in the release after
3191
+ 0.1."""
3192
+ new = os.path.expanduser("~/outerloop-images/agent-py312.sif")
3193
+ old = os.path.expanduser("~/autoresearch-images/agent-py312.sif")
3194
+ return old if (not os.path.isfile(new) and os.path.isfile(old)) else new
3195
+
3196
+
3197
+ def _followup_spec_from_env(root: Path) -> tuple[Any, FollowupSpec | None]:
3198
+ """GitHub client + FollowupSpec from the chain environment, or Nones when
3199
+ the environment is incomplete (the tick then runs without in-review
3200
+ servicing, and logs what is absent)."""
3201
+ pat_file = os.environ.get("OUTERLOOP_PAT_FILE", "")
3202
+ app_file = os.environ.get("OUTERLOOP_GITHUB_APP_FILE", "")
3203
+ account = os.environ.get("OUTERLOOP_ACCOUNT", "")
3204
+ partition = os.environ.get("OUTERLOOP_PARTITION", "")
3205
+ image = os.environ.get("OUTERLOOP_IMAGE", _default_image())
3206
+ home = os.environ.get("OUTERLOOP_HOME", "")
3207
+ # Account and partition are optional on Slurm: empty ones leave the billing
3208
+ # association and the partition to Slurm's defaults, as `start` already
3209
+ # does for the resident. Local compute has no placement at all.
3210
+ target = os.environ.get("OUTERLOOP_TARGET", "")
3211
+ # no default identity (#298): every own-comment and own-PR filter keys on
3212
+ # this login, and a wrong one is worse than none
3213
+ bot_login = os.environ.get("OUTERLOOP_BOT_LOGIN", "").strip()
3214
+ image_ok = Path(image).is_file()
3215
+ panel = os.environ.get("OUTERLOOP_PANEL", "verify,review")
3216
+ if not image_ok and local_mode():
3217
+ # The local loop on a machine with no container: sessions run under
3218
+ # the harness's own sandbox and evaluations run bare, on the
3219
+ # operator's own machine with the operator's own keys. The panel is
3220
+ # off unless the operator opts in, because an uncontained judge holds
3221
+ # a shell next to its own key file. Said once per process. Contained
3222
+ # local mode needs apptainer and the image (docs/install.md).
3223
+ global _UNCONTAINED_WARNED
3224
+ if not _UNCONTAINED_WARNED:
3225
+ _UNCONTAINED_WARNED = True
3226
+ log.warning(
3227
+ "local mode: no container image at %s; sessions run under the harness "
3228
+ "sandbox and evaluations run bare on this machine; the panel is %s; a "
3229
+ "codex author needs the image (docs/install.md, local mode)",
3230
+ image,
3231
+ "on by OUTERLOOP_PANEL_UNCONTAINED=1"
3232
+ if os.environ.get("OUTERLOOP_PANEL_UNCONTAINED") == "1"
3233
+ else "off (OUTERLOOP_PANEL_UNCONTAINED=1 turns it on)",
3234
+ )
3235
+ if os.environ.get("OUTERLOOP_PANEL_UNCONTAINED") != "1":
3236
+ panel = ""
3237
+ # the jobs must not inherit a path to an image that is not there
3238
+ os.environ.pop("OUTERLOOP_IMAGE", None)
3239
+ image, image_ok = "", True
3240
+ if (pat_file or app_file) and home and target and bot_login and image_ok:
3241
+ from outerloop.appauth import resolve_bot_auth
3242
+ from outerloop.github import GitHubClient
3243
+
3244
+ try:
3245
+ github = GitHubClient(auth=resolve_bot_auth(pat_file, app_file))
3246
+ followup_spec = FollowupSpec(
3247
+ account=account,
3248
+ partition=partition,
3249
+ run_root=root,
3250
+ image=image,
3251
+ home=Path(home),
3252
+ pat_file=pat_file,
3253
+ github_app_file=app_file,
3254
+ target=target,
3255
+ steward_key_file=os.environ.get("OUTERLOOP_STEWARD_KEY_FILE", ""),
3256
+ panel=panel,
3257
+ panel_key_file=os.environ.get("OUTERLOOP_PANEL_KEY_FILE", ""),
3258
+ job_partition=os.environ.get("OUTERLOOP_JOB_PARTITION", ""),
3259
+ gpu_partition=os.environ.get("OUTERLOOP_GPU_PARTITION", ""),
3260
+ gpu_account=os.environ.get("OUTERLOOP_GPU_ACCOUNT", ""),
3261
+ max_job_minutes=_max_job_minutes_from_env(),
3262
+ )
3263
+ return github, followup_spec
3264
+ except Exception as exc:
3265
+ log.warning("in-review servicing disabled: %s", exc)
3266
+ return None, None
3267
+ absent = [
3268
+ name
3269
+ for name, value in [
3270
+ ("OUTERLOOP_PAT_FILE or _GITHUB_APP_FILE", pat_file or app_file),
3271
+ ("OUTERLOOP_HOME", home),
3272
+ ("OUTERLOOP_TARGET", target),
3273
+ ("OUTERLOOP_BOT_LOGIN", bot_login),
3274
+ ]
3275
+ if not value
3276
+ ]
3277
+ if not image_ok:
3278
+ absent.append(f"image:{image}")
3279
+ log.info("in-review servicing disabled (missing: %s)", ", ".join(absent))
3280
+ return None, None
3281
+
3282
+
3283
+ def _loop_cadence_s(cadence_min: float) -> float:
3284
+ """The --loop sleep, clamped to [60s, 24h]: argparse accepts inf (which
3285
+ would OverflowError out of time.sleep) and sub-minute values would spin.
3286
+ A non-positive argument defers to OUTERLOOP_CADENCE_MIN."""
3287
+ return min(24 * 3600.0, max(60.0, cadence_min * 60 if cadence_min > 0 else _cadence_s()))
3288
+
3289
+
3290
+ def main() -> int:
3291
+ import argparse
3292
+ import time
3293
+
3294
+ parser = argparse.ArgumentParser(
3295
+ prog="outerloop tick",
3296
+ description="One tick of the loop: service the open runs, launch new work, wake "
3297
+ "parked runs. The chain runs this every cadence; --loop does the same in the "
3298
+ "foreground.",
3299
+ )
3300
+ parser.add_argument("--root", required=True, type=Path, help="state root on the shared FS")
3301
+ parser.add_argument(
3302
+ "--grace-s",
3303
+ type=float,
3304
+ default=DEFAULT_GRACE_S,
3305
+ help="seconds a finished experiment's delivery job gets before the sweep steps in",
3306
+ )
3307
+ parser.add_argument(
3308
+ "--lease-ttl-s",
3309
+ type=float,
3310
+ default=DEFAULT_LEASE_TTL_S,
3311
+ help="seconds after which a held run lease counts as stale",
3312
+ )
3313
+ parser.add_argument(
3314
+ "--min-free-gb",
3315
+ type=float,
3316
+ default=DEFAULT_MIN_FREE_BYTES / 1024**3,
3317
+ help="skip launching new work when the state filesystem has less than this many GB free",
3318
+ )
3319
+ parser.add_argument(
3320
+ "--loop",
3321
+ action="store_true",
3322
+ help="run a tick every cadence in the foreground — the local-mode "
3323
+ "chain (Slurm deployments use tick_chain.sbatch instead)",
3324
+ )
3325
+ parser.add_argument(
3326
+ "--cadence-min",
3327
+ type=float,
3328
+ default=0.0,
3329
+ help="minutes between --loop ticks; unset defers to "
3330
+ "OUTERLOOP_CADENCE_MIN via the chain's own parser (default 30)",
3331
+ )
3332
+ args = parser.parse_args()
3333
+ logging.basicConfig(level=logging.INFO, format="%(asctime)s %(message)s")
3334
+
3335
+ args.root.mkdir(parents=True, exist_ok=True)
3336
+ # The tick's --root is the authority; children (and this process's own
3337
+ # LocalCompute) read OUTERLOOP_ROOT, so a bare `tick --loop --root X`
3338
+ # must not split-brain them: local job states would land nowhere and
3339
+ # every finished job would read GONE until the park deadline.
3340
+ # RESOLVED: local jobs cd into flight checkouts, so a relative root
3341
+ # would scatter their state dirs across working directories
3342
+ if os.environ.get("OUTERLOOP_ROOT", "") != str(args.root.resolve()):
3343
+ os.environ["OUTERLOOP_ROOT"] = str(args.root.resolve())
3344
+ # In-review servicing is LIVE when credentials + image are available in the
3345
+ # chain environment. The waiting-run sweep delivers real wakes only when the
3346
+ # operator arms it — the OUTERLOOP_DISPATCH_WAKE env var or a
3347
+ # <root>/DISPATCH_WAKE sentinel — and the env is complete; by default it
3348
+ # stays dry with the LoggingDispatcher — dispatched climbing lands DARK.
3349
+ # ONE compute for the process: LocalCompute remembers its jobs' states
3350
+ # in memory, so a --loop deployment must not discard them between ticks.
3351
+ compute = compute_from_env()
3352
+
3353
+ def run_once() -> None:
3354
+ github, followup_spec = _followup_spec_from_env(args.root)
3355
+ now = time.time()
3356
+ dispatcher, wake_live = _wake_dispatcher_from_env(compute, followup_spec, now, args.root)
3357
+ # parks arm their own wake from this recipe; without it the sweep delivers.
3358
+ # Local compute never arms: jobs are synchronous, so an afterany wake's
3359
+ # dependencies are terminal before submit returns — the next loop
3360
+ # iteration's sweep delivers every wake instead (wake latency = cadence).
3361
+ if wake_live and followup_spec is not None and not isinstance(compute, LocalCompute):
3362
+ write_wake_spec(args.root, followup_spec)
3363
+ else:
3364
+ remove_wake_spec(args.root)
3365
+
3366
+ report = tick(
3367
+ args.root,
3368
+ compute,
3369
+ dispatcher,
3370
+ now=now,
3371
+ grace_s=args.grace_s,
3372
+ lease_ttl_s=args.lease_ttl_s,
3373
+ dry_run=not wake_live,
3374
+ github=github,
3375
+ followup_spec=followup_spec,
3376
+ followup_dry_run=False,
3377
+ min_free_bytes=int(args.min_free_gb * 1024**3),
3378
+ min_tick_s=_min_tick_s_from_env(),
3379
+ )
3380
+ # Stamp the coalesce marker at REAL completion time (a fresh time.time(),
3381
+ # not the start-of-tick `now`), so a long tick does not leave a stale marker.
3382
+ mark_tick_complete(args.root, report, time.time())
3383
+ log.info(
3384
+ "tick done: paused=%s coalesced=%s swept=%d woken=%d deferred=%d reaped=%d stuck=%d "
3385
+ "impl_ended=%s review_ended=%s followups=%s intake=%s self_initiated=%s steward=%s "
3386
+ "disk=%s launch_blocked=%s shed=%d",
3387
+ report.paused,
3388
+ report.coalesced,
3389
+ report.swept,
3390
+ len(report.woken),
3391
+ len(report.deferred),
3392
+ len(report.reaped_leases),
3393
+ len(report.stuck),
3394
+ report.implementing_ended or "-",
3395
+ report.review_ended,
3396
+ report.followups_submitted,
3397
+ report.intake,
3398
+ report.self_initiated,
3399
+ report.steward,
3400
+ report.disk or "ok",
3401
+ report.launch_blocked,
3402
+ len(report.shed),
3403
+ )
3404
+
3405
+ if not args.loop:
3406
+ run_once()
3407
+ return 0
3408
+ # The local-mode chain: same stateless tick, a foreground loop instead of
3409
+ # sbatch successors. Records on disk carry all state, so killing and
3410
+ # restarting the loop resumes exactly like the Slurm chain would.
3411
+ cadence_s = _loop_cadence_s(args.cadence_min)
3412
+ while True:
3413
+ started = time.time()
3414
+ try:
3415
+ run_once()
3416
+ except Exception:
3417
+ log.exception("tick failed; the loop continues")
3418
+ time.sleep(max(0.0, cadence_s - (time.time() - started)))
3419
+
3420
+
3421
+ if __name__ == "__main__":
3422
+ raise SystemExit(main())