outerloop-science 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. outerloop/__init__.py +18 -0
  2. outerloop/__main__.py +3 -0
  3. outerloop/appauth.py +230 -0
  4. outerloop/appmanifest.py +203 -0
  5. outerloop/attempt.py +3784 -0
  6. outerloop/brief.py +528 -0
  7. outerloop/cli.py +621 -0
  8. outerloop/climbboard.py +1395 -0
  9. outerloop/compute.py +654 -0
  10. outerloop/contract.py +492 -0
  11. outerloop/contract_cli.py +63 -0
  12. outerloop/disk.py +164 -0
  13. outerloop/dispatch.py +631 -0
  14. outerloop/evalcache.py +147 -0
  15. outerloop/followup.py +2172 -0
  16. outerloop/github.py +1531 -0
  17. outerloop/harness.py +1435 -0
  18. outerloop/housekeeping.py +151 -0
  19. outerloop/image.py +368 -0
  20. outerloop/init.py +744 -0
  21. outerloop/intake.py +126 -0
  22. outerloop/launchlog.py +239 -0
  23. outerloop/limits.py +80 -0
  24. outerloop/maintain.py +353 -0
  25. outerloop/maintain_agent_cli.py +81 -0
  26. outerloop/maintain_post_cli.py +140 -0
  27. outerloop/markers.py +48 -0
  28. outerloop/measure.py +529 -0
  29. outerloop/orchestrator.py +2011 -0
  30. outerloop/panel.py +188 -0
  31. outerloop/paths.py +40 -0
  32. outerloop/posting.py +160 -0
  33. outerloop/progress.py +170 -0
  34. outerloop/py.typed +0 -0
  35. outerloop/review.py +615 -0
  36. outerloop/review_agent.py +263 -0
  37. outerloop/review_agent_cli.py +209 -0
  38. outerloop/review_post_cli.py +162 -0
  39. outerloop/review_summarize_cli.py +165 -0
  40. outerloop/role_runner.py +229 -0
  41. outerloop/roles.py +274 -0
  42. outerloop/rolespec.py +91 -0
  43. outerloop/runstate.py +385 -0
  44. outerloop/steward.py +845 -0
  45. outerloop/style.py +12 -0
  46. outerloop/syscall.py +1192 -0
  47. outerloop/syscall_cli.py +762 -0
  48. outerloop/tick.py +3422 -0
  49. outerloop/verifier.py +403 -0
  50. outerloop/verify_agent.py +151 -0
  51. outerloop/verify_agent_cli.py +95 -0
  52. outerloop/verify_post_cli.py +116 -0
  53. outerloop/watcher.py +203 -0
  54. outerloop_science-0.1.0.dist-info/METADATA +152 -0
  55. outerloop_science-0.1.0.dist-info/RECORD +59 -0
  56. outerloop_science-0.1.0.dist-info/WHEEL +4 -0
  57. outerloop_science-0.1.0.dist-info/entry_points.txt +2 -0
  58. outerloop_science-0.1.0.dist-info/licenses/LICENSE +202 -0
  59. outerloop_science-0.1.0.dist-info/licenses/NOTICE +5 -0
@@ -0,0 +1,2011 @@
1
+ """Orchestrator v1: one climb attempt on one benchmark of one target.
2
+
3
+ Deliberately narrow: `attempt_once` runs a single
4
+ implement→evaluate→verify→PR cycle for the configured benchmark. Task
5
+ selection across benchmarks, the planner, experiment sbatch + wakes, and
6
+ notebook reports grow from here — each behind a seam that already exists.
7
+
8
+ The verification stance is the architecture's: the agent's claim is never
9
+ trusted. The orchestrator re-runs the benchmark command itself — baseline at
10
+ the pre-session tree, candidate after — and only a direction-consistent,
11
+ threshold-clearing delta opens a PR.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import contextlib
17
+ import json
18
+ import logging
19
+ import math
20
+ import subprocess
21
+ from collections.abc import Callable, Sequence
22
+ from dataclasses import dataclass, field
23
+ from dataclasses import replace as dc_replace
24
+ from fractions import Fraction
25
+ from pathlib import Path
26
+ from secrets import randbits
27
+ from typing import TYPE_CHECKING, Any, Protocol
28
+
29
+ if TYPE_CHECKING:
30
+ from outerloop.measure import Measure
31
+
32
+ from outerloop.brief import BriefInputs, BudgetState, Task, build_brief, render
33
+ from outerloop.contract import (
34
+ Benchmark,
35
+ Contract,
36
+ _fold,
37
+ load_contract,
38
+ normalize_path,
39
+ path_is_forbidden,
40
+ )
41
+ from outerloop.harness import Harness, SessionResult, budget_exhausted, outage, redact
42
+ from outerloop.panel import PanelVerdict
43
+ from outerloop.role_runner import run_role
44
+ from outerloop.roles import author_spec
45
+ from outerloop.rolespec import RoleSpec
46
+ from outerloop.syscall import (
47
+ MISSING_REPORT,
48
+ SyscallError,
49
+ SyscallRequest,
50
+ clamp_concurrency,
51
+ evals_gpu_hours,
52
+ launches_gpu_hours,
53
+ refresh_tool,
54
+ )
55
+ from outerloop.syscall import budget_error as syscall_budget_error
56
+ from outerloop.syscall import read_request as read_syscall_request
57
+ from outerloop.syscall import render_refusal as render_syscall_refusal
58
+
59
+ log = logging.getLogger(__name__)
60
+
61
+ EVAL_TIMEOUT_S = 1800
62
+ MAX_REPORT_BODY = 20_000
63
+
64
+
65
+ # Environment keys the evaluator manages itself; a contract's seed_env may
66
+ # never name one (validated at load; filtered again at injection).
67
+ PROTECTED_EVAL_ENV = frozenset(
68
+ {
69
+ "HOME",
70
+ "PATH",
71
+ "TMPDIR",
72
+ "LANG",
73
+ "VIRTUAL_ENV",
74
+ # interpreter/loader steering: a random-integer value cannot carry a
75
+ # payload, but a contract naming one of these would silently break
76
+ # every eval in a way that reads as measurement failure
77
+ "PYTHONPATH",
78
+ "PYTHONHOME",
79
+ "PYTHONSTARTUP",
80
+ }
81
+ )
82
+ # ...and whole families: any UV_* steers uv's env/cache resolution, and any
83
+ # APPTAINERENV_* is translated into the CONTAINER's environment by apptainer
84
+ # (APPTAINERENV_HOME becomes HOME inside), so exact-name checks cannot
85
+ # enumerate them.
86
+ # APPTAINER_* configures the HOST-side apptainer CLI (bind paths, home,
87
+ # containment) — same family logic, different side of the boundary.
88
+ # LD_/DYLD_ steer the dynamic loader; GIT_ redirects repo resolution.
89
+ # (PYTHONHASHSEED stays allowed — it IS a seed, and a legitimate seed_env.)
90
+ PROTECTED_ENV_PREFIXES = ("UV_", "APPTAINERENV_", "APPTAINER_", "LD_", "DYLD_", "GIT_")
91
+
92
+
93
+ def managed_eval_env(name: str) -> bool:
94
+ """True when injecting `name` could disturb the eval's own isolation."""
95
+ return name in PROTECTED_EVAL_ENV or name.startswith(PROTECTED_ENV_PREFIXES)
96
+
97
+
98
+ class EvalError(RuntimeError):
99
+ """The benchmark command failed or produced no readable metric."""
100
+
101
+
102
+ class RunParked(Exception):
103
+ """A dispatched climb submitted its measures and must hibernate until they
104
+ finish. `attempt_once` raises it (a park is an exceptional exit); the caller,
105
+ which owns the run record and git, persists the fields below as the WAITING
106
+ stage — `afterany` among them, the dependency set the wake waits on — and
107
+ ends the run's turn, keeping the candidate snapshot alive so the wake can
108
+ read it. `phase` is WHICH park: `candidate` (after the session — the wake
109
+ decides) or `author-sleep` (the author launched work and slept). The caller
110
+ fills in the candidate snapshot ref (which it holds) when writing the
111
+ stage."""
112
+
113
+ def __init__(
114
+ self,
115
+ *,
116
+ phase: str,
117
+ afterany: str,
118
+ base_sha: str,
119
+ seed: int,
120
+ suite_seed: int,
121
+ candidate_sha: str = "",
122
+ session: SessionResult | None = None,
123
+ syscall: SyscallRequest | None = None,
124
+ launches_used: int = 0,
125
+ sleeps_used: int = 0,
126
+ submitted: bool = False,
127
+ gpu_hours_used: float = 0.0,
128
+ eval_minutes: int | None = None,
129
+ judged: tuple[str, AttemptResult] | None = None,
130
+ launch_afterany: str = "",
131
+ ):
132
+ self.phase = phase
133
+ # the author's launch jobs alone (a candidate park's `afterany` also
134
+ # carries the gate's evals): the wake reconciles their charge
135
+ self.launch_afterany = launch_afterany
136
+ # the gate's last negative and the tree it judged, carried across an
137
+ # author-sleep so a wake ending on that tree reuses the verdict
138
+ self.judged = judged
139
+ self.afterany = afterany
140
+ self.base_sha = base_sha
141
+ self.seed = seed
142
+ self.suite_seed = suite_seed
143
+ self.candidate_sha = candidate_sha
144
+ self.session = session
145
+ # GPU-hours drawn so far (launches + gate evals), and the eval
146
+ # walltime the author declared at submit — the wake re-parks and
147
+ # re-floors on the same numbers
148
+ self.gpu_hours_used = gpu_hours_used
149
+ self.eval_minutes = eval_minutes
150
+ # Syscall parks (an author-sleep, or a SUBMITTED candidate — Phase B):
151
+ # the request the wake gathers results for, and the budget counts AFTER
152
+ # this park. `submitted` marks a candidate park raised by `submit`: the
153
+ # wake delivers gate + panel results back to the AUTHOR instead of
154
+ # deciding by policy.
155
+ self.syscall = syscall
156
+ self.launches_used = launches_used
157
+ self.sleeps_used = sleeps_used
158
+ self.submitted = submitted
159
+ super().__init__(f"climb parked at {phase} on {afterany or '(no dep)'}")
160
+
161
+
162
+ class Evaluator(Protocol):
163
+ """Runs a benchmark command in a workspace, returns the metric value."""
164
+
165
+ def evaluate(
166
+ self, workspace: Path, command: str, metric: str, extra_env: dict[str, str] | None = None
167
+ ) -> float: ...
168
+
169
+
170
+ @dataclass
171
+ class SubprocessEvaluator:
172
+ """Runs the contract's benchmark command and reads `metric` from its
173
+ JSON output (the contract requires commands to print their metrics).
174
+
175
+ The command executes AGENT-WRITTEN code — it is session-grade untrusted
176
+ execution and gets session-grade containment: with `container_image` set
177
+ (the production configuration), the command runs under `apptainer exec
178
+ --containall` seeing only the workspace, a throwaway tmpfs HOME, and no
179
+ host environment. Uncontained mode exists for tests and non-cluster dev,
180
+ with a scrubbed env that NEVER includes the real HOME (the orchestrator
181
+ account holds the bot PAT under it)."""
182
+
183
+ timeout_s: int = EVAL_TIMEOUT_S
184
+ container_image: str = ""
185
+ apptainer_binary: str = "apptainer"
186
+
187
+ def evaluate(
188
+ self, workspace: Path, command: str, metric: str, extra_env: dict[str, str] | None = None
189
+ ) -> float:
190
+
191
+ # Throwaway HOME OUTSIDE the clone: never the orchestrator's real home
192
+ # (it shelters the PAT), and never the workspace — eval cache/state
193
+ # artifacts must not masquerade as agent edits in the diff. The
194
+ # CONTAINED eval needs it too (--home): apptainer's tmpfs home is
195
+ # size-capped and uv blows it extracting wheels.
196
+ # Fresh per-EVAL home (never reused): baseline and candidate cannot
197
+ # see each other's writes, and nothing survives to any later run.
198
+ import os
199
+ import tempfile
200
+
201
+ try:
202
+ eval_home = Path(
203
+ tempfile.mkdtemp(
204
+ prefix=f"{workspace.name}-eval-home-", dir=workspace.resolve().parent
205
+ )
206
+ )
207
+ except OSError as exc:
208
+ raise EvalError(f"could not create eval home: {exc}") from exc
209
+ # per-EVAL cache on node-local scratch: local IO (NFS caches flake),
210
+ # no state crossing evals (agent code runs during the candidate eval
211
+ # and must not poison later baselines), and only THIS directory is
212
+ # bound into the container — never the whole host /tmp.
213
+ try:
214
+ cache_dir = Path(
215
+ tempfile.mkdtemp(prefix="uv-cache-", dir=os.environ.get("TMPDIR", "/tmp"))
216
+ )
217
+ except OSError as exc:
218
+ import shutil
219
+
220
+ shutil.rmtree(eval_home, ignore_errors=True)
221
+ raise EvalError(f"could not create eval cache dir: {exc}") from exc
222
+ try:
223
+ return self._measure(workspace, command, metric, eval_home, cache_dir, extra_env)
224
+ finally:
225
+ # bounded disk: each eval's home AND cache die with it
226
+ # (re-downloading wheels per eval is the accepted isolation cost)
227
+ import shutil
228
+
229
+ shutil.rmtree(eval_home, ignore_errors=True)
230
+ shutil.rmtree(cache_dir, ignore_errors=True)
231
+
232
+ def _measure(
233
+ self,
234
+ workspace: Path,
235
+ command: str,
236
+ metric: str,
237
+ eval_home: Path,
238
+ cache_dir: Path,
239
+ extra_env: dict[str, str] | None = None,
240
+ ) -> float:
241
+ return self._parse_measured(
242
+ self._run(workspace, command, eval_home, cache_dir, extra_env), metric
243
+ )
244
+
245
+ def _run(
246
+ self,
247
+ workspace: Path,
248
+ command: str,
249
+ eval_home: Path,
250
+ cache_dir: Path,
251
+ extra_env: dict[str, str] | None = None,
252
+ ) -> str:
253
+ import os
254
+ import signal
255
+
256
+ if self.container_image:
257
+ # per evaluation, and gone with the cache dir the caller removes
258
+ workdir = cache_dir / "work"
259
+ workdir.mkdir(parents=True, exist_ok=True)
260
+ argv = [
261
+ self.apptainer_binary,
262
+ "exec",
263
+ "--containall",
264
+ "--cleanenv",
265
+ "--bind",
266
+ f"{workspace}:{workspace}",
267
+ "--home",
268
+ f"{eval_home}:{eval_home}",
269
+ # node-local scratch for uv's cache: the container's own /tmp
270
+ # is a size-capped tmpfs, and shared-FS caches flake (NFS)
271
+ "--bind",
272
+ f"{cache_dir}:{cache_dir}",
273
+ "--pwd",
274
+ str(workspace),
275
+ # /tmp inside the jail on this evaluation's own scratch, not
276
+ # apptainer's tmpfs (the dispatched job script does the same)
277
+ "--workdir",
278
+ str(workdir),
279
+ self.container_image,
280
+ "sh",
281
+ "-c",
282
+ command,
283
+ ]
284
+ else:
285
+ argv = ["sh", "-c", command]
286
+ env = {k: os.environ[k] for k in ("PATH", "LANG", "TMPDIR") if k in os.environ}
287
+ # uv's cache does heavy small-file IO; on shared filesystems (NFS)
288
+ # that flakes with stale-handle/copy errors. Keep the cache on
289
+ # node-local scratch and copy across filesystems.
290
+ env["UV_CACHE_DIR"] = str(cache_dir)
291
+ env["UV_LINK_MODE"] = "copy"
292
+ # PRIVATE project env per eval: the session builds ws/.venv for its
293
+ # own use, and a second process consuming a venv another process
294
+ # just wrote races NFS close-to-open consistency. The eval builds
295
+ # its own environment from the LOCKFILE on NODE-LOCAL scratch (beside the
296
+ # uv cache: fast IO, zero NFS in the venv path, dies with the
297
+ # eval) — no shared mutable state, and the orchestrator never
298
+ # executes session-authored entrypoints.
299
+ env["UV_PROJECT_ENVIRONMENT"] = str(cache_dir / "venv")
300
+ if self.container_image:
301
+ # --cleanenv drops the host env; APPTAINERENV_* survives it
302
+ env["APPTAINERENV_UV_CACHE_DIR"] = env["UV_CACHE_DIR"]
303
+ env["APPTAINERENV_UV_LINK_MODE"] = "copy"
304
+ env["APPTAINERENV_UV_PROJECT_ENVIRONMENT"] = env["UV_PROJECT_ENVIRONMENT"]
305
+ env["HOME"] = str(eval_home)
306
+ if extra_env:
307
+ # explicit injections only (the base env is a scrubbed
308
+ # allowlist): today this carries the benchmark's run seed.
309
+ # Managed keys are dropped, never overwritten — the contract
310
+ # validator already rejects them, this is defense in depth
311
+ # (an injected HOME/UV_* would defeat per-eval isolation)
312
+ for key, value in extra_env.items():
313
+ if managed_eval_env(key):
314
+ log.warning("refusing extra_env override of managed %s", key)
315
+ continue
316
+ env[key] = value
317
+ if self.container_image:
318
+ env[f"APPTAINERENV_{key}"] = value
319
+ try:
320
+ # process group, like the harness: a timed-out eval must not
321
+ # leave orphans mutating a workspace that later gets committed
322
+ process = subprocess.Popen(
323
+ argv,
324
+ cwd=workspace,
325
+ env=env,
326
+ stdout=subprocess.PIPE,
327
+ stderr=subprocess.PIPE,
328
+ text=True,
329
+ start_new_session=True,
330
+ )
331
+ except OSError as exc:
332
+ raise EvalError(f"eval could not start: {exc}") from exc
333
+ try:
334
+ stdout, stderr = process.communicate(timeout=self.timeout_s)
335
+ except subprocess.TimeoutExpired as exc:
336
+ import contextlib
337
+
338
+ with contextlib.suppress(ProcessLookupError, PermissionError):
339
+ os.killpg(process.pid, signal.SIGKILL)
340
+ try:
341
+ process.communicate(timeout=10)
342
+ except subprocess.TimeoutExpired:
343
+ process.kill()
344
+ with contextlib.suppress(subprocess.TimeoutExpired):
345
+ process.communicate(timeout=5)
346
+ raise EvalError(f"eval timed out after {self.timeout_s}s") from exc
347
+ if process.returncode != 0:
348
+ raise EvalError(f"eval failed ({process.returncode}): {stderr[-500:]}")
349
+ return stdout
350
+
351
+ def check(self, workspace: Path, command: str) -> None:
352
+ """Run `command` with eval-grade containment, requiring only exit 0.
353
+
354
+ The steward's validation suite (pytest, per-benchmark smoke runs)
355
+ executes STEWARD-written env code — same trust level as agent
356
+ code, same containment, no metric parsed."""
357
+ import shutil
358
+ import tempfile
359
+
360
+ try:
361
+ eval_home = Path(
362
+ tempfile.mkdtemp(
363
+ prefix=f"{workspace.name}-check-home-", dir=workspace.resolve().parent
364
+ )
365
+ )
366
+ except OSError as exc:
367
+ raise EvalError(f"could not create check home: {exc}") from exc
368
+ cache_dir = Path(tempfile.mkdtemp(prefix="outerloop-check-cache-"))
369
+ try:
370
+ self._run(workspace, command, eval_home, cache_dir)
371
+ finally:
372
+ shutil.rmtree(eval_home, ignore_errors=True)
373
+ shutil.rmtree(cache_dir, ignore_errors=True)
374
+
375
+ def _parse_measured(self, stdout: str, metric: str) -> float:
376
+ value = metric_from_output(stdout, metric)
377
+ if value is None:
378
+ raise EvalError(f"metric {metric!r} not found in eval output")
379
+ if not math.isfinite(value):
380
+ raise EvalError(f"metric {metric!r} is not finite: {value}")
381
+ return value
382
+
383
+
384
+ def metric_from_output(stdout: str, metric: str) -> float | None:
385
+ """The metric from the LAST single-line JSON object that carries it.
386
+
387
+ No regex fallback: a fuzzy match that reads the wrong number (a progress
388
+ line, a prefixed metric name) is worse than a clean failure — the
389
+ contract requires eval commands to print their metrics as JSON."""
390
+ for line in reversed(stdout.strip().splitlines()):
391
+ line = line.strip()
392
+ if line.startswith("{"):
393
+ try:
394
+ data = json.loads(line)
395
+ except json.JSONDecodeError:
396
+ continue
397
+ if isinstance(data, dict) and metric in data:
398
+ try:
399
+ return float(data[metric])
400
+ except (TypeError, ValueError):
401
+ return None
402
+ # the {"metric": <name>, "value": <v>} shape (what the pilot's
403
+ # eval actually prints)
404
+ if isinstance(data, dict) and data.get("metric") == metric and "value" in data:
405
+ try:
406
+ return float(data["value"])
407
+ except (TypeError, ValueError):
408
+ return None
409
+ return None
410
+
411
+
412
+ def _bot_login_default() -> str:
413
+ """RunConfig's login default, resolved at construction from the one env
414
+ knob; github is imported here on purpose — this module's import graph
415
+ stays free of it."""
416
+ from outerloop.github import bot_login_from_env
417
+
418
+ return bot_login_from_env()
419
+
420
+
421
+ @dataclass(frozen=True)
422
+ class RunConfig:
423
+ target: str # owner/repo
424
+ benchmark: str # the ONE benchmark this loop works on
425
+ agent_id: str = "agent-01"
426
+ # Commits are AUTHORED as the bot account (a real GitHub identity):
427
+ # a bare "agent-01" noreply address links to whoever owns that login.
428
+ # The agent id lives in a commit trailer instead.
429
+ bot_login: str = field(default_factory=_bot_login_default)
430
+ # relative improvement below this is noise, not a PR (ε is contract-
431
+ # configurable later; this is the loop-side floor)
432
+ min_relative_improvement: float = 0.005
433
+ budget: BudgetState = field(default_factory=lambda: BudgetState(0.0, 1))
434
+
435
+ @property
436
+ def branch_prefix(self) -> str:
437
+ # derived, never stored: every call site passed agent_id but left the
438
+ # old field at its default, so every PR branch said agent-01. An id
439
+ # that cannot shape a ref — empty, or a malformed value an old CLI
440
+ # accepted into a record — keeps the old spelling rather than
441
+ # handing git an invalid branch name at publish.
442
+ import re
443
+
444
+ if re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9_-]{0,63}", self.agent_id or ""):
445
+ return f"feat/auto/{self.agent_id}"
446
+ return "feat/auto/agent-01"
447
+
448
+
449
+ @dataclass(frozen=True)
450
+ class SuiteMeasurement:
451
+ """One sibling benchmark's paired measurement from the suite gate."""
452
+
453
+ name: str
454
+ baseline: float
455
+ candidate: float
456
+ regressed: bool
457
+ display_digits: int | None = None
458
+
459
+
460
+ @dataclass(frozen=True)
461
+ class AttemptResult:
462
+ """What one attempt produced — the raw material of the run report."""
463
+
464
+ # improved | no-improvement | session-error | session-budget |
465
+ # session-outage | eval-error | scope-violation | suite-regression
466
+ outcome: str
467
+ baseline: float | None = None
468
+ candidate: float | None = None
469
+ branch: str = ""
470
+ # the exact paths that were scope-checked and then measured — the caller
471
+ # must refuse to commit anything beyond this set
472
+ measured_paths: tuple[str, ...] = ()
473
+ # the sealed candidate snapshot this result was measured on (set on
474
+ # `improved`); a caller can publish THIS tree (the wake path does,
475
+ # attempt.py) and a depth loop can select the best across passes by it.
476
+ # "" on non-improved outcomes.
477
+ candidate_sha: str = ""
478
+ session: SessionResult | None = None
479
+ note: str = ""
480
+ # the author's report at submit (SyscallRequest.report): the PR's research
481
+ # report and what the panel read; empty when the run never submitted one
482
+ submit_report: str = ""
483
+ # the seed both measurements ran under (0 = benchmark has no seed_env):
484
+ # recorded in the ledger row so the number is re-derivable
485
+ run_seed: int = 0
486
+ # sibling measurements when the suite gate ran (shared paths touched);
487
+ # empty when the diff was env-specific or no shared paths are declared
488
+ suite: tuple[SuiteMeasurement, ...] = ()
489
+ # the seed every seeded sibling's pair ran under (0 = gate did not run)
490
+ suite_seed: int = 0
491
+ # the pre-PR panel's record: per-round transcript for the PR body, how
492
+ # many reads ran, whether blocking findings were still open at the cap,
493
+ # and whether the FINAL read was degraded (a lens with no verdict, an
494
+ # unsanitizable tree). Either flag means the caller opens a DRAFT PR
495
+ # and never arms auto-merge.
496
+ panel_transcript: str = ""
497
+ panel_rounds: int = 0
498
+ panel_blocking_open: bool = False
499
+ panel_degraded: bool = False
500
+
501
+ def report(self, config: RunConfig, redact_secrets: tuple[str, ...] = ()) -> str:
502
+ lines = [
503
+ f"# Run report — {config.target} / {config.benchmark}",
504
+ f"Outcome: **{self.outcome}**",
505
+ ]
506
+ if self.baseline is not None:
507
+ lines.append(f"Baseline: {self.baseline}")
508
+ if self.candidate is not None:
509
+ lines.append(f"Candidate: {self.candidate}")
510
+ for row in self.suite:
511
+ verdict = "REGRESSED" if row.regressed else "ok"
512
+ lines.append(f"Suite {row.name}: {row.baseline} -> {row.candidate} ({verdict})")
513
+ if self.panel_rounds:
514
+ if self.panel_blocking_open:
515
+ state = "blocking findings OPEN at the cap"
516
+ elif self.panel_degraded:
517
+ state = "DEGRADED final read (a lens produced no verdict)"
518
+ else:
519
+ state = "clean"
520
+ lines.append(f"Panel: {self.panel_rounds} read(s), {state}")
521
+ if self.note:
522
+ lines.append(f"Note: {self.note}")
523
+ if self.session is not None:
524
+ lines += [
525
+ f"Session: cost=${self.session.cost_usd:.2f}, "
526
+ f"turns={self.session.num_turns}, stop={self.session.stop_reason}",
527
+ "",
528
+ "## Agent's report",
529
+ # redact BEFORE truncating: a secret straddling the cut would
530
+ # otherwise survive as an unmatchable prefix
531
+ redact(self.session.final_text, redact_secrets)[:MAX_REPORT_BODY],
532
+ ]
533
+ return redact("\n".join(lines), redact_secrets)
534
+
535
+
536
+ def _benchmark(contract: Contract, name: str):
537
+ for bench in contract.benchmarks:
538
+ if bench.name == name:
539
+ return bench
540
+ raise ValueError(
541
+ f"benchmark {name!r} not in contract ({[b.name for b in contract.benchmarks]})"
542
+ )
543
+
544
+
545
+ def out_of_scope(paths: Sequence[str], contract: Contract) -> list[str]:
546
+ """Changed paths the contract does not allow the agent to touch.
547
+
548
+ Checked BEFORE the candidate eval: an out-of-scope edit could be to the
549
+ eval harness itself, and measuring a doctored ruler would turn "CI
550
+ re-verifies independently" into re-running the fraud."""
551
+ allowed = [normalize_path(entry) for entry in contract.scope.allowed]
552
+ violations = []
553
+ for path in paths:
554
+ if path_is_forbidden(path, contract):
555
+ violations.append(path)
556
+ continue
557
+ try:
558
+ candidate = normalize_path(path)
559
+ except Exception:
560
+ violations.append(path)
561
+ continue
562
+ if not any(candidate == a or a in candidate.parents for a in allowed):
563
+ violations.append(path)
564
+ return violations
565
+
566
+
567
+ def shared_touched(paths: Sequence[str], contract: Contract) -> list[str]:
568
+ """Changed paths under `scope.shared` — the suite-gate trigger. Runs on
569
+ paths that already passed `out_of_scope`, so unparseable entries are
570
+ simply not shared (they were rejected upstream). Case-folded like the
571
+ forbidden/steward checks: a `Model/` spelling must not dodge the gate."""
572
+ shared = [_fold(normalize_path(entry)) for entry in contract.scope.shared]
573
+ hits = []
574
+ for path in paths:
575
+ try:
576
+ candidate = _fold(normalize_path(path))
577
+ except Exception:
578
+ continue
579
+ if any(candidate == s or s in candidate.parents for s in shared):
580
+ hits.append(path)
581
+ return hits
582
+
583
+
584
+ def steward_out_of_scope(paths: Sequence[str], contract: Contract) -> list[str]:
585
+ """Changed paths the STEWARD may not touch.
586
+
587
+ The inversion of `out_of_scope`: the steward edits the env/ruler
588
+ territory (`contract.steward.allowed`) and may NEVER touch the solver's
589
+ territory (`contract.scope.allowed`) — the roles' separation is what
590
+ makes verifier-checked stewardship trustworthy. The always-forbidden
591
+ set (contract, `.github/`, roadmap) binds here too. No steward section
592
+ in the contract means everything is out of scope.
593
+ """
594
+ if contract.steward is None:
595
+ return list(paths)
596
+ allowed = [_fold(normalize_path(entry)) for entry in contract.steward.allowed]
597
+ solver = [_fold(normalize_path(entry)) for entry in contract.scope.allowed]
598
+ violations = []
599
+ for path in paths:
600
+ if path_is_forbidden(path, contract):
601
+ violations.append(path)
602
+ continue
603
+ try:
604
+ candidate = _fold(normalize_path(path))
605
+ except Exception:
606
+ violations.append(path)
607
+ continue
608
+ # case-folded both directions, like path_is_forbidden: on a
609
+ # case-insensitive checkout, Solvers/ IS solvers/
610
+ if any(candidate == sp or sp in candidate.parents for sp in solver):
611
+ violations.append(path) # solver territory: never the steward's
612
+ continue
613
+ if not any(candidate == a or a in candidate.parents for a in allowed):
614
+ violations.append(path)
615
+ return violations
616
+
617
+
618
+ def draw_run_seed() -> int:
619
+ """A fresh measurement seed, never 0 — zero is the ledger's "no seed
620
+ recorded" sentinel, and the injection guards key off truthiness."""
621
+ return 1 + randbits(30)
622
+
623
+
624
+ def benchmark_floor(
625
+ prior_best: float, min_delta: float | None, min_delta_rel: float | None
626
+ ) -> float:
627
+ """The effective absolute cross-seed floor for a comparison against the
628
+ recorded best. The larger of the absolute floor and the relative floor
629
+ scaled to the level, so a benchmark that sets both gets the more
630
+ conservative one. Returns 0.0 when no floor is declared.
631
+
632
+ A relative-only floor scales to 0 at a recorded level of 0, which means
633
+ no floor. That is a real limit of a relative floor, not a bug: a metric
634
+ that can sit at 0 should pair min_delta_rel with a small absolute
635
+ min_delta as a backstop. Unbounded metrics that use a relative floor
636
+ (wall-clock timing) do not reach 0."""
637
+ floors = []
638
+ if min_delta:
639
+ floors.append(min_delta)
640
+ if min_delta_rel and math.isfinite(prior_best):
641
+ floors.append(min_delta_rel * abs(prior_best))
642
+ return max(floors) if floors else 0.0
643
+
644
+
645
+ def reaches_floor(
646
+ prior: float,
647
+ candidate: float,
648
+ direction: str,
649
+ min_delta: float | None,
650
+ min_delta_rel: float | None,
651
+ ) -> bool:
652
+ """Inclusive floor test in exact decimal arithmetic. Metric values and
653
+ floors arrive as decimal text (eval JSON, the contract's YAML), so the
654
+ comparison is made on the decimals that were written, not on their
655
+ binary approximations: 0.3 - 0.2 is exactly 0.1 here, and there is no
656
+ tolerance for a short delta to hide in at any scale. Non-finite inputs
657
+ fail closed: an infinite floor (`min_delta: .inf` is valid YAML) is
658
+ never reached, and a NaN anywhere is not a measurement. The caller's
659
+ float floor is only for messages."""
660
+ if not all(math.isfinite(v) for v in (prior, candidate, min_delta or 0, min_delta_rel or 0)):
661
+ return False
662
+ p, c = Fraction(repr(prior)), Fraction(repr(candidate))
663
+ delta = c - p if direction == "max" else p - c
664
+ floors = []
665
+ if min_delta:
666
+ floors.append(Fraction(repr(min_delta)))
667
+ if min_delta_rel:
668
+ floors.append(Fraction(repr(min_delta_rel)) * abs(p))
669
+ return delta >= max(floors) if floors else True
670
+
671
+
672
+ def clears_min_delta(
673
+ prior_best: float,
674
+ candidate: float,
675
+ direction: str,
676
+ min_delta: float | None,
677
+ min_delta_rel: float | None = None,
678
+ ) -> bool:
679
+ """Cross-seed comparisons on a resampled pool must reach the
680
+ benchmark's noise floor: the recorded best was measured under a
681
+ different seed, so a delta below the floor is pool luck, not progress.
682
+ The floor is INCLUSIVE — a delta equal to it is credited: the contract
683
+ declares the smallest movement it calls real, and on a quantized metric
684
+ (a step count measured every N steps) the floor IS a reachable value,
685
+ so a strict bar silently demands the next quantum (gpt-speedrun,
686
+ 2026-09-03: three candidates measured exactly one floor better than the
687
+ base were all discarded). Same-seed paired comparisons never call this."""
688
+ if not (min_delta or min_delta_rel):
689
+ return True # no floor declared
690
+ if not (math.isfinite(prior_best) and math.isfinite(candidate)):
691
+ return False # a declared floor with non-finite inputs fails closed
692
+ return reaches_floor(prior_best, candidate, direction, min_delta, min_delta_rel)
693
+
694
+
695
+ def suite_regressed(
696
+ baseline: float,
697
+ candidate: float,
698
+ direction: str,
699
+ min_delta: float | None = None,
700
+ min_delta_rel: float | None = None,
701
+ ) -> bool:
702
+ """Did a sibling benchmark move the WRONG way beyond its own floor?
703
+
704
+ Both sides are same-seed paired, so with no floor declared any wrong-way
705
+ move counts (paired noise is ~0 by construction); a declared floor gives
706
+ a stochastic eval its honest tolerance. Non-finite values fail closed —
707
+ an unmeasurable sibling must never read as "no regression"."""
708
+ if not (math.isfinite(baseline) and math.isfinite(candidate)):
709
+ return True
710
+ drop = baseline - candidate if direction == "max" else candidate - baseline
711
+ if drop <= 0:
712
+ return False
713
+ return drop > benchmark_floor(baseline, min_delta, min_delta_rel)
714
+
715
+
716
+ def improved(baseline: float, candidate: float, direction: str, min_rel: float) -> bool:
717
+ """Direction-aware, threshold-clearing improvement. Non-finite values
718
+ never count (the evaluator rejects them; this is defense in depth)."""
719
+ if not (math.isfinite(baseline) and math.isfinite(candidate)):
720
+ return False
721
+ if baseline == 0:
722
+ # no relative scale exists: apply the threshold absolutely
723
+ return candidate >= min_rel if direction == "max" else candidate <= -min_rel
724
+ rel = (candidate - baseline) / abs(baseline)
725
+ return rel >= min_rel if direction == "max" else rel <= -min_rel
726
+
727
+
728
+ def make_task(
729
+ contract: Contract, benchmark_name: str, baseline: float | None, hypothesis: str = ""
730
+ ) -> Task:
731
+ bench = _benchmark(contract, benchmark_name)
732
+ better = "lower" if bench.direction == "min" else "higher"
733
+ suite_gated = bool(contract.scope.shared) and len(contract.benchmarks) > 1
734
+ # ORIENTATION, not direction: the brief states the current score and how
735
+ # the metric reads as FACTS, and leaves the goal and the finish to the
736
+ # author (research-loop.md, author-directed). The GATE — never the brief —
737
+ # is the real bar: it re-measures both sides after the session, so a
738
+ # missing baseline (a benchmark's first run) just drops the reference
739
+ # number. Naming a target here would only invite optimizing that number.
740
+ current = f"currently {baseline}" if baseline is not None else "no score recorded yet"
741
+ return Task(
742
+ hypothesis=hypothesis
743
+ or (
744
+ f"The {bench.name} solver can be improved: study the current "
745
+ f"implementation and the evaluation, form ONE concrete hypothesis "
746
+ f"for why it underperforms, and implement it."
747
+ ),
748
+ benchmark=bench.name,
749
+ # a fact about the metric, not a target to chase
750
+ expected_effect=f"{bench.metric} ({better} is better), {current}",
751
+ # the finish is the AUTHOR's call; the gate decides what publishes
752
+ done_criteria=(
753
+ "You decide when your result is worth publishing — and a negative "
754
+ "result reported clearly is a success. The orchestrator re-measures "
755
+ f"`{bench.command}` on a private seed to verify any improvement "
756
+ "claim, and the PR's CI runs the repository tests"
757
+ + (
758
+ "; changes touching shared paths are suite-gated, so no sibling "
759
+ "benchmark may regress beyond its floor"
760
+ if suite_gated
761
+ else ""
762
+ )
763
+ + "."
764
+ ),
765
+ )
766
+
767
+
768
+ class Measurer(Protocol):
769
+ """Obtains a climb's measurements. `results` returns every measure's value
770
+ keyed by its name, or — for a dispatched backend — raises
771
+ `MeasurementPending` after submitting the not-yet-done jobs, for the caller
772
+ to park on. `measure.DispatchedMeasurer` runs each measure as a job on a
773
+ fresh checkout of its committed sha — on the cluster, or synchronously via
774
+ `LocalCompute`; the seam is what lets `measure_and_decide` be re-enterable
775
+ without knowing which."""
776
+
777
+ def results(self, measures: list[Measure]) -> dict[str, float]: ...
778
+
779
+
780
+ @dataclass(frozen=True)
781
+ class MeasureOK:
782
+ """A credited measurement: the candidate cleared the improvement threshold
783
+ and, when the diff touched shared code, no sibling regressed. The caller's
784
+ panel/PR path proceeds from here."""
785
+
786
+ baseline: float
787
+ candidate: float
788
+ suite: tuple[SuiteMeasurement, ...] = ()
789
+ suite_seed: int = 0
790
+ # how the baseline number was obtained when it was NOT measured beside
791
+ # this candidate (`baseline: cached`) — surfaces on the credited result
792
+ baseline_note: str = ""
793
+
794
+
795
+ def measure_and_decide(
796
+ contract: Contract,
797
+ bench: Benchmark,
798
+ *,
799
+ base_sha: str,
800
+ candidate_sha: str,
801
+ seed: int,
802
+ suite_seed: int,
803
+ measured_paths: Sequence[str],
804
+ measurer: Measurer,
805
+ min_relative_improvement: float,
806
+ ) -> AttemptResult | MeasureOK:
807
+ """The post-session decision as a PURE function of committed shas and a
808
+ `Measurer` — the re-enterable core a wake reconstructs from the record.
809
+
810
+ Measures baseline@base_sha and candidate@candidate_sha (paired on `seed`,
811
+ common random numbers) and, when the diff touches shared code, each
812
+ sibling's paired base/cand (on the sibling's OWN seed var, all sharing the
813
+ single `suite_seed`), then applies the improvement threshold and the suite
814
+ gate. Returns a TERMINAL `AttemptResult` on any stop (scope-violation,
815
+ eval-error, no-improvement, suite-regression), or a `MeasureOK` carrying
816
+ the credited values. No session and no git: the same shas rebuild the same
817
+ plan and read cached results, so a wake re-enters here unchanged. A
818
+ dispatched measurer's `MeasurementPending` propagates untouched, for the
819
+ caller to write the waiting record and park on.
820
+
821
+ `suite_seed` is an INPUT, not drawn here: a wake must reproduce the seed the
822
+ first pass used, so the caller draws it once and persists it. And because
823
+ the baseline is a committed sha rather than a live pre-session workspace,
824
+ the gate can always measure its siblings.
825
+ """
826
+ # deferred like contract.py's managed_eval_env import: measure -> dispatch
827
+ # -> orchestrator for the eval primitives, so orchestrator imports measure
828
+ # at call time to keep the module graph acyclic.
829
+ from outerloop.measure import (
830
+ SiblingSpec,
831
+ plan_measures,
832
+ read_baseline_cache,
833
+ write_baseline_cache,
834
+ )
835
+
836
+ # Scope BEFORE measurement: an out-of-scope tree is never evaluated,
837
+ # because the out-of-scope edit could be to the ruler itself.
838
+ violations = out_of_scope(list(measured_paths), contract)
839
+ if violations:
840
+ return AttemptResult(
841
+ outcome="scope-violation",
842
+ note=f"out-of-scope paths: {', '.join(sorted(violations)[:10])}",
843
+ run_seed=seed,
844
+ )
845
+
846
+ seed_env = bench.seed_env or ""
847
+ siblings = [b for b in contract.benchmarks if b.name != bench.name]
848
+ # Both seed guards fire UP FRONT, before any (expensive) measurement: a
849
+ # seed of 0 — the "no seed recorded" sentinel — would run a pair UNPAIRED,
850
+ # each side drawing its own internal seed, so eval noise could read as
851
+ # improvement. The caller must draw a real seed (and persist it, for the
852
+ # wake to reuse) whenever a benchmark or a sibling declares seed_env; a
853
+ # seeded sibling's suite_seed is a caller-contract precondition even on a
854
+ # diff that won't touch shared code (the next diff might).
855
+ if seed_env and not seed:
856
+ raise ValueError(
857
+ f"benchmark {bench.name!r} declares seed_env {seed_env!r} but seed is 0: "
858
+ "a seeded benchmark needs a drawn seed, or baseline and candidate run unpaired"
859
+ )
860
+ # only when the suite gate is STRUCTURALLY possible: a contract with an
861
+ # empty scope.shared can never trigger it (shared_touched always empty), so
862
+ # a missing suite_seed there is not a misconfiguration — do not over-reject.
863
+ if contract.scope.shared and any(b.seed_env for b in siblings) and not suite_seed:
864
+ raise ValueError(
865
+ "a seeded sibling needs a nonzero suite_seed (drawn once, persisted for the wake); "
866
+ "suite_seed 0 would run the sibling pair unpaired"
867
+ )
868
+
869
+ # PHASE 1 — baseline + candidate only. Siblings are NOT measured until the
870
+ # candidate has cleared the threshold: a non-improving candidate must never
871
+ # burn the (expensive) sibling evals, matching attempt_once's lazy order.
872
+ # `baseline: cached` — the base tree is measured ONCE per (benchmark, base
873
+ # sha) into the target's cache and reused by every attempt on that base;
874
+ # only the candidate runs. Unpaired, so the contract's floor carries the
875
+ # cross-seed noise (the loader requires one). A miss measures both and
876
+ # records the baseline for the next attempt.
877
+ cache_dir = getattr(measurer, "baseline_cache", None)
878
+ image = str(getattr(measurer, "image", ""))
879
+ cached = (
880
+ read_baseline_cache(
881
+ cache_dir,
882
+ bench.name,
883
+ base_sha,
884
+ image=image,
885
+ command=bench.command,
886
+ metric=bench.metric,
887
+ seed_env=bench.seed_env or "",
888
+ gpus=bench.gpus,
889
+ )
890
+ if bench.baseline == "cached" and cache_dir is not None
891
+ else None
892
+ )
893
+ plan = plan_measures(
894
+ bench.command,
895
+ bench.metric,
896
+ base_sha,
897
+ candidate_sha,
898
+ seed_env,
899
+ seed,
900
+ gpus=bench.gpus,
901
+ )
902
+ if cached is not None:
903
+ plan = [m for m in plan if m.name != "baseline"]
904
+ try:
905
+ main = measurer.results(plan)
906
+ except EvalError as exc:
907
+ return AttemptResult(outcome="eval-error", note=str(exc), run_seed=seed)
908
+
909
+ baseline_note = ""
910
+ if cached is not None:
911
+ baseline = float(cached["value"])
912
+ baseline_note = (
913
+ f"baseline {baseline} reused from the cache (measured at seed "
914
+ f"{cached.get('seed')} by run {cached.get('run')}); candidate at seed {seed}"
915
+ )
916
+ else:
917
+ baseline = main["baseline"]
918
+ if bench.baseline == "cached" and cache_dir is not None:
919
+ write_baseline_cache(
920
+ cache_dir,
921
+ bench.name,
922
+ base_sha,
923
+ value=baseline,
924
+ seed=seed,
925
+ run_tag=str(getattr(measurer, "run_tag", "")),
926
+ image=image,
927
+ command=bench.command,
928
+ metric=bench.metric,
929
+ seed_env=bench.seed_env or "",
930
+ gpus=bench.gpus,
931
+ )
932
+ candidate = main["candidate"]
933
+ if not improved(baseline, candidate, bench.direction, min_relative_improvement):
934
+ return AttemptResult(
935
+ outcome="no-improvement",
936
+ baseline=baseline,
937
+ candidate=candidate,
938
+ run_seed=seed,
939
+ )
940
+ # The contract's OWN significance floor, when the benchmark declares one.
941
+ # Same-seed pairing removes pool noise but not training stochasticity —
942
+ # a benchmark whose eval trains models calibrates min_delta to its
943
+ # cross-run sd, and the gate must speak that language too (yolo#16
944
+ # published at +0.0379 against a declared 0.04 floor because only the
945
+ # followup path read it). No declared floor -> the relative default
946
+ # above remains the whole bar.
947
+ floor = benchmark_floor(baseline, bench.min_delta, bench.min_delta_rel)
948
+ if floor:
949
+ delta = (candidate - baseline) if bench.direction == "max" else (baseline - candidate)
950
+ # INCLUSIVE, matching clears_min_delta: the floor is the smallest
951
+ # movement the contract calls real, so a delta equal to it clears
952
+ if not reaches_floor(
953
+ baseline, candidate, bench.direction, bench.min_delta, bench.min_delta_rel
954
+ ):
955
+ return AttemptResult(
956
+ outcome="no-improvement",
957
+ note=(
958
+ f"delta {delta:+.6g} is inside the contract's significance "
959
+ f"floor ({floor:g}): real movement, not creditable progress"
960
+ ),
961
+ baseline=baseline,
962
+ candidate=candidate,
963
+ run_seed=seed,
964
+ )
965
+
966
+ # PHASE 2 — suite gate, only for a credited candidate whose diff touched
967
+ # shared code. A second measure set (a second park for a dispatched
968
+ # backend): an extra CPU wake, never a wasted GPU sibling eval.
969
+ if not (siblings and shared_touched(measured_paths, contract)):
970
+ return MeasureOK(baseline=baseline, candidate=candidate, baseline_note=baseline_note)
971
+
972
+ # every seeded sibling runs its pair under the ONE suite_seed, read through
973
+ # its own seed var (mirrors the in-job gate).
974
+ sib_specs = tuple(
975
+ SiblingSpec(
976
+ b.name,
977
+ b.command,
978
+ b.metric,
979
+ seed_env=b.seed_env or "",
980
+ seed=suite_seed,
981
+ gpus=b.gpus, # each sibling on ITS lane, not the climbed benchmark's
982
+ )
983
+ for b in siblings
984
+ )
985
+ sib_plan = [
986
+ m
987
+ for m in plan_measures(
988
+ bench.command,
989
+ bench.metric,
990
+ base_sha,
991
+ candidate_sha,
992
+ seed_env,
993
+ seed,
994
+ siblings=sib_specs,
995
+ gpus=bench.gpus,
996
+ )
997
+ if m.name.startswith("sib-") # baseline/candidate already measured (phase 1)
998
+ ]
999
+ try:
1000
+ vals = measurer.results(sib_plan)
1001
+ except EvalError as exc:
1002
+ # phase 1 already credited the main pair — carry it into the report.
1003
+ return AttemptResult(
1004
+ outcome="eval-error",
1005
+ baseline=baseline,
1006
+ candidate=candidate,
1007
+ note=str(exc),
1008
+ run_seed=seed,
1009
+ )
1010
+
1011
+ suite_rows: list[SuiteMeasurement] = []
1012
+ for b in siblings:
1013
+ sib_base = vals[f"sib-{b.name}-base"]
1014
+ sib_cand = vals[f"sib-{b.name}-cand"]
1015
+ suite_rows.append(
1016
+ SuiteMeasurement(
1017
+ name=b.name,
1018
+ baseline=sib_base,
1019
+ candidate=sib_cand,
1020
+ regressed=suite_regressed(
1021
+ sib_base, sib_cand, b.direction, b.min_delta, b.min_delta_rel
1022
+ ),
1023
+ display_digits=b.display_digits,
1024
+ )
1025
+ )
1026
+ suite = tuple(suite_rows)
1027
+ regressed = [r for r in suite if r.regressed]
1028
+ if regressed:
1029
+ named = ", ".join(f"{r.name} {r.baseline} -> {r.candidate}" for r in regressed)
1030
+ return AttemptResult(
1031
+ outcome="suite-regression",
1032
+ baseline=baseline,
1033
+ candidate=candidate,
1034
+ note=f"shared-path diff regressed sibling benchmark(s): {named}",
1035
+ run_seed=seed,
1036
+ suite=suite,
1037
+ suite_seed=suite_seed,
1038
+ )
1039
+
1040
+ return MeasureOK(
1041
+ baseline=baseline,
1042
+ candidate=candidate,
1043
+ suite=suite,
1044
+ suite_seed=suite_seed, # reached only on the suite path
1045
+ baseline_note=baseline_note,
1046
+ )
1047
+
1048
+
1049
+ def resume_attempt(
1050
+ contract: Contract,
1051
+ bench: Benchmark,
1052
+ *,
1053
+ base_sha: str,
1054
+ candidate_sha: str,
1055
+ seed: int,
1056
+ suite_seed: int,
1057
+ measured_paths: Sequence[str],
1058
+ session: SessionResult,
1059
+ measurer: Measurer,
1060
+ min_relative_improvement: float,
1061
+ ) -> AttemptResult:
1062
+ """Re-enter a parked climb's post-session decision — the WAKE side of a
1063
+ dispatched candidate park. The candidate is already committed (its sha is in
1064
+ the record), so the session does NOT re-run: its edits are captured in
1065
+ `candidate_sha` and it was reconstructed by the caller from the record
1066
+ (`session.final_text` is the saved write-up, and its cost/turns the saved
1067
+ spend) so the PR body and panel claim need no live session. `measured_paths`
1068
+ is the caller's re-derivation of the `base_sha..candidate_sha` diff.
1069
+
1070
+ The measurer reads the cached eval results and this returns the decision, OR
1071
+ the decision needs a measure not yet done — the suite pairs after an
1072
+ improving candidate, "another round of experiments" — and it re-parks by
1073
+ raising `RunParked`, exactly as the first pass did.
1074
+
1075
+ This is the re-entry seam the depth axis (docs/design/research-loop.md)
1076
+ builds on.
1077
+ """
1078
+ from outerloop.measure import MeasurementPending
1079
+
1080
+ try:
1081
+ outcome = measure_and_decide(
1082
+ contract,
1083
+ bench,
1084
+ base_sha=base_sha,
1085
+ candidate_sha=candidate_sha,
1086
+ seed=seed,
1087
+ suite_seed=suite_seed,
1088
+ measured_paths=measured_paths,
1089
+ measurer=measurer,
1090
+ min_relative_improvement=min_relative_improvement,
1091
+ )
1092
+ except MeasurementPending as pending:
1093
+ # a measure is not done yet (the suite pairs this wake just dispatched):
1094
+ # re-park on the new afterany set, same shape as the first candidate park
1095
+ raise RunParked(
1096
+ phase="candidate",
1097
+ afterany=pending.afterany(),
1098
+ base_sha=base_sha,
1099
+ seed=seed,
1100
+ suite_seed=suite_seed,
1101
+ candidate_sha=candidate_sha,
1102
+ session=session,
1103
+ ) from None
1104
+ if isinstance(outcome, AttemptResult):
1105
+ # a terminal measurement outcome (no-improvement / suite-regression /
1106
+ # eval-error): carry the reconstructed session, and give a bare
1107
+ # no-improvement the same framing the in-job path sets (a clear
1108
+ # negative is a success), so a resumed negative does not end note-less.
1109
+ note = outcome.note
1110
+ if outcome.outcome == "no-improvement" and not note:
1111
+ # only a BARE negative gets the generic framing — a specific
1112
+ # reason (e.g. inside the contract's significance floor) must
1113
+ # survive to the record and report
1114
+ note = "a negative result reported clearly is a success"
1115
+ return dc_replace(outcome, session=session, note=note)
1116
+ # credited: candidate cleared the threshold and no sibling regressed. The
1117
+ # caller sets `branch` when it opens the PR.
1118
+ return AttemptResult(
1119
+ outcome="improved",
1120
+ baseline=outcome.baseline,
1121
+ candidate=outcome.candidate,
1122
+ session=session,
1123
+ measured_paths=tuple(measured_paths),
1124
+ run_seed=seed,
1125
+ suite=outcome.suite,
1126
+ suite_seed=outcome.suite_seed,
1127
+ candidate_sha=candidate_sha,
1128
+ note=outcome.baseline_note,
1129
+ )
1130
+
1131
+
1132
+ def _merge_afterany(*parts: str) -> str:
1133
+ """Join afterany dependency strings ("afterany:1:2") into one; "" parts
1134
+ drop out. A submit's park waits on the gate's evals AND its sibling
1135
+ launches together."""
1136
+ ids = [t for part in parts for t in part.split(":")[1:] if t]
1137
+ return "afterany:" + ":".join(ids) if ids else ""
1138
+
1139
+
1140
+ def attempt_once(
1141
+ config: RunConfig,
1142
+ contract_text: str,
1143
+ workspace: Path,
1144
+ harness: Harness,
1145
+ measurer: Measurer,
1146
+ base_sha: str,
1147
+ snapshot: Callable[[], str],
1148
+ ruler: str,
1149
+ changed_paths: Callable[[], Sequence[str]],
1150
+ lessons: str = "",
1151
+ recent_reports: tuple[str, ...] = (),
1152
+ report_archive: bool = False,
1153
+ created: str = "",
1154
+ task_hypothesis: str = "",
1155
+ spec: RoleSpec | None = None,
1156
+ panel_runner: Callable[[float, float, str], PanelVerdict] | None = None,
1157
+ brief_baseline: float | None = None,
1158
+ line_ref: str = "",
1159
+ line_memory: str = "",
1160
+ line_divergence: str = "",
1161
+ resume_session_id: str = "",
1162
+ improve_prompt: str = "",
1163
+ launcher: Callable[[str, SyscallRequest], str] | None = None,
1164
+ # the session watcher: a context manager around each harness run (None =
1165
+ # no watcher in this deployment); docs/design/session-watcher.md
1166
+ watcher: Callable[[], contextlib.AbstractContextManager[Any]] | None = None,
1167
+ launches_used: int = 0,
1168
+ sleeps_used: int = 0,
1169
+ gpu_hours_used: float = 0.0,
1170
+ tree_of: Callable[[str], str] | None = None,
1171
+ judged: tuple[str, AttemptResult] | None = None,
1172
+ ) -> AttemptResult:
1173
+ """One implement→evaluate→verify cycle in an existing clean workspace.
1174
+
1175
+ The caller owns the git side (clone before, diff/commit/push/PR after) —
1176
+ same split as the harness: this function owns the science loop only. It
1177
+ measures through a `Measurer` over committed shas: `base_sha` is the
1178
+ pre-session tree, and `snapshot()` — a caller callback, since snapshotting
1179
+ is git — commits the session's current workspace and returns its
1180
+ `candidate_sha`. The caller registers each sha's worktree with the measurer
1181
+ and owns the snapshot ref lifecycle. `changed_paths` reports every path the
1182
+ session touched (the caller wires it to `git add -A` + staged paths); scope
1183
+ is enforced on it BEFORE the candidate eval runs.
1184
+
1185
+ The session runs as the author role on the role-runner (`spec` defaults
1186
+ to `author_spec`; the caller that built the harness passes its spec so
1187
+ manifest and harness agree). Scope enforcement stays HERE, on the
1188
+ contract — the spec's scope is the manifest copy, filled from the same
1189
+ contract.
1190
+
1191
+ With `panel_runner` (docs/design/orchestrator-verify.md), a credited
1192
+ claim is read by the verification panel BEFORE it can become a PR. On a
1193
+ SUBMITTED claim (the author's `submit` syscall), blocking findings and the
1194
+ gate result go back to the AUTHOR — it revises and resubmits, or concludes
1195
+ (research-loop-buildout.md Phase B); on a plain finish, blocking findings
1196
+ set `panel_blocking_open` (the caller posts a DRAFT PR carrying them).
1197
+ The caller supplies the runner because the panel's checkouts are git work
1198
+ (this function owns no git).
1199
+
1200
+ With `launcher` (author syscalls, research-loop-buildout.md Phase A), a
1201
+ session that ends having asked to launch-and-sleep parks this climb as
1202
+ `author-sleep` instead of measuring: the tree is sealed via `snapshot()`,
1203
+ the launcher submits the jobs (caller-owned — it is compute work), and the
1204
+ RunParked carries the request plus the budget counts. A `submit` rides
1205
+ the same request: the gate (and any sibling launches) run on the sealed
1206
+ tree — a dispatched gate parks as a `candidate` with the `submitted`
1207
+ marker, an inline gate feeds back in-session. `launches_used` /
1208
+ `sleeps_used` are the counts so far (a wake passes them from the stage);
1209
+ the budgets come from the benchmark's `depth_k` / `sleep_k`.
1210
+ """
1211
+ contract = load_contract(contract_text, config.target)
1212
+ bench = _benchmark(contract, config.benchmark)
1213
+ spec = spec or author_spec()
1214
+ if not spec.execution.can_execute:
1215
+ raise ValueError("attempt_once runs an editing role; the spec must allow execution")
1216
+ if not spec.scope:
1217
+ spec = dc_replace(spec, scope=tuple(contract.scope.allowed))
1218
+
1219
+ # the resume-entry (cumulative depth) is a COUPLED pair: it needs both a
1220
+ # session to resume and an instruction to resume with. Reject either alone
1221
+ # loudly — a lone improve_prompt would be silently discarded by the fresh-brief
1222
+ # branch (a depth pass turning into a fresh attempt behind the caller's back),
1223
+ # a lone session id would burn a promptless turn. And reject a resume on a
1224
+ # no-resume backend rather than let the climb end as `session-error`. The depth
1225
+ # loop (caller) owns WHEN to resume; this validates that choice — it never
1226
+ # silently falls back to a fresh brief.
1227
+ if bool(resume_session_id) != bool(improve_prompt):
1228
+ raise ValueError("resume_session_id and improve_prompt must be given together")
1229
+ if resume_session_id and not getattr(harness, "supports_resume", True):
1230
+ # same optional-attr idiom as the panel policy
1231
+ raise ValueError("resume_session_id given but the harness does not support resume")
1232
+ if resume_session_id and (
1233
+ task_hypothesis
1234
+ or lessons
1235
+ or recent_reports
1236
+ or report_archive
1237
+ or created
1238
+ or line_ref
1239
+ or line_memory
1240
+ or line_divergence
1241
+ or brief_baseline is not None
1242
+ ):
1243
+ # a resume skips build_brief entirely (the session already carries this
1244
+ # context from its first pass), so these brief-only inputs would be
1245
+ # silently dropped — the same silent-discard hazard the coupling check
1246
+ # above prevents. Reject them loudly; a resume pass is lean by design.
1247
+ # This is the EXHAUSTIVE set of OPTIONAL brief-only params: a new one added
1248
+ # to build_brief must be added here too. Required params that also feed only
1249
+ # the brief are NOT guarded because a caller cannot omit them — `ruler`, and
1250
+ # the brief-only fields of the required `config` (e.g. `config.budget`) — so
1251
+ # they are unavoidably passed and simply ignored on a resume. `contract_text`
1252
+ # is NOT brief-only — scope/gate use it either way.
1253
+ raise ValueError(
1254
+ "resume_session_id resumes an existing session (no fresh brief); the brief-only "
1255
+ "inputs (task_hypothesis/lessons/recent_reports/report_archive/created/"
1256
+ "line_ref/line_memory/line_divergence/brief_baseline) have no effect on a "
1257
+ "resume — omit them"
1258
+ )
1259
+
1260
+ def _watched() -> contextlib.AbstractContextManager[Any]:
1261
+ return watcher() if watcher is not None else contextlib.nullcontext()
1262
+
1263
+ # deferred like measure_and_decide's import (measure -> dispatch ->
1264
+ # orchestrator for the eval primitives).
1265
+ from outerloop.measure import MeasurementPending
1266
+
1267
+ run_seed = draw_run_seed() if bench.seed_env else 0
1268
+ siblings = [b for b in contract.benchmarks if b.name != bench.name]
1269
+ # ONE suite_seed for the whole climb, fixed up front so a wake reproduces it
1270
+ # (never a re-draw), and only when a seeded sibling could gate. It REUSES the
1271
+ # climbed benchmark's run_seed when there is one (as the in-job gate did),
1272
+ # drawing a fresh seed only for an unseeded benchmark.
1273
+ suite_seed = (
1274
+ (run_seed or draw_run_seed())
1275
+ if contract.scope.shared and any(b.seed_env for b in siblings)
1276
+ else 0
1277
+ )
1278
+
1279
+ # The baseline is NOT measured before the session — it is measured by the
1280
+ # GATE (`measure_and_decide`, base_sha vs candidate_sha) after the session,
1281
+ # so a dispatched climb has ONE park (the candidate), never a pre-session
1282
+ # baseline park. `brief_baseline` is the last-known score from the ledger,
1283
+ # for orienting the brief only ("improve from ~13.8"); it is None on a
1284
+ # benchmark's first run, and the gate re-measures either way.
1285
+ baseline: float | None = brief_baseline
1286
+ if resume_session_id:
1287
+ # a cumulative depth pass (research-loop-buildout.md, Phase 2a): resume the
1288
+ # prior session with the improve prompt instead of a fresh brief, so the
1289
+ # author builds on — and sees the measured result of — its own last pass.
1290
+ with _watched():
1291
+ role_result = run_role(
1292
+ spec, harness, improve_prompt, workspace, resume_session_id=resume_session_id
1293
+ )
1294
+ else:
1295
+ task = make_task(contract, config.benchmark, baseline, hypothesis=task_hypothesis)
1296
+ brief = build_brief(
1297
+ BriefInputs(
1298
+ task=task,
1299
+ contract_text=contract_text,
1300
+ ruler=ruler,
1301
+ lessons=lessons,
1302
+ recent_reports=recent_reports,
1303
+ report_archive=report_archive,
1304
+ budget=config.budget,
1305
+ # the launch/sleep tool is advertised ONLY when it is wired
1306
+ # (never a tool the author cannot actually call)
1307
+ launch_budget=bench.depth_k if launcher is not None else 0,
1308
+ sleep_budget=bench.sleep_k if launcher is not None else 0,
1309
+ # GPU benchmarks: the compute meter the author budgets against
1310
+ gpu_hour_budget=(
1311
+ contract.budgets.gpu_hours_per_run
1312
+ if launcher is not None and bench.gpus
1313
+ else 0.0
1314
+ ),
1315
+ eval_minutes_default=bench.eval_minutes or 0,
1316
+ line_ref=line_ref,
1317
+ memory=line_memory,
1318
+ line_divergence=line_divergence,
1319
+ ),
1320
+ created=created,
1321
+ )
1322
+ with _watched():
1323
+ role_result = run_role(spec, harness, render(brief), workspace)
1324
+ session = role_result.session
1325
+ if not role_result.ok:
1326
+ # the role-runner's verdict, not just the raw session flag (for a
1327
+ # schema-less role they coincide today, but any failure the runner
1328
+ # learns to report must not slip through as a clean run).
1329
+ # Our caps running out is a budget ending, not a malfunction; the
1330
+ # API refusing us is an outage — neither is the run's own failure.
1331
+ if outage(session):
1332
+ kind = "session-outage"
1333
+ elif budget_exhausted(session):
1334
+ kind = "session-budget"
1335
+ else:
1336
+ kind = "session-error"
1337
+ return AttemptResult(
1338
+ outcome=kind,
1339
+ baseline=baseline,
1340
+ session=session,
1341
+ note=role_result.error or session.error_detail or session.stop_reason,
1342
+ )
1343
+
1344
+ # --- the decision loop (research-loop.md, "one syscall"; buildout A+B) ---
1345
+ # Each pass: honor the session's syscall request, then measure the tree.
1346
+ # An enabled author (the caller wired a `launcher`) may end its session —
1347
+ # or a resumed leg of it — having asked to LAUNCH work outside the sandbox
1348
+ # and SLEEP on it (seal the tree, submit through the launcher, park; the
1349
+ # wake re-enters THIS function through the resume-entry so the whole tail
1350
+ # composes unchanged), and/or to SUBMIT its candidate: the gate and any
1351
+ # sibling launches run on the sealed tree — a dispatched gate parks as a
1352
+ # `candidate` with the `submitted` marker (the wake delivers gate + panel
1353
+ # back to the author), an inline gate feeds back in-session and the loop
1354
+ # re-reads the author's next move. Over budget: wake the author ONCE with
1355
+ # a refusal so it can conclude honestly; a session that over-asks again
1356
+ # proceeds to measurement with the tree as it stands (bounded, never an
1357
+ # endless refuse/re-ask loop). With no launcher the feature is off and a
1358
+ # stray request file is just an untracked file (excluded from the diff
1359
+ # either way).
1360
+ panel_reads = 0
1361
+ panel_sections: list[str] = []
1362
+ panel_blocking_open = False
1363
+ panel_degraded = False
1364
+ candidate: float | None = baseline
1365
+ suite: tuple[SuiteMeasurement, ...] = ()
1366
+ suite_seed_ran = 0
1367
+ baseline_note = ""
1368
+ measured: tuple[str, ...] = ()
1369
+ refused_once = False
1370
+ # the gate's last negative and the sealed tree it judged: sealing the same
1371
+ # content again (the author concluded, or resubmitted untouched) reuses
1372
+ # that verdict rather than paying for a second, identical measurement
1373
+ # (`judged` is the wake's: the parked candidate the gate turned down)
1374
+ failed_gate: tuple[str, AttemptResult] | None = judged
1375
+ tree = tree_of or (lambda sha: sha)
1376
+
1377
+ def _resume(prompt: str) -> AttemptResult | None:
1378
+ """Resume the author session with `prompt`: None on success (session
1379
+ advanced), else the terminal AttemptResult for the failed resume."""
1380
+ nonlocal session
1381
+ # the tool the author is about to use is this kernel's, whatever the
1382
+ # session started with (a wake refreshed it too; this covers a refusal)
1383
+ with contextlib.suppress(Exception):
1384
+ refresh_tool(workspace)
1385
+ with _watched():
1386
+ wake_result = run_role(
1387
+ spec, harness, prompt, workspace, resume_session_id=session.session_id
1388
+ )
1389
+ session = wake_result.session
1390
+ if wake_result.ok:
1391
+ return None
1392
+ if outage(session):
1393
+ kind = "session-outage"
1394
+ elif budget_exhausted(session):
1395
+ kind = "session-budget"
1396
+ else:
1397
+ kind = "session-error"
1398
+ return AttemptResult(
1399
+ outcome=kind,
1400
+ baseline=baseline,
1401
+ candidate=candidate,
1402
+ session=session,
1403
+ note=wake_result.error or session.error_detail or session.stop_reason,
1404
+ run_seed=run_seed,
1405
+ panel_transcript="\n\n".join(panel_sections),
1406
+ panel_rounds=panel_reads,
1407
+ )
1408
+
1409
+ def _can_resume() -> bool:
1410
+ return bool(session.session_id) and getattr(harness, "supports_resume", True)
1411
+
1412
+ def _budgets_line() -> str:
1413
+ gpu = (
1414
+ f", {max(0.0, contract.budgets.gpu_hours_per_run - gpu_hours_used):.1f} GPU-hours"
1415
+ if bench.gpus
1416
+ else ""
1417
+ )
1418
+ return (
1419
+ f"Budgets: {max(0, bench.depth_k - launches_used)} launches and "
1420
+ f"{max(0, bench.sleep_k - sleeps_used)} sleeps{gpu} remaining."
1421
+ )
1422
+
1423
+ def _not_run_note(request: SyscallRequest | None) -> str:
1424
+ # inline gates never dispatch a submit's sibling launches (nothing
1425
+ # would gather them) — tell the author; their budget was not spent
1426
+ if request is None or not request.launches:
1427
+ return ""
1428
+ return (
1429
+ "Your sibling launches did NOT run (the gate completed inline); "
1430
+ "stage them again if still needed. "
1431
+ )
1432
+
1433
+ while True:
1434
+ # the syscall request the session's last leg left, if any
1435
+ submitted: SyscallRequest | None = None
1436
+ evals_charge = 0.0 # GPU-hours this pass took for gate evals
1437
+ presealed = "" # a seal taken early to compare against the judged tree
1438
+ while launcher is not None:
1439
+ try:
1440
+ request = read_syscall_request(workspace)
1441
+ except SyscallError as exc:
1442
+ # loud, never silent: the author meant something by the file
1443
+ return AttemptResult(
1444
+ outcome="session-error",
1445
+ baseline=baseline,
1446
+ session=session,
1447
+ note=f"unhonorable syscall request: {exc}",
1448
+ )
1449
+ if request is None:
1450
+ break
1451
+ # suite siblings' paired evals are charged as if measured (the
1452
+ # suite phase decides at measurement; a budget over-charges)
1453
+ suite_gpus = tuple(b.gpus for b in contract.benchmarks if b.name != bench.name)
1454
+ # a `baseline: cached` gate with a warm cache runs ONE main eval
1455
+ # (the candidate); charge what will actually run (terra #178)
1456
+ main_evals = 2
1457
+ if request.submit and bench.baseline == "cached":
1458
+ from outerloop.measure import read_baseline_cache
1459
+
1460
+ cache_dir = getattr(measurer, "baseline_cache", None)
1461
+ if cache_dir is not None and read_baseline_cache(
1462
+ cache_dir,
1463
+ bench.name,
1464
+ base_sha,
1465
+ image=str(getattr(measurer, "image", "")),
1466
+ command=bench.command,
1467
+ metric=bench.metric,
1468
+ seed_env=bench.seed_env or "",
1469
+ gpus=bench.gpus,
1470
+ ):
1471
+ main_evals = 1
1472
+ # a report-less resubmit never rides the failed-gate fast path: it
1473
+ # falls through to the refusal below like any other missing report
1474
+ if request.submit and request.report and failed_gate is not None:
1475
+ # a resubmit of the tree the gate already turned down: nothing
1476
+ # to budget or charge — the verdict is reused below (the sleep
1477
+ # still counts, so unchanged resubmits stay bounded). An eval
1478
+ # that ERRORED is the exception: resubmitting is how the author
1479
+ # retries it (with more minutes, say), so that one runs. The
1480
+ # early seal keeps the later seal's guards: scope first, and a
1481
+ # failed snapshot is the eval error it always was.
1482
+ violations = out_of_scope(list(changed_paths()), contract)
1483
+ if violations:
1484
+ return AttemptResult(
1485
+ outcome="scope-violation",
1486
+ baseline=baseline,
1487
+ session=session,
1488
+ note=f"out-of-scope paths: {', '.join(sorted(violations)[:10])}",
1489
+ run_seed=run_seed,
1490
+ panel_transcript="\n\n".join(panel_sections),
1491
+ panel_rounds=panel_reads,
1492
+ )
1493
+ try:
1494
+ presealed = snapshot()
1495
+ except EvalError as exc:
1496
+ return AttemptResult(
1497
+ outcome="eval-error",
1498
+ baseline=baseline,
1499
+ session=session,
1500
+ note=f"snapshot: {exc}",
1501
+ run_seed=run_seed,
1502
+ panel_transcript="\n\n".join(panel_sections),
1503
+ panel_rounds=panel_reads,
1504
+ )
1505
+ if failed_gate[1].outcome != "eval-error" and tree(failed_gate[0]) == tree(
1506
+ presealed
1507
+ ):
1508
+ submitted = request
1509
+ sleeps_used += 1
1510
+ break
1511
+ problem = syscall_budget_error(
1512
+ request,
1513
+ launches_used=launches_used,
1514
+ launch_budget=bench.depth_k,
1515
+ sleeps_used=sleeps_used,
1516
+ sleep_budget=bench.sleep_k,
1517
+ gpu_hours_used=gpu_hours_used,
1518
+ gpu_hour_budget=contract.budgets.gpu_hours_per_run,
1519
+ gpus=bench.gpus,
1520
+ eval_minutes_default=bench.eval_minutes or 0,
1521
+ suite_gpus=suite_gpus,
1522
+ main_evals=main_evals,
1523
+ )
1524
+ if not problem and request.submit and not request.report:
1525
+ # a refusal the author can act on, never a dead run: a session
1526
+ # that started under an older tool learns the flag here (the wake
1527
+ # refreshed its tool) and resubmits with the report
1528
+ problem = MISSING_REPORT
1529
+ # a sweep's pace is clamped to the contract's GPU ceiling here, once,
1530
+ # before either path — an author-sleep launch or a submit's sibling
1531
+ # launches — submits or records it; clamped, never refused
1532
+ request = clamp_concurrency(
1533
+ request, gpus=bench.gpus, max_concurrent_gpus=contract.budgets.max_concurrent_gpus
1534
+ )
1535
+ if not problem:
1536
+ if request.submit:
1537
+ # a submit rides the measurement below on the SEALED tree —
1538
+ # "a launch whose job is the gate" (buildout Phase B). The
1539
+ # sleep it rides on is counted now; sibling launches are
1540
+ # dispatched (and counted, and CHARGED) only if the gate
1541
+ # parks — an inline gate must never orphan launch jobs no
1542
+ # wake would gather. The gate's evals are charged here, at
1543
+ # the walltime THIS submit declares (else the contract's):
1544
+ # a resubmit without a declaration reverts to the default,
1545
+ # never inheriting a prior park's.
1546
+ submitted = request
1547
+ sleeps_used += 1
1548
+ evals_charge = evals_gpu_hours(
1549
+ request,
1550
+ gpus=bench.gpus,
1551
+ eval_minutes_default=bench.eval_minutes or 0,
1552
+ suite_gpus=suite_gpus,
1553
+ main_evals=main_evals,
1554
+ )
1555
+ gpu_hours_used += evals_charge
1556
+ if hasattr(measurer, "eval_minutes"):
1557
+ measurer.eval_minutes = request.eval_minutes or bench.eval_minutes or 0
1558
+ break
1559
+ # a launch park: its launches are dispatched right below, so
1560
+ # they are charged now
1561
+ gpu_hours_used += launches_gpu_hours(request, gpus=bench.gpus)
1562
+ # Scope BEFORE the snapshot, same invariant as the candidate
1563
+ # path below: an out-of-scope tree is never snapshotted OR
1564
+ # executed — the out-of-scope edit could be to the ruler
1565
+ # itself, and a launch runs code from this tree in an external
1566
+ # job. Same ending as the candidate path.
1567
+ violations = out_of_scope(list(changed_paths()), contract)
1568
+ if violations:
1569
+ return AttemptResult(
1570
+ outcome="scope-violation",
1571
+ baseline=baseline,
1572
+ session=session,
1573
+ note=(
1574
+ f"out-of-scope paths at launch: {', '.join(sorted(violations)[:10])}"
1575
+ ),
1576
+ run_seed=run_seed,
1577
+ )
1578
+ sha = snapshot()
1579
+ launch_afterany = launcher(sha, request)
1580
+ raise RunParked(
1581
+ phase="author-sleep",
1582
+ judged=failed_gate,
1583
+ afterany=launch_afterany,
1584
+ launch_afterany=launch_afterany,
1585
+ base_sha=base_sha,
1586
+ seed=run_seed,
1587
+ suite_seed=suite_seed,
1588
+ candidate_sha=sha,
1589
+ session=session,
1590
+ syscall=request,
1591
+ launches_used=launches_used + len(request.launches),
1592
+ sleeps_used=sleeps_used + 1,
1593
+ gpu_hours_used=gpu_hours_used,
1594
+ )
1595
+ if refused_once or not _can_resume():
1596
+ log.warning("syscall request dropped after refusal (%s); measuring as-is", problem)
1597
+ break
1598
+ # the refusal burns no count (nothing was launched, nothing woke a
1599
+ # job); the refused_once bound is what stops a refuse/re-ask loop.
1600
+ refused_once = True
1601
+ presealed = "" # the refused author may edit the tree again
1602
+ failed = _resume(
1603
+ render_syscall_refusal(
1604
+ problem,
1605
+ launches_remaining=max(0, bench.depth_k - launches_used),
1606
+ sleeps_remaining=max(0, bench.sleep_k - sleeps_used),
1607
+ )
1608
+ )
1609
+ if failed is not None:
1610
+ return failed
1611
+ measured = tuple(changed_paths())
1612
+ # Scope BEFORE the snapshot: an out-of-scope tree is never snapshotted
1613
+ # OR measured — the out-of-scope edit could be to the ruler itself. This
1614
+ # early exit keeps the snapshot off a rejected tree; measure_and_decide
1615
+ # re-checks as the authoritative gate on every entry (including a wake,
1616
+ # which re-enters it directly).
1617
+ violations = out_of_scope(list(measured), contract)
1618
+ if violations:
1619
+ return AttemptResult(
1620
+ outcome="scope-violation",
1621
+ baseline=baseline,
1622
+ session=session,
1623
+ note=f"out-of-scope paths: {', '.join(sorted(violations)[:10])}",
1624
+ run_seed=run_seed,
1625
+ panel_transcript="\n\n".join(panel_sections),
1626
+ panel_rounds=panel_reads,
1627
+ )
1628
+ # Snapshot the session's current output to a committed sha the measurer
1629
+ # keys on; the caller (which owns git) registers its worktree and the
1630
+ # ref. A revision re-snapshots -> a NEW candidate_sha -> a fresh eval. A
1631
+ # snapshot failure is an eval failure (the session ran, the tree just
1632
+ # could not be captured), not a climb crash — same as a candidate eval
1633
+ # that raises.
1634
+ try:
1635
+ candidate_sha = presealed or snapshot()
1636
+ except EvalError as exc:
1637
+ return AttemptResult(
1638
+ outcome="eval-error",
1639
+ baseline=baseline,
1640
+ session=session,
1641
+ note=f"snapshot: {exc}",
1642
+ run_seed=run_seed,
1643
+ panel_transcript="\n\n".join(panel_sections),
1644
+ panel_rounds=panel_reads,
1645
+ )
1646
+ # the tree the gate already judged, sealed again: the verdict stands
1647
+ # when the author concluded, or resubmitted an unchanged tree after a
1648
+ # real negative. After an ERRORED eval only a resubmit runs it again.
1649
+ unchanged = (
1650
+ failed_gate is not None
1651
+ and tree(failed_gate[0]) == tree(candidate_sha)
1652
+ and not (failed_gate[1].outcome == "eval-error" and submitted is not None)
1653
+ )
1654
+ try:
1655
+ if unchanged:
1656
+ assert failed_gate is not None
1657
+ gpu_hours_used -= evals_charge # nothing ran
1658
+ outcome: AttemptResult | MeasureOK = failed_gate[1]
1659
+ elif not measured:
1660
+ # nothing to measure: no paths changed against base, so the
1661
+ # gate would compare base against itself (any benchmark,
1662
+ # metered or not; a SUBMIT of the unchanged tree feeds back
1663
+ # to the author below like any failed gate, charge refunded)
1664
+ gpu_hours_used -= evals_charge
1665
+ outcome = AttemptResult(
1666
+ outcome="no-improvement",
1667
+ baseline=baseline,
1668
+ session=session,
1669
+ note="unmeasured: the sealed tree is unchanged from base",
1670
+ run_seed=run_seed,
1671
+ panel_transcript="\n\n".join(panel_sections),
1672
+ panel_rounds=panel_reads,
1673
+ )
1674
+ elif (
1675
+ submitted is None
1676
+ and launcher is not None # feature off = the gate IS the measurement
1677
+ and bench.depth_k > 0
1678
+ and bench.gpus > 0
1679
+ and float(contract.budgets.gpu_hours_per_run or 0) > 0
1680
+ ):
1681
+ # a METERED finish without a submit is panel-only, launches or
1682
+ # not: the author chose not to claim, and a human scientist
1683
+ # does not spend the full experimental budget re-verifying
1684
+ # their own negative before writing it in the notebook. (With
1685
+ # zero launches this also closes the refuse-twice bypass —
1686
+ # dropping a repeated bare submit must not buy the very
1687
+ # measurement the refusal denied.)
1688
+ outcome = AttemptResult(
1689
+ outcome="no-improvement",
1690
+ baseline=baseline,
1691
+ session=session,
1692
+ note=(
1693
+ "unmeasured finish: no submit was made, so the metered "
1694
+ "gate did not run (panel only)"
1695
+ ),
1696
+ run_seed=run_seed,
1697
+ panel_transcript="\n\n".join(panel_sections),
1698
+ panel_rounds=panel_reads,
1699
+ )
1700
+ else:
1701
+ outcome = measure_and_decide(
1702
+ contract,
1703
+ bench,
1704
+ base_sha=base_sha,
1705
+ candidate_sha=candidate_sha,
1706
+ seed=run_seed,
1707
+ suite_seed=suite_seed,
1708
+ measured_paths=measured,
1709
+ measurer=measurer,
1710
+ min_relative_improvement=config.min_relative_improvement,
1711
+ )
1712
+ except MeasurementPending as pending:
1713
+ # PARK 2 (dispatched candidate/suite, after the session): the wake
1714
+ # reads the cached results and decides — or, on a SUBMITTED park,
1715
+ # delivers them back to the author. Carries the candidate sha
1716
+ # and the session so the caller persists the snapshot ref (drop at
1717
+ # the terminal state) and the resume session id. A submit's sibling
1718
+ # launches ride the SAME park, dispatched only HERE — once the gate
1719
+ # has proven dispatched — on the sealed sha (scope was checked
1720
+ # above, so a launch never runs out-of-scope code).
1721
+ launch_afterany = ""
1722
+ if submitted is not None and submitted.launches:
1723
+ assert launcher is not None # a submit only arrives through it
1724
+ launch_afterany = launcher(candidate_sha, submitted)
1725
+ launches_used += len(submitted.launches)
1726
+ gpu_hours_used += launches_gpu_hours(submitted, gpus=bench.gpus)
1727
+ raise RunParked(
1728
+ phase="candidate",
1729
+ afterany=_merge_afterany(pending.afterany(), launch_afterany),
1730
+ launch_afterany=launch_afterany,
1731
+ base_sha=base_sha,
1732
+ seed=run_seed,
1733
+ suite_seed=suite_seed,
1734
+ candidate_sha=candidate_sha,
1735
+ session=session,
1736
+ syscall=submitted,
1737
+ launches_used=launches_used,
1738
+ sleeps_used=sleeps_used,
1739
+ submitted=submitted is not None,
1740
+ gpu_hours_used=gpu_hours_used,
1741
+ eval_minutes=submitted.eval_minutes if submitted is not None else None,
1742
+ ) from None
1743
+ if isinstance(outcome, AttemptResult):
1744
+ if (
1745
+ submitted is not None
1746
+ and outcome.outcome in ("no-improvement", "suite-regression", "eval-error")
1747
+ and _can_resume()
1748
+ and not (unchanged and sleeps_used > bench.sleep_k)
1749
+ ):
1750
+ # a submitted candidate that failed the gate — including an
1751
+ # eval that errored — is FEEDBACK to the author: it revises and
1752
+ # resubmits, or concludes honestly (buildout Phase B) — never a
1753
+ # silent terminal. Rounds stay bounded by sleep_k.
1754
+ failed_gate = (candidate_sha, outcome)
1755
+ verdict_text = (
1756
+ f"{outcome.note or outcome.outcome} "
1757
+ f"(baseline {outcome.baseline}, candidate {outcome.candidate})."
1758
+ )
1759
+ if unchanged:
1760
+ lead = (
1761
+ "Your `submit` sealed a tree identical to the candidate the gate "
1762
+ "already measured, so nothing was run or paid; that verdict "
1763
+ f"stands: {verdict_text} "
1764
+ )
1765
+ else:
1766
+ lead = f"Your `submit` did NOT clear the gate: {verdict_text} "
1767
+ failed = _resume(
1768
+ f"{lead}{_not_run_note(submitted)}{_budgets_line()} "
1769
+ "Revise and submit again, run more "
1770
+ "experiments, or finish with an honest negative report."
1771
+ )
1772
+ if failed is not None:
1773
+ return failed
1774
+ continue
1775
+ # a terminal measurement outcome (scope-violation / eval-error /
1776
+ # no-improvement / suite-regression): add the session + panel
1777
+ # context this function owns. The baseline stays whatever the GATE
1778
+ # measured (None when it never got that far, e.g. scope-violation),
1779
+ # never overwritten with the ledger's brief number.
1780
+ note = outcome.note
1781
+ if outcome.outcome == "no-improvement" and not note:
1782
+ # only a BARE negative gets generic framing — a specific
1783
+ # reason (e.g. inside the contract's significance floor)
1784
+ # must survive to the record and report
1785
+ note = (
1786
+ "the revision addressing panel findings lost the improvement"
1787
+ if panel_reads
1788
+ else "a negative result reported clearly is a success"
1789
+ )
1790
+ return dc_replace(
1791
+ outcome,
1792
+ session=session,
1793
+ note=note,
1794
+ panel_transcript="\n\n".join(panel_sections),
1795
+ panel_rounds=panel_reads,
1796
+ )
1797
+ # credited: candidate cleared the threshold and no sibling regressed.
1798
+ baseline = outcome.baseline
1799
+ candidate = outcome.candidate
1800
+ suite = outcome.suite
1801
+ suite_seed_ran = outcome.suite_seed
1802
+ baseline_note = outcome.baseline_note # cached-baseline provenance, to the report
1803
+
1804
+ if panel_runner is None:
1805
+ break
1806
+ # the panel reads the CREDITED claim: improvement + suite gate passed
1807
+ panel_reads += 1
1808
+ verdict = panel_runner(
1809
+ baseline,
1810
+ candidate,
1811
+ # the author's report at submit is the claim the panel reads; the
1812
+ # session's last words only when no submit carried one
1813
+ submitted.report if submitted is not None and submitted.report else session.final_text,
1814
+ )
1815
+ panel_sections.append(verdict.transcript)
1816
+ # only the FINAL read's degradation matters: an earlier outage that a
1817
+ # later clean read supersedes is history, not state
1818
+ panel_degraded = verdict.degraded
1819
+ if verdict.blocking and submitted is not None and _can_resume():
1820
+ # blocking findings on a SUBMITTED claim go back to the AUTHOR —
1821
+ # it revises and resubmits, runs more experiments, or concludes
1822
+ # (buildout Phase B: the author drives the depth axis). The loop
1823
+ # then re-reads its next syscall; the revision re-measures from
1824
+ # scratch.
1825
+ failed = _resume(f"{verdict.wake_text}\n\n{_not_run_note(submitted)}{_budgets_line()}")
1826
+ if failed is not None:
1827
+ return failed
1828
+ continue
1829
+ # a plain finish (or an unresumable session): blocking findings stay
1830
+ # open — the caller drafts the PR for a human to triage.
1831
+ panel_blocking_open = bool(verdict.blocking)
1832
+ break
1833
+
1834
+ return AttemptResult(
1835
+ outcome="improved",
1836
+ baseline=baseline,
1837
+ candidate=candidate,
1838
+ session=session,
1839
+ branch=f"{config.branch_prefix}/{config.benchmark}",
1840
+ measured_paths=measured,
1841
+ candidate_sha=candidate_sha,
1842
+ run_seed=run_seed,
1843
+ suite=suite,
1844
+ suite_seed=suite_seed_ran,
1845
+ note=baseline_note,
1846
+ submit_report=submitted.report if submitted is not None else "",
1847
+ panel_transcript="\n\n".join(panel_sections),
1848
+ panel_rounds=panel_reads,
1849
+ panel_blocking_open=panel_blocking_open,
1850
+ panel_degraded=panel_degraded,
1851
+ )
1852
+
1853
+
1854
+ MAX_EXPERIMENT_ROWS = 60
1855
+
1856
+
1857
+ def _cell(text: object, cap: int = 160) -> str:
1858
+ """One markdown table cell: one line, bounded, then pipes escaped — in that
1859
+ order, so a cut never leaves a bare backslash before the row's separator."""
1860
+ return " ".join(str(text).split())[:cap].replace("|", "\\|")
1861
+
1862
+
1863
+ def _ended(row: dict[str, Any]) -> str:
1864
+ if not row.get("back"):
1865
+ return "not back"
1866
+ code = row.get("exit_code")
1867
+ how = f"exit {code}" if code is not None else (str(row.get("state") or "no exit code"))
1868
+ secs = row.get("elapsed")
1869
+ if isinstance(secs, int | float) and secs > 0:
1870
+ h, m = divmod(int(secs) // 60, 60)
1871
+ how += f", {h}h{m:02d}m" if h else f", {m}m"
1872
+ return how
1873
+
1874
+
1875
+ def _experiments_section(rows: list[dict[str, Any]]) -> list[str]:
1876
+ """The run's launches as the ledger recorded them: what ran, how each job
1877
+ ended, what it printed last. Empty when the run launched nothing."""
1878
+ if not rows:
1879
+ return []
1880
+ lines = [
1881
+ "",
1882
+ "## Experiments",
1883
+ "",
1884
+ f"{len(rows)} job(s) launched by the author this run, from the kernel's ledger; "
1885
+ "the result column is the last line each job printed.",
1886
+ "",
1887
+ "| sleep | launch | why | job | ended | result |",
1888
+ "| --- | --- | --- | --- | --- | --- |",
1889
+ ]
1890
+ for row in rows[:MAX_EXPERIMENT_ROWS]:
1891
+ pace = ""
1892
+ if int(row.get("array") or 1) > 1:
1893
+ k = int(row.get("concurrency") or 0) or int(row["array"])
1894
+ pace = f" (x{row['array']}, {k} at a time)"
1895
+ lines.append(
1896
+ f"| {row.get('sleep', '')} | {_cell(row.get('launch', ''), 48)}{pace} | "
1897
+ f"{_cell(row.get('why', ''), 120)} | {_cell(row.get('job', ''), 48)} | "
1898
+ f"{_ended(row)} | {_cell(row.get('result', ''), 160)} |"
1899
+ )
1900
+ rest = rows[MAX_EXPERIMENT_ROWS:]
1901
+ if rest:
1902
+ ok = sum(1 for r in rest if r.get("back") and r.get("exit_code") == 0)
1903
+ lines.append(
1904
+ f"| | … {len(rest)} more job(s): {ok} exit 0, {len(rest) - ok} otherwise | | | | |"
1905
+ )
1906
+ return lines
1907
+
1908
+
1909
+ def pr_body(
1910
+ result: AttemptResult,
1911
+ config: RunConfig,
1912
+ redact_secrets: tuple[str, ...],
1913
+ display_digits: int | None = None,
1914
+ experiments: list[dict[str, Any]] | None = None,
1915
+ ) -> str:
1916
+ """The PR body for an improved run: the author's report, the experiments
1917
+ the run actually ran (from the launch ledger), the measured table, and
1918
+ the panel's transcript.
1919
+
1920
+ Human surfaces render at the benchmark's conventional precision;
1921
+ full precision lives only in results/leader.json, and every
1922
+ comparison runs on full floats.
1923
+ """
1924
+ from outerloop.progress import fmt_metric
1925
+
1926
+ if result.outcome != "improved" or result.baseline is None or result.candidate is None:
1927
+ raise ValueError("pr_body requires an improved result with both measurements")
1928
+ suite_lines: list[str] = []
1929
+ if result.suite:
1930
+ suite_lines = [
1931
+ "",
1932
+ "Shared code was touched, so every sibling benchmark was re-measured "
1933
+ "on both sides (paired seed): none regressed beyond its floor.",
1934
+ "",
1935
+ "| suite benchmark | baseline | candidate |",
1936
+ "| --- | --- | --- |",
1937
+ ] + [
1938
+ f"| {row.name} | {fmt_metric(row.baseline, row.display_digits)} "
1939
+ f"| {fmt_metric(row.candidate, row.display_digits)} |"
1940
+ for row in result.suite
1941
+ ]
1942
+ if result.panel_blocking_open:
1943
+ banner = [
1944
+ "> **Draft — the verification panel capped out with blocking "
1945
+ "findings still open.** They are listed under Pre-PR "
1946
+ "verification below; the human decides.",
1947
+ "",
1948
+ ]
1949
+ elif result.panel_degraded:
1950
+ banner = [
1951
+ "> **Draft — the final panel read was degraded (a lens produced "
1952
+ "no verdict).** Not a certified pass; see Pre-PR verification "
1953
+ "below.",
1954
+ "",
1955
+ ]
1956
+ else:
1957
+ banner = []
1958
+ panel_section = (
1959
+ ["", "## Pre-PR verification", "", result.panel_transcript[:MAX_REPORT_BODY]]
1960
+ if result.panel_transcript
1961
+ else []
1962
+ )
1963
+ if result.submit_report:
1964
+ report_lines = [
1965
+ "*Written by the author at submit, before the orchestrator measured; the "
1966
+ "panel read it against the diff and the experiments below.*",
1967
+ "",
1968
+ redact(result.submit_report, redact_secrets)[:MAX_REPORT_BODY],
1969
+ ]
1970
+ else:
1971
+ report_lines = [
1972
+ (
1973
+ "*This report came from the previous session in this line — no "
1974
+ "agent session ran for this attempt. It was written before the "
1975
+ "orchestrator measured; the table below contains the measured "
1976
+ "results.*"
1977
+ if result.session and result.session.stop_reason == "resumed"
1978
+ else "*Session prose, written before the orchestrator measured; "
1979
+ "the table below contains the measured results.*"
1980
+ ),
1981
+ "",
1982
+ redact(result.session.final_text, redact_secrets)[:MAX_REPORT_BODY]
1983
+ if result.session
1984
+ else "",
1985
+ ]
1986
+ body = "\n".join(
1987
+ [
1988
+ *banner,
1989
+ f"Automated improvement attempt on `{config.benchmark}` "
1990
+ f"(agent `{config.agent_id}`, one hypothesis per PR).",
1991
+ "",
1992
+ "## Research report",
1993
+ "",
1994
+ *report_lines,
1995
+ *_experiments_section(experiments or []),
1996
+ "",
1997
+ "## Measured",
1998
+ "",
1999
+ "| | value |",
2000
+ "| --- | --- |",
2001
+ f"| baseline ({config.benchmark}) | {fmt_metric(result.baseline, display_digits)} |",
2002
+ f"| candidate | {fmt_metric(result.candidate, display_digits)} |",
2003
+ *suite_lines,
2004
+ "",
2005
+ "Both numbers were measured by the orchestrator re-running the "
2006
+ "contract's eval command — not taken from the session. CI "
2007
+ "re-verifies independently.",
2008
+ *panel_section,
2009
+ ]
2010
+ )
2011
+ return redact(body, redact_secrets)