outerloop-science 0.1.0.dev2__py3-none-any.whl → 0.1.0.dev4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. outerloop/__init__.py +2 -2
  2. outerloop/appauth.py +17 -0
  3. outerloop/attempt.py +376 -101
  4. outerloop/brief.py +38 -25
  5. outerloop/cli.py +104 -6
  6. outerloop/climbboard.py +67 -22
  7. outerloop/compute.py +148 -53
  8. outerloop/contract.py +8 -0
  9. outerloop/dispatch.py +63 -18
  10. outerloop/evalcache.py +147 -0
  11. outerloop/followup.py +40 -25
  12. outerloop/github.py +67 -22
  13. outerloop/harness.py +22 -47
  14. outerloop/housekeeping.py +1 -17
  15. outerloop/image.py +0 -4
  16. outerloop/init.py +45 -2
  17. outerloop/intake.py +4 -7
  18. outerloop/launchlog.py +239 -0
  19. outerloop/maintain.py +353 -0
  20. outerloop/maintain_agent_cli.py +81 -0
  21. outerloop/maintain_post_cli.py +140 -0
  22. outerloop/measure.py +6 -0
  23. outerloop/orchestrator.py +141 -31
  24. outerloop/panel.py +3 -3
  25. outerloop/review.py +4 -0
  26. outerloop/review_agent.py +7 -7
  27. outerloop/review_agent_cli.py +2 -2
  28. outerloop/review_post_cli.py +2 -2
  29. outerloop/review_summarize_cli.py +8 -6
  30. outerloop/roles.py +27 -0
  31. outerloop/rolespec.py +3 -1
  32. outerloop/steward.py +7 -14
  33. outerloop/syscall.py +261 -47
  34. outerloop/syscall_cli.py +243 -12
  35. outerloop/tick.py +274 -313
  36. outerloop/verify_agent.py +8 -6
  37. outerloop/verify_post_cli.py +2 -2
  38. outerloop/watcher.py +203 -0
  39. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev4.dist-info}/METADATA +4 -1
  40. outerloop_science-0.1.0.dev4.dist-info/RECORD +59 -0
  41. outerloop_science-0.1.0.dev2.dist-info/RECORD +0 -53
  42. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev4.dist-info}/WHEEL +0 -0
  43. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev4.dist-info}/entry_points.txt +0 -0
  44. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev4.dist-info}/licenses/LICENSE +0 -0
  45. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev4.dist-info}/licenses/NOTICE +0 -0
outerloop/attempt.py CHANGED
@@ -18,6 +18,7 @@ import logging
18
18
  import os
19
19
  import re
20
20
  import shutil
21
+ import time
21
22
  from collections.abc import Callable, Iterable
22
23
  from dataclasses import dataclass
23
24
  from dataclasses import replace as dc_replace
@@ -25,9 +26,9 @@ from functools import partial
25
26
  from pathlib import Path
26
27
  from typing import Any, cast
27
28
 
28
- from outerloop.appauth import resolve_bot_auth
29
+ from outerloop.appauth import add_credential_args, resolve_bot_auth
29
30
  from outerloop.brief import BudgetState, distill_lessons
30
- from outerloop.compute import LocalCompute, local_mode
31
+ from outerloop.compute import LocalCompute
31
32
  from outerloop.contract import Benchmark, Contract, contract_text_in_tree, load_contract
32
33
  from outerloop.dispatch import (
33
34
  Snapshot,
@@ -36,6 +37,7 @@ from outerloop.dispatch import (
36
37
  should_dispatch,
37
38
  snapshot_tree,
38
39
  )
40
+ from outerloop.evalcache import seed_dir
39
41
  from outerloop.github import (
40
42
  GitError,
41
43
  GitHubClient,
@@ -43,8 +45,10 @@ from outerloop.github import (
43
45
  Workspace,
44
46
  contract_at,
45
47
  ensure_regular_git_dir,
48
+ git_identity,
46
49
  )
47
50
  from outerloop.harness import Harness, SessionResult, default_binary, redact
51
+ from outerloop.launchlog import append_ended, append_submitted, experiments_rows
48
52
  from outerloop.markers import has_marker
49
53
  from outerloop.measure import DispatchedMeasurer, DispatchSettings
50
54
  from outerloop.orchestrator import (
@@ -84,9 +88,20 @@ from outerloop.runstate import (
84
88
  save_record,
85
89
  stamp_outage,
86
90
  )
87
- from outerloop.syscall import CHANNEL_DIR_NAMES, MAX_ARTIFACT_BYTES, SyscallRequest, channel_dir
91
+ from outerloop.runstate import (
92
+ run_dir as run_dir_of,
93
+ )
94
+ from outerloop.syscall import (
95
+ CHANNEL_DIR_NAMES,
96
+ MAX_ARTIFACT_BYTES,
97
+ SyscallRequest,
98
+ channel_dir,
99
+ launch_task_ids,
100
+ tool_update_note,
101
+ )
88
102
  from outerloop.syscall import ensure_excluded as syscall_excluded
89
103
  from outerloop.syscall import install_tool as syscall_install_tool
104
+ from outerloop.syscall import refresh_tool as syscall_refresh_tool
90
105
  from outerloop.syscall import write_budget as syscall_write_budget
91
106
  from outerloop.syscall import write_siblings as syscall_write_siblings
92
107
  from outerloop.verifier import MAX_CLAIM_CHARS
@@ -186,7 +201,7 @@ class WorkspaceDrift(RuntimeError):
186
201
  """The tree changed between measurement and commit."""
187
202
 
188
203
 
189
- def _target_clone_url(target: str) -> str:
204
+ def target_clone_url(target: str) -> str:
190
205
  """The canonical HTTPS clone URL for `owner/repo`. The one source of truth
191
206
  for where a run's git pushes go — derived from the target, never read from
192
207
  the session-writable `remote.origin.url`."""
@@ -206,7 +221,8 @@ def _blessed_head(ws: Workspace, result: Any, contract: Any) -> str:
206
221
  return ""
207
222
  try:
208
223
  return ws.git("rev-parse", "HEAD").strip()
209
- except Exception:
224
+ except Exception as exc:
225
+ log.warning("could not read HEAD; not arming self-merge: %s", exc)
210
226
  return ""
211
227
 
212
228
 
@@ -425,9 +441,14 @@ def _park_run(
425
441
  # it lands in the durable record, like every other persisted final_text
426
442
  # — a session that echoed a credential must not leave it in record.json.
427
443
  # Empty/zero for a baseline park (the session has not run yet).
428
- "report": redact(parked.session.final_text, secrets)[:MAX_CLAIM_CHARS]
429
- if parked.session
430
- else "",
444
+ # the author's report at submit, else the session's last words (a park
445
+ # with no submit); either way what the wake's panel and the PR body read
446
+ "report": redact(
447
+ parked.syscall.report
448
+ if parked.syscall is not None and parked.syscall.report
449
+ else (parked.session.final_text if parked.session else ""),
450
+ secrets,
451
+ )[:MAX_CLAIM_CHARS],
431
452
  "session_cost_usd": parked.session.cost_usd if parked.session else 0.0,
432
453
  "session_turns": parked.session.num_turns if parked.session else 0,
433
454
  }
@@ -452,10 +473,35 @@ def _park_run(
452
473
  "minutes": launch.minutes,
453
474
  "artifacts": list(launch.artifacts),
454
475
  **({"array": launch.array} if launch.array > 1 else {}),
476
+ **({"why": redact(launch.why, secrets)} if launch.why else {}),
477
+ **({"concurrency": launch.concurrency} if launch.concurrency else {}),
455
478
  }
456
479
  for launch in parked.syscall.launches
457
480
  ]
458
481
  stage["syscall_note"] = redact(parked.syscall.note, secrets)
482
+ if parked.syscall.launches:
483
+ # the run's launch ledger (`history`, and the queue view's labels):
484
+ # ids align with launch_jobs order, as stage_launch_job_ids reads them
485
+ if parked.launch_afterany:
486
+ launch_ids = afterany_ids(parked.launch_afterany)
487
+ elif parked.phase == "author-sleep":
488
+ launch_ids = list(job_ids)
489
+ else:
490
+ launch_ids = []
491
+ ledger_launches = tuple(
492
+ dc_replace(launch, why=redact(launch.why, secrets))
493
+ for launch in parked.syscall.launches
494
+ )
495
+ _best_effort(
496
+ "launch ledger",
497
+ lambda: append_submitted(
498
+ run_dir_of(run_root, record.run_id),
499
+ sleep=parked.sleeps_used,
500
+ launches=ledger_launches,
501
+ job_ids=launch_ids,
502
+ at=now,
503
+ ),
504
+ )
459
505
  # (the session id the wake resumes is the record's own
460
506
  # resume_session_id, set below for every park — no stage duplicate)
461
507
  stage["launches_used"] = parked.launches_used
@@ -586,6 +632,7 @@ def _dispatch_settings(args: argparse.Namespace) -> DispatchSettings:
586
632
  #174: the wake dropped the GPU lane)."""
587
633
  from outerloop.compute import compute_from_env
588
634
 
635
+ target = getattr(args, "target", "") or ""
589
636
  return DispatchSettings(
590
637
  compute=compute_from_env(),
591
638
  image=args.image,
@@ -593,9 +640,34 @@ def _dispatch_settings(args: argparse.Namespace) -> DispatchSettings:
593
640
  partition=args.partition,
594
641
  gpu_partition=getattr(args, "gpu_partition", ""),
595
642
  gpu_account=getattr(args, "gpu_account", ""),
643
+ seed_cache=seed_dir(Path(args.run_root), target) if target else None,
596
644
  )
597
645
 
598
646
 
647
+ # Experiments yield to verification. Every kernel job is one Slurm user, so
648
+ # among the kernel's own pending jobs the priority order is ours: launches
649
+ # carry this nice so a gate eval or a follow-up re-measure (nice 0) starts
650
+ # first when the cap frees a slot. Sized above the factors that differ between
651
+ # our jobs on Torch — age tops out at 1000 after a week, job size at 1000, the
652
+ # per-GPU TRES share stays in the hundreds — so the order holds however long a
653
+ # launch has waited. Other users' jobs and the group cap are untouched.
654
+ LAUNCH_NICE = 5000
655
+
656
+
657
+ def with_seed(dispatch: DispatchSettings, run_root: Path, target: str) -> DispatchSettings:
658
+ """These settings with the target's seed cache filled in from the record's
659
+ target when the CLI gave none (wake and follow-up jobs carry the run id,
660
+ not the target)."""
661
+ # tolerant of any settings object: a backend that knows no seed (or a
662
+ # test double) is left exactly as it is
663
+ if not target or getattr(dispatch, "seed_cache", "unknown") is not None:
664
+ return dispatch
665
+ try:
666
+ return dc_replace(dispatch, seed_cache=seed_dir(run_root, target))
667
+ except TypeError:
668
+ return dispatch
669
+
670
+
599
671
  def _make_launcher(
600
672
  dispatch: DispatchSettings, run_dir: Path, workspace: Path, run_id: str, gpus: int = 0
601
673
  ):
@@ -605,45 +677,42 @@ def _make_launcher(
605
677
  partially-submitted batch is reaped rather than orphaned. `gpus` is the
606
678
  benchmark's: an author's experiments run on the same lane as its evals."""
607
679
  account, partition = dispatch.placement(gpus)
608
- # under launch admission a GPU launch enters the queue held; the tick
609
- # releases it when the user's GPUs fit under the cap (tick.service_admission)
610
- from outerloop.tick import max_launch_gpus_from_env
611
-
612
- hold = gpus > 0 and not local_mode() and max_launch_gpus_from_env() > 0
613
680
 
614
681
  def launcher(sha: str, request: SyscallRequest) -> str:
615
- from dataclasses import replace as _replace
616
-
617
682
  from outerloop.dispatch import eval_job_spec, write_eval_job
618
- from outerloop.syscall import launch_jobs
683
+ from outerloop.syscall import array_spec
619
684
 
620
685
  ids: list[str] = []
621
686
  try:
622
687
  for launch in request.launches:
623
- # an array launch is N jobs of one command, each with its
624
- # SWEEP_INDEX; one afterany wake covers them all
625
- for job_name, extra_env in launch_jobs(launch):
626
- script = write_eval_job(
627
- run_dir,
628
- f"launch-{job_name}",
629
- repo_root=workspace,
630
- snapshot_sha=sha,
631
- command=launch.command,
632
- image=dispatch.image,
633
- extra_env=extra_env,
634
- artifacts=launch.artifacts,
635
- artifact_max_bytes=MAX_ARTIFACT_BYTES,
636
- gpus=gpus,
637
- )
638
- spec = eval_job_spec(
639
- script,
640
- job_name=f"{run_id}-launch-{job_name}",
641
- account=account,
642
- partition=partition,
643
- eval_minutes=launch.minutes,
644
- gpus=gpus,
645
- )
646
- ids.append(dispatch.compute.submit(_replace(spec, hold=hold)))
688
+ # a sweep is ONE Slurm job array (`--array=0-N%K`): the queue
689
+ # holds one entry, Slurm runs at most K tasks at once, each task
690
+ # derives its job dir and SWEEP_INDEX from its array index, and
691
+ # one afterany on the array id covers every task
692
+ script = write_eval_job(
693
+ run_dir,
694
+ f"launch-{launch.name}",
695
+ repo_root=workspace,
696
+ snapshot_sha=sha,
697
+ command=launch.command,
698
+ image=dispatch.image,
699
+ artifacts=launch.artifacts,
700
+ artifact_max_bytes=MAX_ARTIFACT_BYTES,
701
+ gpus=gpus,
702
+ array=launch.array,
703
+ seed_cache=dispatch.seed_cache,
704
+ )
705
+ spec = eval_job_spec(
706
+ script,
707
+ job_name=f"{run_id}-launch-{launch.name}",
708
+ account=account,
709
+ partition=partition,
710
+ eval_minutes=launch.minutes,
711
+ gpus=gpus,
712
+ nice=LAUNCH_NICE,
713
+ array=array_spec(launch),
714
+ )
715
+ ids.append(dispatch.compute.submit(spec))
647
716
  except Exception:
648
717
  # a partial batch must not orphan: no park record was written yet,
649
718
  # so nothing would ever wake or cancel the jobs that DID submit —
@@ -661,6 +730,26 @@ def _make_launcher(
661
730
  return launcher
662
731
 
663
732
 
733
+ def _make_watcher(
734
+ dispatch: DispatchSettings, run_root: Path, run_id: str, workspace: Path, config: RunConfig
735
+ ) -> Callable[[], Any]:
736
+ """The session watcher for one run: a thread beside the harness that
737
+ answers `queue` and `history` from the channel (docs/design/session-watcher.md).
738
+ Shared by the first pass and every wake leg."""
739
+ from outerloop.watcher import SessionWatcher, WatcherContext
740
+
741
+ ctx = WatcherContext(
742
+ workspace=workspace,
743
+ run_root=run_root,
744
+ run_id=run_id,
745
+ target=config.target,
746
+ agent_id=config.agent_id,
747
+ compute=dispatch.compute,
748
+ gpu_partition=dispatch.gpu_partition,
749
+ )
750
+ return lambda: SessionWatcher(ctx)
751
+
752
+
664
753
  def _wake_author_sleep(
665
754
  *,
666
755
  run_root: Path,
@@ -713,7 +802,12 @@ def _wake_author_sleep(
713
802
  # the same ending shape every other terminal takes. The line notebook
714
803
  # records it first, while the tree is still the session's final tree.
715
804
  _push_line_snapshot(
716
- ws, _line_ref_for(bench, config.agent_id), run_id, result.outcome, secrets
805
+ ws,
806
+ _line_ref_for(bench, config.agent_id),
807
+ run_id,
808
+ result.outcome,
809
+ secrets,
810
+ bot_login=config.bot_login,
717
811
  )
718
812
  for ref in drop_refs:
719
813
  drop_snapshot(ws, Snapshot(commit="", tree="", ref=ref))
@@ -776,6 +870,8 @@ def _wake_author_sleep(
776
870
  minutes=int(item.get("minutes") or 1),
777
871
  artifacts=tuple(str(a) for a in item.get("artifacts", [])),
778
872
  array=int(item.get("array") or 1),
873
+ why=str(item.get("why") or ""),
874
+ concurrency=int(item.get("concurrency") or 0),
779
875
  )
780
876
  for item in _stage_launches(record)
781
877
  )
@@ -786,12 +882,28 @@ def _wake_author_sleep(
786
882
  # of a blank "job failure". The park's launch job ids align positionally
787
883
  # with the results (same launch/array order). Best-effort — the wake never
788
884
  # blocks on the scheduler query.
885
+ task_ids = launch_task_ids(launches, stage_launch_job_ids(record))
789
886
  status_of = getattr(dispatch.compute, "status", None)
790
887
  if status_of is not None:
791
- results = annotate_launch_states(results, _stage_launch_job_ids(record), status_of)
888
+ results = annotate_launch_states(results, task_ids, status_of)
792
889
  launches_used = int(record.stage.get("launches_used", 0)) # type: ignore[call-overload]
793
890
  sleeps_used = int(record.stage.get("sleeps_used", 0)) # type: ignore[call-overload]
794
- gpu_hours_used = _reconcile_launch_hours(record, dispatch, bench.gpus, launches)
891
+ elapsed = _launch_elapsed(dispatch, task_ids) if task_ids else None
892
+ _best_effort(
893
+ "launch ledger",
894
+ lambda: append_ended(
895
+ run_dir, sleep=sleeps_used, results=results, at=time.time(), elapsed_seconds=elapsed
896
+ ),
897
+ )
898
+ gpu_hours_used = _reconcile_launch_hours(record, dispatch, bench.gpus, launches, elapsed)
899
+ # the tool the session invokes comes from THIS kernel: a session that
900
+ # started under an older one gets today's verbs and flags at its wake, and
901
+ # is told what is new
902
+ tool_changed = False
903
+ try:
904
+ tool_changed = syscall_refresh_tool(workspace)
905
+ except Exception as exc:
906
+ log.warning("tool refresh failed: %s", redact(f"{type(exc).__name__}: {exc}", secrets))
795
907
  wake_text = render_wake(
796
908
  results,
797
909
  str(record.stage.get("syscall_note", "")),
@@ -804,9 +916,34 @@ def _wake_author_sleep(
804
916
  ),
805
917
  gpus=bench.gpus,
806
918
  )
919
+ pacing = [
920
+ f"sweep `{la.name}`: {la.array} tasks, at most {la.concurrency or la.array} at a time"
921
+ for la in launches
922
+ if la.array > 1
923
+ ]
924
+ if pacing:
925
+ wake_text = f"{wake_text}\n\n" + "\n".join(pacing) + " (the contract's ceiling applies)."
926
+ if tool_changed:
927
+ wake_text = f"{wake_text}\n\n{tool_update_note(channel_dir(workspace))}"
807
928
  if extra_update:
808
929
  # a submitted park's gate/panel feedback leads; launch results follow
809
930
  wake_text = f"{extra_update}\n\n{wake_text}"
931
+ # A research line whose base moved while it slept RE-PINS to the fresh base
932
+ # and is told to merge it: the agent does the merge (mirroring the in-review
933
+ # conflict wake, followup.py), the kernel only fetches and re-pins. Re-pinning
934
+ # base_sha to the fresh head is what makes the gate baseline and the scope
935
+ # base the CURRENT base (like followup's base_sha_at_fetch), so a sibling's
936
+ # merged work is never credited to this line and a forbidden conflict
937
+ # resolution (differing from the fresh base) is still scope-checked. Non-line
938
+ # runs and an unmoved base are untouched.
939
+ if _line_ref_for(bench, config.agent_id):
940
+ fresh_base = _line_base_advanced(ws, base_branch, base_sha)
941
+ if fresh_base:
942
+ digest = _reintegration_digest(ws, base_sha, fresh_base)
943
+ base_sha = fresh_base
944
+ wake_text = (
945
+ REINTEGRATE_PROMPT.format(base_branch=base_branch, digest=digest) + wake_text
946
+ )
810
947
  _best_effort(
811
948
  "budget refresh",
812
949
  lambda: write_budget(
@@ -829,7 +966,9 @@ def _wake_author_sleep(
829
966
  wake_line = _line_ref_for(bench, config.agent_id)
830
967
 
831
968
  def snapshot() -> str:
832
- snap = snapshot_tree(ws, base_sha, exclude=LINE_MEMORY_PATHS if wake_line else ())
969
+ snap = snapshot_tree(
970
+ ws, base_sha, exclude=LINE_MEMORY_PATHS if wake_line else (), author=config.bot_login
971
+ )
833
972
  snapshots.append(snap)
834
973
  return snap.commit
835
974
 
@@ -850,6 +989,7 @@ def _wake_author_sleep(
850
989
  config.bot_login,
851
990
  _utc_date(now),
852
991
  exclude=LINE_MEMORY_PATHS if wake_line else (),
992
+ secrets=secrets,
853
993
  )
854
994
  if panel_lenses
855
995
  else None
@@ -873,6 +1013,7 @@ def _wake_author_sleep(
873
1013
  resume_session_id=record.resume_session_id,
874
1014
  improve_prompt=wake_text,
875
1015
  launcher=_make_launcher(dispatch, run_dir, workspace, run_id, gpus=bench.gpus),
1016
+ watcher=_make_watcher(dispatch, run_root, run_id, workspace, config),
876
1017
  tree_of=lambda sha: ws.git("rev-parse", f"{sha}^{{tree}}").strip(),
877
1018
  judged=judged or _stage_judged(record),
878
1019
  launches_used=launches_used,
@@ -937,6 +1078,32 @@ def _stage_judged(record: RunRecord) -> tuple[str, AttemptResult] | None:
937
1078
  )
938
1079
 
939
1080
 
1081
+ def _ledger_ended(
1082
+ run_dir: Path,
1083
+ record: RunRecord,
1084
+ launches: tuple,
1085
+ task_ids: list[str],
1086
+ dispatch: DispatchSettings,
1087
+ elapsed: list[int | None] | None,
1088
+ ) -> None:
1089
+ """Record a park's finished launches in the ledger from the run dir alone
1090
+ (no delivery into a workspace), with the scheduler's state for jobs that
1091
+ left no exit code."""
1092
+ from outerloop.syscall import annotate_launch_states, read_results
1093
+
1094
+ results = read_results(run_dir, launches)
1095
+ status_of = getattr(dispatch.compute, "status", None)
1096
+ if status_of is not None:
1097
+ results = annotate_launch_states(results, task_ids, status_of)
1098
+ append_ended(
1099
+ run_dir,
1100
+ sleep=int(record.stage.get("sleeps_used", 0)), # type: ignore[call-overload]
1101
+ results=results,
1102
+ at=time.time(),
1103
+ elapsed_seconds=elapsed,
1104
+ )
1105
+
1106
+
940
1107
  def _stage_syscall_launches(record: RunRecord) -> tuple:
941
1108
  """The park's launches as `Launch` values (command elided: they ran)."""
942
1109
  from outerloop.syscall import Launch
@@ -948,12 +1115,14 @@ def _stage_syscall_launches(record: RunRecord) -> tuple:
948
1115
  minutes=int(item.get("minutes") or 1),
949
1116
  artifacts=tuple(str(a) for a in item.get("artifacts", [])),
950
1117
  array=int(item.get("array") or 1),
1118
+ why=str(item.get("why") or ""),
1119
+ concurrency=int(item.get("concurrency") or 0),
951
1120
  )
952
1121
  for item in _stage_launches(record)
953
1122
  )
954
1123
 
955
1124
 
956
- def _stage_launch_job_ids(record: RunRecord) -> list[str]:
1125
+ def stage_launch_job_ids(record: RunRecord) -> list[str]:
957
1126
  """The park's launch jobs: `launch_afterany` when the park recorded it;
958
1127
  for an older author-sleep park every waited job was a launch; for an
959
1128
  older candidate park the gate's evals are mixed in, so none."""
@@ -966,7 +1135,11 @@ def _stage_launch_job_ids(record: RunRecord) -> list[str]:
966
1135
 
967
1136
 
968
1137
  def _reconcile_launch_hours(
969
- record: RunRecord, dispatch: DispatchSettings, gpus: int, launches: tuple
1138
+ record: RunRecord,
1139
+ dispatch: DispatchSettings,
1140
+ gpus: int,
1141
+ launches: tuple,
1142
+ elapsed: list[int | None] | None = None,
970
1143
  ) -> float:
971
1144
  """The run's GPU-hours after handing back the unused walltime of the
972
1145
  park's launch jobs — once: the stage remembers the refund, so a wake
@@ -976,7 +1149,9 @@ def _reconcile_launch_hours(
976
1149
  used = float(stage.get("gpu_hours_used", 0.0)) # type: ignore[arg-type]
977
1150
  if not gpus or stage.get("launch_hours_refunded"):
978
1151
  return used
979
- refund = _launch_refund(dispatch, launches, _stage_launch_job_ids(record), gpus)
1152
+ refund = _launch_refund(
1153
+ dispatch, launches, launch_task_ids(launches, stage_launch_job_ids(record)), gpus, elapsed
1154
+ )
980
1155
  if refund > 0:
981
1156
  log.info("%s: refunding %.2f GPU-hours of unused launch walltime", record.run_id, refund)
982
1157
  used = max(0.0, used - refund)
@@ -985,21 +1160,35 @@ def _reconcile_launch_hours(
985
1160
  return used
986
1161
 
987
1162
 
1163
+ def _launch_elapsed(dispatch: DispatchSettings, job_ids: list[str]) -> list[int | None] | None:
1164
+ """How long each launch job ran, from the compute, aligned with `job_ids`;
1165
+ None when the compute cannot say (nothing is refunded or recorded on a
1166
+ guess)."""
1167
+ query = getattr(dispatch.compute, "elapsed_seconds", None)
1168
+ if query is None or not job_ids:
1169
+ return None
1170
+ try:
1171
+ return [query(jid) for jid in job_ids]
1172
+ except Exception as exc:
1173
+ log.warning("launch walltime unknown (%s: %s)", type(exc).__name__, exc)
1174
+ return None
1175
+
1176
+
988
1177
  def _launch_refund(
989
- dispatch: DispatchSettings, launches: tuple, job_ids: list[str], gpus: int
1178
+ dispatch: DispatchSettings,
1179
+ launches: tuple,
1180
+ job_ids: list[str],
1181
+ gpus: int,
1182
+ elapsed: list[int | None] | None = None,
990
1183
  ) -> float:
991
1184
  """The unused walltime of a park's launch jobs, in GPU-hours, or 0 when
992
1185
  the compute cannot say how long they ran (nothing is refunded on a
993
- guess)."""
1186
+ guess). `elapsed` may be handed in when the caller already asked."""
994
1187
  from outerloop.syscall import launch_hours_refund
995
1188
 
996
- query = getattr(dispatch.compute, "elapsed_seconds", None)
997
- if query is None or not job_ids:
998
- return 0.0
999
- try:
1000
- elapsed = [query(jid) for jid in job_ids]
1001
- except Exception as exc:
1002
- log.warning("launch walltime unknown (%s: %s); nothing refunded", type(exc).__name__, exc)
1189
+ if elapsed is None:
1190
+ elapsed = _launch_elapsed(dispatch, job_ids)
1191
+ if elapsed is None:
1003
1192
  return 0.0
1004
1193
  return launch_hours_refund(launches, elapsed, gpus=gpus)
1005
1194
 
@@ -1123,8 +1312,61 @@ def _line_ref_for(bench: Benchmark | None, agent_id: str) -> str:
1123
1312
  return f"agents/{agent_id}"
1124
1313
 
1125
1314
 
1315
+ # The base moved under a line while it slept: rather than the kernel doing a
1316
+ # git merge (which mishandles the agent's in-flight tree, its conflicts, and
1317
+ # the scope gate), the wake mirrors the in-review conflict wake — fetch the
1318
+ # fresh base into the workspace and TELL THE AGENT to merge it. The agent has
1319
+ # git and already resolves the run-start merge as its first task; only the
1320
+ # credential-bearing fetch/push are the kernel's. No commit text goes in the
1321
+ # prompt (no cross-agent prompt-injection surface); the agent reads what
1322
+ # landed from git itself.
1323
+ REINTEGRATE_PROMPT = (
1324
+ "# The base moved while you were asleep\n"
1325
+ "`origin/{base_branch}` advanced since your last run and is fetched into "
1326
+ "your workspace. What landed:\n{digest}\n"
1327
+ "Merge it into your line and resolve any conflicts honestly, then decide "
1328
+ "what to re-run given what landed — if a sibling took your direction "
1329
+ "further, pivot or say so plainly rather than pushing on. Your change is "
1330
+ "measured against the current base.\n\n"
1331
+ )
1332
+
1333
+
1334
+ def _reintegration_digest(ws: Workspace, base_sha: str, fresh_head: str) -> str:
1335
+ """A short 'what landed' list: the subjects of the commits merged into the
1336
+ base since this line's base, newest first, capped. These are MERGED commits
1337
+ — vetted by the human-merge gate — so they are context, not untrusted input;
1338
+ the agent also has them in git to read in full."""
1339
+ try:
1340
+ out = ws.git("log", "--no-merges", "--format=%s", f"{base_sha}..{fresh_head}")
1341
+ except Exception:
1342
+ return " (recent changes on the base; see `git log`)"
1343
+ subjects = [ln.strip() for ln in out.splitlines() if ln.strip()][:12]
1344
+ return "\n".join(f" - {s}" for s in subjects) or " (a merge on the base; see `git log`)"
1345
+
1346
+
1347
+ def _line_base_advanced(ws: Workspace, base_branch: str, base_sha: str) -> str:
1348
+ """Fetch `origin/<base_branch>` and return its head when it has advanced
1349
+ past the line's pinned base, else "". The fetch doubles as making the
1350
+ fresh base available for the agent to merge and refreshes the origin refs
1351
+ the scope/measure path pairs against. Best-effort: any git failure returns
1352
+ "" and the wake proceeds exactly as today."""
1353
+ try:
1354
+ ws.fetch_origin()
1355
+ new = ws.git("rev-parse", f"refs/remotes/origin/{base_branch}").strip()
1356
+ merge_base = ws.git("merge-base", new, base_sha).strip()
1357
+ return new if new and merge_base != new else ""
1358
+ except Exception as exc:
1359
+ log.warning("base-moved check failed (%s); wake proceeds unchanged", type(exc).__name__)
1360
+ return ""
1361
+
1362
+
1126
1363
  def _push_line_snapshot(
1127
- ws: Workspace, line_ref: str, run_id: str, outcome: str, secrets: tuple[str, ...] = ()
1364
+ ws: Workspace,
1365
+ line_ref: str,
1366
+ run_id: str,
1367
+ outcome: str,
1368
+ secrets: tuple[str, ...] = (),
1369
+ bot_login: str = "",
1128
1370
  ) -> None:
1129
1371
  """Publish the session's final tree to the agent's line as a sealed
1130
1372
  snapshot commit — every terminal path, any outcome
@@ -1167,7 +1409,7 @@ def _push_line_snapshot(
1167
1409
  fork = parent = remote
1168
1410
  except Exception as exc:
1169
1411
  log.info("line %s: sealing on the local ref (%s)", line_ref, type(exc).__name__)
1170
- snap = snapshot_tree(ws, parent, force=memory)
1412
+ snap = snapshot_tree(ws, parent, force=memory, author=bot_login)
1171
1413
  try:
1172
1414
  # seal only when the tree moved past the parent; the PUSH runs
1173
1415
  # either way — a session that COMMITTED its work advanced the
@@ -1176,10 +1418,7 @@ def _push_line_snapshot(
1176
1418
  sealed = parent
1177
1419
  if snap.tree != ws.git("rev-parse", f"{parent}^{{tree}}").strip():
1178
1420
  sealed = ws.git(
1179
- "-c",
1180
- "user.name=autoresearch",
1181
- "-c",
1182
- "user.email=autoresearch@localhost",
1421
+ *git_identity(bot_login),
1183
1422
  "commit-tree",
1184
1423
  snap.tree,
1185
1424
  "-p",
@@ -1239,7 +1478,9 @@ def _reconcile_with_remote(ws: Workspace, old: str, new: str) -> None:
1239
1478
  ws.git("checkout", new, "--", path)
1240
1479
 
1241
1480
 
1242
- def _checkout_line(ws: Workspace, workspace: Path, agent_id: str, base_branch: str) -> str:
1481
+ def _checkout_line(
1482
+ ws: Workspace, workspace: Path, agent_id: str, base_branch: str, bot_login: str = ""
1483
+ ) -> str:
1243
1484
  """Check out the agent's research line: the persistent branch
1244
1485
  `agents/<agent-id>`, created from the base branch when absent, with the
1245
1486
  base branch merged in when it exists — a conflicted merge is left in the
@@ -1259,10 +1500,7 @@ def _checkout_line(ws: Workspace, workspace: Path, agent_id: str, base_branch: s
1259
1500
  conflicted = False
1260
1501
  try:
1261
1502
  ws.git(
1262
- "-c",
1263
- "user.name=autoresearch",
1264
- "-c",
1265
- "user.email=autoresearch@localhost",
1503
+ *git_identity(bot_login),
1266
1504
  "merge",
1267
1505
  "--no-edit",
1268
1506
  base_ref,
@@ -1277,10 +1515,7 @@ def _checkout_line(ws: Workspace, workspace: Path, agent_id: str, base_branch: s
1277
1515
  ws.git("add", "-A")
1278
1516
  if ws.git("status", "--porcelain").strip():
1279
1517
  ws.git(
1280
- "-c",
1281
- "user.name=autoresearch",
1282
- "-c",
1283
- "user.email=autoresearch@localhost",
1518
+ *git_identity(bot_login),
1284
1519
  "commit",
1285
1520
  "-q",
1286
1521
  "-m",
@@ -1466,12 +1701,13 @@ def resume_run(
1466
1701
  run_dir = run_root / "runs" / run_id
1467
1702
  workspace = run_dir / "ws"
1468
1703
  record = load_record(run_root, run_id)
1704
+ dispatch = with_seed(dispatch, run_root, record.target)
1469
1705
  stage = record.stage
1470
1706
  # Push to the CANONICAL target URL, never the workspace's remote.origin.url:
1471
1707
  # the session could have rewritten that config to exfil the bot token / code
1472
1708
  # to another remote. Passing `url` here means `Workspace.push` uses it
1473
1709
  # instead of reading `remote.origin.url`.
1474
- ws = Workspace(root=workspace, auth=bot_auth, url=_target_clone_url(record.target))
1710
+ ws = Workspace(root=workspace, auth=bot_auth, url=target_clone_url(record.target))
1475
1711
  # A session reshaped .git (symlinked object store, gitdir file, FIFO) is
1476
1712
  # refused BEFORE anything writes through it: the exclude below opens
1477
1713
  # .git/info/exclude, and every ws.git call re-checks. The refusal ENDS
@@ -1643,11 +1879,16 @@ def resume_run(
1643
1879
  minutes=int(item.get("minutes") or 1),
1644
1880
  artifacts=tuple(str(a) for a in item.get("artifacts", [])),
1645
1881
  array=int(item.get("array") or 1),
1882
+ why=str(item.get("why") or ""),
1883
+ concurrency=int(item.get("concurrency") or 0),
1646
1884
  )
1647
1885
  for item in _stage_launches(record)
1648
1886
  ),
1649
1887
  note=str(stage.get("syscall_note", "")),
1650
1888
  submit=True,
1889
+ # the author's report rides every re-park: a suite fan-out
1890
+ # must not drop what the panel and the PR read
1891
+ report=str(stage.get("report", "")),
1651
1892
  )
1652
1893
  old_afterany = str(record.stage.get("afterany", ""))
1653
1894
  made_progress = bool(parked.afterany) and parked.afterany != old_afterany
@@ -1682,7 +1923,19 @@ def resume_run(
1682
1923
  # the park's sibling launches are done too: settle their charge before
1683
1924
  # any path — publish or hand back to the author — reads the budget
1684
1925
  if _stage_launches(record):
1685
- _reconcile_launch_hours(record, dispatch, bench.gpus, _stage_syscall_launches(record))
1926
+ sibling_launches = _stage_syscall_launches(record)
1927
+ sibling_ids = launch_task_ids(sibling_launches, stage_launch_job_ids(record))
1928
+ sibling_elapsed = _launch_elapsed(dispatch, sibling_ids) if sibling_ids else None
1929
+ _reconcile_launch_hours(record, dispatch, bench.gpus, sibling_launches, sibling_elapsed)
1930
+ # the ledger's ended records for the sibling launches, whether or not
1931
+ # the author is woken: the PR's experiments table reads them (an
1932
+ # author wake that follows records nothing twice)
1933
+ _best_effort(
1934
+ "launch ledger",
1935
+ lambda: _ledger_ended(
1936
+ run_dir, record, sibling_launches, sibling_ids, dispatch, sibling_elapsed
1937
+ ),
1938
+ )
1686
1939
 
1687
1940
  def _wake_author(
1688
1941
  extra_update: str, judged: tuple[str, AttemptResult] | None = None
@@ -1741,7 +1994,14 @@ def resume_run(
1741
1994
  # Research lines: record the tree AS OF THIS DECIDED TERMINAL — never
1742
1995
  # earlier, because a blocking panel verdict can still resume the
1743
1996
  # author (a continuation, not a terminal).
1744
- _push_line_snapshot(ws, _line_ref_for(bench, config.agent_id), run_id, outcome, secrets)
1997
+ _push_line_snapshot(
1998
+ ws,
1999
+ _line_ref_for(bench, config.agent_id),
2000
+ run_id,
2001
+ outcome,
2002
+ secrets,
2003
+ bot_login=config.bot_login,
2004
+ )
1745
2005
 
1746
2006
  if result.outcome == "improved":
1747
2007
  # Publish: branch the SEALED candidate sha, fold in the ledger, push,
@@ -1888,6 +2148,7 @@ def resume_run(
1888
2148
  exclude=(
1889
2149
  LINE_MEMORY_PATHS if _line_ref_for(bench, config.agent_id) else ()
1890
2150
  ),
2151
+ secrets=secrets,
1891
2152
  )(baseline, candidate, str(stage.get("report", "")))
1892
2153
  except Exception as exc:
1893
2154
  if isinstance(exc, GitError) and _is_git_tamper(exc):
@@ -1955,18 +2216,23 @@ def resume_run(
1955
2216
  # so push the candidate as-is rather than an empty commit.
1956
2217
  if staged:
1957
2218
  ws.git(
1958
- "-c",
1959
- f"user.name={config.bot_login}",
1960
- "-c",
1961
- f"user.email={config.bot_login}@users.noreply.github.com",
2219
+ *git_identity(config.bot_login),
1962
2220
  "commit",
1963
2221
  "-m",
1964
2222
  f"agent: improve {config.benchmark} ({_title_pair(baseline, candidate)})"
1965
2223
  f"\n\nAgent: {config.agent_id}",
1966
2224
  )
1967
2225
  ws.push(branch)
2226
+ if record.stage.get("submitted"):
2227
+ # the author's report at submit rides the stage: the PR shows it
2228
+ # as the research report, over the ledger's experiments
2229
+ result = dc_replace(result, submit_report=str(record.stage.get("report") or ""))
1968
2230
  body = pr_body(
1969
- result, config, redact_secrets=secrets, display_digits=bench.display_digits
2231
+ result,
2232
+ config,
2233
+ redact_secrets=secrets,
2234
+ display_digits=bench.display_digits,
2235
+ experiments=experiments_rows(run_dir),
1970
2236
  )
1971
2237
  if issue_number:
1972
2238
  body = f"Addresses #{issue_number}.\n\n{body}"
@@ -2264,6 +2530,7 @@ def build_panel_runner(
2264
2530
  start_round: int = 0,
2265
2531
  exclude: tuple[str, ...] = (),
2266
2532
  claim_body: Callable[[float, float, str], str] | None = None,
2533
+ secrets: tuple[str, ...] = (),
2267
2534
  ) -> Callable[[float, float, str], PanelVerdict]:
2268
2535
  """The git half of the pre-PR panel: prepare the two read-only checkouts
2269
2536
  and the synthetic claim, then hand off to `run_panel` (which owns no git).
@@ -2287,6 +2554,10 @@ def build_panel_runner(
2287
2554
  )
2288
2555
 
2289
2556
  def runner(baseline: float, candidate: float, report: str) -> PanelVerdict:
2557
+ # the claim is author text (the report at submit, or the session's
2558
+ # last words): redacted before any lens sees it, like the record and
2559
+ # the PR body
2560
+ report = redact(report, secrets)
2290
2561
  reads["n"] += 1
2291
2562
  panel_ws = run_dir / "panel"
2292
2563
  shutil.rmtree(panel_ws, ignore_errors=True)
@@ -2299,10 +2570,7 @@ def build_panel_runner(
2299
2570
  tree = ws.git("write-tree").strip()
2300
2571
  ws.git("reset")
2301
2572
  snapshot = ws.git(
2302
- "-c",
2303
- "user.name=panel",
2304
- "-c",
2305
- "user.email=panel@localhost",
2573
+ *git_identity(bot_login),
2306
2574
  "commit-tree",
2307
2575
  tree,
2308
2576
  "-p",
@@ -2432,7 +2700,7 @@ def live_attempt(
2432
2700
  # exception path cannot rely on names bound inside the try
2433
2701
  salvage: dict[str, object] = {}
2434
2702
  try:
2435
- ws = Workspace.clone(_target_clone_url(config.target), workspace, auth=bot_auth)
2703
+ ws = Workspace.clone(target_clone_url(config.target), workspace, auth=bot_auth)
2436
2704
  # Build ON the requested PR base: the clone checks out the remote
2437
2705
  # DEFAULT branch, which need not be `base_branch` — the session must
2438
2706
  # edit, and the gate must measure, the tree the PR will land on.
@@ -2496,7 +2764,9 @@ def live_attempt(
2496
2764
  line_ref = ""
2497
2765
  if lines_active:
2498
2766
  try:
2499
- line_ref = _checkout_line(ws, workspace, config.agent_id, base_branch)
2767
+ line_ref = _checkout_line(
2768
+ ws, workspace, config.agent_id, base_branch, config.bot_login
2769
+ )
2500
2770
  except Exception as exc:
2501
2771
  log.warning(
2502
2772
  "line checkout failed (%s); running on %s",
@@ -2639,6 +2909,7 @@ def live_attempt(
2639
2909
  config.bot_login,
2640
2910
  created[:10],
2641
2911
  exclude=LINE_MEMORY_PATHS if lines_active else (),
2912
+ secrets=secrets,
2642
2913
  )
2643
2914
  if panel_lenses
2644
2915
  else None
@@ -2672,12 +2943,16 @@ def live_attempt(
2672
2943
  run_tag=run_id,
2673
2944
  # an inline gate shares the same target-wide baseline cache
2674
2945
  baseline_cache=run_dir.parent / "baselines",
2946
+ seed_cache=dispatch.seed_cache if dispatch is not None else None,
2675
2947
  )
2676
2948
  snapshots: list[Snapshot] = []
2677
2949
 
2678
2950
  def snapshot() -> str:
2679
2951
  snap = snapshot_tree(
2680
- ws, pre_session_sha, exclude=LINE_MEMORY_PATHS if lines_active else ()
2952
+ ws,
2953
+ pre_session_sha,
2954
+ exclude=LINE_MEMORY_PATHS if lines_active else (),
2955
+ author=config.bot_login,
2681
2956
  )
2682
2957
  snapshots.append(snap)
2683
2958
  return snap.commit
@@ -2722,6 +2997,11 @@ def live_attempt(
2722
2997
  line_memory=line_memory,
2723
2998
  line_divergence=line_divergence,
2724
2999
  launcher=launcher,
3000
+ watcher=(
3001
+ _make_watcher(dispatch, run_root, run_id, workspace, config)
3002
+ if dispatch is not None
3003
+ else None
3004
+ ),
2725
3005
  tree_of=lambda sha: ws.git("rev-parse", f"{sha}^{{tree}}").strip(),
2726
3006
  )
2727
3007
  except RunParked as p:
@@ -2786,6 +3066,7 @@ def live_attempt(
2786
3066
  run_id,
2787
3067
  "attempt-error",
2788
3068
  secrets,
3069
+ bot_login=config.bot_login,
2789
3070
  )
2790
3071
  failed = RunRecord(
2791
3072
  **{
@@ -2843,7 +3124,7 @@ def live_attempt(
2843
3124
  # memory (it is excluded from measurable seals by design). The label is
2844
3125
  # the GATE outcome, correct at this moment; a publish failure appends a
2845
3126
  # publish-error snapshot at the tail.
2846
- _push_line_snapshot(ws, line_ref, run_id, result.outcome, secrets)
3127
+ _push_line_snapshot(ws, line_ref, run_id, result.outcome, secrets, bot_login=config.bot_login)
2847
3128
 
2848
3129
  pr_url = ""
2849
3130
  outcome_name = result.outcome
@@ -2895,10 +3176,7 @@ def live_attempt(
2895
3176
  raise WorkspaceDrift(f"publish would stage non-ledger paths: {extra[:10]}")
2896
3177
  if staged:
2897
3178
  ws.git(
2898
- "-c",
2899
- f"user.name={config.bot_login}",
2900
- "-c",
2901
- f"user.email={config.bot_login}@users.noreply.github.com",
3179
+ *git_identity(config.bot_login),
2902
3180
  "commit",
2903
3181
  "-m",
2904
3182
  f"agent: improve {config.benchmark} ({_title_pair(baseline, candidate)})"
@@ -2907,7 +3185,11 @@ def live_attempt(
2907
3185
  ws.push(branch)
2908
3186
  pushed = True
2909
3187
  body = pr_body(
2910
- result, config, redact_secrets=secrets, display_digits=bench.display_digits
3188
+ result,
3189
+ config,
3190
+ redact_secrets=secrets,
3191
+ display_digits=bench.display_digits,
3192
+ experiments=experiments_rows(run_dir),
2911
3193
  )
2912
3194
  if issue_number:
2913
3195
  body = f"Addresses #{issue_number}.\n\n{body}"
@@ -3022,7 +3304,7 @@ def live_attempt(
3022
3304
  # the publish failed after the gate credited the tree: the improved
3023
3305
  # snapshot above stands (the measurement was real); append the
3024
3306
  # publish-error marker so the notebook records how the run ended
3025
- _push_line_snapshot(ws, line_ref, run_id, outcome_name, secrets)
3307
+ _push_line_snapshot(ws, line_ref, run_id, outcome_name, secrets, bot_login=config.bot_login)
3026
3308
  log.info("run %s: %s %s", run_id, outcome_name, pr_url)
3027
3309
  return AttemptOutcome(
3028
3310
  run_id=run_id,
@@ -3115,7 +3397,6 @@ def arm_sigterm_containment() -> None:
3115
3397
  def main() -> int:
3116
3398
  import argparse
3117
3399
  import os
3118
- import time
3119
3400
  from datetime import UTC, datetime
3120
3401
 
3121
3402
  arm_sigterm_containment()
@@ -3230,13 +3511,7 @@ def main() -> int:
3230
3511
  default=120.0,
3231
3512
  help="how long before the walltime the self-deadline fires (floor 60)",
3232
3513
  )
3233
- parser.add_argument("--pat-file", default=str(CONFIG_DIR / "bot_pat"))
3234
- parser.add_argument(
3235
- "--github-app-file",
3236
- default=os.environ.get("OUTERLOOP_GITHUB_APP_FILE", ""),
3237
- help="GitHub App config (JSON: app_id, installation_id, private_key); "
3238
- "when set, installation tokens replace the PAT",
3239
- )
3514
+ add_credential_args(parser)
3240
3515
  parser.add_argument(
3241
3516
  "--key-file",
3242
3517
  default="",