outerloop-science 0.1.0.dev2__py3-none-any.whl → 0.1.0.dev3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. outerloop/__init__.py +2 -2
  2. outerloop/attempt.py +310 -93
  3. outerloop/brief.py +38 -25
  4. outerloop/cli.py +40 -5
  5. outerloop/climbboard.py +3 -0
  6. outerloop/compute.py +148 -53
  7. outerloop/contract.py +8 -0
  8. outerloop/dispatch.py +63 -18
  9. outerloop/evalcache.py +147 -0
  10. outerloop/followup.py +38 -16
  11. outerloop/github.py +38 -13
  12. outerloop/harness.py +1 -18
  13. outerloop/housekeeping.py +1 -17
  14. outerloop/image.py +0 -4
  15. outerloop/init.py +19 -1
  16. outerloop/intake.py +4 -7
  17. outerloop/launchlog.py +239 -0
  18. outerloop/maintain.py +325 -0
  19. outerloop/maintain_agent_cli.py +81 -0
  20. outerloop/maintain_post_cli.py +140 -0
  21. outerloop/measure.py +6 -0
  22. outerloop/orchestrator.py +141 -31
  23. outerloop/panel.py +3 -3
  24. outerloop/review.py +4 -0
  25. outerloop/review_agent.py +7 -7
  26. outerloop/review_agent_cli.py +2 -2
  27. outerloop/review_post_cli.py +2 -2
  28. outerloop/review_summarize_cli.py +7 -5
  29. outerloop/roles.py +27 -0
  30. outerloop/rolespec.py +3 -1
  31. outerloop/steward.py +5 -5
  32. outerloop/syscall.py +261 -47
  33. outerloop/syscall_cli.py +243 -12
  34. outerloop/tick.py +90 -178
  35. outerloop/verify_agent.py +8 -6
  36. outerloop/verify_post_cli.py +2 -2
  37. outerloop/watcher.py +203 -0
  38. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/METADATA +4 -1
  39. outerloop_science-0.1.0.dev3.dist-info/RECORD +59 -0
  40. outerloop_science-0.1.0.dev2.dist-info/RECORD +0 -53
  41. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/WHEEL +0 -0
  42. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/entry_points.txt +0 -0
  43. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/licenses/LICENSE +0 -0
  44. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/licenses/NOTICE +0 -0
outerloop/attempt.py CHANGED
@@ -18,6 +18,7 @@ import logging
18
18
  import os
19
19
  import re
20
20
  import shutil
21
+ import time
21
22
  from collections.abc import Callable, Iterable
22
23
  from dataclasses import dataclass
23
24
  from dataclasses import replace as dc_replace
@@ -27,7 +28,7 @@ from typing import Any, cast
27
28
 
28
29
  from outerloop.appauth import resolve_bot_auth
29
30
  from outerloop.brief import BudgetState, distill_lessons
30
- from outerloop.compute import LocalCompute, local_mode
31
+ from outerloop.compute import LocalCompute
31
32
  from outerloop.contract import Benchmark, Contract, contract_text_in_tree, load_contract
32
33
  from outerloop.dispatch import (
33
34
  Snapshot,
@@ -36,6 +37,7 @@ from outerloop.dispatch import (
36
37
  should_dispatch,
37
38
  snapshot_tree,
38
39
  )
40
+ from outerloop.evalcache import seed_dir
39
41
  from outerloop.github import (
40
42
  GitError,
41
43
  GitHubClient,
@@ -43,8 +45,10 @@ from outerloop.github import (
43
45
  Workspace,
44
46
  contract_at,
45
47
  ensure_regular_git_dir,
48
+ git_identity,
46
49
  )
47
50
  from outerloop.harness import Harness, SessionResult, default_binary, redact
51
+ from outerloop.launchlog import append_ended, append_submitted, experiments_rows
48
52
  from outerloop.markers import has_marker
49
53
  from outerloop.measure import DispatchedMeasurer, DispatchSettings
50
54
  from outerloop.orchestrator import (
@@ -84,9 +88,20 @@ from outerloop.runstate import (
84
88
  save_record,
85
89
  stamp_outage,
86
90
  )
87
- from outerloop.syscall import CHANNEL_DIR_NAMES, MAX_ARTIFACT_BYTES, SyscallRequest, channel_dir
91
+ from outerloop.runstate import (
92
+ run_dir as run_dir_of,
93
+ )
94
+ from outerloop.syscall import (
95
+ CHANNEL_DIR_NAMES,
96
+ MAX_ARTIFACT_BYTES,
97
+ SyscallRequest,
98
+ channel_dir,
99
+ launch_task_ids,
100
+ tool_update_note,
101
+ )
88
102
  from outerloop.syscall import ensure_excluded as syscall_excluded
89
103
  from outerloop.syscall import install_tool as syscall_install_tool
104
+ from outerloop.syscall import refresh_tool as syscall_refresh_tool
90
105
  from outerloop.syscall import write_budget as syscall_write_budget
91
106
  from outerloop.syscall import write_siblings as syscall_write_siblings
92
107
  from outerloop.verifier import MAX_CLAIM_CHARS
@@ -186,7 +201,7 @@ class WorkspaceDrift(RuntimeError):
186
201
  """The tree changed between measurement and commit."""
187
202
 
188
203
 
189
- def _target_clone_url(target: str) -> str:
204
+ def target_clone_url(target: str) -> str:
190
205
  """The canonical HTTPS clone URL for `owner/repo`. The one source of truth
191
206
  for where a run's git pushes go — derived from the target, never read from
192
207
  the session-writable `remote.origin.url`."""
@@ -206,7 +221,8 @@ def _blessed_head(ws: Workspace, result: Any, contract: Any) -> str:
206
221
  return ""
207
222
  try:
208
223
  return ws.git("rev-parse", "HEAD").strip()
209
- except Exception:
224
+ except Exception as exc:
225
+ log.warning("could not read HEAD; not arming self-merge: %s", exc)
210
226
  return ""
211
227
 
212
228
 
@@ -425,9 +441,14 @@ def _park_run(
425
441
  # it lands in the durable record, like every other persisted final_text
426
442
  # — a session that echoed a credential must not leave it in record.json.
427
443
  # Empty/zero for a baseline park (the session has not run yet).
428
- "report": redact(parked.session.final_text, secrets)[:MAX_CLAIM_CHARS]
429
- if parked.session
430
- else "",
444
+ # the author's report at submit, else the session's last words (a park
445
+ # with no submit); either way what the wake's panel and the PR body read
446
+ "report": redact(
447
+ parked.syscall.report
448
+ if parked.syscall is not None and parked.syscall.report
449
+ else (parked.session.final_text if parked.session else ""),
450
+ secrets,
451
+ )[:MAX_CLAIM_CHARS],
431
452
  "session_cost_usd": parked.session.cost_usd if parked.session else 0.0,
432
453
  "session_turns": parked.session.num_turns if parked.session else 0,
433
454
  }
@@ -452,10 +473,35 @@ def _park_run(
452
473
  "minutes": launch.minutes,
453
474
  "artifacts": list(launch.artifacts),
454
475
  **({"array": launch.array} if launch.array > 1 else {}),
476
+ **({"why": redact(launch.why, secrets)} if launch.why else {}),
477
+ **({"concurrency": launch.concurrency} if launch.concurrency else {}),
455
478
  }
456
479
  for launch in parked.syscall.launches
457
480
  ]
458
481
  stage["syscall_note"] = redact(parked.syscall.note, secrets)
482
+ if parked.syscall.launches:
483
+ # the run's launch ledger (`history`, and the queue view's labels):
484
+ # ids align with launch_jobs order, as stage_launch_job_ids reads them
485
+ if parked.launch_afterany:
486
+ launch_ids = afterany_ids(parked.launch_afterany)
487
+ elif parked.phase == "author-sleep":
488
+ launch_ids = list(job_ids)
489
+ else:
490
+ launch_ids = []
491
+ ledger_launches = tuple(
492
+ dc_replace(launch, why=redact(launch.why, secrets))
493
+ for launch in parked.syscall.launches
494
+ )
495
+ _best_effort(
496
+ "launch ledger",
497
+ lambda: append_submitted(
498
+ run_dir_of(run_root, record.run_id),
499
+ sleep=parked.sleeps_used,
500
+ launches=ledger_launches,
501
+ job_ids=launch_ids,
502
+ at=now,
503
+ ),
504
+ )
459
505
  # (the session id the wake resumes is the record's own
460
506
  # resume_session_id, set below for every park — no stage duplicate)
461
507
  stage["launches_used"] = parked.launches_used
@@ -586,6 +632,7 @@ def _dispatch_settings(args: argparse.Namespace) -> DispatchSettings:
586
632
  #174: the wake dropped the GPU lane)."""
587
633
  from outerloop.compute import compute_from_env
588
634
 
635
+ target = getattr(args, "target", "") or ""
589
636
  return DispatchSettings(
590
637
  compute=compute_from_env(),
591
638
  image=args.image,
@@ -593,9 +640,34 @@ def _dispatch_settings(args: argparse.Namespace) -> DispatchSettings:
593
640
  partition=args.partition,
594
641
  gpu_partition=getattr(args, "gpu_partition", ""),
595
642
  gpu_account=getattr(args, "gpu_account", ""),
643
+ seed_cache=seed_dir(Path(args.run_root), target) if target else None,
596
644
  )
597
645
 
598
646
 
647
+ # Experiments yield to verification. Every kernel job is one Slurm user, so
648
+ # among the kernel's own pending jobs the priority order is ours: launches
649
+ # carry this nice so a gate eval or a follow-up re-measure (nice 0) starts
650
+ # first when the cap frees a slot. Sized above the factors that differ between
651
+ # our jobs on Torch — age tops out at 1000 after a week, job size at 1000, the
652
+ # per-GPU TRES share stays in the hundreds — so the order holds however long a
653
+ # launch has waited. Other users' jobs and the group cap are untouched.
654
+ LAUNCH_NICE = 5000
655
+
656
+
657
+ def with_seed(dispatch: DispatchSettings, run_root: Path, target: str) -> DispatchSettings:
658
+ """These settings with the target's seed cache filled in from the record's
659
+ target when the CLI gave none (wake and follow-up jobs carry the run id,
660
+ not the target)."""
661
+ # tolerant of any settings object: a backend that knows no seed (or a
662
+ # test double) is left exactly as it is
663
+ if not target or getattr(dispatch, "seed_cache", "unknown") is not None:
664
+ return dispatch
665
+ try:
666
+ return dc_replace(dispatch, seed_cache=seed_dir(run_root, target))
667
+ except TypeError:
668
+ return dispatch
669
+
670
+
599
671
  def _make_launcher(
600
672
  dispatch: DispatchSettings, run_dir: Path, workspace: Path, run_id: str, gpus: int = 0
601
673
  ):
@@ -605,45 +677,42 @@ def _make_launcher(
605
677
  partially-submitted batch is reaped rather than orphaned. `gpus` is the
606
678
  benchmark's: an author's experiments run on the same lane as its evals."""
607
679
  account, partition = dispatch.placement(gpus)
608
- # under launch admission a GPU launch enters the queue held; the tick
609
- # releases it when the user's GPUs fit under the cap (tick.service_admission)
610
- from outerloop.tick import max_launch_gpus_from_env
611
-
612
- hold = gpus > 0 and not local_mode() and max_launch_gpus_from_env() > 0
613
680
 
614
681
  def launcher(sha: str, request: SyscallRequest) -> str:
615
- from dataclasses import replace as _replace
616
-
617
682
  from outerloop.dispatch import eval_job_spec, write_eval_job
618
- from outerloop.syscall import launch_jobs
683
+ from outerloop.syscall import array_spec
619
684
 
620
685
  ids: list[str] = []
621
686
  try:
622
687
  for launch in request.launches:
623
- # an array launch is N jobs of one command, each with its
624
- # SWEEP_INDEX; one afterany wake covers them all
625
- for job_name, extra_env in launch_jobs(launch):
626
- script = write_eval_job(
627
- run_dir,
628
- f"launch-{job_name}",
629
- repo_root=workspace,
630
- snapshot_sha=sha,
631
- command=launch.command,
632
- image=dispatch.image,
633
- extra_env=extra_env,
634
- artifacts=launch.artifacts,
635
- artifact_max_bytes=MAX_ARTIFACT_BYTES,
636
- gpus=gpus,
637
- )
638
- spec = eval_job_spec(
639
- script,
640
- job_name=f"{run_id}-launch-{job_name}",
641
- account=account,
642
- partition=partition,
643
- eval_minutes=launch.minutes,
644
- gpus=gpus,
645
- )
646
- ids.append(dispatch.compute.submit(_replace(spec, hold=hold)))
688
+ # a sweep is ONE Slurm job array (`--array=0-N%K`): the queue
689
+ # holds one entry, Slurm runs at most K tasks at once, each task
690
+ # derives its job dir and SWEEP_INDEX from its array index, and
691
+ # one afterany on the array id covers every task
692
+ script = write_eval_job(
693
+ run_dir,
694
+ f"launch-{launch.name}",
695
+ repo_root=workspace,
696
+ snapshot_sha=sha,
697
+ command=launch.command,
698
+ image=dispatch.image,
699
+ artifacts=launch.artifacts,
700
+ artifact_max_bytes=MAX_ARTIFACT_BYTES,
701
+ gpus=gpus,
702
+ array=launch.array,
703
+ seed_cache=dispatch.seed_cache,
704
+ )
705
+ spec = eval_job_spec(
706
+ script,
707
+ job_name=f"{run_id}-launch-{launch.name}",
708
+ account=account,
709
+ partition=partition,
710
+ eval_minutes=launch.minutes,
711
+ gpus=gpus,
712
+ nice=LAUNCH_NICE,
713
+ array=array_spec(launch),
714
+ )
715
+ ids.append(dispatch.compute.submit(spec))
647
716
  except Exception:
648
717
  # a partial batch must not orphan: no park record was written yet,
649
718
  # so nothing would ever wake or cancel the jobs that DID submit —
@@ -661,6 +730,26 @@ def _make_launcher(
661
730
  return launcher
662
731
 
663
732
 
733
+ def _make_watcher(
734
+ dispatch: DispatchSettings, run_root: Path, run_id: str, workspace: Path, config: RunConfig
735
+ ) -> Callable[[], Any]:
736
+ """The session watcher for one run: a thread beside the harness that
737
+ answers `queue` and `history` from the channel (docs/design/session-watcher.md).
738
+ Shared by the first pass and every wake leg."""
739
+ from outerloop.watcher import SessionWatcher, WatcherContext
740
+
741
+ ctx = WatcherContext(
742
+ workspace=workspace,
743
+ run_root=run_root,
744
+ run_id=run_id,
745
+ target=config.target,
746
+ agent_id=config.agent_id,
747
+ compute=dispatch.compute,
748
+ gpu_partition=dispatch.gpu_partition,
749
+ )
750
+ return lambda: SessionWatcher(ctx)
751
+
752
+
664
753
  def _wake_author_sleep(
665
754
  *,
666
755
  run_root: Path,
@@ -713,7 +802,12 @@ def _wake_author_sleep(
713
802
  # the same ending shape every other terminal takes. The line notebook
714
803
  # records it first, while the tree is still the session's final tree.
715
804
  _push_line_snapshot(
716
- ws, _line_ref_for(bench, config.agent_id), run_id, result.outcome, secrets
805
+ ws,
806
+ _line_ref_for(bench, config.agent_id),
807
+ run_id,
808
+ result.outcome,
809
+ secrets,
810
+ bot_login=config.bot_login,
717
811
  )
718
812
  for ref in drop_refs:
719
813
  drop_snapshot(ws, Snapshot(commit="", tree="", ref=ref))
@@ -776,6 +870,8 @@ def _wake_author_sleep(
776
870
  minutes=int(item.get("minutes") or 1),
777
871
  artifacts=tuple(str(a) for a in item.get("artifacts", [])),
778
872
  array=int(item.get("array") or 1),
873
+ why=str(item.get("why") or ""),
874
+ concurrency=int(item.get("concurrency") or 0),
779
875
  )
780
876
  for item in _stage_launches(record)
781
877
  )
@@ -786,12 +882,28 @@ def _wake_author_sleep(
786
882
  # of a blank "job failure". The park's launch job ids align positionally
787
883
  # with the results (same launch/array order). Best-effort — the wake never
788
884
  # blocks on the scheduler query.
885
+ task_ids = launch_task_ids(launches, stage_launch_job_ids(record))
789
886
  status_of = getattr(dispatch.compute, "status", None)
790
887
  if status_of is not None:
791
- results = annotate_launch_states(results, _stage_launch_job_ids(record), status_of)
888
+ results = annotate_launch_states(results, task_ids, status_of)
792
889
  launches_used = int(record.stage.get("launches_used", 0)) # type: ignore[call-overload]
793
890
  sleeps_used = int(record.stage.get("sleeps_used", 0)) # type: ignore[call-overload]
794
- gpu_hours_used = _reconcile_launch_hours(record, dispatch, bench.gpus, launches)
891
+ elapsed = _launch_elapsed(dispatch, task_ids) if task_ids else None
892
+ _best_effort(
893
+ "launch ledger",
894
+ lambda: append_ended(
895
+ run_dir, sleep=sleeps_used, results=results, at=time.time(), elapsed_seconds=elapsed
896
+ ),
897
+ )
898
+ gpu_hours_used = _reconcile_launch_hours(record, dispatch, bench.gpus, launches, elapsed)
899
+ # the tool the session invokes comes from THIS kernel: a session that
900
+ # started under an older one gets today's verbs and flags at its wake, and
901
+ # is told what is new
902
+ tool_changed = False
903
+ try:
904
+ tool_changed = syscall_refresh_tool(workspace)
905
+ except Exception as exc:
906
+ log.warning("tool refresh failed: %s", redact(f"{type(exc).__name__}: {exc}", secrets))
795
907
  wake_text = render_wake(
796
908
  results,
797
909
  str(record.stage.get("syscall_note", "")),
@@ -804,6 +916,15 @@ def _wake_author_sleep(
804
916
  ),
805
917
  gpus=bench.gpus,
806
918
  )
919
+ pacing = [
920
+ f"sweep `{la.name}`: {la.array} tasks, at most {la.concurrency or la.array} at a time"
921
+ for la in launches
922
+ if la.array > 1
923
+ ]
924
+ if pacing:
925
+ wake_text = f"{wake_text}\n\n" + "\n".join(pacing) + " (the contract's ceiling applies)."
926
+ if tool_changed:
927
+ wake_text = f"{wake_text}\n\n{tool_update_note(channel_dir(workspace))}"
807
928
  if extra_update:
808
929
  # a submitted park's gate/panel feedback leads; launch results follow
809
930
  wake_text = f"{extra_update}\n\n{wake_text}"
@@ -829,7 +950,9 @@ def _wake_author_sleep(
829
950
  wake_line = _line_ref_for(bench, config.agent_id)
830
951
 
831
952
  def snapshot() -> str:
832
- snap = snapshot_tree(ws, base_sha, exclude=LINE_MEMORY_PATHS if wake_line else ())
953
+ snap = snapshot_tree(
954
+ ws, base_sha, exclude=LINE_MEMORY_PATHS if wake_line else (), author=config.bot_login
955
+ )
833
956
  snapshots.append(snap)
834
957
  return snap.commit
835
958
 
@@ -850,6 +973,7 @@ def _wake_author_sleep(
850
973
  config.bot_login,
851
974
  _utc_date(now),
852
975
  exclude=LINE_MEMORY_PATHS if wake_line else (),
976
+ secrets=secrets,
853
977
  )
854
978
  if panel_lenses
855
979
  else None
@@ -873,6 +997,7 @@ def _wake_author_sleep(
873
997
  resume_session_id=record.resume_session_id,
874
998
  improve_prompt=wake_text,
875
999
  launcher=_make_launcher(dispatch, run_dir, workspace, run_id, gpus=bench.gpus),
1000
+ watcher=_make_watcher(dispatch, run_root, run_id, workspace, config),
876
1001
  tree_of=lambda sha: ws.git("rev-parse", f"{sha}^{{tree}}").strip(),
877
1002
  judged=judged or _stage_judged(record),
878
1003
  launches_used=launches_used,
@@ -937,6 +1062,32 @@ def _stage_judged(record: RunRecord) -> tuple[str, AttemptResult] | None:
937
1062
  )
938
1063
 
939
1064
 
1065
+ def _ledger_ended(
1066
+ run_dir: Path,
1067
+ record: RunRecord,
1068
+ launches: tuple,
1069
+ task_ids: list[str],
1070
+ dispatch: DispatchSettings,
1071
+ elapsed: list[int | None] | None,
1072
+ ) -> None:
1073
+ """Record a park's finished launches in the ledger from the run dir alone
1074
+ (no delivery into a workspace), with the scheduler's state for jobs that
1075
+ left no exit code."""
1076
+ from outerloop.syscall import annotate_launch_states, read_results
1077
+
1078
+ results = read_results(run_dir, launches)
1079
+ status_of = getattr(dispatch.compute, "status", None)
1080
+ if status_of is not None:
1081
+ results = annotate_launch_states(results, task_ids, status_of)
1082
+ append_ended(
1083
+ run_dir,
1084
+ sleep=int(record.stage.get("sleeps_used", 0)), # type: ignore[call-overload]
1085
+ results=results,
1086
+ at=time.time(),
1087
+ elapsed_seconds=elapsed,
1088
+ )
1089
+
1090
+
940
1091
  def _stage_syscall_launches(record: RunRecord) -> tuple:
941
1092
  """The park's launches as `Launch` values (command elided: they ran)."""
942
1093
  from outerloop.syscall import Launch
@@ -948,12 +1099,14 @@ def _stage_syscall_launches(record: RunRecord) -> tuple:
948
1099
  minutes=int(item.get("minutes") or 1),
949
1100
  artifacts=tuple(str(a) for a in item.get("artifacts", [])),
950
1101
  array=int(item.get("array") or 1),
1102
+ why=str(item.get("why") or ""),
1103
+ concurrency=int(item.get("concurrency") or 0),
951
1104
  )
952
1105
  for item in _stage_launches(record)
953
1106
  )
954
1107
 
955
1108
 
956
- def _stage_launch_job_ids(record: RunRecord) -> list[str]:
1109
+ def stage_launch_job_ids(record: RunRecord) -> list[str]:
957
1110
  """The park's launch jobs: `launch_afterany` when the park recorded it;
958
1111
  for an older author-sleep park every waited job was a launch; for an
959
1112
  older candidate park the gate's evals are mixed in, so none."""
@@ -966,7 +1119,11 @@ def _stage_launch_job_ids(record: RunRecord) -> list[str]:
966
1119
 
967
1120
 
968
1121
  def _reconcile_launch_hours(
969
- record: RunRecord, dispatch: DispatchSettings, gpus: int, launches: tuple
1122
+ record: RunRecord,
1123
+ dispatch: DispatchSettings,
1124
+ gpus: int,
1125
+ launches: tuple,
1126
+ elapsed: list[int | None] | None = None,
970
1127
  ) -> float:
971
1128
  """The run's GPU-hours after handing back the unused walltime of the
972
1129
  park's launch jobs — once: the stage remembers the refund, so a wake
@@ -976,7 +1133,9 @@ def _reconcile_launch_hours(
976
1133
  used = float(stage.get("gpu_hours_used", 0.0)) # type: ignore[arg-type]
977
1134
  if not gpus or stage.get("launch_hours_refunded"):
978
1135
  return used
979
- refund = _launch_refund(dispatch, launches, _stage_launch_job_ids(record), gpus)
1136
+ refund = _launch_refund(
1137
+ dispatch, launches, launch_task_ids(launches, stage_launch_job_ids(record)), gpus, elapsed
1138
+ )
980
1139
  if refund > 0:
981
1140
  log.info("%s: refunding %.2f GPU-hours of unused launch walltime", record.run_id, refund)
982
1141
  used = max(0.0, used - refund)
@@ -985,21 +1144,35 @@ def _reconcile_launch_hours(
985
1144
  return used
986
1145
 
987
1146
 
1147
+ def _launch_elapsed(dispatch: DispatchSettings, job_ids: list[str]) -> list[int | None] | None:
1148
+ """How long each launch job ran, from the compute, aligned with `job_ids`;
1149
+ None when the compute cannot say (nothing is refunded or recorded on a
1150
+ guess)."""
1151
+ query = getattr(dispatch.compute, "elapsed_seconds", None)
1152
+ if query is None or not job_ids:
1153
+ return None
1154
+ try:
1155
+ return [query(jid) for jid in job_ids]
1156
+ except Exception as exc:
1157
+ log.warning("launch walltime unknown (%s: %s)", type(exc).__name__, exc)
1158
+ return None
1159
+
1160
+
988
1161
  def _launch_refund(
989
- dispatch: DispatchSettings, launches: tuple, job_ids: list[str], gpus: int
1162
+ dispatch: DispatchSettings,
1163
+ launches: tuple,
1164
+ job_ids: list[str],
1165
+ gpus: int,
1166
+ elapsed: list[int | None] | None = None,
990
1167
  ) -> float:
991
1168
  """The unused walltime of a park's launch jobs, in GPU-hours, or 0 when
992
1169
  the compute cannot say how long they ran (nothing is refunded on a
993
- guess)."""
1170
+ guess). `elapsed` may be handed in when the caller already asked."""
994
1171
  from outerloop.syscall import launch_hours_refund
995
1172
 
996
- query = getattr(dispatch.compute, "elapsed_seconds", None)
997
- if query is None or not job_ids:
998
- return 0.0
999
- try:
1000
- elapsed = [query(jid) for jid in job_ids]
1001
- except Exception as exc:
1002
- log.warning("launch walltime unknown (%s: %s); nothing refunded", type(exc).__name__, exc)
1173
+ if elapsed is None:
1174
+ elapsed = _launch_elapsed(dispatch, job_ids)
1175
+ if elapsed is None:
1003
1176
  return 0.0
1004
1177
  return launch_hours_refund(launches, elapsed, gpus=gpus)
1005
1178
 
@@ -1124,7 +1297,12 @@ def _line_ref_for(bench: Benchmark | None, agent_id: str) -> str:
1124
1297
 
1125
1298
 
1126
1299
  def _push_line_snapshot(
1127
- ws: Workspace, line_ref: str, run_id: str, outcome: str, secrets: tuple[str, ...] = ()
1300
+ ws: Workspace,
1301
+ line_ref: str,
1302
+ run_id: str,
1303
+ outcome: str,
1304
+ secrets: tuple[str, ...] = (),
1305
+ bot_login: str = "",
1128
1306
  ) -> None:
1129
1307
  """Publish the session's final tree to the agent's line as a sealed
1130
1308
  snapshot commit — every terminal path, any outcome
@@ -1167,7 +1345,7 @@ def _push_line_snapshot(
1167
1345
  fork = parent = remote
1168
1346
  except Exception as exc:
1169
1347
  log.info("line %s: sealing on the local ref (%s)", line_ref, type(exc).__name__)
1170
- snap = snapshot_tree(ws, parent, force=memory)
1348
+ snap = snapshot_tree(ws, parent, force=memory, author=bot_login)
1171
1349
  try:
1172
1350
  # seal only when the tree moved past the parent; the PUSH runs
1173
1351
  # either way — a session that COMMITTED its work advanced the
@@ -1176,10 +1354,7 @@ def _push_line_snapshot(
1176
1354
  sealed = parent
1177
1355
  if snap.tree != ws.git("rev-parse", f"{parent}^{{tree}}").strip():
1178
1356
  sealed = ws.git(
1179
- "-c",
1180
- "user.name=autoresearch",
1181
- "-c",
1182
- "user.email=autoresearch@localhost",
1357
+ *git_identity(bot_login),
1183
1358
  "commit-tree",
1184
1359
  snap.tree,
1185
1360
  "-p",
@@ -1239,7 +1414,9 @@ def _reconcile_with_remote(ws: Workspace, old: str, new: str) -> None:
1239
1414
  ws.git("checkout", new, "--", path)
1240
1415
 
1241
1416
 
1242
- def _checkout_line(ws: Workspace, workspace: Path, agent_id: str, base_branch: str) -> str:
1417
+ def _checkout_line(
1418
+ ws: Workspace, workspace: Path, agent_id: str, base_branch: str, bot_login: str = ""
1419
+ ) -> str:
1243
1420
  """Check out the agent's research line: the persistent branch
1244
1421
  `agents/<agent-id>`, created from the base branch when absent, with the
1245
1422
  base branch merged in when it exists — a conflicted merge is left in the
@@ -1259,10 +1436,7 @@ def _checkout_line(ws: Workspace, workspace: Path, agent_id: str, base_branch: s
1259
1436
  conflicted = False
1260
1437
  try:
1261
1438
  ws.git(
1262
- "-c",
1263
- "user.name=autoresearch",
1264
- "-c",
1265
- "user.email=autoresearch@localhost",
1439
+ *git_identity(bot_login),
1266
1440
  "merge",
1267
1441
  "--no-edit",
1268
1442
  base_ref,
@@ -1277,10 +1451,7 @@ def _checkout_line(ws: Workspace, workspace: Path, agent_id: str, base_branch: s
1277
1451
  ws.git("add", "-A")
1278
1452
  if ws.git("status", "--porcelain").strip():
1279
1453
  ws.git(
1280
- "-c",
1281
- "user.name=autoresearch",
1282
- "-c",
1283
- "user.email=autoresearch@localhost",
1454
+ *git_identity(bot_login),
1284
1455
  "commit",
1285
1456
  "-q",
1286
1457
  "-m",
@@ -1466,12 +1637,13 @@ def resume_run(
1466
1637
  run_dir = run_root / "runs" / run_id
1467
1638
  workspace = run_dir / "ws"
1468
1639
  record = load_record(run_root, run_id)
1640
+ dispatch = with_seed(dispatch, run_root, record.target)
1469
1641
  stage = record.stage
1470
1642
  # Push to the CANONICAL target URL, never the workspace's remote.origin.url:
1471
1643
  # the session could have rewritten that config to exfil the bot token / code
1472
1644
  # to another remote. Passing `url` here means `Workspace.push` uses it
1473
1645
  # instead of reading `remote.origin.url`.
1474
- ws = Workspace(root=workspace, auth=bot_auth, url=_target_clone_url(record.target))
1646
+ ws = Workspace(root=workspace, auth=bot_auth, url=target_clone_url(record.target))
1475
1647
  # A session reshaped .git (symlinked object store, gitdir file, FIFO) is
1476
1648
  # refused BEFORE anything writes through it: the exclude below opens
1477
1649
  # .git/info/exclude, and every ws.git call re-checks. The refusal ENDS
@@ -1643,11 +1815,16 @@ def resume_run(
1643
1815
  minutes=int(item.get("minutes") or 1),
1644
1816
  artifacts=tuple(str(a) for a in item.get("artifacts", [])),
1645
1817
  array=int(item.get("array") or 1),
1818
+ why=str(item.get("why") or ""),
1819
+ concurrency=int(item.get("concurrency") or 0),
1646
1820
  )
1647
1821
  for item in _stage_launches(record)
1648
1822
  ),
1649
1823
  note=str(stage.get("syscall_note", "")),
1650
1824
  submit=True,
1825
+ # the author's report rides every re-park: a suite fan-out
1826
+ # must not drop what the panel and the PR read
1827
+ report=str(stage.get("report", "")),
1651
1828
  )
1652
1829
  old_afterany = str(record.stage.get("afterany", ""))
1653
1830
  made_progress = bool(parked.afterany) and parked.afterany != old_afterany
@@ -1682,7 +1859,19 @@ def resume_run(
1682
1859
  # the park's sibling launches are done too: settle their charge before
1683
1860
  # any path — publish or hand back to the author — reads the budget
1684
1861
  if _stage_launches(record):
1685
- _reconcile_launch_hours(record, dispatch, bench.gpus, _stage_syscall_launches(record))
1862
+ sibling_launches = _stage_syscall_launches(record)
1863
+ sibling_ids = launch_task_ids(sibling_launches, stage_launch_job_ids(record))
1864
+ sibling_elapsed = _launch_elapsed(dispatch, sibling_ids) if sibling_ids else None
1865
+ _reconcile_launch_hours(record, dispatch, bench.gpus, sibling_launches, sibling_elapsed)
1866
+ # the ledger's ended records for the sibling launches, whether or not
1867
+ # the author is woken: the PR's experiments table reads them (an
1868
+ # author wake that follows records nothing twice)
1869
+ _best_effort(
1870
+ "launch ledger",
1871
+ lambda: _ledger_ended(
1872
+ run_dir, record, sibling_launches, sibling_ids, dispatch, sibling_elapsed
1873
+ ),
1874
+ )
1686
1875
 
1687
1876
  def _wake_author(
1688
1877
  extra_update: str, judged: tuple[str, AttemptResult] | None = None
@@ -1741,7 +1930,14 @@ def resume_run(
1741
1930
  # Research lines: record the tree AS OF THIS DECIDED TERMINAL — never
1742
1931
  # earlier, because a blocking panel verdict can still resume the
1743
1932
  # author (a continuation, not a terminal).
1744
- _push_line_snapshot(ws, _line_ref_for(bench, config.agent_id), run_id, outcome, secrets)
1933
+ _push_line_snapshot(
1934
+ ws,
1935
+ _line_ref_for(bench, config.agent_id),
1936
+ run_id,
1937
+ outcome,
1938
+ secrets,
1939
+ bot_login=config.bot_login,
1940
+ )
1745
1941
 
1746
1942
  if result.outcome == "improved":
1747
1943
  # Publish: branch the SEALED candidate sha, fold in the ledger, push,
@@ -1888,6 +2084,7 @@ def resume_run(
1888
2084
  exclude=(
1889
2085
  LINE_MEMORY_PATHS if _line_ref_for(bench, config.agent_id) else ()
1890
2086
  ),
2087
+ secrets=secrets,
1891
2088
  )(baseline, candidate, str(stage.get("report", "")))
1892
2089
  except Exception as exc:
1893
2090
  if isinstance(exc, GitError) and _is_git_tamper(exc):
@@ -1955,18 +2152,23 @@ def resume_run(
1955
2152
  # so push the candidate as-is rather than an empty commit.
1956
2153
  if staged:
1957
2154
  ws.git(
1958
- "-c",
1959
- f"user.name={config.bot_login}",
1960
- "-c",
1961
- f"user.email={config.bot_login}@users.noreply.github.com",
2155
+ *git_identity(config.bot_login),
1962
2156
  "commit",
1963
2157
  "-m",
1964
2158
  f"agent: improve {config.benchmark} ({_title_pair(baseline, candidate)})"
1965
2159
  f"\n\nAgent: {config.agent_id}",
1966
2160
  )
1967
2161
  ws.push(branch)
2162
+ if record.stage.get("submitted"):
2163
+ # the author's report at submit rides the stage: the PR shows it
2164
+ # as the research report, over the ledger's experiments
2165
+ result = dc_replace(result, submit_report=str(record.stage.get("report") or ""))
1968
2166
  body = pr_body(
1969
- result, config, redact_secrets=secrets, display_digits=bench.display_digits
2167
+ result,
2168
+ config,
2169
+ redact_secrets=secrets,
2170
+ display_digits=bench.display_digits,
2171
+ experiments=experiments_rows(run_dir),
1970
2172
  )
1971
2173
  if issue_number:
1972
2174
  body = f"Addresses #{issue_number}.\n\n{body}"
@@ -2264,6 +2466,7 @@ def build_panel_runner(
2264
2466
  start_round: int = 0,
2265
2467
  exclude: tuple[str, ...] = (),
2266
2468
  claim_body: Callable[[float, float, str], str] | None = None,
2469
+ secrets: tuple[str, ...] = (),
2267
2470
  ) -> Callable[[float, float, str], PanelVerdict]:
2268
2471
  """The git half of the pre-PR panel: prepare the two read-only checkouts
2269
2472
  and the synthetic claim, then hand off to `run_panel` (which owns no git).
@@ -2287,6 +2490,10 @@ def build_panel_runner(
2287
2490
  )
2288
2491
 
2289
2492
  def runner(baseline: float, candidate: float, report: str) -> PanelVerdict:
2493
+ # the claim is author text (the report at submit, or the session's
2494
+ # last words): redacted before any lens sees it, like the record and
2495
+ # the PR body
2496
+ report = redact(report, secrets)
2290
2497
  reads["n"] += 1
2291
2498
  panel_ws = run_dir / "panel"
2292
2499
  shutil.rmtree(panel_ws, ignore_errors=True)
@@ -2299,10 +2506,7 @@ def build_panel_runner(
2299
2506
  tree = ws.git("write-tree").strip()
2300
2507
  ws.git("reset")
2301
2508
  snapshot = ws.git(
2302
- "-c",
2303
- "user.name=panel",
2304
- "-c",
2305
- "user.email=panel@localhost",
2509
+ *git_identity(bot_login),
2306
2510
  "commit-tree",
2307
2511
  tree,
2308
2512
  "-p",
@@ -2432,7 +2636,7 @@ def live_attempt(
2432
2636
  # exception path cannot rely on names bound inside the try
2433
2637
  salvage: dict[str, object] = {}
2434
2638
  try:
2435
- ws = Workspace.clone(_target_clone_url(config.target), workspace, auth=bot_auth)
2639
+ ws = Workspace.clone(target_clone_url(config.target), workspace, auth=bot_auth)
2436
2640
  # Build ON the requested PR base: the clone checks out the remote
2437
2641
  # DEFAULT branch, which need not be `base_branch` — the session must
2438
2642
  # edit, and the gate must measure, the tree the PR will land on.
@@ -2496,7 +2700,9 @@ def live_attempt(
2496
2700
  line_ref = ""
2497
2701
  if lines_active:
2498
2702
  try:
2499
- line_ref = _checkout_line(ws, workspace, config.agent_id, base_branch)
2703
+ line_ref = _checkout_line(
2704
+ ws, workspace, config.agent_id, base_branch, config.bot_login
2705
+ )
2500
2706
  except Exception as exc:
2501
2707
  log.warning(
2502
2708
  "line checkout failed (%s); running on %s",
@@ -2639,6 +2845,7 @@ def live_attempt(
2639
2845
  config.bot_login,
2640
2846
  created[:10],
2641
2847
  exclude=LINE_MEMORY_PATHS if lines_active else (),
2848
+ secrets=secrets,
2642
2849
  )
2643
2850
  if panel_lenses
2644
2851
  else None
@@ -2672,12 +2879,16 @@ def live_attempt(
2672
2879
  run_tag=run_id,
2673
2880
  # an inline gate shares the same target-wide baseline cache
2674
2881
  baseline_cache=run_dir.parent / "baselines",
2882
+ seed_cache=dispatch.seed_cache if dispatch is not None else None,
2675
2883
  )
2676
2884
  snapshots: list[Snapshot] = []
2677
2885
 
2678
2886
  def snapshot() -> str:
2679
2887
  snap = snapshot_tree(
2680
- ws, pre_session_sha, exclude=LINE_MEMORY_PATHS if lines_active else ()
2888
+ ws,
2889
+ pre_session_sha,
2890
+ exclude=LINE_MEMORY_PATHS if lines_active else (),
2891
+ author=config.bot_login,
2681
2892
  )
2682
2893
  snapshots.append(snap)
2683
2894
  return snap.commit
@@ -2722,6 +2933,11 @@ def live_attempt(
2722
2933
  line_memory=line_memory,
2723
2934
  line_divergence=line_divergence,
2724
2935
  launcher=launcher,
2936
+ watcher=(
2937
+ _make_watcher(dispatch, run_root, run_id, workspace, config)
2938
+ if dispatch is not None
2939
+ else None
2940
+ ),
2725
2941
  tree_of=lambda sha: ws.git("rev-parse", f"{sha}^{{tree}}").strip(),
2726
2942
  )
2727
2943
  except RunParked as p:
@@ -2786,6 +3002,7 @@ def live_attempt(
2786
3002
  run_id,
2787
3003
  "attempt-error",
2788
3004
  secrets,
3005
+ bot_login=config.bot_login,
2789
3006
  )
2790
3007
  failed = RunRecord(
2791
3008
  **{
@@ -2843,7 +3060,7 @@ def live_attempt(
2843
3060
  # memory (it is excluded from measurable seals by design). The label is
2844
3061
  # the GATE outcome, correct at this moment; a publish failure appends a
2845
3062
  # publish-error snapshot at the tail.
2846
- _push_line_snapshot(ws, line_ref, run_id, result.outcome, secrets)
3063
+ _push_line_snapshot(ws, line_ref, run_id, result.outcome, secrets, bot_login=config.bot_login)
2847
3064
 
2848
3065
  pr_url = ""
2849
3066
  outcome_name = result.outcome
@@ -2895,10 +3112,7 @@ def live_attempt(
2895
3112
  raise WorkspaceDrift(f"publish would stage non-ledger paths: {extra[:10]}")
2896
3113
  if staged:
2897
3114
  ws.git(
2898
- "-c",
2899
- f"user.name={config.bot_login}",
2900
- "-c",
2901
- f"user.email={config.bot_login}@users.noreply.github.com",
3115
+ *git_identity(config.bot_login),
2902
3116
  "commit",
2903
3117
  "-m",
2904
3118
  f"agent: improve {config.benchmark} ({_title_pair(baseline, candidate)})"
@@ -2907,7 +3121,11 @@ def live_attempt(
2907
3121
  ws.push(branch)
2908
3122
  pushed = True
2909
3123
  body = pr_body(
2910
- result, config, redact_secrets=secrets, display_digits=bench.display_digits
3124
+ result,
3125
+ config,
3126
+ redact_secrets=secrets,
3127
+ display_digits=bench.display_digits,
3128
+ experiments=experiments_rows(run_dir),
2911
3129
  )
2912
3130
  if issue_number:
2913
3131
  body = f"Addresses #{issue_number}.\n\n{body}"
@@ -3022,7 +3240,7 @@ def live_attempt(
3022
3240
  # the publish failed after the gate credited the tree: the improved
3023
3241
  # snapshot above stands (the measurement was real); append the
3024
3242
  # publish-error marker so the notebook records how the run ended
3025
- _push_line_snapshot(ws, line_ref, run_id, outcome_name, secrets)
3243
+ _push_line_snapshot(ws, line_ref, run_id, outcome_name, secrets, bot_login=config.bot_login)
3026
3244
  log.info("run %s: %s %s", run_id, outcome_name, pr_url)
3027
3245
  return AttemptOutcome(
3028
3246
  run_id=run_id,
@@ -3115,7 +3333,6 @@ def arm_sigterm_containment() -> None:
3115
3333
  def main() -> int:
3116
3334
  import argparse
3117
3335
  import os
3118
- import time
3119
3336
  from datetime import UTC, datetime
3120
3337
 
3121
3338
  arm_sigterm_containment()