outerloop-science 0.1.0.dev2__py3-none-any.whl → 0.1.0.dev3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- outerloop/__init__.py +2 -2
- outerloop/attempt.py +310 -93
- outerloop/brief.py +38 -25
- outerloop/cli.py +40 -5
- outerloop/climbboard.py +3 -0
- outerloop/compute.py +148 -53
- outerloop/contract.py +8 -0
- outerloop/dispatch.py +63 -18
- outerloop/evalcache.py +147 -0
- outerloop/followup.py +38 -16
- outerloop/github.py +38 -13
- outerloop/harness.py +1 -18
- outerloop/housekeeping.py +1 -17
- outerloop/image.py +0 -4
- outerloop/init.py +19 -1
- outerloop/intake.py +4 -7
- outerloop/launchlog.py +239 -0
- outerloop/maintain.py +325 -0
- outerloop/maintain_agent_cli.py +81 -0
- outerloop/maintain_post_cli.py +140 -0
- outerloop/measure.py +6 -0
- outerloop/orchestrator.py +141 -31
- outerloop/panel.py +3 -3
- outerloop/review.py +4 -0
- outerloop/review_agent.py +7 -7
- outerloop/review_agent_cli.py +2 -2
- outerloop/review_post_cli.py +2 -2
- outerloop/review_summarize_cli.py +7 -5
- outerloop/roles.py +27 -0
- outerloop/rolespec.py +3 -1
- outerloop/steward.py +5 -5
- outerloop/syscall.py +261 -47
- outerloop/syscall_cli.py +243 -12
- outerloop/tick.py +90 -178
- outerloop/verify_agent.py +8 -6
- outerloop/verify_post_cli.py +2 -2
- outerloop/watcher.py +203 -0
- {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/METADATA +4 -1
- outerloop_science-0.1.0.dev3.dist-info/RECORD +59 -0
- outerloop_science-0.1.0.dev2.dist-info/RECORD +0 -53
- {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/WHEEL +0 -0
- {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/entry_points.txt +0 -0
- {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/licenses/LICENSE +0 -0
- {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/licenses/NOTICE +0 -0
outerloop/attempt.py
CHANGED
|
@@ -18,6 +18,7 @@ import logging
|
|
|
18
18
|
import os
|
|
19
19
|
import re
|
|
20
20
|
import shutil
|
|
21
|
+
import time
|
|
21
22
|
from collections.abc import Callable, Iterable
|
|
22
23
|
from dataclasses import dataclass
|
|
23
24
|
from dataclasses import replace as dc_replace
|
|
@@ -27,7 +28,7 @@ from typing import Any, cast
|
|
|
27
28
|
|
|
28
29
|
from outerloop.appauth import resolve_bot_auth
|
|
29
30
|
from outerloop.brief import BudgetState, distill_lessons
|
|
30
|
-
from outerloop.compute import LocalCompute
|
|
31
|
+
from outerloop.compute import LocalCompute
|
|
31
32
|
from outerloop.contract import Benchmark, Contract, contract_text_in_tree, load_contract
|
|
32
33
|
from outerloop.dispatch import (
|
|
33
34
|
Snapshot,
|
|
@@ -36,6 +37,7 @@ from outerloop.dispatch import (
|
|
|
36
37
|
should_dispatch,
|
|
37
38
|
snapshot_tree,
|
|
38
39
|
)
|
|
40
|
+
from outerloop.evalcache import seed_dir
|
|
39
41
|
from outerloop.github import (
|
|
40
42
|
GitError,
|
|
41
43
|
GitHubClient,
|
|
@@ -43,8 +45,10 @@ from outerloop.github import (
|
|
|
43
45
|
Workspace,
|
|
44
46
|
contract_at,
|
|
45
47
|
ensure_regular_git_dir,
|
|
48
|
+
git_identity,
|
|
46
49
|
)
|
|
47
50
|
from outerloop.harness import Harness, SessionResult, default_binary, redact
|
|
51
|
+
from outerloop.launchlog import append_ended, append_submitted, experiments_rows
|
|
48
52
|
from outerloop.markers import has_marker
|
|
49
53
|
from outerloop.measure import DispatchedMeasurer, DispatchSettings
|
|
50
54
|
from outerloop.orchestrator import (
|
|
@@ -84,9 +88,20 @@ from outerloop.runstate import (
|
|
|
84
88
|
save_record,
|
|
85
89
|
stamp_outage,
|
|
86
90
|
)
|
|
87
|
-
from outerloop.
|
|
91
|
+
from outerloop.runstate import (
|
|
92
|
+
run_dir as run_dir_of,
|
|
93
|
+
)
|
|
94
|
+
from outerloop.syscall import (
|
|
95
|
+
CHANNEL_DIR_NAMES,
|
|
96
|
+
MAX_ARTIFACT_BYTES,
|
|
97
|
+
SyscallRequest,
|
|
98
|
+
channel_dir,
|
|
99
|
+
launch_task_ids,
|
|
100
|
+
tool_update_note,
|
|
101
|
+
)
|
|
88
102
|
from outerloop.syscall import ensure_excluded as syscall_excluded
|
|
89
103
|
from outerloop.syscall import install_tool as syscall_install_tool
|
|
104
|
+
from outerloop.syscall import refresh_tool as syscall_refresh_tool
|
|
90
105
|
from outerloop.syscall import write_budget as syscall_write_budget
|
|
91
106
|
from outerloop.syscall import write_siblings as syscall_write_siblings
|
|
92
107
|
from outerloop.verifier import MAX_CLAIM_CHARS
|
|
@@ -186,7 +201,7 @@ class WorkspaceDrift(RuntimeError):
|
|
|
186
201
|
"""The tree changed between measurement and commit."""
|
|
187
202
|
|
|
188
203
|
|
|
189
|
-
def
|
|
204
|
+
def target_clone_url(target: str) -> str:
|
|
190
205
|
"""The canonical HTTPS clone URL for `owner/repo`. The one source of truth
|
|
191
206
|
for where a run's git pushes go — derived from the target, never read from
|
|
192
207
|
the session-writable `remote.origin.url`."""
|
|
@@ -206,7 +221,8 @@ def _blessed_head(ws: Workspace, result: Any, contract: Any) -> str:
|
|
|
206
221
|
return ""
|
|
207
222
|
try:
|
|
208
223
|
return ws.git("rev-parse", "HEAD").strip()
|
|
209
|
-
except Exception:
|
|
224
|
+
except Exception as exc:
|
|
225
|
+
log.warning("could not read HEAD; not arming self-merge: %s", exc)
|
|
210
226
|
return ""
|
|
211
227
|
|
|
212
228
|
|
|
@@ -425,9 +441,14 @@ def _park_run(
|
|
|
425
441
|
# it lands in the durable record, like every other persisted final_text
|
|
426
442
|
# — a session that echoed a credential must not leave it in record.json.
|
|
427
443
|
# Empty/zero for a baseline park (the session has not run yet).
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
444
|
+
# the author's report at submit, else the session's last words (a park
|
|
445
|
+
# with no submit); either way what the wake's panel and the PR body read
|
|
446
|
+
"report": redact(
|
|
447
|
+
parked.syscall.report
|
|
448
|
+
if parked.syscall is not None and parked.syscall.report
|
|
449
|
+
else (parked.session.final_text if parked.session else ""),
|
|
450
|
+
secrets,
|
|
451
|
+
)[:MAX_CLAIM_CHARS],
|
|
431
452
|
"session_cost_usd": parked.session.cost_usd if parked.session else 0.0,
|
|
432
453
|
"session_turns": parked.session.num_turns if parked.session else 0,
|
|
433
454
|
}
|
|
@@ -452,10 +473,35 @@ def _park_run(
|
|
|
452
473
|
"minutes": launch.minutes,
|
|
453
474
|
"artifacts": list(launch.artifacts),
|
|
454
475
|
**({"array": launch.array} if launch.array > 1 else {}),
|
|
476
|
+
**({"why": redact(launch.why, secrets)} if launch.why else {}),
|
|
477
|
+
**({"concurrency": launch.concurrency} if launch.concurrency else {}),
|
|
455
478
|
}
|
|
456
479
|
for launch in parked.syscall.launches
|
|
457
480
|
]
|
|
458
481
|
stage["syscall_note"] = redact(parked.syscall.note, secrets)
|
|
482
|
+
if parked.syscall.launches:
|
|
483
|
+
# the run's launch ledger (`history`, and the queue view's labels):
|
|
484
|
+
# ids align with launch_jobs order, as stage_launch_job_ids reads them
|
|
485
|
+
if parked.launch_afterany:
|
|
486
|
+
launch_ids = afterany_ids(parked.launch_afterany)
|
|
487
|
+
elif parked.phase == "author-sleep":
|
|
488
|
+
launch_ids = list(job_ids)
|
|
489
|
+
else:
|
|
490
|
+
launch_ids = []
|
|
491
|
+
ledger_launches = tuple(
|
|
492
|
+
dc_replace(launch, why=redact(launch.why, secrets))
|
|
493
|
+
for launch in parked.syscall.launches
|
|
494
|
+
)
|
|
495
|
+
_best_effort(
|
|
496
|
+
"launch ledger",
|
|
497
|
+
lambda: append_submitted(
|
|
498
|
+
run_dir_of(run_root, record.run_id),
|
|
499
|
+
sleep=parked.sleeps_used,
|
|
500
|
+
launches=ledger_launches,
|
|
501
|
+
job_ids=launch_ids,
|
|
502
|
+
at=now,
|
|
503
|
+
),
|
|
504
|
+
)
|
|
459
505
|
# (the session id the wake resumes is the record's own
|
|
460
506
|
# resume_session_id, set below for every park — no stage duplicate)
|
|
461
507
|
stage["launches_used"] = parked.launches_used
|
|
@@ -586,6 +632,7 @@ def _dispatch_settings(args: argparse.Namespace) -> DispatchSettings:
|
|
|
586
632
|
#174: the wake dropped the GPU lane)."""
|
|
587
633
|
from outerloop.compute import compute_from_env
|
|
588
634
|
|
|
635
|
+
target = getattr(args, "target", "") or ""
|
|
589
636
|
return DispatchSettings(
|
|
590
637
|
compute=compute_from_env(),
|
|
591
638
|
image=args.image,
|
|
@@ -593,9 +640,34 @@ def _dispatch_settings(args: argparse.Namespace) -> DispatchSettings:
|
|
|
593
640
|
partition=args.partition,
|
|
594
641
|
gpu_partition=getattr(args, "gpu_partition", ""),
|
|
595
642
|
gpu_account=getattr(args, "gpu_account", ""),
|
|
643
|
+
seed_cache=seed_dir(Path(args.run_root), target) if target else None,
|
|
596
644
|
)
|
|
597
645
|
|
|
598
646
|
|
|
647
|
+
# Experiments yield to verification. Every kernel job is one Slurm user, so
|
|
648
|
+
# among the kernel's own pending jobs the priority order is ours: launches
|
|
649
|
+
# carry this nice so a gate eval or a follow-up re-measure (nice 0) starts
|
|
650
|
+
# first when the cap frees a slot. Sized above the factors that differ between
|
|
651
|
+
# our jobs on Torch — age tops out at 1000 after a week, job size at 1000, the
|
|
652
|
+
# per-GPU TRES share stays in the hundreds — so the order holds however long a
|
|
653
|
+
# launch has waited. Other users' jobs and the group cap are untouched.
|
|
654
|
+
LAUNCH_NICE = 5000
|
|
655
|
+
|
|
656
|
+
|
|
657
|
+
def with_seed(dispatch: DispatchSettings, run_root: Path, target: str) -> DispatchSettings:
|
|
658
|
+
"""These settings with the target's seed cache filled in from the record's
|
|
659
|
+
target when the CLI gave none (wake and follow-up jobs carry the run id,
|
|
660
|
+
not the target)."""
|
|
661
|
+
# tolerant of any settings object: a backend that knows no seed (or a
|
|
662
|
+
# test double) is left exactly as it is
|
|
663
|
+
if not target or getattr(dispatch, "seed_cache", "unknown") is not None:
|
|
664
|
+
return dispatch
|
|
665
|
+
try:
|
|
666
|
+
return dc_replace(dispatch, seed_cache=seed_dir(run_root, target))
|
|
667
|
+
except TypeError:
|
|
668
|
+
return dispatch
|
|
669
|
+
|
|
670
|
+
|
|
599
671
|
def _make_launcher(
|
|
600
672
|
dispatch: DispatchSettings, run_dir: Path, workspace: Path, run_id: str, gpus: int = 0
|
|
601
673
|
):
|
|
@@ -605,45 +677,42 @@ def _make_launcher(
|
|
|
605
677
|
partially-submitted batch is reaped rather than orphaned. `gpus` is the
|
|
606
678
|
benchmark's: an author's experiments run on the same lane as its evals."""
|
|
607
679
|
account, partition = dispatch.placement(gpus)
|
|
608
|
-
# under launch admission a GPU launch enters the queue held; the tick
|
|
609
|
-
# releases it when the user's GPUs fit under the cap (tick.service_admission)
|
|
610
|
-
from outerloop.tick import max_launch_gpus_from_env
|
|
611
|
-
|
|
612
|
-
hold = gpus > 0 and not local_mode() and max_launch_gpus_from_env() > 0
|
|
613
680
|
|
|
614
681
|
def launcher(sha: str, request: SyscallRequest) -> str:
|
|
615
|
-
from dataclasses import replace as _replace
|
|
616
|
-
|
|
617
682
|
from outerloop.dispatch import eval_job_spec, write_eval_job
|
|
618
|
-
from outerloop.syscall import
|
|
683
|
+
from outerloop.syscall import array_spec
|
|
619
684
|
|
|
620
685
|
ids: list[str] = []
|
|
621
686
|
try:
|
|
622
687
|
for launch in request.launches:
|
|
623
|
-
#
|
|
624
|
-
#
|
|
625
|
-
|
|
626
|
-
|
|
627
|
-
|
|
628
|
-
|
|
629
|
-
|
|
630
|
-
|
|
631
|
-
|
|
632
|
-
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
|
|
644
|
-
|
|
645
|
-
|
|
646
|
-
|
|
688
|
+
# a sweep is ONE Slurm job array (`--array=0-N%K`): the queue
|
|
689
|
+
# holds one entry, Slurm runs at most K tasks at once, each task
|
|
690
|
+
# derives its job dir and SWEEP_INDEX from its array index, and
|
|
691
|
+
# one afterany on the array id covers every task
|
|
692
|
+
script = write_eval_job(
|
|
693
|
+
run_dir,
|
|
694
|
+
f"launch-{launch.name}",
|
|
695
|
+
repo_root=workspace,
|
|
696
|
+
snapshot_sha=sha,
|
|
697
|
+
command=launch.command,
|
|
698
|
+
image=dispatch.image,
|
|
699
|
+
artifacts=launch.artifacts,
|
|
700
|
+
artifact_max_bytes=MAX_ARTIFACT_BYTES,
|
|
701
|
+
gpus=gpus,
|
|
702
|
+
array=launch.array,
|
|
703
|
+
seed_cache=dispatch.seed_cache,
|
|
704
|
+
)
|
|
705
|
+
spec = eval_job_spec(
|
|
706
|
+
script,
|
|
707
|
+
job_name=f"{run_id}-launch-{launch.name}",
|
|
708
|
+
account=account,
|
|
709
|
+
partition=partition,
|
|
710
|
+
eval_minutes=launch.minutes,
|
|
711
|
+
gpus=gpus,
|
|
712
|
+
nice=LAUNCH_NICE,
|
|
713
|
+
array=array_spec(launch),
|
|
714
|
+
)
|
|
715
|
+
ids.append(dispatch.compute.submit(spec))
|
|
647
716
|
except Exception:
|
|
648
717
|
# a partial batch must not orphan: no park record was written yet,
|
|
649
718
|
# so nothing would ever wake or cancel the jobs that DID submit —
|
|
@@ -661,6 +730,26 @@ def _make_launcher(
|
|
|
661
730
|
return launcher
|
|
662
731
|
|
|
663
732
|
|
|
733
|
+
def _make_watcher(
|
|
734
|
+
dispatch: DispatchSettings, run_root: Path, run_id: str, workspace: Path, config: RunConfig
|
|
735
|
+
) -> Callable[[], Any]:
|
|
736
|
+
"""The session watcher for one run: a thread beside the harness that
|
|
737
|
+
answers `queue` and `history` from the channel (docs/design/session-watcher.md).
|
|
738
|
+
Shared by the first pass and every wake leg."""
|
|
739
|
+
from outerloop.watcher import SessionWatcher, WatcherContext
|
|
740
|
+
|
|
741
|
+
ctx = WatcherContext(
|
|
742
|
+
workspace=workspace,
|
|
743
|
+
run_root=run_root,
|
|
744
|
+
run_id=run_id,
|
|
745
|
+
target=config.target,
|
|
746
|
+
agent_id=config.agent_id,
|
|
747
|
+
compute=dispatch.compute,
|
|
748
|
+
gpu_partition=dispatch.gpu_partition,
|
|
749
|
+
)
|
|
750
|
+
return lambda: SessionWatcher(ctx)
|
|
751
|
+
|
|
752
|
+
|
|
664
753
|
def _wake_author_sleep(
|
|
665
754
|
*,
|
|
666
755
|
run_root: Path,
|
|
@@ -713,7 +802,12 @@ def _wake_author_sleep(
|
|
|
713
802
|
# the same ending shape every other terminal takes. The line notebook
|
|
714
803
|
# records it first, while the tree is still the session's final tree.
|
|
715
804
|
_push_line_snapshot(
|
|
716
|
-
ws,
|
|
805
|
+
ws,
|
|
806
|
+
_line_ref_for(bench, config.agent_id),
|
|
807
|
+
run_id,
|
|
808
|
+
result.outcome,
|
|
809
|
+
secrets,
|
|
810
|
+
bot_login=config.bot_login,
|
|
717
811
|
)
|
|
718
812
|
for ref in drop_refs:
|
|
719
813
|
drop_snapshot(ws, Snapshot(commit="", tree="", ref=ref))
|
|
@@ -776,6 +870,8 @@ def _wake_author_sleep(
|
|
|
776
870
|
minutes=int(item.get("minutes") or 1),
|
|
777
871
|
artifacts=tuple(str(a) for a in item.get("artifacts", [])),
|
|
778
872
|
array=int(item.get("array") or 1),
|
|
873
|
+
why=str(item.get("why") or ""),
|
|
874
|
+
concurrency=int(item.get("concurrency") or 0),
|
|
779
875
|
)
|
|
780
876
|
for item in _stage_launches(record)
|
|
781
877
|
)
|
|
@@ -786,12 +882,28 @@ def _wake_author_sleep(
|
|
|
786
882
|
# of a blank "job failure". The park's launch job ids align positionally
|
|
787
883
|
# with the results (same launch/array order). Best-effort — the wake never
|
|
788
884
|
# blocks on the scheduler query.
|
|
885
|
+
task_ids = launch_task_ids(launches, stage_launch_job_ids(record))
|
|
789
886
|
status_of = getattr(dispatch.compute, "status", None)
|
|
790
887
|
if status_of is not None:
|
|
791
|
-
results = annotate_launch_states(results,
|
|
888
|
+
results = annotate_launch_states(results, task_ids, status_of)
|
|
792
889
|
launches_used = int(record.stage.get("launches_used", 0)) # type: ignore[call-overload]
|
|
793
890
|
sleeps_used = int(record.stage.get("sleeps_used", 0)) # type: ignore[call-overload]
|
|
794
|
-
|
|
891
|
+
elapsed = _launch_elapsed(dispatch, task_ids) if task_ids else None
|
|
892
|
+
_best_effort(
|
|
893
|
+
"launch ledger",
|
|
894
|
+
lambda: append_ended(
|
|
895
|
+
run_dir, sleep=sleeps_used, results=results, at=time.time(), elapsed_seconds=elapsed
|
|
896
|
+
),
|
|
897
|
+
)
|
|
898
|
+
gpu_hours_used = _reconcile_launch_hours(record, dispatch, bench.gpus, launches, elapsed)
|
|
899
|
+
# the tool the session invokes comes from THIS kernel: a session that
|
|
900
|
+
# started under an older one gets today's verbs and flags at its wake, and
|
|
901
|
+
# is told what is new
|
|
902
|
+
tool_changed = False
|
|
903
|
+
try:
|
|
904
|
+
tool_changed = syscall_refresh_tool(workspace)
|
|
905
|
+
except Exception as exc:
|
|
906
|
+
log.warning("tool refresh failed: %s", redact(f"{type(exc).__name__}: {exc}", secrets))
|
|
795
907
|
wake_text = render_wake(
|
|
796
908
|
results,
|
|
797
909
|
str(record.stage.get("syscall_note", "")),
|
|
@@ -804,6 +916,15 @@ def _wake_author_sleep(
|
|
|
804
916
|
),
|
|
805
917
|
gpus=bench.gpus,
|
|
806
918
|
)
|
|
919
|
+
pacing = [
|
|
920
|
+
f"sweep `{la.name}`: {la.array} tasks, at most {la.concurrency or la.array} at a time"
|
|
921
|
+
for la in launches
|
|
922
|
+
if la.array > 1
|
|
923
|
+
]
|
|
924
|
+
if pacing:
|
|
925
|
+
wake_text = f"{wake_text}\n\n" + "\n".join(pacing) + " (the contract's ceiling applies)."
|
|
926
|
+
if tool_changed:
|
|
927
|
+
wake_text = f"{wake_text}\n\n{tool_update_note(channel_dir(workspace))}"
|
|
807
928
|
if extra_update:
|
|
808
929
|
# a submitted park's gate/panel feedback leads; launch results follow
|
|
809
930
|
wake_text = f"{extra_update}\n\n{wake_text}"
|
|
@@ -829,7 +950,9 @@ def _wake_author_sleep(
|
|
|
829
950
|
wake_line = _line_ref_for(bench, config.agent_id)
|
|
830
951
|
|
|
831
952
|
def snapshot() -> str:
|
|
832
|
-
snap = snapshot_tree(
|
|
953
|
+
snap = snapshot_tree(
|
|
954
|
+
ws, base_sha, exclude=LINE_MEMORY_PATHS if wake_line else (), author=config.bot_login
|
|
955
|
+
)
|
|
833
956
|
snapshots.append(snap)
|
|
834
957
|
return snap.commit
|
|
835
958
|
|
|
@@ -850,6 +973,7 @@ def _wake_author_sleep(
|
|
|
850
973
|
config.bot_login,
|
|
851
974
|
_utc_date(now),
|
|
852
975
|
exclude=LINE_MEMORY_PATHS if wake_line else (),
|
|
976
|
+
secrets=secrets,
|
|
853
977
|
)
|
|
854
978
|
if panel_lenses
|
|
855
979
|
else None
|
|
@@ -873,6 +997,7 @@ def _wake_author_sleep(
|
|
|
873
997
|
resume_session_id=record.resume_session_id,
|
|
874
998
|
improve_prompt=wake_text,
|
|
875
999
|
launcher=_make_launcher(dispatch, run_dir, workspace, run_id, gpus=bench.gpus),
|
|
1000
|
+
watcher=_make_watcher(dispatch, run_root, run_id, workspace, config),
|
|
876
1001
|
tree_of=lambda sha: ws.git("rev-parse", f"{sha}^{{tree}}").strip(),
|
|
877
1002
|
judged=judged or _stage_judged(record),
|
|
878
1003
|
launches_used=launches_used,
|
|
@@ -937,6 +1062,32 @@ def _stage_judged(record: RunRecord) -> tuple[str, AttemptResult] | None:
|
|
|
937
1062
|
)
|
|
938
1063
|
|
|
939
1064
|
|
|
1065
|
+
def _ledger_ended(
|
|
1066
|
+
run_dir: Path,
|
|
1067
|
+
record: RunRecord,
|
|
1068
|
+
launches: tuple,
|
|
1069
|
+
task_ids: list[str],
|
|
1070
|
+
dispatch: DispatchSettings,
|
|
1071
|
+
elapsed: list[int | None] | None,
|
|
1072
|
+
) -> None:
|
|
1073
|
+
"""Record a park's finished launches in the ledger from the run dir alone
|
|
1074
|
+
(no delivery into a workspace), with the scheduler's state for jobs that
|
|
1075
|
+
left no exit code."""
|
|
1076
|
+
from outerloop.syscall import annotate_launch_states, read_results
|
|
1077
|
+
|
|
1078
|
+
results = read_results(run_dir, launches)
|
|
1079
|
+
status_of = getattr(dispatch.compute, "status", None)
|
|
1080
|
+
if status_of is not None:
|
|
1081
|
+
results = annotate_launch_states(results, task_ids, status_of)
|
|
1082
|
+
append_ended(
|
|
1083
|
+
run_dir,
|
|
1084
|
+
sleep=int(record.stage.get("sleeps_used", 0)), # type: ignore[call-overload]
|
|
1085
|
+
results=results,
|
|
1086
|
+
at=time.time(),
|
|
1087
|
+
elapsed_seconds=elapsed,
|
|
1088
|
+
)
|
|
1089
|
+
|
|
1090
|
+
|
|
940
1091
|
def _stage_syscall_launches(record: RunRecord) -> tuple:
|
|
941
1092
|
"""The park's launches as `Launch` values (command elided: they ran)."""
|
|
942
1093
|
from outerloop.syscall import Launch
|
|
@@ -948,12 +1099,14 @@ def _stage_syscall_launches(record: RunRecord) -> tuple:
|
|
|
948
1099
|
minutes=int(item.get("minutes") or 1),
|
|
949
1100
|
artifacts=tuple(str(a) for a in item.get("artifacts", [])),
|
|
950
1101
|
array=int(item.get("array") or 1),
|
|
1102
|
+
why=str(item.get("why") or ""),
|
|
1103
|
+
concurrency=int(item.get("concurrency") or 0),
|
|
951
1104
|
)
|
|
952
1105
|
for item in _stage_launches(record)
|
|
953
1106
|
)
|
|
954
1107
|
|
|
955
1108
|
|
|
956
|
-
def
|
|
1109
|
+
def stage_launch_job_ids(record: RunRecord) -> list[str]:
|
|
957
1110
|
"""The park's launch jobs: `launch_afterany` when the park recorded it;
|
|
958
1111
|
for an older author-sleep park every waited job was a launch; for an
|
|
959
1112
|
older candidate park the gate's evals are mixed in, so none."""
|
|
@@ -966,7 +1119,11 @@ def _stage_launch_job_ids(record: RunRecord) -> list[str]:
|
|
|
966
1119
|
|
|
967
1120
|
|
|
968
1121
|
def _reconcile_launch_hours(
|
|
969
|
-
record: RunRecord,
|
|
1122
|
+
record: RunRecord,
|
|
1123
|
+
dispatch: DispatchSettings,
|
|
1124
|
+
gpus: int,
|
|
1125
|
+
launches: tuple,
|
|
1126
|
+
elapsed: list[int | None] | None = None,
|
|
970
1127
|
) -> float:
|
|
971
1128
|
"""The run's GPU-hours after handing back the unused walltime of the
|
|
972
1129
|
park's launch jobs — once: the stage remembers the refund, so a wake
|
|
@@ -976,7 +1133,9 @@ def _reconcile_launch_hours(
|
|
|
976
1133
|
used = float(stage.get("gpu_hours_used", 0.0)) # type: ignore[arg-type]
|
|
977
1134
|
if not gpus or stage.get("launch_hours_refunded"):
|
|
978
1135
|
return used
|
|
979
|
-
refund = _launch_refund(
|
|
1136
|
+
refund = _launch_refund(
|
|
1137
|
+
dispatch, launches, launch_task_ids(launches, stage_launch_job_ids(record)), gpus, elapsed
|
|
1138
|
+
)
|
|
980
1139
|
if refund > 0:
|
|
981
1140
|
log.info("%s: refunding %.2f GPU-hours of unused launch walltime", record.run_id, refund)
|
|
982
1141
|
used = max(0.0, used - refund)
|
|
@@ -985,21 +1144,35 @@ def _reconcile_launch_hours(
|
|
|
985
1144
|
return used
|
|
986
1145
|
|
|
987
1146
|
|
|
1147
|
+
def _launch_elapsed(dispatch: DispatchSettings, job_ids: list[str]) -> list[int | None] | None:
|
|
1148
|
+
"""How long each launch job ran, from the compute, aligned with `job_ids`;
|
|
1149
|
+
None when the compute cannot say (nothing is refunded or recorded on a
|
|
1150
|
+
guess)."""
|
|
1151
|
+
query = getattr(dispatch.compute, "elapsed_seconds", None)
|
|
1152
|
+
if query is None or not job_ids:
|
|
1153
|
+
return None
|
|
1154
|
+
try:
|
|
1155
|
+
return [query(jid) for jid in job_ids]
|
|
1156
|
+
except Exception as exc:
|
|
1157
|
+
log.warning("launch walltime unknown (%s: %s)", type(exc).__name__, exc)
|
|
1158
|
+
return None
|
|
1159
|
+
|
|
1160
|
+
|
|
988
1161
|
def _launch_refund(
|
|
989
|
-
dispatch: DispatchSettings,
|
|
1162
|
+
dispatch: DispatchSettings,
|
|
1163
|
+
launches: tuple,
|
|
1164
|
+
job_ids: list[str],
|
|
1165
|
+
gpus: int,
|
|
1166
|
+
elapsed: list[int | None] | None = None,
|
|
990
1167
|
) -> float:
|
|
991
1168
|
"""The unused walltime of a park's launch jobs, in GPU-hours, or 0 when
|
|
992
1169
|
the compute cannot say how long they ran (nothing is refunded on a
|
|
993
|
-
guess)."""
|
|
1170
|
+
guess). `elapsed` may be handed in when the caller already asked."""
|
|
994
1171
|
from outerloop.syscall import launch_hours_refund
|
|
995
1172
|
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
|
|
999
|
-
try:
|
|
1000
|
-
elapsed = [query(jid) for jid in job_ids]
|
|
1001
|
-
except Exception as exc:
|
|
1002
|
-
log.warning("launch walltime unknown (%s: %s); nothing refunded", type(exc).__name__, exc)
|
|
1173
|
+
if elapsed is None:
|
|
1174
|
+
elapsed = _launch_elapsed(dispatch, job_ids)
|
|
1175
|
+
if elapsed is None:
|
|
1003
1176
|
return 0.0
|
|
1004
1177
|
return launch_hours_refund(launches, elapsed, gpus=gpus)
|
|
1005
1178
|
|
|
@@ -1124,7 +1297,12 @@ def _line_ref_for(bench: Benchmark | None, agent_id: str) -> str:
|
|
|
1124
1297
|
|
|
1125
1298
|
|
|
1126
1299
|
def _push_line_snapshot(
|
|
1127
|
-
ws: Workspace,
|
|
1300
|
+
ws: Workspace,
|
|
1301
|
+
line_ref: str,
|
|
1302
|
+
run_id: str,
|
|
1303
|
+
outcome: str,
|
|
1304
|
+
secrets: tuple[str, ...] = (),
|
|
1305
|
+
bot_login: str = "",
|
|
1128
1306
|
) -> None:
|
|
1129
1307
|
"""Publish the session's final tree to the agent's line as a sealed
|
|
1130
1308
|
snapshot commit — every terminal path, any outcome
|
|
@@ -1167,7 +1345,7 @@ def _push_line_snapshot(
|
|
|
1167
1345
|
fork = parent = remote
|
|
1168
1346
|
except Exception as exc:
|
|
1169
1347
|
log.info("line %s: sealing on the local ref (%s)", line_ref, type(exc).__name__)
|
|
1170
|
-
snap = snapshot_tree(ws, parent, force=memory)
|
|
1348
|
+
snap = snapshot_tree(ws, parent, force=memory, author=bot_login)
|
|
1171
1349
|
try:
|
|
1172
1350
|
# seal only when the tree moved past the parent; the PUSH runs
|
|
1173
1351
|
# either way — a session that COMMITTED its work advanced the
|
|
@@ -1176,10 +1354,7 @@ def _push_line_snapshot(
|
|
|
1176
1354
|
sealed = parent
|
|
1177
1355
|
if snap.tree != ws.git("rev-parse", f"{parent}^{{tree}}").strip():
|
|
1178
1356
|
sealed = ws.git(
|
|
1179
|
-
|
|
1180
|
-
"user.name=autoresearch",
|
|
1181
|
-
"-c",
|
|
1182
|
-
"user.email=autoresearch@localhost",
|
|
1357
|
+
*git_identity(bot_login),
|
|
1183
1358
|
"commit-tree",
|
|
1184
1359
|
snap.tree,
|
|
1185
1360
|
"-p",
|
|
@@ -1239,7 +1414,9 @@ def _reconcile_with_remote(ws: Workspace, old: str, new: str) -> None:
|
|
|
1239
1414
|
ws.git("checkout", new, "--", path)
|
|
1240
1415
|
|
|
1241
1416
|
|
|
1242
|
-
def _checkout_line(
|
|
1417
|
+
def _checkout_line(
|
|
1418
|
+
ws: Workspace, workspace: Path, agent_id: str, base_branch: str, bot_login: str = ""
|
|
1419
|
+
) -> str:
|
|
1243
1420
|
"""Check out the agent's research line: the persistent branch
|
|
1244
1421
|
`agents/<agent-id>`, created from the base branch when absent, with the
|
|
1245
1422
|
base branch merged in when it exists — a conflicted merge is left in the
|
|
@@ -1259,10 +1436,7 @@ def _checkout_line(ws: Workspace, workspace: Path, agent_id: str, base_branch: s
|
|
|
1259
1436
|
conflicted = False
|
|
1260
1437
|
try:
|
|
1261
1438
|
ws.git(
|
|
1262
|
-
|
|
1263
|
-
"user.name=autoresearch",
|
|
1264
|
-
"-c",
|
|
1265
|
-
"user.email=autoresearch@localhost",
|
|
1439
|
+
*git_identity(bot_login),
|
|
1266
1440
|
"merge",
|
|
1267
1441
|
"--no-edit",
|
|
1268
1442
|
base_ref,
|
|
@@ -1277,10 +1451,7 @@ def _checkout_line(ws: Workspace, workspace: Path, agent_id: str, base_branch: s
|
|
|
1277
1451
|
ws.git("add", "-A")
|
|
1278
1452
|
if ws.git("status", "--porcelain").strip():
|
|
1279
1453
|
ws.git(
|
|
1280
|
-
|
|
1281
|
-
"user.name=autoresearch",
|
|
1282
|
-
"-c",
|
|
1283
|
-
"user.email=autoresearch@localhost",
|
|
1454
|
+
*git_identity(bot_login),
|
|
1284
1455
|
"commit",
|
|
1285
1456
|
"-q",
|
|
1286
1457
|
"-m",
|
|
@@ -1466,12 +1637,13 @@ def resume_run(
|
|
|
1466
1637
|
run_dir = run_root / "runs" / run_id
|
|
1467
1638
|
workspace = run_dir / "ws"
|
|
1468
1639
|
record = load_record(run_root, run_id)
|
|
1640
|
+
dispatch = with_seed(dispatch, run_root, record.target)
|
|
1469
1641
|
stage = record.stage
|
|
1470
1642
|
# Push to the CANONICAL target URL, never the workspace's remote.origin.url:
|
|
1471
1643
|
# the session could have rewritten that config to exfil the bot token / code
|
|
1472
1644
|
# to another remote. Passing `url` here means `Workspace.push` uses it
|
|
1473
1645
|
# instead of reading `remote.origin.url`.
|
|
1474
|
-
ws = Workspace(root=workspace, auth=bot_auth, url=
|
|
1646
|
+
ws = Workspace(root=workspace, auth=bot_auth, url=target_clone_url(record.target))
|
|
1475
1647
|
# A session reshaped .git (symlinked object store, gitdir file, FIFO) is
|
|
1476
1648
|
# refused BEFORE anything writes through it: the exclude below opens
|
|
1477
1649
|
# .git/info/exclude, and every ws.git call re-checks. The refusal ENDS
|
|
@@ -1643,11 +1815,16 @@ def resume_run(
|
|
|
1643
1815
|
minutes=int(item.get("minutes") or 1),
|
|
1644
1816
|
artifacts=tuple(str(a) for a in item.get("artifacts", [])),
|
|
1645
1817
|
array=int(item.get("array") or 1),
|
|
1818
|
+
why=str(item.get("why") or ""),
|
|
1819
|
+
concurrency=int(item.get("concurrency") or 0),
|
|
1646
1820
|
)
|
|
1647
1821
|
for item in _stage_launches(record)
|
|
1648
1822
|
),
|
|
1649
1823
|
note=str(stage.get("syscall_note", "")),
|
|
1650
1824
|
submit=True,
|
|
1825
|
+
# the author's report rides every re-park: a suite fan-out
|
|
1826
|
+
# must not drop what the panel and the PR read
|
|
1827
|
+
report=str(stage.get("report", "")),
|
|
1651
1828
|
)
|
|
1652
1829
|
old_afterany = str(record.stage.get("afterany", ""))
|
|
1653
1830
|
made_progress = bool(parked.afterany) and parked.afterany != old_afterany
|
|
@@ -1682,7 +1859,19 @@ def resume_run(
|
|
|
1682
1859
|
# the park's sibling launches are done too: settle their charge before
|
|
1683
1860
|
# any path — publish or hand back to the author — reads the budget
|
|
1684
1861
|
if _stage_launches(record):
|
|
1685
|
-
|
|
1862
|
+
sibling_launches = _stage_syscall_launches(record)
|
|
1863
|
+
sibling_ids = launch_task_ids(sibling_launches, stage_launch_job_ids(record))
|
|
1864
|
+
sibling_elapsed = _launch_elapsed(dispatch, sibling_ids) if sibling_ids else None
|
|
1865
|
+
_reconcile_launch_hours(record, dispatch, bench.gpus, sibling_launches, sibling_elapsed)
|
|
1866
|
+
# the ledger's ended records for the sibling launches, whether or not
|
|
1867
|
+
# the author is woken: the PR's experiments table reads them (an
|
|
1868
|
+
# author wake that follows records nothing twice)
|
|
1869
|
+
_best_effort(
|
|
1870
|
+
"launch ledger",
|
|
1871
|
+
lambda: _ledger_ended(
|
|
1872
|
+
run_dir, record, sibling_launches, sibling_ids, dispatch, sibling_elapsed
|
|
1873
|
+
),
|
|
1874
|
+
)
|
|
1686
1875
|
|
|
1687
1876
|
def _wake_author(
|
|
1688
1877
|
extra_update: str, judged: tuple[str, AttemptResult] | None = None
|
|
@@ -1741,7 +1930,14 @@ def resume_run(
|
|
|
1741
1930
|
# Research lines: record the tree AS OF THIS DECIDED TERMINAL — never
|
|
1742
1931
|
# earlier, because a blocking panel verdict can still resume the
|
|
1743
1932
|
# author (a continuation, not a terminal).
|
|
1744
|
-
_push_line_snapshot(
|
|
1933
|
+
_push_line_snapshot(
|
|
1934
|
+
ws,
|
|
1935
|
+
_line_ref_for(bench, config.agent_id),
|
|
1936
|
+
run_id,
|
|
1937
|
+
outcome,
|
|
1938
|
+
secrets,
|
|
1939
|
+
bot_login=config.bot_login,
|
|
1940
|
+
)
|
|
1745
1941
|
|
|
1746
1942
|
if result.outcome == "improved":
|
|
1747
1943
|
# Publish: branch the SEALED candidate sha, fold in the ledger, push,
|
|
@@ -1888,6 +2084,7 @@ def resume_run(
|
|
|
1888
2084
|
exclude=(
|
|
1889
2085
|
LINE_MEMORY_PATHS if _line_ref_for(bench, config.agent_id) else ()
|
|
1890
2086
|
),
|
|
2087
|
+
secrets=secrets,
|
|
1891
2088
|
)(baseline, candidate, str(stage.get("report", "")))
|
|
1892
2089
|
except Exception as exc:
|
|
1893
2090
|
if isinstance(exc, GitError) and _is_git_tamper(exc):
|
|
@@ -1955,18 +2152,23 @@ def resume_run(
|
|
|
1955
2152
|
# so push the candidate as-is rather than an empty commit.
|
|
1956
2153
|
if staged:
|
|
1957
2154
|
ws.git(
|
|
1958
|
-
|
|
1959
|
-
f"user.name={config.bot_login}",
|
|
1960
|
-
"-c",
|
|
1961
|
-
f"user.email={config.bot_login}@users.noreply.github.com",
|
|
2155
|
+
*git_identity(config.bot_login),
|
|
1962
2156
|
"commit",
|
|
1963
2157
|
"-m",
|
|
1964
2158
|
f"agent: improve {config.benchmark} ({_title_pair(baseline, candidate)})"
|
|
1965
2159
|
f"\n\nAgent: {config.agent_id}",
|
|
1966
2160
|
)
|
|
1967
2161
|
ws.push(branch)
|
|
2162
|
+
if record.stage.get("submitted"):
|
|
2163
|
+
# the author's report at submit rides the stage: the PR shows it
|
|
2164
|
+
# as the research report, over the ledger's experiments
|
|
2165
|
+
result = dc_replace(result, submit_report=str(record.stage.get("report") or ""))
|
|
1968
2166
|
body = pr_body(
|
|
1969
|
-
result,
|
|
2167
|
+
result,
|
|
2168
|
+
config,
|
|
2169
|
+
redact_secrets=secrets,
|
|
2170
|
+
display_digits=bench.display_digits,
|
|
2171
|
+
experiments=experiments_rows(run_dir),
|
|
1970
2172
|
)
|
|
1971
2173
|
if issue_number:
|
|
1972
2174
|
body = f"Addresses #{issue_number}.\n\n{body}"
|
|
@@ -2264,6 +2466,7 @@ def build_panel_runner(
|
|
|
2264
2466
|
start_round: int = 0,
|
|
2265
2467
|
exclude: tuple[str, ...] = (),
|
|
2266
2468
|
claim_body: Callable[[float, float, str], str] | None = None,
|
|
2469
|
+
secrets: tuple[str, ...] = (),
|
|
2267
2470
|
) -> Callable[[float, float, str], PanelVerdict]:
|
|
2268
2471
|
"""The git half of the pre-PR panel: prepare the two read-only checkouts
|
|
2269
2472
|
and the synthetic claim, then hand off to `run_panel` (which owns no git).
|
|
@@ -2287,6 +2490,10 @@ def build_panel_runner(
|
|
|
2287
2490
|
)
|
|
2288
2491
|
|
|
2289
2492
|
def runner(baseline: float, candidate: float, report: str) -> PanelVerdict:
|
|
2493
|
+
# the claim is author text (the report at submit, or the session's
|
|
2494
|
+
# last words): redacted before any lens sees it, like the record and
|
|
2495
|
+
# the PR body
|
|
2496
|
+
report = redact(report, secrets)
|
|
2290
2497
|
reads["n"] += 1
|
|
2291
2498
|
panel_ws = run_dir / "panel"
|
|
2292
2499
|
shutil.rmtree(panel_ws, ignore_errors=True)
|
|
@@ -2299,10 +2506,7 @@ def build_panel_runner(
|
|
|
2299
2506
|
tree = ws.git("write-tree").strip()
|
|
2300
2507
|
ws.git("reset")
|
|
2301
2508
|
snapshot = ws.git(
|
|
2302
|
-
|
|
2303
|
-
"user.name=panel",
|
|
2304
|
-
"-c",
|
|
2305
|
-
"user.email=panel@localhost",
|
|
2509
|
+
*git_identity(bot_login),
|
|
2306
2510
|
"commit-tree",
|
|
2307
2511
|
tree,
|
|
2308
2512
|
"-p",
|
|
@@ -2432,7 +2636,7 @@ def live_attempt(
|
|
|
2432
2636
|
# exception path cannot rely on names bound inside the try
|
|
2433
2637
|
salvage: dict[str, object] = {}
|
|
2434
2638
|
try:
|
|
2435
|
-
ws = Workspace.clone(
|
|
2639
|
+
ws = Workspace.clone(target_clone_url(config.target), workspace, auth=bot_auth)
|
|
2436
2640
|
# Build ON the requested PR base: the clone checks out the remote
|
|
2437
2641
|
# DEFAULT branch, which need not be `base_branch` — the session must
|
|
2438
2642
|
# edit, and the gate must measure, the tree the PR will land on.
|
|
@@ -2496,7 +2700,9 @@ def live_attempt(
|
|
|
2496
2700
|
line_ref = ""
|
|
2497
2701
|
if lines_active:
|
|
2498
2702
|
try:
|
|
2499
|
-
line_ref = _checkout_line(
|
|
2703
|
+
line_ref = _checkout_line(
|
|
2704
|
+
ws, workspace, config.agent_id, base_branch, config.bot_login
|
|
2705
|
+
)
|
|
2500
2706
|
except Exception as exc:
|
|
2501
2707
|
log.warning(
|
|
2502
2708
|
"line checkout failed (%s); running on %s",
|
|
@@ -2639,6 +2845,7 @@ def live_attempt(
|
|
|
2639
2845
|
config.bot_login,
|
|
2640
2846
|
created[:10],
|
|
2641
2847
|
exclude=LINE_MEMORY_PATHS if lines_active else (),
|
|
2848
|
+
secrets=secrets,
|
|
2642
2849
|
)
|
|
2643
2850
|
if panel_lenses
|
|
2644
2851
|
else None
|
|
@@ -2672,12 +2879,16 @@ def live_attempt(
|
|
|
2672
2879
|
run_tag=run_id,
|
|
2673
2880
|
# an inline gate shares the same target-wide baseline cache
|
|
2674
2881
|
baseline_cache=run_dir.parent / "baselines",
|
|
2882
|
+
seed_cache=dispatch.seed_cache if dispatch is not None else None,
|
|
2675
2883
|
)
|
|
2676
2884
|
snapshots: list[Snapshot] = []
|
|
2677
2885
|
|
|
2678
2886
|
def snapshot() -> str:
|
|
2679
2887
|
snap = snapshot_tree(
|
|
2680
|
-
ws,
|
|
2888
|
+
ws,
|
|
2889
|
+
pre_session_sha,
|
|
2890
|
+
exclude=LINE_MEMORY_PATHS if lines_active else (),
|
|
2891
|
+
author=config.bot_login,
|
|
2681
2892
|
)
|
|
2682
2893
|
snapshots.append(snap)
|
|
2683
2894
|
return snap.commit
|
|
@@ -2722,6 +2933,11 @@ def live_attempt(
|
|
|
2722
2933
|
line_memory=line_memory,
|
|
2723
2934
|
line_divergence=line_divergence,
|
|
2724
2935
|
launcher=launcher,
|
|
2936
|
+
watcher=(
|
|
2937
|
+
_make_watcher(dispatch, run_root, run_id, workspace, config)
|
|
2938
|
+
if dispatch is not None
|
|
2939
|
+
else None
|
|
2940
|
+
),
|
|
2725
2941
|
tree_of=lambda sha: ws.git("rev-parse", f"{sha}^{{tree}}").strip(),
|
|
2726
2942
|
)
|
|
2727
2943
|
except RunParked as p:
|
|
@@ -2786,6 +3002,7 @@ def live_attempt(
|
|
|
2786
3002
|
run_id,
|
|
2787
3003
|
"attempt-error",
|
|
2788
3004
|
secrets,
|
|
3005
|
+
bot_login=config.bot_login,
|
|
2789
3006
|
)
|
|
2790
3007
|
failed = RunRecord(
|
|
2791
3008
|
**{
|
|
@@ -2843,7 +3060,7 @@ def live_attempt(
|
|
|
2843
3060
|
# memory (it is excluded from measurable seals by design). The label is
|
|
2844
3061
|
# the GATE outcome, correct at this moment; a publish failure appends a
|
|
2845
3062
|
# publish-error snapshot at the tail.
|
|
2846
|
-
_push_line_snapshot(ws, line_ref, run_id, result.outcome, secrets)
|
|
3063
|
+
_push_line_snapshot(ws, line_ref, run_id, result.outcome, secrets, bot_login=config.bot_login)
|
|
2847
3064
|
|
|
2848
3065
|
pr_url = ""
|
|
2849
3066
|
outcome_name = result.outcome
|
|
@@ -2895,10 +3112,7 @@ def live_attempt(
|
|
|
2895
3112
|
raise WorkspaceDrift(f"publish would stage non-ledger paths: {extra[:10]}")
|
|
2896
3113
|
if staged:
|
|
2897
3114
|
ws.git(
|
|
2898
|
-
|
|
2899
|
-
f"user.name={config.bot_login}",
|
|
2900
|
-
"-c",
|
|
2901
|
-
f"user.email={config.bot_login}@users.noreply.github.com",
|
|
3115
|
+
*git_identity(config.bot_login),
|
|
2902
3116
|
"commit",
|
|
2903
3117
|
"-m",
|
|
2904
3118
|
f"agent: improve {config.benchmark} ({_title_pair(baseline, candidate)})"
|
|
@@ -2907,7 +3121,11 @@ def live_attempt(
|
|
|
2907
3121
|
ws.push(branch)
|
|
2908
3122
|
pushed = True
|
|
2909
3123
|
body = pr_body(
|
|
2910
|
-
result,
|
|
3124
|
+
result,
|
|
3125
|
+
config,
|
|
3126
|
+
redact_secrets=secrets,
|
|
3127
|
+
display_digits=bench.display_digits,
|
|
3128
|
+
experiments=experiments_rows(run_dir),
|
|
2911
3129
|
)
|
|
2912
3130
|
if issue_number:
|
|
2913
3131
|
body = f"Addresses #{issue_number}.\n\n{body}"
|
|
@@ -3022,7 +3240,7 @@ def live_attempt(
|
|
|
3022
3240
|
# the publish failed after the gate credited the tree: the improved
|
|
3023
3241
|
# snapshot above stands (the measurement was real); append the
|
|
3024
3242
|
# publish-error marker so the notebook records how the run ended
|
|
3025
|
-
_push_line_snapshot(ws, line_ref, run_id, outcome_name, secrets)
|
|
3243
|
+
_push_line_snapshot(ws, line_ref, run_id, outcome_name, secrets, bot_login=config.bot_login)
|
|
3026
3244
|
log.info("run %s: %s %s", run_id, outcome_name, pr_url)
|
|
3027
3245
|
return AttemptOutcome(
|
|
3028
3246
|
run_id=run_id,
|
|
@@ -3115,7 +3333,6 @@ def arm_sigterm_containment() -> None:
|
|
|
3115
3333
|
def main() -> int:
|
|
3116
3334
|
import argparse
|
|
3117
3335
|
import os
|
|
3118
|
-
import time
|
|
3119
3336
|
from datetime import UTC, datetime
|
|
3120
3337
|
|
|
3121
3338
|
arm_sigterm_containment()
|