outerloop-science 0.1.0.dev2__py3-none-any.whl → 0.1.0.dev4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- outerloop/__init__.py +2 -2
- outerloop/appauth.py +17 -0
- outerloop/attempt.py +376 -101
- outerloop/brief.py +38 -25
- outerloop/cli.py +104 -6
- outerloop/climbboard.py +67 -22
- outerloop/compute.py +148 -53
- outerloop/contract.py +8 -0
- outerloop/dispatch.py +63 -18
- outerloop/evalcache.py +147 -0
- outerloop/followup.py +40 -25
- outerloop/github.py +67 -22
- outerloop/harness.py +22 -47
- outerloop/housekeeping.py +1 -17
- outerloop/image.py +0 -4
- outerloop/init.py +45 -2
- outerloop/intake.py +4 -7
- outerloop/launchlog.py +239 -0
- outerloop/maintain.py +353 -0
- outerloop/maintain_agent_cli.py +81 -0
- outerloop/maintain_post_cli.py +140 -0
- outerloop/measure.py +6 -0
- outerloop/orchestrator.py +141 -31
- outerloop/panel.py +3 -3
- outerloop/review.py +4 -0
- outerloop/review_agent.py +7 -7
- outerloop/review_agent_cli.py +2 -2
- outerloop/review_post_cli.py +2 -2
- outerloop/review_summarize_cli.py +8 -6
- outerloop/roles.py +27 -0
- outerloop/rolespec.py +3 -1
- outerloop/steward.py +7 -14
- outerloop/syscall.py +261 -47
- outerloop/syscall_cli.py +243 -12
- outerloop/tick.py +274 -313
- outerloop/verify_agent.py +8 -6
- outerloop/verify_post_cli.py +2 -2
- outerloop/watcher.py +203 -0
- {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev4.dist-info}/METADATA +4 -1
- outerloop_science-0.1.0.dev4.dist-info/RECORD +59 -0
- outerloop_science-0.1.0.dev2.dist-info/RECORD +0 -53
- {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev4.dist-info}/WHEEL +0 -0
- {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev4.dist-info}/entry_points.txt +0 -0
- {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev4.dist-info}/licenses/LICENSE +0 -0
- {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev4.dist-info}/licenses/NOTICE +0 -0
outerloop/attempt.py
CHANGED
|
@@ -18,6 +18,7 @@ import logging
|
|
|
18
18
|
import os
|
|
19
19
|
import re
|
|
20
20
|
import shutil
|
|
21
|
+
import time
|
|
21
22
|
from collections.abc import Callable, Iterable
|
|
22
23
|
from dataclasses import dataclass
|
|
23
24
|
from dataclasses import replace as dc_replace
|
|
@@ -25,9 +26,9 @@ from functools import partial
|
|
|
25
26
|
from pathlib import Path
|
|
26
27
|
from typing import Any, cast
|
|
27
28
|
|
|
28
|
-
from outerloop.appauth import resolve_bot_auth
|
|
29
|
+
from outerloop.appauth import add_credential_args, resolve_bot_auth
|
|
29
30
|
from outerloop.brief import BudgetState, distill_lessons
|
|
30
|
-
from outerloop.compute import LocalCompute
|
|
31
|
+
from outerloop.compute import LocalCompute
|
|
31
32
|
from outerloop.contract import Benchmark, Contract, contract_text_in_tree, load_contract
|
|
32
33
|
from outerloop.dispatch import (
|
|
33
34
|
Snapshot,
|
|
@@ -36,6 +37,7 @@ from outerloop.dispatch import (
|
|
|
36
37
|
should_dispatch,
|
|
37
38
|
snapshot_tree,
|
|
38
39
|
)
|
|
40
|
+
from outerloop.evalcache import seed_dir
|
|
39
41
|
from outerloop.github import (
|
|
40
42
|
GitError,
|
|
41
43
|
GitHubClient,
|
|
@@ -43,8 +45,10 @@ from outerloop.github import (
|
|
|
43
45
|
Workspace,
|
|
44
46
|
contract_at,
|
|
45
47
|
ensure_regular_git_dir,
|
|
48
|
+
git_identity,
|
|
46
49
|
)
|
|
47
50
|
from outerloop.harness import Harness, SessionResult, default_binary, redact
|
|
51
|
+
from outerloop.launchlog import append_ended, append_submitted, experiments_rows
|
|
48
52
|
from outerloop.markers import has_marker
|
|
49
53
|
from outerloop.measure import DispatchedMeasurer, DispatchSettings
|
|
50
54
|
from outerloop.orchestrator import (
|
|
@@ -84,9 +88,20 @@ from outerloop.runstate import (
|
|
|
84
88
|
save_record,
|
|
85
89
|
stamp_outage,
|
|
86
90
|
)
|
|
87
|
-
from outerloop.
|
|
91
|
+
from outerloop.runstate import (
|
|
92
|
+
run_dir as run_dir_of,
|
|
93
|
+
)
|
|
94
|
+
from outerloop.syscall import (
|
|
95
|
+
CHANNEL_DIR_NAMES,
|
|
96
|
+
MAX_ARTIFACT_BYTES,
|
|
97
|
+
SyscallRequest,
|
|
98
|
+
channel_dir,
|
|
99
|
+
launch_task_ids,
|
|
100
|
+
tool_update_note,
|
|
101
|
+
)
|
|
88
102
|
from outerloop.syscall import ensure_excluded as syscall_excluded
|
|
89
103
|
from outerloop.syscall import install_tool as syscall_install_tool
|
|
104
|
+
from outerloop.syscall import refresh_tool as syscall_refresh_tool
|
|
90
105
|
from outerloop.syscall import write_budget as syscall_write_budget
|
|
91
106
|
from outerloop.syscall import write_siblings as syscall_write_siblings
|
|
92
107
|
from outerloop.verifier import MAX_CLAIM_CHARS
|
|
@@ -186,7 +201,7 @@ class WorkspaceDrift(RuntimeError):
|
|
|
186
201
|
"""The tree changed between measurement and commit."""
|
|
187
202
|
|
|
188
203
|
|
|
189
|
-
def
|
|
204
|
+
def target_clone_url(target: str) -> str:
|
|
190
205
|
"""The canonical HTTPS clone URL for `owner/repo`. The one source of truth
|
|
191
206
|
for where a run's git pushes go — derived from the target, never read from
|
|
192
207
|
the session-writable `remote.origin.url`."""
|
|
@@ -206,7 +221,8 @@ def _blessed_head(ws: Workspace, result: Any, contract: Any) -> str:
|
|
|
206
221
|
return ""
|
|
207
222
|
try:
|
|
208
223
|
return ws.git("rev-parse", "HEAD").strip()
|
|
209
|
-
except Exception:
|
|
224
|
+
except Exception as exc:
|
|
225
|
+
log.warning("could not read HEAD; not arming self-merge: %s", exc)
|
|
210
226
|
return ""
|
|
211
227
|
|
|
212
228
|
|
|
@@ -425,9 +441,14 @@ def _park_run(
|
|
|
425
441
|
# it lands in the durable record, like every other persisted final_text
|
|
426
442
|
# — a session that echoed a credential must not leave it in record.json.
|
|
427
443
|
# Empty/zero for a baseline park (the session has not run yet).
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
444
|
+
# the author's report at submit, else the session's last words (a park
|
|
445
|
+
# with no submit); either way what the wake's panel and the PR body read
|
|
446
|
+
"report": redact(
|
|
447
|
+
parked.syscall.report
|
|
448
|
+
if parked.syscall is not None and parked.syscall.report
|
|
449
|
+
else (parked.session.final_text if parked.session else ""),
|
|
450
|
+
secrets,
|
|
451
|
+
)[:MAX_CLAIM_CHARS],
|
|
431
452
|
"session_cost_usd": parked.session.cost_usd if parked.session else 0.0,
|
|
432
453
|
"session_turns": parked.session.num_turns if parked.session else 0,
|
|
433
454
|
}
|
|
@@ -452,10 +473,35 @@ def _park_run(
|
|
|
452
473
|
"minutes": launch.minutes,
|
|
453
474
|
"artifacts": list(launch.artifacts),
|
|
454
475
|
**({"array": launch.array} if launch.array > 1 else {}),
|
|
476
|
+
**({"why": redact(launch.why, secrets)} if launch.why else {}),
|
|
477
|
+
**({"concurrency": launch.concurrency} if launch.concurrency else {}),
|
|
455
478
|
}
|
|
456
479
|
for launch in parked.syscall.launches
|
|
457
480
|
]
|
|
458
481
|
stage["syscall_note"] = redact(parked.syscall.note, secrets)
|
|
482
|
+
if parked.syscall.launches:
|
|
483
|
+
# the run's launch ledger (`history`, and the queue view's labels):
|
|
484
|
+
# ids align with launch_jobs order, as stage_launch_job_ids reads them
|
|
485
|
+
if parked.launch_afterany:
|
|
486
|
+
launch_ids = afterany_ids(parked.launch_afterany)
|
|
487
|
+
elif parked.phase == "author-sleep":
|
|
488
|
+
launch_ids = list(job_ids)
|
|
489
|
+
else:
|
|
490
|
+
launch_ids = []
|
|
491
|
+
ledger_launches = tuple(
|
|
492
|
+
dc_replace(launch, why=redact(launch.why, secrets))
|
|
493
|
+
for launch in parked.syscall.launches
|
|
494
|
+
)
|
|
495
|
+
_best_effort(
|
|
496
|
+
"launch ledger",
|
|
497
|
+
lambda: append_submitted(
|
|
498
|
+
run_dir_of(run_root, record.run_id),
|
|
499
|
+
sleep=parked.sleeps_used,
|
|
500
|
+
launches=ledger_launches,
|
|
501
|
+
job_ids=launch_ids,
|
|
502
|
+
at=now,
|
|
503
|
+
),
|
|
504
|
+
)
|
|
459
505
|
# (the session id the wake resumes is the record's own
|
|
460
506
|
# resume_session_id, set below for every park — no stage duplicate)
|
|
461
507
|
stage["launches_used"] = parked.launches_used
|
|
@@ -586,6 +632,7 @@ def _dispatch_settings(args: argparse.Namespace) -> DispatchSettings:
|
|
|
586
632
|
#174: the wake dropped the GPU lane)."""
|
|
587
633
|
from outerloop.compute import compute_from_env
|
|
588
634
|
|
|
635
|
+
target = getattr(args, "target", "") or ""
|
|
589
636
|
return DispatchSettings(
|
|
590
637
|
compute=compute_from_env(),
|
|
591
638
|
image=args.image,
|
|
@@ -593,9 +640,34 @@ def _dispatch_settings(args: argparse.Namespace) -> DispatchSettings:
|
|
|
593
640
|
partition=args.partition,
|
|
594
641
|
gpu_partition=getattr(args, "gpu_partition", ""),
|
|
595
642
|
gpu_account=getattr(args, "gpu_account", ""),
|
|
643
|
+
seed_cache=seed_dir(Path(args.run_root), target) if target else None,
|
|
596
644
|
)
|
|
597
645
|
|
|
598
646
|
|
|
647
|
+
# Experiments yield to verification. Every kernel job is one Slurm user, so
|
|
648
|
+
# among the kernel's own pending jobs the priority order is ours: launches
|
|
649
|
+
# carry this nice so a gate eval or a follow-up re-measure (nice 0) starts
|
|
650
|
+
# first when the cap frees a slot. Sized above the factors that differ between
|
|
651
|
+
# our jobs on Torch — age tops out at 1000 after a week, job size at 1000, the
|
|
652
|
+
# per-GPU TRES share stays in the hundreds — so the order holds however long a
|
|
653
|
+
# launch has waited. Other users' jobs and the group cap are untouched.
|
|
654
|
+
LAUNCH_NICE = 5000
|
|
655
|
+
|
|
656
|
+
|
|
657
|
+
def with_seed(dispatch: DispatchSettings, run_root: Path, target: str) -> DispatchSettings:
|
|
658
|
+
"""These settings with the target's seed cache filled in from the record's
|
|
659
|
+
target when the CLI gave none (wake and follow-up jobs carry the run id,
|
|
660
|
+
not the target)."""
|
|
661
|
+
# tolerant of any settings object: a backend that knows no seed (or a
|
|
662
|
+
# test double) is left exactly as it is
|
|
663
|
+
if not target or getattr(dispatch, "seed_cache", "unknown") is not None:
|
|
664
|
+
return dispatch
|
|
665
|
+
try:
|
|
666
|
+
return dc_replace(dispatch, seed_cache=seed_dir(run_root, target))
|
|
667
|
+
except TypeError:
|
|
668
|
+
return dispatch
|
|
669
|
+
|
|
670
|
+
|
|
599
671
|
def _make_launcher(
|
|
600
672
|
dispatch: DispatchSettings, run_dir: Path, workspace: Path, run_id: str, gpus: int = 0
|
|
601
673
|
):
|
|
@@ -605,45 +677,42 @@ def _make_launcher(
|
|
|
605
677
|
partially-submitted batch is reaped rather than orphaned. `gpus` is the
|
|
606
678
|
benchmark's: an author's experiments run on the same lane as its evals."""
|
|
607
679
|
account, partition = dispatch.placement(gpus)
|
|
608
|
-
# under launch admission a GPU launch enters the queue held; the tick
|
|
609
|
-
# releases it when the user's GPUs fit under the cap (tick.service_admission)
|
|
610
|
-
from outerloop.tick import max_launch_gpus_from_env
|
|
611
|
-
|
|
612
|
-
hold = gpus > 0 and not local_mode() and max_launch_gpus_from_env() > 0
|
|
613
680
|
|
|
614
681
|
def launcher(sha: str, request: SyscallRequest) -> str:
|
|
615
|
-
from dataclasses import replace as _replace
|
|
616
|
-
|
|
617
682
|
from outerloop.dispatch import eval_job_spec, write_eval_job
|
|
618
|
-
from outerloop.syscall import
|
|
683
|
+
from outerloop.syscall import array_spec
|
|
619
684
|
|
|
620
685
|
ids: list[str] = []
|
|
621
686
|
try:
|
|
622
687
|
for launch in request.launches:
|
|
623
|
-
#
|
|
624
|
-
#
|
|
625
|
-
|
|
626
|
-
|
|
627
|
-
|
|
628
|
-
|
|
629
|
-
|
|
630
|
-
|
|
631
|
-
|
|
632
|
-
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
|
|
644
|
-
|
|
645
|
-
|
|
646
|
-
|
|
688
|
+
# a sweep is ONE Slurm job array (`--array=0-N%K`): the queue
|
|
689
|
+
# holds one entry, Slurm runs at most K tasks at once, each task
|
|
690
|
+
# derives its job dir and SWEEP_INDEX from its array index, and
|
|
691
|
+
# one afterany on the array id covers every task
|
|
692
|
+
script = write_eval_job(
|
|
693
|
+
run_dir,
|
|
694
|
+
f"launch-{launch.name}",
|
|
695
|
+
repo_root=workspace,
|
|
696
|
+
snapshot_sha=sha,
|
|
697
|
+
command=launch.command,
|
|
698
|
+
image=dispatch.image,
|
|
699
|
+
artifacts=launch.artifacts,
|
|
700
|
+
artifact_max_bytes=MAX_ARTIFACT_BYTES,
|
|
701
|
+
gpus=gpus,
|
|
702
|
+
array=launch.array,
|
|
703
|
+
seed_cache=dispatch.seed_cache,
|
|
704
|
+
)
|
|
705
|
+
spec = eval_job_spec(
|
|
706
|
+
script,
|
|
707
|
+
job_name=f"{run_id}-launch-{launch.name}",
|
|
708
|
+
account=account,
|
|
709
|
+
partition=partition,
|
|
710
|
+
eval_minutes=launch.minutes,
|
|
711
|
+
gpus=gpus,
|
|
712
|
+
nice=LAUNCH_NICE,
|
|
713
|
+
array=array_spec(launch),
|
|
714
|
+
)
|
|
715
|
+
ids.append(dispatch.compute.submit(spec))
|
|
647
716
|
except Exception:
|
|
648
717
|
# a partial batch must not orphan: no park record was written yet,
|
|
649
718
|
# so nothing would ever wake or cancel the jobs that DID submit —
|
|
@@ -661,6 +730,26 @@ def _make_launcher(
|
|
|
661
730
|
return launcher
|
|
662
731
|
|
|
663
732
|
|
|
733
|
+
def _make_watcher(
|
|
734
|
+
dispatch: DispatchSettings, run_root: Path, run_id: str, workspace: Path, config: RunConfig
|
|
735
|
+
) -> Callable[[], Any]:
|
|
736
|
+
"""The session watcher for one run: a thread beside the harness that
|
|
737
|
+
answers `queue` and `history` from the channel (docs/design/session-watcher.md).
|
|
738
|
+
Shared by the first pass and every wake leg."""
|
|
739
|
+
from outerloop.watcher import SessionWatcher, WatcherContext
|
|
740
|
+
|
|
741
|
+
ctx = WatcherContext(
|
|
742
|
+
workspace=workspace,
|
|
743
|
+
run_root=run_root,
|
|
744
|
+
run_id=run_id,
|
|
745
|
+
target=config.target,
|
|
746
|
+
agent_id=config.agent_id,
|
|
747
|
+
compute=dispatch.compute,
|
|
748
|
+
gpu_partition=dispatch.gpu_partition,
|
|
749
|
+
)
|
|
750
|
+
return lambda: SessionWatcher(ctx)
|
|
751
|
+
|
|
752
|
+
|
|
664
753
|
def _wake_author_sleep(
|
|
665
754
|
*,
|
|
666
755
|
run_root: Path,
|
|
@@ -713,7 +802,12 @@ def _wake_author_sleep(
|
|
|
713
802
|
# the same ending shape every other terminal takes. The line notebook
|
|
714
803
|
# records it first, while the tree is still the session's final tree.
|
|
715
804
|
_push_line_snapshot(
|
|
716
|
-
ws,
|
|
805
|
+
ws,
|
|
806
|
+
_line_ref_for(bench, config.agent_id),
|
|
807
|
+
run_id,
|
|
808
|
+
result.outcome,
|
|
809
|
+
secrets,
|
|
810
|
+
bot_login=config.bot_login,
|
|
717
811
|
)
|
|
718
812
|
for ref in drop_refs:
|
|
719
813
|
drop_snapshot(ws, Snapshot(commit="", tree="", ref=ref))
|
|
@@ -776,6 +870,8 @@ def _wake_author_sleep(
|
|
|
776
870
|
minutes=int(item.get("minutes") or 1),
|
|
777
871
|
artifacts=tuple(str(a) for a in item.get("artifacts", [])),
|
|
778
872
|
array=int(item.get("array") or 1),
|
|
873
|
+
why=str(item.get("why") or ""),
|
|
874
|
+
concurrency=int(item.get("concurrency") or 0),
|
|
779
875
|
)
|
|
780
876
|
for item in _stage_launches(record)
|
|
781
877
|
)
|
|
@@ -786,12 +882,28 @@ def _wake_author_sleep(
|
|
|
786
882
|
# of a blank "job failure". The park's launch job ids align positionally
|
|
787
883
|
# with the results (same launch/array order). Best-effort — the wake never
|
|
788
884
|
# blocks on the scheduler query.
|
|
885
|
+
task_ids = launch_task_ids(launches, stage_launch_job_ids(record))
|
|
789
886
|
status_of = getattr(dispatch.compute, "status", None)
|
|
790
887
|
if status_of is not None:
|
|
791
|
-
results = annotate_launch_states(results,
|
|
888
|
+
results = annotate_launch_states(results, task_ids, status_of)
|
|
792
889
|
launches_used = int(record.stage.get("launches_used", 0)) # type: ignore[call-overload]
|
|
793
890
|
sleeps_used = int(record.stage.get("sleeps_used", 0)) # type: ignore[call-overload]
|
|
794
|
-
|
|
891
|
+
elapsed = _launch_elapsed(dispatch, task_ids) if task_ids else None
|
|
892
|
+
_best_effort(
|
|
893
|
+
"launch ledger",
|
|
894
|
+
lambda: append_ended(
|
|
895
|
+
run_dir, sleep=sleeps_used, results=results, at=time.time(), elapsed_seconds=elapsed
|
|
896
|
+
),
|
|
897
|
+
)
|
|
898
|
+
gpu_hours_used = _reconcile_launch_hours(record, dispatch, bench.gpus, launches, elapsed)
|
|
899
|
+
# the tool the session invokes comes from THIS kernel: a session that
|
|
900
|
+
# started under an older one gets today's verbs and flags at its wake, and
|
|
901
|
+
# is told what is new
|
|
902
|
+
tool_changed = False
|
|
903
|
+
try:
|
|
904
|
+
tool_changed = syscall_refresh_tool(workspace)
|
|
905
|
+
except Exception as exc:
|
|
906
|
+
log.warning("tool refresh failed: %s", redact(f"{type(exc).__name__}: {exc}", secrets))
|
|
795
907
|
wake_text = render_wake(
|
|
796
908
|
results,
|
|
797
909
|
str(record.stage.get("syscall_note", "")),
|
|
@@ -804,9 +916,34 @@ def _wake_author_sleep(
|
|
|
804
916
|
),
|
|
805
917
|
gpus=bench.gpus,
|
|
806
918
|
)
|
|
919
|
+
pacing = [
|
|
920
|
+
f"sweep `{la.name}`: {la.array} tasks, at most {la.concurrency or la.array} at a time"
|
|
921
|
+
for la in launches
|
|
922
|
+
if la.array > 1
|
|
923
|
+
]
|
|
924
|
+
if pacing:
|
|
925
|
+
wake_text = f"{wake_text}\n\n" + "\n".join(pacing) + " (the contract's ceiling applies)."
|
|
926
|
+
if tool_changed:
|
|
927
|
+
wake_text = f"{wake_text}\n\n{tool_update_note(channel_dir(workspace))}"
|
|
807
928
|
if extra_update:
|
|
808
929
|
# a submitted park's gate/panel feedback leads; launch results follow
|
|
809
930
|
wake_text = f"{extra_update}\n\n{wake_text}"
|
|
931
|
+
# A research line whose base moved while it slept RE-PINS to the fresh base
|
|
932
|
+
# and is told to merge it: the agent does the merge (mirroring the in-review
|
|
933
|
+
# conflict wake, followup.py), the kernel only fetches and re-pins. Re-pinning
|
|
934
|
+
# base_sha to the fresh head is what makes the gate baseline and the scope
|
|
935
|
+
# base the CURRENT base (like followup's base_sha_at_fetch), so a sibling's
|
|
936
|
+
# merged work is never credited to this line and a forbidden conflict
|
|
937
|
+
# resolution (differing from the fresh base) is still scope-checked. Non-line
|
|
938
|
+
# runs and an unmoved base are untouched.
|
|
939
|
+
if _line_ref_for(bench, config.agent_id):
|
|
940
|
+
fresh_base = _line_base_advanced(ws, base_branch, base_sha)
|
|
941
|
+
if fresh_base:
|
|
942
|
+
digest = _reintegration_digest(ws, base_sha, fresh_base)
|
|
943
|
+
base_sha = fresh_base
|
|
944
|
+
wake_text = (
|
|
945
|
+
REINTEGRATE_PROMPT.format(base_branch=base_branch, digest=digest) + wake_text
|
|
946
|
+
)
|
|
810
947
|
_best_effort(
|
|
811
948
|
"budget refresh",
|
|
812
949
|
lambda: write_budget(
|
|
@@ -829,7 +966,9 @@ def _wake_author_sleep(
|
|
|
829
966
|
wake_line = _line_ref_for(bench, config.agent_id)
|
|
830
967
|
|
|
831
968
|
def snapshot() -> str:
|
|
832
|
-
snap = snapshot_tree(
|
|
969
|
+
snap = snapshot_tree(
|
|
970
|
+
ws, base_sha, exclude=LINE_MEMORY_PATHS if wake_line else (), author=config.bot_login
|
|
971
|
+
)
|
|
833
972
|
snapshots.append(snap)
|
|
834
973
|
return snap.commit
|
|
835
974
|
|
|
@@ -850,6 +989,7 @@ def _wake_author_sleep(
|
|
|
850
989
|
config.bot_login,
|
|
851
990
|
_utc_date(now),
|
|
852
991
|
exclude=LINE_MEMORY_PATHS if wake_line else (),
|
|
992
|
+
secrets=secrets,
|
|
853
993
|
)
|
|
854
994
|
if panel_lenses
|
|
855
995
|
else None
|
|
@@ -873,6 +1013,7 @@ def _wake_author_sleep(
|
|
|
873
1013
|
resume_session_id=record.resume_session_id,
|
|
874
1014
|
improve_prompt=wake_text,
|
|
875
1015
|
launcher=_make_launcher(dispatch, run_dir, workspace, run_id, gpus=bench.gpus),
|
|
1016
|
+
watcher=_make_watcher(dispatch, run_root, run_id, workspace, config),
|
|
876
1017
|
tree_of=lambda sha: ws.git("rev-parse", f"{sha}^{{tree}}").strip(),
|
|
877
1018
|
judged=judged or _stage_judged(record),
|
|
878
1019
|
launches_used=launches_used,
|
|
@@ -937,6 +1078,32 @@ def _stage_judged(record: RunRecord) -> tuple[str, AttemptResult] | None:
|
|
|
937
1078
|
)
|
|
938
1079
|
|
|
939
1080
|
|
|
1081
|
+
def _ledger_ended(
|
|
1082
|
+
run_dir: Path,
|
|
1083
|
+
record: RunRecord,
|
|
1084
|
+
launches: tuple,
|
|
1085
|
+
task_ids: list[str],
|
|
1086
|
+
dispatch: DispatchSettings,
|
|
1087
|
+
elapsed: list[int | None] | None,
|
|
1088
|
+
) -> None:
|
|
1089
|
+
"""Record a park's finished launches in the ledger from the run dir alone
|
|
1090
|
+
(no delivery into a workspace), with the scheduler's state for jobs that
|
|
1091
|
+
left no exit code."""
|
|
1092
|
+
from outerloop.syscall import annotate_launch_states, read_results
|
|
1093
|
+
|
|
1094
|
+
results = read_results(run_dir, launches)
|
|
1095
|
+
status_of = getattr(dispatch.compute, "status", None)
|
|
1096
|
+
if status_of is not None:
|
|
1097
|
+
results = annotate_launch_states(results, task_ids, status_of)
|
|
1098
|
+
append_ended(
|
|
1099
|
+
run_dir,
|
|
1100
|
+
sleep=int(record.stage.get("sleeps_used", 0)), # type: ignore[call-overload]
|
|
1101
|
+
results=results,
|
|
1102
|
+
at=time.time(),
|
|
1103
|
+
elapsed_seconds=elapsed,
|
|
1104
|
+
)
|
|
1105
|
+
|
|
1106
|
+
|
|
940
1107
|
def _stage_syscall_launches(record: RunRecord) -> tuple:
|
|
941
1108
|
"""The park's launches as `Launch` values (command elided: they ran)."""
|
|
942
1109
|
from outerloop.syscall import Launch
|
|
@@ -948,12 +1115,14 @@ def _stage_syscall_launches(record: RunRecord) -> tuple:
|
|
|
948
1115
|
minutes=int(item.get("minutes") or 1),
|
|
949
1116
|
artifacts=tuple(str(a) for a in item.get("artifacts", [])),
|
|
950
1117
|
array=int(item.get("array") or 1),
|
|
1118
|
+
why=str(item.get("why") or ""),
|
|
1119
|
+
concurrency=int(item.get("concurrency") or 0),
|
|
951
1120
|
)
|
|
952
1121
|
for item in _stage_launches(record)
|
|
953
1122
|
)
|
|
954
1123
|
|
|
955
1124
|
|
|
956
|
-
def
|
|
1125
|
+
def stage_launch_job_ids(record: RunRecord) -> list[str]:
|
|
957
1126
|
"""The park's launch jobs: `launch_afterany` when the park recorded it;
|
|
958
1127
|
for an older author-sleep park every waited job was a launch; for an
|
|
959
1128
|
older candidate park the gate's evals are mixed in, so none."""
|
|
@@ -966,7 +1135,11 @@ def _stage_launch_job_ids(record: RunRecord) -> list[str]:
|
|
|
966
1135
|
|
|
967
1136
|
|
|
968
1137
|
def _reconcile_launch_hours(
|
|
969
|
-
record: RunRecord,
|
|
1138
|
+
record: RunRecord,
|
|
1139
|
+
dispatch: DispatchSettings,
|
|
1140
|
+
gpus: int,
|
|
1141
|
+
launches: tuple,
|
|
1142
|
+
elapsed: list[int | None] | None = None,
|
|
970
1143
|
) -> float:
|
|
971
1144
|
"""The run's GPU-hours after handing back the unused walltime of the
|
|
972
1145
|
park's launch jobs — once: the stage remembers the refund, so a wake
|
|
@@ -976,7 +1149,9 @@ def _reconcile_launch_hours(
|
|
|
976
1149
|
used = float(stage.get("gpu_hours_used", 0.0)) # type: ignore[arg-type]
|
|
977
1150
|
if not gpus or stage.get("launch_hours_refunded"):
|
|
978
1151
|
return used
|
|
979
|
-
refund = _launch_refund(
|
|
1152
|
+
refund = _launch_refund(
|
|
1153
|
+
dispatch, launches, launch_task_ids(launches, stage_launch_job_ids(record)), gpus, elapsed
|
|
1154
|
+
)
|
|
980
1155
|
if refund > 0:
|
|
981
1156
|
log.info("%s: refunding %.2f GPU-hours of unused launch walltime", record.run_id, refund)
|
|
982
1157
|
used = max(0.0, used - refund)
|
|
@@ -985,21 +1160,35 @@ def _reconcile_launch_hours(
|
|
|
985
1160
|
return used
|
|
986
1161
|
|
|
987
1162
|
|
|
1163
|
+
def _launch_elapsed(dispatch: DispatchSettings, job_ids: list[str]) -> list[int | None] | None:
|
|
1164
|
+
"""How long each launch job ran, from the compute, aligned with `job_ids`;
|
|
1165
|
+
None when the compute cannot say (nothing is refunded or recorded on a
|
|
1166
|
+
guess)."""
|
|
1167
|
+
query = getattr(dispatch.compute, "elapsed_seconds", None)
|
|
1168
|
+
if query is None or not job_ids:
|
|
1169
|
+
return None
|
|
1170
|
+
try:
|
|
1171
|
+
return [query(jid) for jid in job_ids]
|
|
1172
|
+
except Exception as exc:
|
|
1173
|
+
log.warning("launch walltime unknown (%s: %s)", type(exc).__name__, exc)
|
|
1174
|
+
return None
|
|
1175
|
+
|
|
1176
|
+
|
|
988
1177
|
def _launch_refund(
|
|
989
|
-
dispatch: DispatchSettings,
|
|
1178
|
+
dispatch: DispatchSettings,
|
|
1179
|
+
launches: tuple,
|
|
1180
|
+
job_ids: list[str],
|
|
1181
|
+
gpus: int,
|
|
1182
|
+
elapsed: list[int | None] | None = None,
|
|
990
1183
|
) -> float:
|
|
991
1184
|
"""The unused walltime of a park's launch jobs, in GPU-hours, or 0 when
|
|
992
1185
|
the compute cannot say how long they ran (nothing is refunded on a
|
|
993
|
-
guess)."""
|
|
1186
|
+
guess). `elapsed` may be handed in when the caller already asked."""
|
|
994
1187
|
from outerloop.syscall import launch_hours_refund
|
|
995
1188
|
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
|
|
999
|
-
try:
|
|
1000
|
-
elapsed = [query(jid) for jid in job_ids]
|
|
1001
|
-
except Exception as exc:
|
|
1002
|
-
log.warning("launch walltime unknown (%s: %s); nothing refunded", type(exc).__name__, exc)
|
|
1189
|
+
if elapsed is None:
|
|
1190
|
+
elapsed = _launch_elapsed(dispatch, job_ids)
|
|
1191
|
+
if elapsed is None:
|
|
1003
1192
|
return 0.0
|
|
1004
1193
|
return launch_hours_refund(launches, elapsed, gpus=gpus)
|
|
1005
1194
|
|
|
@@ -1123,8 +1312,61 @@ def _line_ref_for(bench: Benchmark | None, agent_id: str) -> str:
|
|
|
1123
1312
|
return f"agents/{agent_id}"
|
|
1124
1313
|
|
|
1125
1314
|
|
|
1315
|
+
# The base moved under a line while it slept: rather than the kernel doing a
|
|
1316
|
+
# git merge (which mishandles the agent's in-flight tree, its conflicts, and
|
|
1317
|
+
# the scope gate), the wake mirrors the in-review conflict wake — fetch the
|
|
1318
|
+
# fresh base into the workspace and TELL THE AGENT to merge it. The agent has
|
|
1319
|
+
# git and already resolves the run-start merge as its first task; only the
|
|
1320
|
+
# credential-bearing fetch/push are the kernel's. No commit text goes in the
|
|
1321
|
+
# prompt (no cross-agent prompt-injection surface); the agent reads what
|
|
1322
|
+
# landed from git itself.
|
|
1323
|
+
REINTEGRATE_PROMPT = (
|
|
1324
|
+
"# The base moved while you were asleep\n"
|
|
1325
|
+
"`origin/{base_branch}` advanced since your last run and is fetched into "
|
|
1326
|
+
"your workspace. What landed:\n{digest}\n"
|
|
1327
|
+
"Merge it into your line and resolve any conflicts honestly, then decide "
|
|
1328
|
+
"what to re-run given what landed — if a sibling took your direction "
|
|
1329
|
+
"further, pivot or say so plainly rather than pushing on. Your change is "
|
|
1330
|
+
"measured against the current base.\n\n"
|
|
1331
|
+
)
|
|
1332
|
+
|
|
1333
|
+
|
|
1334
|
+
def _reintegration_digest(ws: Workspace, base_sha: str, fresh_head: str) -> str:
|
|
1335
|
+
"""A short 'what landed' list: the subjects of the commits merged into the
|
|
1336
|
+
base since this line's base, newest first, capped. These are MERGED commits
|
|
1337
|
+
— vetted by the human-merge gate — so they are context, not untrusted input;
|
|
1338
|
+
the agent also has them in git to read in full."""
|
|
1339
|
+
try:
|
|
1340
|
+
out = ws.git("log", "--no-merges", "--format=%s", f"{base_sha}..{fresh_head}")
|
|
1341
|
+
except Exception:
|
|
1342
|
+
return " (recent changes on the base; see `git log`)"
|
|
1343
|
+
subjects = [ln.strip() for ln in out.splitlines() if ln.strip()][:12]
|
|
1344
|
+
return "\n".join(f" - {s}" for s in subjects) or " (a merge on the base; see `git log`)"
|
|
1345
|
+
|
|
1346
|
+
|
|
1347
|
+
def _line_base_advanced(ws: Workspace, base_branch: str, base_sha: str) -> str:
|
|
1348
|
+
"""Fetch `origin/<base_branch>` and return its head when it has advanced
|
|
1349
|
+
past the line's pinned base, else "". The fetch doubles as making the
|
|
1350
|
+
fresh base available for the agent to merge and refreshes the origin refs
|
|
1351
|
+
the scope/measure path pairs against. Best-effort: any git failure returns
|
|
1352
|
+
"" and the wake proceeds exactly as today."""
|
|
1353
|
+
try:
|
|
1354
|
+
ws.fetch_origin()
|
|
1355
|
+
new = ws.git("rev-parse", f"refs/remotes/origin/{base_branch}").strip()
|
|
1356
|
+
merge_base = ws.git("merge-base", new, base_sha).strip()
|
|
1357
|
+
return new if new and merge_base != new else ""
|
|
1358
|
+
except Exception as exc:
|
|
1359
|
+
log.warning("base-moved check failed (%s); wake proceeds unchanged", type(exc).__name__)
|
|
1360
|
+
return ""
|
|
1361
|
+
|
|
1362
|
+
|
|
1126
1363
|
def _push_line_snapshot(
|
|
1127
|
-
ws: Workspace,
|
|
1364
|
+
ws: Workspace,
|
|
1365
|
+
line_ref: str,
|
|
1366
|
+
run_id: str,
|
|
1367
|
+
outcome: str,
|
|
1368
|
+
secrets: tuple[str, ...] = (),
|
|
1369
|
+
bot_login: str = "",
|
|
1128
1370
|
) -> None:
|
|
1129
1371
|
"""Publish the session's final tree to the agent's line as a sealed
|
|
1130
1372
|
snapshot commit — every terminal path, any outcome
|
|
@@ -1167,7 +1409,7 @@ def _push_line_snapshot(
|
|
|
1167
1409
|
fork = parent = remote
|
|
1168
1410
|
except Exception as exc:
|
|
1169
1411
|
log.info("line %s: sealing on the local ref (%s)", line_ref, type(exc).__name__)
|
|
1170
|
-
snap = snapshot_tree(ws, parent, force=memory)
|
|
1412
|
+
snap = snapshot_tree(ws, parent, force=memory, author=bot_login)
|
|
1171
1413
|
try:
|
|
1172
1414
|
# seal only when the tree moved past the parent; the PUSH runs
|
|
1173
1415
|
# either way — a session that COMMITTED its work advanced the
|
|
@@ -1176,10 +1418,7 @@ def _push_line_snapshot(
|
|
|
1176
1418
|
sealed = parent
|
|
1177
1419
|
if snap.tree != ws.git("rev-parse", f"{parent}^{{tree}}").strip():
|
|
1178
1420
|
sealed = ws.git(
|
|
1179
|
-
|
|
1180
|
-
"user.name=autoresearch",
|
|
1181
|
-
"-c",
|
|
1182
|
-
"user.email=autoresearch@localhost",
|
|
1421
|
+
*git_identity(bot_login),
|
|
1183
1422
|
"commit-tree",
|
|
1184
1423
|
snap.tree,
|
|
1185
1424
|
"-p",
|
|
@@ -1239,7 +1478,9 @@ def _reconcile_with_remote(ws: Workspace, old: str, new: str) -> None:
|
|
|
1239
1478
|
ws.git("checkout", new, "--", path)
|
|
1240
1479
|
|
|
1241
1480
|
|
|
1242
|
-
def _checkout_line(
|
|
1481
|
+
def _checkout_line(
|
|
1482
|
+
ws: Workspace, workspace: Path, agent_id: str, base_branch: str, bot_login: str = ""
|
|
1483
|
+
) -> str:
|
|
1243
1484
|
"""Check out the agent's research line: the persistent branch
|
|
1244
1485
|
`agents/<agent-id>`, created from the base branch when absent, with the
|
|
1245
1486
|
base branch merged in when it exists — a conflicted merge is left in the
|
|
@@ -1259,10 +1500,7 @@ def _checkout_line(ws: Workspace, workspace: Path, agent_id: str, base_branch: s
|
|
|
1259
1500
|
conflicted = False
|
|
1260
1501
|
try:
|
|
1261
1502
|
ws.git(
|
|
1262
|
-
|
|
1263
|
-
"user.name=autoresearch",
|
|
1264
|
-
"-c",
|
|
1265
|
-
"user.email=autoresearch@localhost",
|
|
1503
|
+
*git_identity(bot_login),
|
|
1266
1504
|
"merge",
|
|
1267
1505
|
"--no-edit",
|
|
1268
1506
|
base_ref,
|
|
@@ -1277,10 +1515,7 @@ def _checkout_line(ws: Workspace, workspace: Path, agent_id: str, base_branch: s
|
|
|
1277
1515
|
ws.git("add", "-A")
|
|
1278
1516
|
if ws.git("status", "--porcelain").strip():
|
|
1279
1517
|
ws.git(
|
|
1280
|
-
|
|
1281
|
-
"user.name=autoresearch",
|
|
1282
|
-
"-c",
|
|
1283
|
-
"user.email=autoresearch@localhost",
|
|
1518
|
+
*git_identity(bot_login),
|
|
1284
1519
|
"commit",
|
|
1285
1520
|
"-q",
|
|
1286
1521
|
"-m",
|
|
@@ -1466,12 +1701,13 @@ def resume_run(
|
|
|
1466
1701
|
run_dir = run_root / "runs" / run_id
|
|
1467
1702
|
workspace = run_dir / "ws"
|
|
1468
1703
|
record = load_record(run_root, run_id)
|
|
1704
|
+
dispatch = with_seed(dispatch, run_root, record.target)
|
|
1469
1705
|
stage = record.stage
|
|
1470
1706
|
# Push to the CANONICAL target URL, never the workspace's remote.origin.url:
|
|
1471
1707
|
# the session could have rewritten that config to exfil the bot token / code
|
|
1472
1708
|
# to another remote. Passing `url` here means `Workspace.push` uses it
|
|
1473
1709
|
# instead of reading `remote.origin.url`.
|
|
1474
|
-
ws = Workspace(root=workspace, auth=bot_auth, url=
|
|
1710
|
+
ws = Workspace(root=workspace, auth=bot_auth, url=target_clone_url(record.target))
|
|
1475
1711
|
# A session reshaped .git (symlinked object store, gitdir file, FIFO) is
|
|
1476
1712
|
# refused BEFORE anything writes through it: the exclude below opens
|
|
1477
1713
|
# .git/info/exclude, and every ws.git call re-checks. The refusal ENDS
|
|
@@ -1643,11 +1879,16 @@ def resume_run(
|
|
|
1643
1879
|
minutes=int(item.get("minutes") or 1),
|
|
1644
1880
|
artifacts=tuple(str(a) for a in item.get("artifacts", [])),
|
|
1645
1881
|
array=int(item.get("array") or 1),
|
|
1882
|
+
why=str(item.get("why") or ""),
|
|
1883
|
+
concurrency=int(item.get("concurrency") or 0),
|
|
1646
1884
|
)
|
|
1647
1885
|
for item in _stage_launches(record)
|
|
1648
1886
|
),
|
|
1649
1887
|
note=str(stage.get("syscall_note", "")),
|
|
1650
1888
|
submit=True,
|
|
1889
|
+
# the author's report rides every re-park: a suite fan-out
|
|
1890
|
+
# must not drop what the panel and the PR read
|
|
1891
|
+
report=str(stage.get("report", "")),
|
|
1651
1892
|
)
|
|
1652
1893
|
old_afterany = str(record.stage.get("afterany", ""))
|
|
1653
1894
|
made_progress = bool(parked.afterany) and parked.afterany != old_afterany
|
|
@@ -1682,7 +1923,19 @@ def resume_run(
|
|
|
1682
1923
|
# the park's sibling launches are done too: settle their charge before
|
|
1683
1924
|
# any path — publish or hand back to the author — reads the budget
|
|
1684
1925
|
if _stage_launches(record):
|
|
1685
|
-
|
|
1926
|
+
sibling_launches = _stage_syscall_launches(record)
|
|
1927
|
+
sibling_ids = launch_task_ids(sibling_launches, stage_launch_job_ids(record))
|
|
1928
|
+
sibling_elapsed = _launch_elapsed(dispatch, sibling_ids) if sibling_ids else None
|
|
1929
|
+
_reconcile_launch_hours(record, dispatch, bench.gpus, sibling_launches, sibling_elapsed)
|
|
1930
|
+
# the ledger's ended records for the sibling launches, whether or not
|
|
1931
|
+
# the author is woken: the PR's experiments table reads them (an
|
|
1932
|
+
# author wake that follows records nothing twice)
|
|
1933
|
+
_best_effort(
|
|
1934
|
+
"launch ledger",
|
|
1935
|
+
lambda: _ledger_ended(
|
|
1936
|
+
run_dir, record, sibling_launches, sibling_ids, dispatch, sibling_elapsed
|
|
1937
|
+
),
|
|
1938
|
+
)
|
|
1686
1939
|
|
|
1687
1940
|
def _wake_author(
|
|
1688
1941
|
extra_update: str, judged: tuple[str, AttemptResult] | None = None
|
|
@@ -1741,7 +1994,14 @@ def resume_run(
|
|
|
1741
1994
|
# Research lines: record the tree AS OF THIS DECIDED TERMINAL — never
|
|
1742
1995
|
# earlier, because a blocking panel verdict can still resume the
|
|
1743
1996
|
# author (a continuation, not a terminal).
|
|
1744
|
-
_push_line_snapshot(
|
|
1997
|
+
_push_line_snapshot(
|
|
1998
|
+
ws,
|
|
1999
|
+
_line_ref_for(bench, config.agent_id),
|
|
2000
|
+
run_id,
|
|
2001
|
+
outcome,
|
|
2002
|
+
secrets,
|
|
2003
|
+
bot_login=config.bot_login,
|
|
2004
|
+
)
|
|
1745
2005
|
|
|
1746
2006
|
if result.outcome == "improved":
|
|
1747
2007
|
# Publish: branch the SEALED candidate sha, fold in the ledger, push,
|
|
@@ -1888,6 +2148,7 @@ def resume_run(
|
|
|
1888
2148
|
exclude=(
|
|
1889
2149
|
LINE_MEMORY_PATHS if _line_ref_for(bench, config.agent_id) else ()
|
|
1890
2150
|
),
|
|
2151
|
+
secrets=secrets,
|
|
1891
2152
|
)(baseline, candidate, str(stage.get("report", "")))
|
|
1892
2153
|
except Exception as exc:
|
|
1893
2154
|
if isinstance(exc, GitError) and _is_git_tamper(exc):
|
|
@@ -1955,18 +2216,23 @@ def resume_run(
|
|
|
1955
2216
|
# so push the candidate as-is rather than an empty commit.
|
|
1956
2217
|
if staged:
|
|
1957
2218
|
ws.git(
|
|
1958
|
-
|
|
1959
|
-
f"user.name={config.bot_login}",
|
|
1960
|
-
"-c",
|
|
1961
|
-
f"user.email={config.bot_login}@users.noreply.github.com",
|
|
2219
|
+
*git_identity(config.bot_login),
|
|
1962
2220
|
"commit",
|
|
1963
2221
|
"-m",
|
|
1964
2222
|
f"agent: improve {config.benchmark} ({_title_pair(baseline, candidate)})"
|
|
1965
2223
|
f"\n\nAgent: {config.agent_id}",
|
|
1966
2224
|
)
|
|
1967
2225
|
ws.push(branch)
|
|
2226
|
+
if record.stage.get("submitted"):
|
|
2227
|
+
# the author's report at submit rides the stage: the PR shows it
|
|
2228
|
+
# as the research report, over the ledger's experiments
|
|
2229
|
+
result = dc_replace(result, submit_report=str(record.stage.get("report") or ""))
|
|
1968
2230
|
body = pr_body(
|
|
1969
|
-
result,
|
|
2231
|
+
result,
|
|
2232
|
+
config,
|
|
2233
|
+
redact_secrets=secrets,
|
|
2234
|
+
display_digits=bench.display_digits,
|
|
2235
|
+
experiments=experiments_rows(run_dir),
|
|
1970
2236
|
)
|
|
1971
2237
|
if issue_number:
|
|
1972
2238
|
body = f"Addresses #{issue_number}.\n\n{body}"
|
|
@@ -2264,6 +2530,7 @@ def build_panel_runner(
|
|
|
2264
2530
|
start_round: int = 0,
|
|
2265
2531
|
exclude: tuple[str, ...] = (),
|
|
2266
2532
|
claim_body: Callable[[float, float, str], str] | None = None,
|
|
2533
|
+
secrets: tuple[str, ...] = (),
|
|
2267
2534
|
) -> Callable[[float, float, str], PanelVerdict]:
|
|
2268
2535
|
"""The git half of the pre-PR panel: prepare the two read-only checkouts
|
|
2269
2536
|
and the synthetic claim, then hand off to `run_panel` (which owns no git).
|
|
@@ -2287,6 +2554,10 @@ def build_panel_runner(
|
|
|
2287
2554
|
)
|
|
2288
2555
|
|
|
2289
2556
|
def runner(baseline: float, candidate: float, report: str) -> PanelVerdict:
|
|
2557
|
+
# the claim is author text (the report at submit, or the session's
|
|
2558
|
+
# last words): redacted before any lens sees it, like the record and
|
|
2559
|
+
# the PR body
|
|
2560
|
+
report = redact(report, secrets)
|
|
2290
2561
|
reads["n"] += 1
|
|
2291
2562
|
panel_ws = run_dir / "panel"
|
|
2292
2563
|
shutil.rmtree(panel_ws, ignore_errors=True)
|
|
@@ -2299,10 +2570,7 @@ def build_panel_runner(
|
|
|
2299
2570
|
tree = ws.git("write-tree").strip()
|
|
2300
2571
|
ws.git("reset")
|
|
2301
2572
|
snapshot = ws.git(
|
|
2302
|
-
|
|
2303
|
-
"user.name=panel",
|
|
2304
|
-
"-c",
|
|
2305
|
-
"user.email=panel@localhost",
|
|
2573
|
+
*git_identity(bot_login),
|
|
2306
2574
|
"commit-tree",
|
|
2307
2575
|
tree,
|
|
2308
2576
|
"-p",
|
|
@@ -2432,7 +2700,7 @@ def live_attempt(
|
|
|
2432
2700
|
# exception path cannot rely on names bound inside the try
|
|
2433
2701
|
salvage: dict[str, object] = {}
|
|
2434
2702
|
try:
|
|
2435
|
-
ws = Workspace.clone(
|
|
2703
|
+
ws = Workspace.clone(target_clone_url(config.target), workspace, auth=bot_auth)
|
|
2436
2704
|
# Build ON the requested PR base: the clone checks out the remote
|
|
2437
2705
|
# DEFAULT branch, which need not be `base_branch` — the session must
|
|
2438
2706
|
# edit, and the gate must measure, the tree the PR will land on.
|
|
@@ -2496,7 +2764,9 @@ def live_attempt(
|
|
|
2496
2764
|
line_ref = ""
|
|
2497
2765
|
if lines_active:
|
|
2498
2766
|
try:
|
|
2499
|
-
line_ref = _checkout_line(
|
|
2767
|
+
line_ref = _checkout_line(
|
|
2768
|
+
ws, workspace, config.agent_id, base_branch, config.bot_login
|
|
2769
|
+
)
|
|
2500
2770
|
except Exception as exc:
|
|
2501
2771
|
log.warning(
|
|
2502
2772
|
"line checkout failed (%s); running on %s",
|
|
@@ -2639,6 +2909,7 @@ def live_attempt(
|
|
|
2639
2909
|
config.bot_login,
|
|
2640
2910
|
created[:10],
|
|
2641
2911
|
exclude=LINE_MEMORY_PATHS if lines_active else (),
|
|
2912
|
+
secrets=secrets,
|
|
2642
2913
|
)
|
|
2643
2914
|
if panel_lenses
|
|
2644
2915
|
else None
|
|
@@ -2672,12 +2943,16 @@ def live_attempt(
|
|
|
2672
2943
|
run_tag=run_id,
|
|
2673
2944
|
# an inline gate shares the same target-wide baseline cache
|
|
2674
2945
|
baseline_cache=run_dir.parent / "baselines",
|
|
2946
|
+
seed_cache=dispatch.seed_cache if dispatch is not None else None,
|
|
2675
2947
|
)
|
|
2676
2948
|
snapshots: list[Snapshot] = []
|
|
2677
2949
|
|
|
2678
2950
|
def snapshot() -> str:
|
|
2679
2951
|
snap = snapshot_tree(
|
|
2680
|
-
ws,
|
|
2952
|
+
ws,
|
|
2953
|
+
pre_session_sha,
|
|
2954
|
+
exclude=LINE_MEMORY_PATHS if lines_active else (),
|
|
2955
|
+
author=config.bot_login,
|
|
2681
2956
|
)
|
|
2682
2957
|
snapshots.append(snap)
|
|
2683
2958
|
return snap.commit
|
|
@@ -2722,6 +2997,11 @@ def live_attempt(
|
|
|
2722
2997
|
line_memory=line_memory,
|
|
2723
2998
|
line_divergence=line_divergence,
|
|
2724
2999
|
launcher=launcher,
|
|
3000
|
+
watcher=(
|
|
3001
|
+
_make_watcher(dispatch, run_root, run_id, workspace, config)
|
|
3002
|
+
if dispatch is not None
|
|
3003
|
+
else None
|
|
3004
|
+
),
|
|
2725
3005
|
tree_of=lambda sha: ws.git("rev-parse", f"{sha}^{{tree}}").strip(),
|
|
2726
3006
|
)
|
|
2727
3007
|
except RunParked as p:
|
|
@@ -2786,6 +3066,7 @@ def live_attempt(
|
|
|
2786
3066
|
run_id,
|
|
2787
3067
|
"attempt-error",
|
|
2788
3068
|
secrets,
|
|
3069
|
+
bot_login=config.bot_login,
|
|
2789
3070
|
)
|
|
2790
3071
|
failed = RunRecord(
|
|
2791
3072
|
**{
|
|
@@ -2843,7 +3124,7 @@ def live_attempt(
|
|
|
2843
3124
|
# memory (it is excluded from measurable seals by design). The label is
|
|
2844
3125
|
# the GATE outcome, correct at this moment; a publish failure appends a
|
|
2845
3126
|
# publish-error snapshot at the tail.
|
|
2846
|
-
_push_line_snapshot(ws, line_ref, run_id, result.outcome, secrets)
|
|
3127
|
+
_push_line_snapshot(ws, line_ref, run_id, result.outcome, secrets, bot_login=config.bot_login)
|
|
2847
3128
|
|
|
2848
3129
|
pr_url = ""
|
|
2849
3130
|
outcome_name = result.outcome
|
|
@@ -2895,10 +3176,7 @@ def live_attempt(
|
|
|
2895
3176
|
raise WorkspaceDrift(f"publish would stage non-ledger paths: {extra[:10]}")
|
|
2896
3177
|
if staged:
|
|
2897
3178
|
ws.git(
|
|
2898
|
-
|
|
2899
|
-
f"user.name={config.bot_login}",
|
|
2900
|
-
"-c",
|
|
2901
|
-
f"user.email={config.bot_login}@users.noreply.github.com",
|
|
3179
|
+
*git_identity(config.bot_login),
|
|
2902
3180
|
"commit",
|
|
2903
3181
|
"-m",
|
|
2904
3182
|
f"agent: improve {config.benchmark} ({_title_pair(baseline, candidate)})"
|
|
@@ -2907,7 +3185,11 @@ def live_attempt(
|
|
|
2907
3185
|
ws.push(branch)
|
|
2908
3186
|
pushed = True
|
|
2909
3187
|
body = pr_body(
|
|
2910
|
-
result,
|
|
3188
|
+
result,
|
|
3189
|
+
config,
|
|
3190
|
+
redact_secrets=secrets,
|
|
3191
|
+
display_digits=bench.display_digits,
|
|
3192
|
+
experiments=experiments_rows(run_dir),
|
|
2911
3193
|
)
|
|
2912
3194
|
if issue_number:
|
|
2913
3195
|
body = f"Addresses #{issue_number}.\n\n{body}"
|
|
@@ -3022,7 +3304,7 @@ def live_attempt(
|
|
|
3022
3304
|
# the publish failed after the gate credited the tree: the improved
|
|
3023
3305
|
# snapshot above stands (the measurement was real); append the
|
|
3024
3306
|
# publish-error marker so the notebook records how the run ended
|
|
3025
|
-
_push_line_snapshot(ws, line_ref, run_id, outcome_name, secrets)
|
|
3307
|
+
_push_line_snapshot(ws, line_ref, run_id, outcome_name, secrets, bot_login=config.bot_login)
|
|
3026
3308
|
log.info("run %s: %s %s", run_id, outcome_name, pr_url)
|
|
3027
3309
|
return AttemptOutcome(
|
|
3028
3310
|
run_id=run_id,
|
|
@@ -3115,7 +3397,6 @@ def arm_sigterm_containment() -> None:
|
|
|
3115
3397
|
def main() -> int:
|
|
3116
3398
|
import argparse
|
|
3117
3399
|
import os
|
|
3118
|
-
import time
|
|
3119
3400
|
from datetime import UTC, datetime
|
|
3120
3401
|
|
|
3121
3402
|
arm_sigterm_containment()
|
|
@@ -3230,13 +3511,7 @@ def main() -> int:
|
|
|
3230
3511
|
default=120.0,
|
|
3231
3512
|
help="how long before the walltime the self-deadline fires (floor 60)",
|
|
3232
3513
|
)
|
|
3233
|
-
parser
|
|
3234
|
-
parser.add_argument(
|
|
3235
|
-
"--github-app-file",
|
|
3236
|
-
default=os.environ.get("OUTERLOOP_GITHUB_APP_FILE", ""),
|
|
3237
|
-
help="GitHub App config (JSON: app_id, installation_id, private_key); "
|
|
3238
|
-
"when set, installation tokens replace the PAT",
|
|
3239
|
-
)
|
|
3514
|
+
add_credential_args(parser)
|
|
3240
3515
|
parser.add_argument(
|
|
3241
3516
|
"--key-file",
|
|
3242
3517
|
default="",
|