outerloop-science 0.1.0.dev3__py3-none-any.whl → 0.1.0.dev4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
outerloop/maintain.py CHANGED
@@ -85,6 +85,17 @@ MAINTENANCE_LENSES: dict[str, str] = {
85
85
  "whose status no longer matches the code, knobs and flags the docs "
86
86
  "never name, wording that disagrees between two documents."
87
87
  ),
88
+ "architecture": (
89
+ "LENS — abstraction and extensibility, forward-looking rather than "
90
+ "cleanup: where two abstractions could be unified or a layer dropped "
91
+ "so the system is simpler to reason about; and, holding the principle "
92
+ "that a new backend, benchmark, or role should need zero kernel "
93
+ "change, where an extension point is missing so adding one today "
94
+ "forces a kernel edit. Name the files and propose the merge or the "
95
+ "seam; mark these decisions — a refactor of an abstraction many "
96
+ "parts of the system depend on is the maintainer's call, not a "
97
+ "mechanical change."
98
+ ),
88
99
  }
89
100
 
90
101
  SYSTEM_PROMPT = (
@@ -182,6 +193,14 @@ def _item(finding: Finding, repo: str, ref: str) -> str:
182
193
  return f"- {label}**{summary}.** {detail} ([{where}]({link}); {finding.confidence})"
183
194
 
184
195
 
196
+ def digest_title(today: str) -> str:
197
+ """The rolling issue's title, carrying the date of the digest it shows so
198
+ its freshness reads from the issue list. A successful scan sets it to that
199
+ scan's date; a scan that could not run keeps the last good digest — and this
200
+ date — so a stalled scan reads as stale rather than falsely fresh."""
201
+ return f"{DIGEST_TITLE} — {today}"
202
+
203
+
185
204
  def render_digest(
186
205
  result: ReviewResult,
187
206
  *,
@@ -210,7 +229,16 @@ def render_digest(
210
229
  "",
211
230
  ]
212
231
  if result.notes:
213
- lines += [result.notes, ""]
232
+ # keep the top a short summary (title, advisory, counts); the scan's
233
+ # own verdict and its rejected-findings reasoning fold away below it
234
+ lines += [
235
+ "<details><summary>Scan verdict and rejected findings</summary>",
236
+ "",
237
+ result.notes,
238
+ "",
239
+ "</details>",
240
+ "",
241
+ ]
214
242
  by_section: dict[str, list[Finding]] = {}
215
243
  for f in findings:
216
244
  section = f.category if f.category in MAINTENANCE_LENSES else "other"
@@ -18,7 +18,7 @@ from datetime import UTC, datetime
18
18
  from pathlib import Path
19
19
 
20
20
  from outerloop.github import EnvTokenProvider, GitHubClient
21
- from outerloop.maintain import DIGEST_TITLE, MARKER, render_digest, render_stub
21
+ from outerloop.maintain import MARKER, digest_title, render_digest, render_stub
22
22
  from outerloop.posting import EXPECTED_FAILURES
23
23
  from outerloop.review import result_from_data
24
24
 
@@ -84,7 +84,7 @@ def post_digest(
84
84
  str(envelope.get("detail", "")), repo=repo, ref=ref, today=today, who=who
85
85
  )
86
86
  if number is None:
87
- client.create_issue(repo, DIGEST_TITLE, body)
87
+ client.create_issue(repo, digest_title(today), body)
88
88
  else:
89
89
  client.comment(repo, number, body)
90
90
  return "skip-stub"
@@ -92,10 +92,10 @@ def post_digest(
92
92
  result = result_from_data(data if isinstance(data, dict) else {})
93
93
  body = render_digest(result, repo=repo, ref=ref, today=today, reviewed_by=who)
94
94
  if number is None:
95
- number = client.create_issue(repo, DIGEST_TITLE, body)
95
+ number = client.create_issue(repo, digest_title(today), body)
96
96
  log.info("opened the digest issue %s#%s", repo, number)
97
97
  return "created"
98
- client.update_issue(repo, number, body)
98
+ client.update_issue(repo, number, body, title=digest_title(today))
99
99
  # a body edit notifies nobody; the comment does
100
100
  client.comment(
101
101
  repo,
@@ -30,7 +30,7 @@ from outerloop.roles import summarizer_spec
30
30
 
31
31
  log = logging.getLogger(__name__)
32
32
 
33
- MAX_OPINIONS = 8 # artifacts are workflow-authored, but cap the read anyway
33
+ MAX_OPINIONS = 12 # the maintenance scan fans out 9 lenses; cap the read, generously
34
34
 
35
35
 
36
36
  def _load_envelopes(root: Path, repo: str, number: int) -> list[dict]:
outerloop/steward.py CHANGED
@@ -26,7 +26,7 @@ from dataclasses import replace as dc_replace
26
26
  from pathlib import Path
27
27
  from typing import Any, Protocol
28
28
 
29
- from outerloop.appauth import resolve_bot_auth
29
+ from outerloop.appauth import add_credential_args, resolve_bot_auth
30
30
  from outerloop.attempt import (
31
31
  AttemptOutcome,
32
32
  Terminated,
@@ -750,7 +750,6 @@ def live_steward(
750
750
  def main() -> int:
751
751
  import argparse
752
752
  import base64
753
- import os
754
753
  import time
755
754
  from datetime import UTC, datetime
756
755
 
@@ -774,13 +773,7 @@ def main() -> int:
774
773
  parser.add_argument("--session-minutes", type=int, default=60)
775
774
  parser.add_argument("--job-minutes", type=int, default=0)
776
775
  parser.add_argument("--deadline-margin-s", type=float, default=120.0)
777
- parser.add_argument("--pat-file", default=str(CONFIG_DIR / "bot_pat"))
778
- parser.add_argument(
779
- "--github-app-file",
780
- default=os.environ.get("OUTERLOOP_GITHUB_APP_FILE", ""),
781
- help="GitHub App config (JSON: app_id, installation_id, private_key); "
782
- "when set, installation tokens replace the PAT",
783
- )
776
+ add_credential_args(parser)
784
777
  parser.add_argument(
785
778
  "--key-file",
786
779
  default=str(CONFIG_DIR / "steward_key"),
outerloop/tick.py CHANGED
@@ -573,6 +573,7 @@ def service_in_review(
573
573
  dry_run: bool = False,
574
574
  allow_submit: bool = True,
575
575
  contract: Any = None,
576
+ records: list[RunRecord] | None = None,
576
577
  ) -> tuple[list[tuple[str, str]], list[tuple[str, str]]]:
577
578
  """PR-state transitions + follow-up job submission for in-review runs.
578
579
 
@@ -593,13 +594,15 @@ def service_in_review(
593
594
  panel_wake_pending,
594
595
  )
595
596
 
597
+ if records is None:
598
+ records = list_runs(root)
596
599
  ended: list[tuple[str, str]] = []
597
600
  submitted: list[tuple[str, str]] = []
598
- for record in list_runs(root):
601
+ for record in records:
599
602
  if record.state != IN_REVIEW or not record.pr_url:
600
603
  continue
601
604
  try:
602
- ending = close_if_done(root, record, github, now)
605
+ ending = close_if_done(root, load_record(root, record.run_id), github, now)
603
606
  if ending:
604
607
  ended.append((record.run_id, ending))
605
608
  continue
@@ -1268,7 +1271,13 @@ def _ledger_issue_cache(root: Path, target: str) -> Path:
1268
1271
  return root / ("research-log-issue-" + target.replace("/", "__"))
1269
1272
 
1270
1273
 
1271
- def service_research_log(root: Path, github: Any, spec: FollowupSpec, now: float) -> int:
1274
+ def service_research_log(
1275
+ root: Path,
1276
+ github: Any,
1277
+ spec: FollowupSpec,
1278
+ now: float,
1279
+ records: list[RunRecord] | None = None,
1280
+ ) -> int:
1272
1281
  """STATE-driven ledger (terra #170 r1: wiring publisher calls at terminal
1273
1282
  sites missed four terminal paths — attempt error, zero-change resume,
1274
1283
  the direct live terminal, steward): any run of this target whose
@@ -1284,13 +1293,15 @@ def service_research_log(root: Path, github: Any, spec: FollowupSpec, now: float
1284
1293
  before the feature's first pass are marker-stamped silently (no
1285
1294
  backfill spam).
1286
1295
  """
1296
+ if records is None:
1297
+ records = list_runs(root)
1287
1298
  since_path = _ledger_since(root, spec.target)
1288
1299
  first_pass = not since_path.exists()
1289
1300
  if first_pass:
1290
1301
  with contextlib.suppress(OSError):
1291
1302
  since_path.write_text(str(now))
1292
1303
  published = 0
1293
- for record in list_runs(root):
1304
+ for record in records:
1294
1305
  if record.target != spec.target or record.state not in (ENDED, IN_REVIEW):
1295
1306
  continue
1296
1307
  marker = _ledger_marker(root, record.run_id)
@@ -1603,139 +1614,138 @@ def _sweep_one(
1603
1614
  reaped: list[str],
1604
1615
  stuck: list[str],
1605
1616
  ) -> None:
1606
- if True:
1607
- # Leases first: a LIVE wake in flight owns this run — even the stuck
1608
- # verdict must wait for it (its session may be the one that succeeds).
1609
- lease = read_lease(root, record.run_id)
1610
- if lease is not None:
1611
- alive = _holder_alive(compute, lease.holder_job_id)
1612
- if not lease_is_stale(lease, now, lease_ttl_s, alive):
1613
- if not _armed_wake_lost(root, compute, record, lease, now, grace_s, dry_run):
1614
- return
1615
- record = load_record(root, record.run_id)
1616
- if dry_run:
1617
- reaped.append(record.run_id)
1617
+ # Leases first: a LIVE wake in flight owns this run — even the stuck
1618
+ # verdict must wait for it (its session may be the one that succeeds).
1619
+ lease = read_lease(root, record.run_id)
1620
+ if lease is not None:
1621
+ alive = _holder_alive(compute, lease.holder_job_id)
1622
+ if not lease_is_stale(lease, now, lease_ttl_s, alive):
1623
+ if not _armed_wake_lost(root, compute, record, lease, now, grace_s, dry_run):
1618
1624
  return
1619
- if not reap_lease(root, record.run_id, reaper=f"{os.getpid()}-{now}", expected=lease):
1620
- return # a concurrent tick reaped it first; it owns redelivery
1625
+ record = load_record(root, record.run_id)
1626
+ if dry_run:
1621
1627
  reaped.append(record.run_id)
1622
-
1623
- # Layer 5: too many failed attempts is a terminal, reported state.
1624
- if record.wake_attempts >= MAX_WAKE_ATTEMPTS:
1625
- if not dry_run:
1626
- ended = replace(
1627
- record,
1628
- state=ENDED,
1629
- ending=STUCK,
1630
- ending_note=(
1631
- f"{record.wake_attempts} wake attempts without the run leaving 'waiting'"
1632
- ),
1633
- )
1634
- save_record(root, ended, now)
1635
- stuck.append(record.run_id)
1636
1628
  return
1629
+ if not reap_lease(root, record.run_id, reaper=f"{os.getpid()}-{now}", expected=lease):
1630
+ return # a concurrent tick reaped it first; it owns redelivery
1631
+ reaped.append(record.run_id)
1632
+
1633
+ # Layer 5: too many failed attempts is a terminal, reported state.
1634
+ if record.wake_attempts >= MAX_WAKE_ATTEMPTS:
1635
+ if not dry_run:
1636
+ ended = replace(
1637
+ record,
1638
+ state=ENDED,
1639
+ ending=STUCK,
1640
+ ending_note=(
1641
+ f"{record.wake_attempts} wake attempts without the run leaving 'waiting'"
1642
+ ),
1643
+ )
1644
+ save_record(root, ended, now)
1645
+ stuck.append(record.run_id)
1646
+ return
1637
1647
 
1638
- job_ids = _poll_targets(record)
1639
- if not job_ids:
1640
- # No job ids to poll. A BLIND PARK (the measurer could not read Slurm,
1641
- # so `MeasurementPending` carried no ids) still hibernated with a
1642
- # deadline — the deadline floor is its ONLY wake, so fire on it. A
1643
- # genuinely mid-write record has no deadline and is left alone.
1644
- # (A jobless CHECKPOINT SLEEP arrives here too — its deadline is
1645
- # near-term by construction, attempt.py sizes it to the next sweep
1646
- # pass, not the 12h queue slack that protects queued jobs.)
1647
- if record.deadline > 0 and now > record.deadline:
1648
- wake(record, "blind park past deadline", "deadline")
1649
- return
1648
+ job_ids = _poll_targets(record)
1649
+ if not job_ids:
1650
+ # No job ids to poll. A BLIND PARK (the measurer could not read Slurm,
1651
+ # so `MeasurementPending` carried no ids) still hibernated with a
1652
+ # deadline — the deadline floor is its ONLY wake, so fire on it. A
1653
+ # genuinely mid-write record has no deadline and is left alone.
1654
+ # (A jobless CHECKPOINT SLEEP arrives here too — its deadline is
1655
+ # near-term by construction, attempt.py sizes it to the next sweep
1656
+ # pass, not the 12h queue slack that protects queued jobs.)
1657
+ if record.deadline > 0 and now > record.deadline:
1658
+ wake(record, "blind park past deadline", "deadline")
1659
+ return
1650
1660
 
1651
- try:
1652
- states = [compute.status(jid) for jid in job_ids]
1653
- except SlurmQueryError:
1654
- # Layer 4's rule: query failure is "Slurm unknown", never "gone".
1655
- deferred.append(record.run_id)
1656
- return
1661
+ try:
1662
+ states = [compute.status(jid) for jid in job_ids]
1663
+ except SlurmQueryError:
1664
+ # Layer 4's rule: query failure is "Slurm unknown", never "gone".
1665
+ deferred.append(record.run_id)
1666
+ return
1657
1667
 
1658
- # deadline <= 0 cannot be written by save_record for waiting runs;
1659
- # if one exists anyway (legacy/hand-edited), treat it as already past
1660
- # for GONE — a vanished-experiment wake is safe — but never for
1661
- # PENDING, where the consequence would be cancelling a healthy job.
1662
- past_deadline = record.deadline <= 0 or now > record.deadline
1663
-
1664
- if all(is_terminal(s) for s in states):
1665
- state = ",".join(sorted(set(states)))
1666
- # Layer 3, with real grace: time runs from when the sweep FIRST
1667
- # saw every job terminal, not from submission — the afterany
1668
- # job gets the full window to deliver before the backup steps in.
1669
- # Local compute has no afterany jobs to wait for (jobs are
1670
- # terminal at submit): the sweep IS the delivery, so grace would
1671
- # only cost a whole extra loop iteration.
1672
- if local_mode():
1673
- wake(record, f"experiment {state}", state)
1674
- return
1675
- if record.terminal_seen <= 0:
1676
- if dry_run:
1677
- # no writes in dry-run: report the would-wake now so the
1678
- # terminal path is visible to live plumbing checks
1679
- wake(record, f"experiment {state}", state)
1680
- else:
1681
- save_record(
1682
- root,
1683
- replace(
1684
- record,
1685
- terminal_seen=now,
1686
- # repair legacy records as we touch them (see _wake)
1687
- deadline=record.deadline if record.deadline > 0 else now,
1688
- ),
1689
- now,
1690
- )
1691
- return
1692
- if now - record.terminal_seen >= grace_s:
1668
+ # deadline <= 0 cannot be written by save_record for waiting runs;
1669
+ # if one exists anyway (legacy/hand-edited), treat it as already past
1670
+ # for GONE — a vanished-experiment wake is safe — but never for
1671
+ # PENDING, where the consequence would be cancelling a healthy job.
1672
+ past_deadline = record.deadline <= 0 or now > record.deadline
1673
+
1674
+ if all(is_terminal(s) for s in states):
1675
+ state = ",".join(sorted(set(states)))
1676
+ # Layer 3, with real grace: time runs from when the sweep FIRST
1677
+ # saw every job terminal, not from submission — the afterany
1678
+ # job gets the full window to deliver before the backup steps in.
1679
+ # Local compute has no afterany jobs to wait for (jobs are
1680
+ # terminal at submit): the sweep IS the delivery, so grace would
1681
+ # only cost a whole extra loop iteration.
1682
+ if local_mode():
1683
+ wake(record, f"experiment {state}", state)
1684
+ return
1685
+ if record.terminal_seen <= 0:
1686
+ if dry_run:
1687
+ # no writes in dry-run: report the would-wake now so the
1688
+ # terminal path is visible to live plumbing checks
1693
1689
  wake(record, f"experiment {state}", state)
1694
- elif any(is_pending(s) for s in states) and record.deadline > 0 and now > record.deadline:
1695
- # Past the deadline with a job still queued. Ask Slurm WHY before
1696
- # calling it unschedulable: a busy queue or the account's own cap is
1697
- # a wait a scientist would sit out, so the deadline moves out by one
1698
- # slack window instead (Torch 2026-09-06: four launches pending on
1699
- # QOSMaxGRESPerUser were about to be cancelled and re-launched into
1700
- # the same cap). Local compute has no queue and no reasons.
1701
- pending = [j for j, s in zip(job_ids, states, strict=True) if is_pending(s)]
1702
- reason_of = getattr(compute, "pending_reason", None)
1703
- reasons: dict[str, str] = {}
1704
- if reason_of is not None:
1705
- try:
1706
- reasons = {j: str(reason_of(j)) for j in pending}
1707
- except SlurmQueryError:
1708
- deferred.append(record.run_id) # unknown is never "cancel"
1709
- return
1710
- if reasons and all(is_queue_wait(r) for r in reasons.values()):
1711
- extended = now + BLIND_PARK_SLACK_MIN * 60
1712
- if not dry_run:
1713
- save_record(root, replace(record, deadline=extended), now)
1714
- log.info(
1715
- "sweep: %s waits in the queue (%s); deadline extended by %d min",
1716
- record.run_id,
1717
- ", ".join(f"{j}={r}" for j, r in reasons.items()),
1718
- BLIND_PARK_SLACK_MIN,
1690
+ else:
1691
+ save_record(
1692
+ root,
1693
+ replace(
1694
+ record,
1695
+ terminal_seen=now,
1696
+ # repair legacy records as we touch them (see _wake)
1697
+ deadline=record.deadline if record.deadline > 0 else now,
1698
+ ),
1699
+ now,
1719
1700
  )
1701
+ return
1702
+ if now - record.terminal_seen >= grace_s:
1703
+ wake(record, f"experiment {state}", state)
1704
+ elif any(is_pending(s) for s in states) and record.deadline > 0 and now > record.deadline:
1705
+ # Past the deadline with a job still queued. Ask Slurm WHY before
1706
+ # calling it unschedulable: a busy queue or the account's own cap is
1707
+ # a wait a scientist would sit out, so the deadline moves out by one
1708
+ # slack window instead (Torch 2026-09-06: four launches pending on
1709
+ # QOSMaxGRESPerUser were about to be cancelled and re-launched into
1710
+ # the same cap). Local compute has no queue and no reasons.
1711
+ pending = [j for j, s in zip(job_ids, states, strict=True) if is_pending(s)]
1712
+ reason_of = getattr(compute, "pending_reason", None)
1713
+ reasons: dict[str, str] = {}
1714
+ if reason_of is not None:
1715
+ try:
1716
+ reasons = {j: str(reason_of(j)) for j in pending}
1717
+ except SlurmQueryError:
1718
+ deferred.append(record.run_id) # unknown is never "cancel"
1720
1719
  return
1721
- # Unschedulable in practice: cancel every non-terminal job
1722
- # (best-effort — scancel trouble must not abort the sweep), then
1723
- # wake with that fact.
1720
+ if reasons and all(is_queue_wait(r) for r in reasons.values()):
1721
+ extended = now + BLIND_PARK_SLACK_MIN * 60
1724
1722
  if not dry_run:
1725
- for jid, s in zip(job_ids, states, strict=True):
1726
- if is_terminal(s) or s == GONE:
1727
- continue
1728
- try:
1729
- compute.cancel(jid)
1730
- except Exception as exc: # scancel trouble is never fatal here
1731
- log.warning("cancel %s failed: %s", jid, exc)
1732
- wake(record, "experiment unschedulable (pending past deadline)", "unschedulable")
1733
- elif all(is_terminal(s) or s == GONE for s in states):
1734
- # done-or-vanished, at least one GONE (all-terminal handled above)
1735
- if past_deadline:
1736
- wake(record, "experiment vanished from Slurm", "vanished")
1737
- # else: sacct lag right after submission is normal; wait.
1738
- # something RUNNING (or recently pending): nothing to do yet.
1723
+ save_record(root, replace(record, deadline=extended), now)
1724
+ log.info(
1725
+ "sweep: %s waits in the queue (%s); deadline extended by %d min",
1726
+ record.run_id,
1727
+ ", ".join(f"{j}={r}" for j, r in reasons.items()),
1728
+ BLIND_PARK_SLACK_MIN,
1729
+ )
1730
+ return
1731
+ # Unschedulable in practice: cancel every non-terminal job
1732
+ # (best-effort — scancel trouble must not abort the sweep), then
1733
+ # wake with that fact.
1734
+ if not dry_run:
1735
+ for jid, s in zip(job_ids, states, strict=True):
1736
+ if is_terminal(s) or s == GONE:
1737
+ continue
1738
+ try:
1739
+ compute.cancel(jid)
1740
+ except Exception as exc: # scancel trouble is never fatal here
1741
+ log.warning("cancel %s failed: %s", jid, exc)
1742
+ wake(record, "experiment unschedulable (pending past deadline)", "unschedulable")
1743
+ elif all(is_terminal(s) or s == GONE for s in states):
1744
+ # done-or-vanished, at least one GONE (all-terminal handled above)
1745
+ if past_deadline:
1746
+ wake(record, "experiment vanished from Slurm", "vanished")
1747
+ # else: sacct lag right after submission is normal; wait.
1748
+ # something RUNNING (or recently pending): nothing to do yet.
1739
1749
 
1740
1750
 
1741
1751
  def tick(
@@ -1838,7 +1848,19 @@ def tick(
1838
1848
  # no contract), and a live session waiting on `sync` must not depend on
1839
1849
  # whether github/contract loaded this tick.
1840
1850
  if followup_spec is not None:
1841
- service_syncs(root, followup_spec, now)
1851
+ # ONE run-record snapshot for every READ-heavy service that follows the
1852
+ # mutation phase (sweep/park/reap and housekeeping have already run and
1853
+ # written what they will). Sharing it means walking runs/ once, not once
1854
+ # per service. The launch lanes read a stale-but-conservative picture on
1855
+ # purpose: the only record-writer between here and them is
1856
+ # service_in_review, and it only ENDS runs (frees slots), so the
1857
+ # snapshot can only OVER-count active runs — never launch a duplicate.
1858
+ # The one exception is service_research_log, which PUBLISHES terminal
1859
+ # outcomes: it reads fresh (below), never the snapshot. Any service
1860
+ # that acts on a single record's CURRENT state re-reads that one
1861
+ # record fresh via load_record (the freshness guard).
1862
+ tick_records = list_runs(root)
1863
+ service_syncs(root, followup_spec, now, tick_records)
1842
1864
  # a dry run reports and writes nothing: no download, no seed
1843
1865
  if github is not None and followup_spec is not None and followup_spec.target and not dry_run:
1844
1866
  try:
@@ -1914,8 +1936,13 @@ def tick(
1914
1936
  dry_run=followup_dry_run,
1915
1937
  allow_submit=launch_ok,
1916
1938
  contract=contract,
1939
+ records=tick_records,
1917
1940
  )
1918
1941
  try:
1942
+ # research_log reads FRESH, not the shared snapshot: it publishes
1943
+ # terminal outcomes and writes done markers, and service_in_review
1944
+ # just above may have ended a run this tick — a stale in-review
1945
+ # record would be published as 'improved' and locked, wrong.
1919
1946
  service_research_log(root, github, spec, now)
1920
1947
  except Exception as exc: # the ledger is advisory; the tick continues
1921
1948
  log.warning("research-log service failed: %s", exc)
@@ -1928,7 +1955,15 @@ def tick(
1928
1955
  )
1929
1956
  steward_job = (
1930
1957
  service_steward(
1931
- root, github, compute, spec, now, contract, limits, dry_run=followup_dry_run
1958
+ root,
1959
+ github,
1960
+ compute,
1961
+ spec,
1962
+ now,
1963
+ contract,
1964
+ limits,
1965
+ dry_run=followup_dry_run,
1966
+ records=tick_records,
1932
1967
  )
1933
1968
  if launch_ok and intake_job is None and contract is not None
1934
1969
  else None
@@ -1944,6 +1979,7 @@ def tick(
1944
1979
  now,
1945
1980
  limits=limits,
1946
1981
  dry_run=followup_dry_run,
1982
+ records=tick_records,
1947
1983
  )
1948
1984
  except Exception as exc:
1949
1985
  log.warning("self-initiated selection failed: %s", exc)
@@ -1978,7 +2014,9 @@ def service_eval_cache(root: Path, github: Any, target: str) -> str:
1978
2014
  return status
1979
2015
 
1980
2016
 
1981
- def service_syncs(root: Path, spec: Any, now: float) -> None:
2017
+ def service_syncs(
2018
+ root: Path, spec: Any, now: float, records: list[RunRecord] | None = None
2019
+ ) -> None:
1982
2020
  """Honor mid-leg sync requests: a LIVE session asked for fresh origin/*
1983
2021
  refs and is waiting inside its own clock. The fetch pins the canonical
1984
2022
  URL (never the workspace's mutable remote config) and only refs/remotes
@@ -1989,7 +2027,9 @@ def service_syncs(root: Path, spec: Any, now: float) -> None:
1989
2027
  from outerloop.github import Workspace
1990
2028
  from outerloop.syscall import mark_synced, sync_requested
1991
2029
 
1992
- for record in list_runs(root):
2030
+ if records is None:
2031
+ records = list_runs(root)
2032
+ for record in records:
1993
2033
  if record.state != IMPLEMENTING:
1994
2034
  continue
1995
2035
  workspace = run_dir(root, record.run_id) / "ws"
@@ -2022,10 +2062,15 @@ def service_boards(
2022
2062
  publish from the first tick (before any run ends), and a failure never
2023
2063
  stops the tick. With a compute backend, the strip also carries the
2024
2064
  kernel's queue (its own Slurm jobs, attributed to agents)."""
2065
+ # ONE snapshot for the whole board pass: the climb rows, the live strip,
2066
+ # and the queue's owner maps all read the SAME records, so they share it
2067
+ # rather than each re-walking runs/. Taken here (end of tick) so a run
2068
+ # ended earlier this tick already shows.
2069
+ records = list_runs(root)
2025
2070
  try:
2026
2071
  from outerloop.climbboard import contract_directions, service_climb_board
2027
2072
 
2028
- service_climb_board(root, github, target, contract_directions(contract))
2073
+ service_climb_board(root, github, target, contract_directions(contract), records)
2029
2074
  except Exception as exc:
2030
2075
  log.warning("climb board service failed: %s", exc)
2031
2076
  try:
@@ -2037,7 +2082,7 @@ def service_boards(
2037
2082
  queue = compute.queue_snapshot()
2038
2083
  except Exception as exc: # blind this tick: the strip publishes without a queue
2039
2084
  log.warning("queue snapshot failed: %s", exc)
2040
- service_status(root, github, target, now, contract, queue=queue)
2085
+ service_status(root, github, target, now, contract, queue=queue, records=records)
2041
2086
  except Exception as exc: # each is advisory ALONE: one failing never mutes the other
2042
2087
  log.warning("status strip service failed: %s", exc)
2043
2088
 
@@ -2490,6 +2535,7 @@ def service_self_initiated(
2490
2535
  now: float,
2491
2536
  limits: EffectiveLimits | None = None,
2492
2537
  dry_run: bool = False,
2538
+ records: list[RunRecord] | None = None,
2493
2539
  ) -> tuple[str, str] | None:
2494
2540
  """The default background mode: when nothing else needs doing, climb the
2495
2541
  least-recently-attempted benchmark.
@@ -2504,7 +2550,8 @@ def service_self_initiated(
2504
2550
  log.info("self-initiated lane paused (api outage: %s)", paused)
2505
2551
  return None
2506
2552
  try:
2507
- records = list_runs(root)
2553
+ if records is None:
2554
+ records = list_runs(root)
2508
2555
  width = _attempt_width(contract)
2509
2556
  # WIDTH: every live pending marker occupies a slot; landed ones
2510
2557
  # clear; dead ones become per-benchmark tombstones and free theirs.
@@ -2656,6 +2703,7 @@ def service_steward(
2656
2703
  contract: Any,
2657
2704
  limits: EffectiveLimits,
2658
2705
  dry_run: bool = False,
2706
+ records: list[RunRecord] | None = None,
2659
2707
  ) -> tuple[str, str] | None:
2660
2708
  """The steward lane: claim at most ONE labeled work-order issue per tick
2661
2709
  and submit a stewardship job. Off until the operator provisions the
@@ -2673,7 +2721,8 @@ def service_steward(
2673
2721
 
2674
2722
  # ONE active run per target covers stewardships too: an env rewrite
2675
2723
  # must not fly alongside a solver climb or another stewardship.
2676
- records = list_runs(root)
2724
+ if records is None:
2725
+ records = list_runs(root)
2677
2726
  # reconcile first: killed jobs never post their own release — and
2678
2727
  # BEFORE the outage pause below, because a claim orphaned by the
2679
2728
  # very session the outage killed must not stay held all cooldown
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: outerloop-science
3
- Version: 0.1.0.dev3
3
+ Version: 0.1.0.dev4
4
4
  Summary: Autonomous research agents that improve the benchmark you point them at, one verified pull request at a time
5
5
  Project-URL: Homepage, https://outerloop.science
6
6
  Project-URL: Repository, https://github.com/outerloop-science/outerloop