outerloop-science 0.1.0.dev2__py3-none-any.whl → 0.1.0.dev3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- outerloop/__init__.py +2 -2
- outerloop/attempt.py +310 -93
- outerloop/brief.py +38 -25
- outerloop/cli.py +40 -5
- outerloop/climbboard.py +3 -0
- outerloop/compute.py +148 -53
- outerloop/contract.py +8 -0
- outerloop/dispatch.py +63 -18
- outerloop/evalcache.py +147 -0
- outerloop/followup.py +38 -16
- outerloop/github.py +38 -13
- outerloop/harness.py +1 -18
- outerloop/housekeeping.py +1 -17
- outerloop/image.py +0 -4
- outerloop/init.py +19 -1
- outerloop/intake.py +4 -7
- outerloop/launchlog.py +239 -0
- outerloop/maintain.py +325 -0
- outerloop/maintain_agent_cli.py +81 -0
- outerloop/maintain_post_cli.py +140 -0
- outerloop/measure.py +6 -0
- outerloop/orchestrator.py +141 -31
- outerloop/panel.py +3 -3
- outerloop/review.py +4 -0
- outerloop/review_agent.py +7 -7
- outerloop/review_agent_cli.py +2 -2
- outerloop/review_post_cli.py +2 -2
- outerloop/review_summarize_cli.py +7 -5
- outerloop/roles.py +27 -0
- outerloop/rolespec.py +3 -1
- outerloop/steward.py +5 -5
- outerloop/syscall.py +261 -47
- outerloop/syscall_cli.py +243 -12
- outerloop/tick.py +90 -178
- outerloop/verify_agent.py +8 -6
- outerloop/verify_post_cli.py +2 -2
- outerloop/watcher.py +203 -0
- {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/METADATA +4 -1
- outerloop_science-0.1.0.dev3.dist-info/RECORD +59 -0
- outerloop_science-0.1.0.dev2.dist-info/RECORD +0 -53
- {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/WHEEL +0 -0
- {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/entry_points.txt +0 -0
- {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/licenses/LICENSE +0 -0
- {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/licenses/NOTICE +0 -0
outerloop/init.py
CHANGED
|
@@ -19,6 +19,7 @@ from __future__ import annotations
|
|
|
19
19
|
import argparse
|
|
20
20
|
import getpass
|
|
21
21
|
import json
|
|
22
|
+
import logging
|
|
22
23
|
import os
|
|
23
24
|
import shutil
|
|
24
25
|
import sys
|
|
@@ -33,6 +34,8 @@ from outerloop.cli import ENV_FILE
|
|
|
33
34
|
from outerloop.image import ensure_image
|
|
34
35
|
from outerloop.paths import write_private
|
|
35
36
|
|
|
37
|
+
log = logging.getLogger(__name__)
|
|
38
|
+
|
|
36
39
|
CONFIG_DIR = ENV_FILE.parent
|
|
37
40
|
DEFAULT_PAT_FILE = CONFIG_DIR / "bot_pat"
|
|
38
41
|
API = "https://api.github.com"
|
|
@@ -69,6 +72,9 @@ def render_env(
|
|
|
69
72
|
lines.append(f"OUTERLOOP_ACCOUNT={a.account}")
|
|
70
73
|
if a.partition: # optional: unset lets Slurm pick its default partition
|
|
71
74
|
lines.append(f"OUTERLOOP_PARTITION={a.partition}")
|
|
75
|
+
# the checkout stays put until the operator upgrades; `release` follows
|
|
76
|
+
# the release tags, `main` every merge (docs/install.md)
|
|
77
|
+
lines.append("OUTERLOOP_AUTO_UPDATE=off")
|
|
72
78
|
if a.image:
|
|
73
79
|
lines.append(f"OUTERLOOP_IMAGE={a.image}")
|
|
74
80
|
elif a.uncontained:
|
|
@@ -356,7 +362,8 @@ def _owner_type(owner: str) -> str:
|
|
|
356
362
|
try:
|
|
357
363
|
with urllib.request.urlopen(req, timeout=15) as resp:
|
|
358
364
|
return str(json.loads(resp.read()).get("type", ""))
|
|
359
|
-
except Exception:
|
|
365
|
+
except Exception as exc:
|
|
366
|
+
log.warning("could not look up the account type of %s: %s", owner, exc)
|
|
360
367
|
return ""
|
|
361
368
|
|
|
362
369
|
|
|
@@ -684,6 +691,17 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
684
691
|
if login:
|
|
685
692
|
write_private(env_path, render_env(answers, effective_pat, bot_login=login))
|
|
686
693
|
print(f" posting as {login} (OUTERLOOP_BOT_LOGIN)")
|
|
694
|
+
else:
|
|
695
|
+
print(
|
|
696
|
+
" the token's login could not be read — set OUTERLOOP_BOT_LOGIN in "
|
|
697
|
+
f"{env_path} before `outerloop start` (the tick skips a target without it)"
|
|
698
|
+
)
|
|
699
|
+
else:
|
|
700
|
+
print(
|
|
701
|
+
" OUTERLOOP_BOT_LOGIN not recorded (the check did not pass) — rerun "
|
|
702
|
+
"`outerloop init --force` with network access, or set it in "
|
|
703
|
+
f"{env_path} (the tick skips a target without it)"
|
|
704
|
+
)
|
|
687
705
|
else:
|
|
688
706
|
print(" no PAT set — add OUTERLOOP_PAT_FILE before the agents can open PRs")
|
|
689
707
|
_author_key_hint(answers)
|
outerloop/intake.py
CHANGED
|
@@ -13,11 +13,11 @@ from __future__ import annotations
|
|
|
13
13
|
import logging
|
|
14
14
|
from dataclasses import dataclass
|
|
15
15
|
|
|
16
|
-
from outerloop.brief import MAX_TASK_CHARS,
|
|
16
|
+
from outerloop.brief import MAX_TASK_CHARS, cap, code_fence
|
|
17
17
|
from outerloop.contract import Contract
|
|
18
18
|
from outerloop.followup import QUALIFYING_ASSOCIATIONS
|
|
19
19
|
from outerloop.github import is_own_login
|
|
20
|
-
from outerloop.markers import has_label, has_marker,
|
|
20
|
+
from outerloop.markers import has_label, has_marker, marker
|
|
21
21
|
|
|
22
22
|
log = logging.getLogger(__name__)
|
|
23
23
|
|
|
@@ -29,9 +29,6 @@ RELEASE_MARKER = marker("claim-released")
|
|
|
29
29
|
# failure must not claim/release (and comment) forever. Same idea as the
|
|
30
30
|
# steward lane's MAX_STEWARD_ATTEMPTS.
|
|
31
31
|
MAX_INTAKE_ATTEMPTS = 3
|
|
32
|
-
# steward work orders carry this label; they are the STEWARD lane's,
|
|
33
|
-
# never the solver's (a solver climb cannot touch env paths anyway)
|
|
34
|
-
STEWARD_LABEL = label_name("steward")
|
|
35
32
|
|
|
36
33
|
|
|
37
34
|
@dataclass(frozen=True)
|
|
@@ -119,8 +116,8 @@ def issue_hypothesis(task: IssueTask) -> str:
|
|
|
119
116
|
The author passed the standing gate, so the REQUEST is legitimate; the
|
|
120
117
|
fence marks where quoted text ends and the harness's authority resumes.
|
|
121
118
|
"""
|
|
122
|
-
quoted =
|
|
123
|
-
fence =
|
|
119
|
+
quoted = cap(f"{task.title}\n\n{task.body}".strip(), MAX_TASK_CHARS - 400)
|
|
120
|
+
fence = code_fence(quoted)
|
|
124
121
|
return (
|
|
125
122
|
f"A maintainer (@{task.author}) opened issue #{task.number} requesting "
|
|
126
123
|
f"work on the `{task.benchmark}` benchmark. Their request:\n"
|
outerloop/launchlog.py
ADDED
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
"""The per-run launch ledger: append-only JSON lines in the run directory, one
|
|
2
|
+
record when a sleep's launches are submitted and one when each job's result
|
|
3
|
+
comes back at the wake. It is what `history` reads and what labels a launch in
|
|
4
|
+
the queue view (docs/design/session-watcher.md, "History"). Kernel-owned: the
|
|
5
|
+
run directory is never the session's to write."""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from outerloop.syscall import Launch, LaunchResult, launch_jobs
|
|
14
|
+
|
|
15
|
+
LEDGER = "launches.jsonl"
|
|
16
|
+
# generous: a run is depth_k launches x sleep_k sleeps x the array width, far below this
|
|
17
|
+
MAX_LEDGER_BYTES = 4_000_000
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _append(run_dir: Path, rows: list[dict[str, Any]]) -> None:
|
|
21
|
+
if not rows:
|
|
22
|
+
return
|
|
23
|
+
run_dir.mkdir(parents=True, exist_ok=True)
|
|
24
|
+
path = run_dir / LEDGER
|
|
25
|
+
# a crash mid-append leaves a torn last line; start on a fresh one so the
|
|
26
|
+
# torn line is the only record lost, never the next one too
|
|
27
|
+
torn = False
|
|
28
|
+
try:
|
|
29
|
+
with path.open("rb") as fh:
|
|
30
|
+
fh.seek(-1, 2)
|
|
31
|
+
torn = fh.read(1) != b"\n"
|
|
32
|
+
except OSError:
|
|
33
|
+
pass
|
|
34
|
+
with path.open("a", encoding="utf-8") as fh:
|
|
35
|
+
if torn:
|
|
36
|
+
fh.write("\n")
|
|
37
|
+
for row in rows:
|
|
38
|
+
fh.write(json.dumps(row, sort_keys=True) + "\n")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def append_submitted(
|
|
42
|
+
run_dir: Path, *, sleep: int, launches: tuple[Launch, ...], job_ids: list[str], at: float
|
|
43
|
+
) -> None:
|
|
44
|
+
"""One record per launch of a sleep, with the job ids it fanned out to. The
|
|
45
|
+
ids are positional over `launch_jobs` order, exactly as the park recorded
|
|
46
|
+
them; a launch whose ids are missing (an older park) gets none. A launch
|
|
47
|
+
already recorded under this (sleep, name) is not written again: a run
|
|
48
|
+
re-parks the same sleep through a multi-stage gate, and the first record
|
|
49
|
+
is the one with the author's words."""
|
|
50
|
+
known = {
|
|
51
|
+
(int(row.get("sleep") or 0), str(row.get("name") or ""))
|
|
52
|
+
for row in read_ledger(run_dir)
|
|
53
|
+
if row.get("event") == "submitted"
|
|
54
|
+
}
|
|
55
|
+
# one id per launch (a sweep is one Slurm job array), or one per task for
|
|
56
|
+
# a park recorded when arrays were separate jobs
|
|
57
|
+
per_launch = len(job_ids) == len(launches)
|
|
58
|
+
rows: list[dict[str, Any]] = []
|
|
59
|
+
k = 0
|
|
60
|
+
for launch in launches:
|
|
61
|
+
n = 1 if per_launch else len(launch_jobs(launch))
|
|
62
|
+
ids = job_ids[k : k + n]
|
|
63
|
+
k += n
|
|
64
|
+
if (sleep, launch.name) in known:
|
|
65
|
+
continue
|
|
66
|
+
rows.append(
|
|
67
|
+
{
|
|
68
|
+
"event": "submitted",
|
|
69
|
+
"sleep": sleep,
|
|
70
|
+
"name": launch.name,
|
|
71
|
+
"why": launch.why,
|
|
72
|
+
"minutes": launch.minutes,
|
|
73
|
+
"array": launch.array,
|
|
74
|
+
"concurrency": launch.concurrency,
|
|
75
|
+
"job_ids": list(ids),
|
|
76
|
+
"at": at,
|
|
77
|
+
}
|
|
78
|
+
)
|
|
79
|
+
_append(run_dir, rows)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def append_ended(
|
|
83
|
+
run_dir: Path,
|
|
84
|
+
*,
|
|
85
|
+
sleep: int,
|
|
86
|
+
results: tuple[LaunchResult, ...],
|
|
87
|
+
at: float,
|
|
88
|
+
elapsed_seconds: list[int | None] | None = None,
|
|
89
|
+
) -> None:
|
|
90
|
+
"""One record per job that came back at the wake, keyed to its sleep, with
|
|
91
|
+
how long it ran when the compute could say (aligned with `results`) and
|
|
92
|
+
the last line it printed — captured now, because a later launch with the
|
|
93
|
+
same name overwrites the job dir. A job already recorded as ended under
|
|
94
|
+
this (sleep, name) is not written again: a submitted park's wake and the
|
|
95
|
+
author's wake may both see the same jobs."""
|
|
96
|
+
known = {
|
|
97
|
+
(int(row.get("sleep") or 0), str(row.get("name") or ""))
|
|
98
|
+
for row in read_ledger(run_dir)
|
|
99
|
+
if row.get("event") == "ended"
|
|
100
|
+
}
|
|
101
|
+
_append(
|
|
102
|
+
run_dir,
|
|
103
|
+
[
|
|
104
|
+
{
|
|
105
|
+
"event": "ended",
|
|
106
|
+
"sleep": sleep,
|
|
107
|
+
"name": r.name,
|
|
108
|
+
"exit_code": r.exit_code,
|
|
109
|
+
"state": r.slurm_state,
|
|
110
|
+
"elapsed": (
|
|
111
|
+
elapsed_seconds[i]
|
|
112
|
+
if elapsed_seconds is not None and i < len(elapsed_seconds)
|
|
113
|
+
else None
|
|
114
|
+
),
|
|
115
|
+
"last_line": last_line(r.stdout_tail),
|
|
116
|
+
"at": at,
|
|
117
|
+
}
|
|
118
|
+
for i, r in enumerate(results)
|
|
119
|
+
if (sleep, r.name) not in known
|
|
120
|
+
],
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def last_line(text: str, cap: int = 160) -> str:
|
|
125
|
+
"""The last non-empty line of a job's output — its result line, as the
|
|
126
|
+
wake shows the author — on one line and bounded. "" when there is none."""
|
|
127
|
+
for line in reversed(text.splitlines()):
|
|
128
|
+
flat = " ".join(line.split())
|
|
129
|
+
if flat:
|
|
130
|
+
return flat[:cap]
|
|
131
|
+
return ""
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def read_ledger(run_dir: Path) -> list[dict[str, Any]]:
|
|
135
|
+
"""Every record, oldest first; a malformed line is skipped, a missing file
|
|
136
|
+
is an empty history."""
|
|
137
|
+
try:
|
|
138
|
+
with (run_dir / LEDGER).open("rb") as fh:
|
|
139
|
+
raw = fh.read(MAX_LEDGER_BYTES)
|
|
140
|
+
except OSError:
|
|
141
|
+
return []
|
|
142
|
+
rows: list[dict[str, Any]] = []
|
|
143
|
+
for line in raw.decode("utf-8", "replace").splitlines():
|
|
144
|
+
try:
|
|
145
|
+
row = json.loads(line)
|
|
146
|
+
except ValueError:
|
|
147
|
+
continue
|
|
148
|
+
if isinstance(row, dict):
|
|
149
|
+
rows.append(row)
|
|
150
|
+
return rows
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def history(run_dir: Path) -> list[dict[str, Any]]:
|
|
154
|
+
"""Submitted launches in order, each with the ended records of its jobs.
|
|
155
|
+
The identity is (sleep, name): a name is unique within one sleep only."""
|
|
156
|
+
rows = read_ledger(run_dir)
|
|
157
|
+
entries: dict[tuple[int, str], dict[str, Any]] = {}
|
|
158
|
+
for row in rows:
|
|
159
|
+
if row.get("event") != "submitted":
|
|
160
|
+
continue
|
|
161
|
+
key = (int(row.get("sleep") or 0), str(row.get("name") or ""))
|
|
162
|
+
entries[key] = {
|
|
163
|
+
"sleep": row.get("sleep"),
|
|
164
|
+
"name": row.get("name"),
|
|
165
|
+
"why": row.get("why", ""),
|
|
166
|
+
"minutes": row.get("minutes"),
|
|
167
|
+
"array": row.get("array", 1),
|
|
168
|
+
"concurrency": row.get("concurrency", 0),
|
|
169
|
+
"job_ids": list(row.get("job_ids") or []),
|
|
170
|
+
"submitted_at": row.get("at"),
|
|
171
|
+
"jobs": [],
|
|
172
|
+
}
|
|
173
|
+
for row in rows:
|
|
174
|
+
if row.get("event") != "ended":
|
|
175
|
+
continue
|
|
176
|
+
# an array member is `<name>.<i>`; the dot is outside the name alphabet
|
|
177
|
+
launch_name = str(row.get("name") or "").split(".", 1)[0]
|
|
178
|
+
entry = entries.get((int(row.get("sleep") or 0), launch_name))
|
|
179
|
+
if entry is not None:
|
|
180
|
+
entry["jobs"].append(
|
|
181
|
+
{
|
|
182
|
+
"name": row.get("name"),
|
|
183
|
+
"exit_code": row.get("exit_code"),
|
|
184
|
+
"state": row.get("state", ""),
|
|
185
|
+
"elapsed": row.get("elapsed"),
|
|
186
|
+
"last_line": str(row.get("last_line") or ""),
|
|
187
|
+
"ended_at": row.get("at"),
|
|
188
|
+
}
|
|
189
|
+
)
|
|
190
|
+
return list(entries.values())
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def experiments_rows(run_dir: Path) -> list[dict[str, Any]]:
|
|
194
|
+
"""The pull request's experiments table: one row per job of every launch
|
|
195
|
+
this run — its sleep, name and why, how it ended and how long it ran, and
|
|
196
|
+
the last line it printed, all from the ledger (the job dir is overwritten
|
|
197
|
+
by a later launch of the same name). A job not back yet says so."""
|
|
198
|
+
rows: list[dict[str, Any]] = []
|
|
199
|
+
for entry in history(run_dir):
|
|
200
|
+
ended = {str(j.get("name")): j for j in entry["jobs"]}
|
|
201
|
+
array = int(entry.get("array") or 1)
|
|
202
|
+
name = str(entry.get("name") or "")
|
|
203
|
+
jobs = [name] if array <= 1 else [f"{name}.{k}" for k in range(array)]
|
|
204
|
+
for job in jobs:
|
|
205
|
+
j = ended.get(job)
|
|
206
|
+
rows.append(
|
|
207
|
+
{
|
|
208
|
+
"sleep": entry.get("sleep"),
|
|
209
|
+
"launch": name,
|
|
210
|
+
"why": str(entry.get("why") or ""),
|
|
211
|
+
"array": array,
|
|
212
|
+
"concurrency": int(entry.get("concurrency") or 0),
|
|
213
|
+
"job": job,
|
|
214
|
+
"back": j is not None,
|
|
215
|
+
"exit_code": j.get("exit_code") if j else None,
|
|
216
|
+
"state": str(j.get("state") or "") if j else "",
|
|
217
|
+
"elapsed": j.get("elapsed") if j else None,
|
|
218
|
+
"result": str(j.get("last_line") or "") if j else "",
|
|
219
|
+
}
|
|
220
|
+
)
|
|
221
|
+
return rows
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def why_by_job(run_dir: Path) -> dict[str, dict[str, Any]]:
|
|
225
|
+
"""Slurm job id -> the launch it belongs to (name, why, sleep): how the
|
|
226
|
+
queue view labels a launch job for every agent."""
|
|
227
|
+
out: dict[str, dict[str, Any]] = {}
|
|
228
|
+
for row in read_ledger(run_dir):
|
|
229
|
+
if row.get("event") != "submitted":
|
|
230
|
+
continue
|
|
231
|
+
for job_id in row.get("job_ids") or []:
|
|
232
|
+
out[str(job_id)] = {
|
|
233
|
+
"name": row.get("name", ""),
|
|
234
|
+
"why": row.get("why", ""),
|
|
235
|
+
"sleep": row.get("sleep"),
|
|
236
|
+
"array": int(row.get("array") or 1),
|
|
237
|
+
"concurrency": int(row.get("concurrency") or 0),
|
|
238
|
+
}
|
|
239
|
+
return out
|
outerloop/maintain.py
ADDED
|
@@ -0,0 +1,325 @@
|
|
|
1
|
+
"""The maintenance scan: a read-only agent session over a checkout of a
|
|
2
|
+
repository's default branch that records cleanup, upgrade, test-health and
|
|
3
|
+
performance items as findings, and the digest those findings render into.
|
|
4
|
+
|
|
5
|
+
It reuses the reviewer's machinery — the FINDINGS_SCHEMA verdict through the
|
|
6
|
+
syscall tool, lenses fanned out and merged by the summarizer, the emit/post
|
|
7
|
+
split — and differs in three places: the brief scans a tree instead of a
|
|
8
|
+
diff, nothing is blocking, and the destination is one rolling issue
|
|
9
|
+
(docs/design/reviewer-infra.md, "Maintenance scan"). Any repository can run
|
|
10
|
+
it from the reusable workflow; the brief assumes nothing about this one."""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import contextlib
|
|
15
|
+
import logging
|
|
16
|
+
import urllib.parse
|
|
17
|
+
from collections.abc import Iterable
|
|
18
|
+
from datetime import UTC, datetime
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from outerloop.harness import Harness, backend_id
|
|
22
|
+
from outerloop.markers import marker
|
|
23
|
+
from outerloop.posting import EXPECTED_FAILURES
|
|
24
|
+
from outerloop.review import DEFAULT_SYSCALL_CMD, Finding, ReviewResult, sanitize
|
|
25
|
+
from outerloop.review_agent import emit_envelope
|
|
26
|
+
from outerloop.role_runner import run_role
|
|
27
|
+
from outerloop.rolespec import RoleSpec
|
|
28
|
+
|
|
29
|
+
log = logging.getLogger(__name__)
|
|
30
|
+
|
|
31
|
+
MARKER = marker("maintenance-digest")
|
|
32
|
+
DIGEST_TITLE = "Maintainer digest"
|
|
33
|
+
ADVISORY = (
|
|
34
|
+
"*Advisory findings from `outerloop`. The maintainer decides: items marked "
|
|
35
|
+
"**Decision** need a call before any change; the rest are mechanical and may "
|
|
36
|
+
"be taken as work orders. The scan edits nothing.*"
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
# Each lens is one section of the digest; `general` is the whole checklist.
|
|
40
|
+
# The library lives here; which lenses run is the caller workflow's matrix.
|
|
41
|
+
MAINTENANCE_LENSES: dict[str, str] = {
|
|
42
|
+
"pathways": (
|
|
43
|
+
"LENS — dead and unused pathways: symbols, CLI flags, config keys and "
|
|
44
|
+
"environment knobs with no caller or no documentation; compatibility "
|
|
45
|
+
"shims and what each still guards; test-only code living in the "
|
|
46
|
+
"package. Grep across source, tests, scripts and docs before calling "
|
|
47
|
+
"anything unused."
|
|
48
|
+
),
|
|
49
|
+
"duplication": (
|
|
50
|
+
"LENS — logic with more than one owner: the same rule implemented in "
|
|
51
|
+
"two places (two parsers of one file format, two copies of one "
|
|
52
|
+
"sequence), private helpers imported across modules, argument groups "
|
|
53
|
+
"or fixtures copied between entry points or test files, version pins "
|
|
54
|
+
"repeated in several files."
|
|
55
|
+
),
|
|
56
|
+
"structure": (
|
|
57
|
+
"LENS — size and shape: the largest modules and longest functions "
|
|
58
|
+
"(measure them), import cycles and the in-function imports that hide "
|
|
59
|
+
"them, templates or data embedded in code, and the natural seams a "
|
|
60
|
+
"split would follow."
|
|
61
|
+
),
|
|
62
|
+
"upgrades": (
|
|
63
|
+
"LENS — dependencies and tooling: pinned versions against the latest "
|
|
64
|
+
"available (the package index, GitHub releases), CI action versions, "
|
|
65
|
+
"runner images, linter and type-checker settings that could be "
|
|
66
|
+
"tightened cheaply, interpreter versions exercised. Name the pin and "
|
|
67
|
+
"the current upstream for each."
|
|
68
|
+
),
|
|
69
|
+
"tests": (
|
|
70
|
+
"LENS — test-suite health: the slowest tests and why, real sleeps and "
|
|
71
|
+
"real subprocesses where a fake would do, fixtures and fakes defined "
|
|
72
|
+
"several times, tests that no longer pin the behavior they name, "
|
|
73
|
+
"markers declared but unused."
|
|
74
|
+
),
|
|
75
|
+
"performance": (
|
|
76
|
+
"LENS — repeated work on the hot path: find the loop or entry point "
|
|
77
|
+
"the repository runs most often and count what it re-reads, re-lists "
|
|
78
|
+
"or re-fetches per iteration and per record; caching that is missing, "
|
|
79
|
+
"network calls without conditional requests, files parsed more than "
|
|
80
|
+
"once."
|
|
81
|
+
),
|
|
82
|
+
"docs": (
|
|
83
|
+
"LENS — documentation drift: comments that narrate history instead of "
|
|
84
|
+
"intent, changelog sections to consolidate, roadmap or design notes "
|
|
85
|
+
"whose status no longer matches the code, knobs and flags the docs "
|
|
86
|
+
"never name, wording that disagrees between two documents."
|
|
87
|
+
),
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
SYSTEM_PROMPT = (
|
|
91
|
+
"You are the maintainer's periodic scan of this repository. You read the whole "
|
|
92
|
+
"tree, measure rather than guess, and record each item worth doing as a finding. "
|
|
93
|
+
"Nothing you find blocks anything: the maintainer reads the digest and decides.\n\n"
|
|
94
|
+
"What to record: cleanup, simplification, upgrade, test-health and performance "
|
|
95
|
+
"items — what a careful maintainer would put on their own list after a week away. "
|
|
96
|
+
"Skip style nits a linter already reports and work the repository's own roadmap "
|
|
97
|
+
"already tracks as planned.\n\n"
|
|
98
|
+
"Evidence: every item names a file and a line, and a measured fact where one "
|
|
99
|
+
"exists (a line count, a call count, the pinned and the latest version, a test "
|
|
100
|
+
"duration). Say what you ran.\n\n"
|
|
101
|
+
"Shape of each finding:\n"
|
|
102
|
+
"- --kind change: mechanical, safe for an agent to do in a pull request without a "
|
|
103
|
+
"design call.\n"
|
|
104
|
+
"- --kind question: needs the maintainer's decision first (when to drop a "
|
|
105
|
+
"compatibility path, whether to re-verify a pinned tool, which module owns a "
|
|
106
|
+
"duplicated rule).\n"
|
|
107
|
+
"- --kind note: worth knowing, not worth a change.\n"
|
|
108
|
+
"- --category: the digest section the item belongs to, one of {sections}.\n"
|
|
109
|
+
'- --detail: start with effort and risk, for example "S, low." (S is under an '
|
|
110
|
+
"hour, M an afternoon, L a day or more; risk is what could break), then the "
|
|
111
|
+
"evidence.\n"
|
|
112
|
+
"- never --blocking.\n\n"
|
|
113
|
+
"Your concluding notes open the digest: one line on what is healthy, then the "
|
|
114
|
+
"three items most worth doing, one sentence each."
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _investigation(ref: str, syscall_cmd: str) -> str:
|
|
119
|
+
return (
|
|
120
|
+
f"The repository is checked out in your working directory at commit {ref}. "
|
|
121
|
+
"Use Read, Grep and Glob, and run read-only commands in the shell: line "
|
|
122
|
+
"counts, the test suite with durations, the package manager's outdated "
|
|
123
|
+
"list, the linter's statistics. Do not modify the tree, do not install "
|
|
124
|
+
"anything beyond what its own lockfile describes, and do not push or "
|
|
125
|
+
"post anything — your only product is the verdict.\n\n"
|
|
126
|
+
"Record each item as you confirm it, one command per item:\n"
|
|
127
|
+
f" {syscall_cmd} finding --file <path> [--line N] "
|
|
128
|
+
"--confidence <low|medium|high> --category <section> --summary <one line> "
|
|
129
|
+
"--detail <effort, risk, then the evidence> --kind <change|question|note>\n"
|
|
130
|
+
"When you are done, commit your verdict and end your turn:\n"
|
|
131
|
+
f" {syscall_cmd} conclude --notes <what is healthy; the three items most worth doing>\n"
|
|
132
|
+
"The verdict you commit is your final answer — do not also restate it in a message."
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def build_maintenance_brief(
|
|
137
|
+
repo: str,
|
|
138
|
+
ref: str,
|
|
139
|
+
today: str | None = None,
|
|
140
|
+
*,
|
|
141
|
+
syscall_cmd: str = DEFAULT_SYSCALL_CMD,
|
|
142
|
+
lens: str = "",
|
|
143
|
+
) -> str:
|
|
144
|
+
"""The scan brief: the standing prompt, the lens (or every section for
|
|
145
|
+
`general`), the investigation instruction and the repository line. An
|
|
146
|
+
unknown lens fails loudly, as the reviewer's does."""
|
|
147
|
+
if lens and lens != "general" and lens not in MAINTENANCE_LENSES:
|
|
148
|
+
raise ValueError(f"unknown maintenance lens {lens!r} (have: {sorted(MAINTENANCE_LENSES)})")
|
|
149
|
+
sections = ", ".join(MAINTENANCE_LENSES)
|
|
150
|
+
if lens and lens != "general":
|
|
151
|
+
focus = MAINTENANCE_LENSES[lens]
|
|
152
|
+
else:
|
|
153
|
+
focus = "Cover every section:\n\n" + "\n\n".join(MAINTENANCE_LENSES.values())
|
|
154
|
+
header = f"Today's date: {today}\n" if today else ""
|
|
155
|
+
header += f"Repository: {repo} at {ref}"
|
|
156
|
+
return (
|
|
157
|
+
f"{SYSTEM_PROMPT.format(sections=sections)}\n\n{focus}\n\n"
|
|
158
|
+
f"{_investigation(ref, syscall_cmd)}\n\n{header}\n"
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
_KIND_ORDER = {"question": 0, "change": 1, "suggestion": 1, "note": 2}
|
|
163
|
+
_CONFIDENCE_ORDER = {"high": 0, "medium": 1, "low": 2}
|
|
164
|
+
_LABEL = {"question": "**Decision.** ", "note": "*Note.* "}
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _item(finding: Finding, repo: str, ref: str) -> str:
|
|
168
|
+
# backticks stripped: a file value containing one would close the code
|
|
169
|
+
# span and render model markdown inline (same rule as the review body)
|
|
170
|
+
safe_file = finding.file.replace("`", "")
|
|
171
|
+
where = f"`{safe_file}`" + (f":{finding.line}" if finding.line else "")
|
|
172
|
+
link = (
|
|
173
|
+
f"https://github.com/{repo}/blob/{urllib.parse.quote(ref)}/{urllib.parse.quote(safe_file)}"
|
|
174
|
+
)
|
|
175
|
+
if finding.line:
|
|
176
|
+
link += f"#L{finding.line}"
|
|
177
|
+
summary = finding.summary.rstrip(".!?…")
|
|
178
|
+
if summary.count("`") % 2:
|
|
179
|
+
summary += "`"
|
|
180
|
+
detail = finding.detail + ("`" if finding.detail.count("`") % 2 else "")
|
|
181
|
+
label = _LABEL.get(finding.kind, "")
|
|
182
|
+
return f"- {label}**{summary}.** {detail} ([{where}]({link}); {finding.confidence})"
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def render_digest(
|
|
186
|
+
result: ReviewResult,
|
|
187
|
+
*,
|
|
188
|
+
repo: str,
|
|
189
|
+
ref: str,
|
|
190
|
+
today: str,
|
|
191
|
+
reviewed_by: str,
|
|
192
|
+
) -> str:
|
|
193
|
+
"""The rolling issue's body: marker first, the header and the advisory
|
|
194
|
+
line, the counts, the scan's own summary, then one section per category
|
|
195
|
+
with decisions first. Every string in `result` is already sanitized by
|
|
196
|
+
`result_from_data`; the marker leads so the poster can find the issue."""
|
|
197
|
+
findings = result.findings
|
|
198
|
+
decisions = sum(1 for f in findings if f.kind == "question")
|
|
199
|
+
notes = sum(1 for f in findings if f.kind == "note")
|
|
200
|
+
mechanical = len(findings) - decisions - notes
|
|
201
|
+
who = sanitize(reviewed_by, 120) or "unattributed"
|
|
202
|
+
lines = [
|
|
203
|
+
MARKER,
|
|
204
|
+
f"**{DIGEST_TITLE}** — {repo} at `{ref[:8]}` on {today}; scanned by `{who}`.",
|
|
205
|
+
"",
|
|
206
|
+
ADVISORY,
|
|
207
|
+
"",
|
|
208
|
+
f"{len(findings)} items: {decisions} need a decision, {mechanical} are "
|
|
209
|
+
f"mechanical, {notes} are notes.",
|
|
210
|
+
"",
|
|
211
|
+
]
|
|
212
|
+
if result.notes:
|
|
213
|
+
lines += [result.notes, ""]
|
|
214
|
+
by_section: dict[str, list[Finding]] = {}
|
|
215
|
+
for f in findings:
|
|
216
|
+
section = f.category if f.category in MAINTENANCE_LENSES else "other"
|
|
217
|
+
by_section.setdefault(section, []).append(f)
|
|
218
|
+
for section in [*MAINTENANCE_LENSES, "other"]:
|
|
219
|
+
items = by_section.get(section)
|
|
220
|
+
if not items:
|
|
221
|
+
continue
|
|
222
|
+
items.sort(key=lambda f: (_KIND_ORDER.get(f.kind, 2), _CONFIDENCE_ORDER[f.confidence]))
|
|
223
|
+
lines += [f"### {section}", ""]
|
|
224
|
+
lines += [_item(f, repo, ref) for f in items]
|
|
225
|
+
lines.append("")
|
|
226
|
+
lines.append(
|
|
227
|
+
"_Each scan replaces this body; earlier digests are in the edit history. "
|
|
228
|
+
"Run a scan by hand from the Actions tab (maintenance → Run workflow)._"
|
|
229
|
+
)
|
|
230
|
+
return "\n".join(lines).rstrip() + "\n"
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def render_stub(detail: str, *, repo: str, ref: str, today: str, who: str) -> str:
|
|
234
|
+
"""What the poster writes when the scan could not run: the reason, on the
|
|
235
|
+
digest issue, never silence."""
|
|
236
|
+
reason = sanitize(detail, 300)
|
|
237
|
+
by = f" ({sanitize(who, 120)})" if who else ""
|
|
238
|
+
return (
|
|
239
|
+
f"{MARKER}\n**{DIGEST_TITLE}** — the scan of {repo} at `{ref[:8]}` on {today} "
|
|
240
|
+
f"could not run{by}: {reason}"
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def run_maintenance_scan(
|
|
245
|
+
repo: str,
|
|
246
|
+
ref: str,
|
|
247
|
+
harness: Harness,
|
|
248
|
+
workspace: Path,
|
|
249
|
+
*,
|
|
250
|
+
spec: RoleSpec | None = None,
|
|
251
|
+
emit_path: Path,
|
|
252
|
+
today: str | None = None,
|
|
253
|
+
lens: str = "",
|
|
254
|
+
) -> str | None:
|
|
255
|
+
"""One lens session over `workspace` (a default-branch checkout the caller
|
|
256
|
+
prepared and sanitized). EVERY outcome writes an envelope for the posting
|
|
257
|
+
job — findings, or a skip-stub naming why — so a missing artifact always
|
|
258
|
+
means a broken session. Returns "emitted", or None when it could not
|
|
259
|
+
produce a verdict. Advisory: never raises the expected failures."""
|
|
260
|
+
from outerloop.roles import maintainer_spec
|
|
261
|
+
|
|
262
|
+
spec = spec or maintainer_spec()
|
|
263
|
+
today = today or datetime.now(UTC).date().isoformat()
|
|
264
|
+
try:
|
|
265
|
+
from outerloop.syscall import tool_command
|
|
266
|
+
|
|
267
|
+
brief = build_maintenance_brief(
|
|
268
|
+
repo, ref, today, syscall_cmd=tool_command(workspace), lens=lens
|
|
269
|
+
)
|
|
270
|
+
role_result = run_role(spec, harness, brief, workspace)
|
|
271
|
+
if not role_result.ok or role_result.data is None:
|
|
272
|
+
detail = role_result.error or role_result.session.stop_reason
|
|
273
|
+
log.warning("maintenance scan produced no verdict on %s (%s): %s", repo, lens, detail)
|
|
274
|
+
emit_envelope(
|
|
275
|
+
emit_path,
|
|
276
|
+
repo,
|
|
277
|
+
0,
|
|
278
|
+
kind="skip-stub",
|
|
279
|
+
detail=detail,
|
|
280
|
+
reviewed_by=backend_id(harness),
|
|
281
|
+
lens=lens,
|
|
282
|
+
)
|
|
283
|
+
return None
|
|
284
|
+
emit_envelope(
|
|
285
|
+
emit_path,
|
|
286
|
+
repo,
|
|
287
|
+
0,
|
|
288
|
+
kind="findings",
|
|
289
|
+
data=role_result.data,
|
|
290
|
+
reviewed_by=backend_id(harness),
|
|
291
|
+
lens=lens,
|
|
292
|
+
)
|
|
293
|
+
cost = role_result.session.cost_usd
|
|
294
|
+
log.info(
|
|
295
|
+
"emitted maintenance findings for %s (%s; cost=%s turns=%d)",
|
|
296
|
+
repo,
|
|
297
|
+
lens or "general",
|
|
298
|
+
f"${cost:.2f}" if cost else "unreported",
|
|
299
|
+
role_result.session.num_turns,
|
|
300
|
+
)
|
|
301
|
+
return "emitted"
|
|
302
|
+
except EXPECTED_FAILURES as exc: # advisory: never red the repository's Actions
|
|
303
|
+
log.warning("maintenance scan did not complete: %s: %s", type(exc).__name__, exc)
|
|
304
|
+
with contextlib.suppress(Exception):
|
|
305
|
+
emit_envelope(
|
|
306
|
+
emit_path,
|
|
307
|
+
repo,
|
|
308
|
+
0,
|
|
309
|
+
kind="skip-stub",
|
|
310
|
+
detail=f"{type(exc).__name__}: {exc}",
|
|
311
|
+
reviewed_by=backend_id(harness),
|
|
312
|
+
lens=lens,
|
|
313
|
+
)
|
|
314
|
+
return None
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def lens_names(lenses: Iterable[str]) -> list[str]:
|
|
318
|
+
"""The lens names a caller configured, `general` included, unknown ones
|
|
319
|
+
refused — so a misspelled matrix entry fails at configuration time."""
|
|
320
|
+
out = []
|
|
321
|
+
for name in lenses:
|
|
322
|
+
if name != "general" and name not in MAINTENANCE_LENSES:
|
|
323
|
+
raise ValueError(f"unknown maintenance lens {name!r}")
|
|
324
|
+
out.append(name)
|
|
325
|
+
return out
|