outerloop-science 0.1.0.dev2__py3-none-any.whl → 0.1.0.dev3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. outerloop/__init__.py +2 -2
  2. outerloop/attempt.py +310 -93
  3. outerloop/brief.py +38 -25
  4. outerloop/cli.py +40 -5
  5. outerloop/climbboard.py +3 -0
  6. outerloop/compute.py +148 -53
  7. outerloop/contract.py +8 -0
  8. outerloop/dispatch.py +63 -18
  9. outerloop/evalcache.py +147 -0
  10. outerloop/followup.py +38 -16
  11. outerloop/github.py +38 -13
  12. outerloop/harness.py +1 -18
  13. outerloop/housekeeping.py +1 -17
  14. outerloop/image.py +0 -4
  15. outerloop/init.py +19 -1
  16. outerloop/intake.py +4 -7
  17. outerloop/launchlog.py +239 -0
  18. outerloop/maintain.py +325 -0
  19. outerloop/maintain_agent_cli.py +81 -0
  20. outerloop/maintain_post_cli.py +140 -0
  21. outerloop/measure.py +6 -0
  22. outerloop/orchestrator.py +141 -31
  23. outerloop/panel.py +3 -3
  24. outerloop/review.py +4 -0
  25. outerloop/review_agent.py +7 -7
  26. outerloop/review_agent_cli.py +2 -2
  27. outerloop/review_post_cli.py +2 -2
  28. outerloop/review_summarize_cli.py +7 -5
  29. outerloop/roles.py +27 -0
  30. outerloop/rolespec.py +3 -1
  31. outerloop/steward.py +5 -5
  32. outerloop/syscall.py +261 -47
  33. outerloop/syscall_cli.py +243 -12
  34. outerloop/tick.py +90 -178
  35. outerloop/verify_agent.py +8 -6
  36. outerloop/verify_post_cli.py +2 -2
  37. outerloop/watcher.py +203 -0
  38. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/METADATA +4 -1
  39. outerloop_science-0.1.0.dev3.dist-info/RECORD +59 -0
  40. outerloop_science-0.1.0.dev2.dist-info/RECORD +0 -53
  41. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/WHEEL +0 -0
  42. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/entry_points.txt +0 -0
  43. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/licenses/LICENSE +0 -0
  44. {outerloop_science-0.1.0.dev2.dist-info → outerloop_science-0.1.0.dev3.dist-info}/licenses/NOTICE +0 -0
outerloop/init.py CHANGED
@@ -19,6 +19,7 @@ from __future__ import annotations
19
19
  import argparse
20
20
  import getpass
21
21
  import json
22
+ import logging
22
23
  import os
23
24
  import shutil
24
25
  import sys
@@ -33,6 +34,8 @@ from outerloop.cli import ENV_FILE
33
34
  from outerloop.image import ensure_image
34
35
  from outerloop.paths import write_private
35
36
 
37
+ log = logging.getLogger(__name__)
38
+
36
39
  CONFIG_DIR = ENV_FILE.parent
37
40
  DEFAULT_PAT_FILE = CONFIG_DIR / "bot_pat"
38
41
  API = "https://api.github.com"
@@ -69,6 +72,9 @@ def render_env(
69
72
  lines.append(f"OUTERLOOP_ACCOUNT={a.account}")
70
73
  if a.partition: # optional: unset lets Slurm pick its default partition
71
74
  lines.append(f"OUTERLOOP_PARTITION={a.partition}")
75
+ # the checkout stays put until the operator upgrades; `release` follows
76
+ # the release tags, `main` every merge (docs/install.md)
77
+ lines.append("OUTERLOOP_AUTO_UPDATE=off")
72
78
  if a.image:
73
79
  lines.append(f"OUTERLOOP_IMAGE={a.image}")
74
80
  elif a.uncontained:
@@ -356,7 +362,8 @@ def _owner_type(owner: str) -> str:
356
362
  try:
357
363
  with urllib.request.urlopen(req, timeout=15) as resp:
358
364
  return str(json.loads(resp.read()).get("type", ""))
359
- except Exception:
365
+ except Exception as exc:
366
+ log.warning("could not look up the account type of %s: %s", owner, exc)
360
367
  return ""
361
368
 
362
369
 
@@ -684,6 +691,17 @@ def main(argv: list[str] | None = None) -> int:
684
691
  if login:
685
692
  write_private(env_path, render_env(answers, effective_pat, bot_login=login))
686
693
  print(f" posting as {login} (OUTERLOOP_BOT_LOGIN)")
694
+ else:
695
+ print(
696
+ " the token's login could not be read — set OUTERLOOP_BOT_LOGIN in "
697
+ f"{env_path} before `outerloop start` (the tick skips a target without it)"
698
+ )
699
+ else:
700
+ print(
701
+ " OUTERLOOP_BOT_LOGIN not recorded (the check did not pass) — rerun "
702
+ "`outerloop init --force` with network access, or set it in "
703
+ f"{env_path} (the tick skips a target without it)"
704
+ )
687
705
  else:
688
706
  print(" no PAT set — add OUTERLOOP_PAT_FILE before the agents can open PRs")
689
707
  _author_key_hint(answers)
outerloop/intake.py CHANGED
@@ -13,11 +13,11 @@ from __future__ import annotations
13
13
  import logging
14
14
  from dataclasses import dataclass
15
15
 
16
- from outerloop.brief import MAX_TASK_CHARS, _cap, _fence
16
+ from outerloop.brief import MAX_TASK_CHARS, cap, code_fence
17
17
  from outerloop.contract import Contract
18
18
  from outerloop.followup import QUALIFYING_ASSOCIATIONS
19
19
  from outerloop.github import is_own_login
20
- from outerloop.markers import has_label, has_marker, label_name, marker
20
+ from outerloop.markers import has_label, has_marker, marker
21
21
 
22
22
  log = logging.getLogger(__name__)
23
23
 
@@ -29,9 +29,6 @@ RELEASE_MARKER = marker("claim-released")
29
29
  # failure must not claim/release (and comment) forever. Same idea as the
30
30
  # steward lane's MAX_STEWARD_ATTEMPTS.
31
31
  MAX_INTAKE_ATTEMPTS = 3
32
- # steward work orders carry this label; they are the STEWARD lane's,
33
- # never the solver's (a solver climb cannot touch env paths anyway)
34
- STEWARD_LABEL = label_name("steward")
35
32
 
36
33
 
37
34
  @dataclass(frozen=True)
@@ -119,8 +116,8 @@ def issue_hypothesis(task: IssueTask) -> str:
119
116
  The author passed the standing gate, so the REQUEST is legitimate; the
120
117
  fence marks where quoted text ends and the harness's authority resumes.
121
118
  """
122
- quoted = _cap(f"{task.title}\n\n{task.body}".strip(), MAX_TASK_CHARS - 400)
123
- fence = _fence(quoted)
119
+ quoted = cap(f"{task.title}\n\n{task.body}".strip(), MAX_TASK_CHARS - 400)
120
+ fence = code_fence(quoted)
124
121
  return (
125
122
  f"A maintainer (@{task.author}) opened issue #{task.number} requesting "
126
123
  f"work on the `{task.benchmark}` benchmark. Their request:\n"
outerloop/launchlog.py ADDED
@@ -0,0 +1,239 @@
1
+ """The per-run launch ledger: append-only JSON lines in the run directory, one
2
+ record when a sleep's launches are submitted and one when each job's result
3
+ comes back at the wake. It is what `history` reads and what labels a launch in
4
+ the queue view (docs/design/session-watcher.md, "History"). Kernel-owned: the
5
+ run directory is never the session's to write."""
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ from pathlib import Path
11
+ from typing import Any
12
+
13
+ from outerloop.syscall import Launch, LaunchResult, launch_jobs
14
+
15
+ LEDGER = "launches.jsonl"
16
+ # generous: a run is depth_k launches x sleep_k sleeps x the array width, far below this
17
+ MAX_LEDGER_BYTES = 4_000_000
18
+
19
+
20
+ def _append(run_dir: Path, rows: list[dict[str, Any]]) -> None:
21
+ if not rows:
22
+ return
23
+ run_dir.mkdir(parents=True, exist_ok=True)
24
+ path = run_dir / LEDGER
25
+ # a crash mid-append leaves a torn last line; start on a fresh one so the
26
+ # torn line is the only record lost, never the next one too
27
+ torn = False
28
+ try:
29
+ with path.open("rb") as fh:
30
+ fh.seek(-1, 2)
31
+ torn = fh.read(1) != b"\n"
32
+ except OSError:
33
+ pass
34
+ with path.open("a", encoding="utf-8") as fh:
35
+ if torn:
36
+ fh.write("\n")
37
+ for row in rows:
38
+ fh.write(json.dumps(row, sort_keys=True) + "\n")
39
+
40
+
41
+ def append_submitted(
42
+ run_dir: Path, *, sleep: int, launches: tuple[Launch, ...], job_ids: list[str], at: float
43
+ ) -> None:
44
+ """One record per launch of a sleep, with the job ids it fanned out to. The
45
+ ids are positional over `launch_jobs` order, exactly as the park recorded
46
+ them; a launch whose ids are missing (an older park) gets none. A launch
47
+ already recorded under this (sleep, name) is not written again: a run
48
+ re-parks the same sleep through a multi-stage gate, and the first record
49
+ is the one with the author's words."""
50
+ known = {
51
+ (int(row.get("sleep") or 0), str(row.get("name") or ""))
52
+ for row in read_ledger(run_dir)
53
+ if row.get("event") == "submitted"
54
+ }
55
+ # one id per launch (a sweep is one Slurm job array), or one per task for
56
+ # a park recorded when arrays were separate jobs
57
+ per_launch = len(job_ids) == len(launches)
58
+ rows: list[dict[str, Any]] = []
59
+ k = 0
60
+ for launch in launches:
61
+ n = 1 if per_launch else len(launch_jobs(launch))
62
+ ids = job_ids[k : k + n]
63
+ k += n
64
+ if (sleep, launch.name) in known:
65
+ continue
66
+ rows.append(
67
+ {
68
+ "event": "submitted",
69
+ "sleep": sleep,
70
+ "name": launch.name,
71
+ "why": launch.why,
72
+ "minutes": launch.minutes,
73
+ "array": launch.array,
74
+ "concurrency": launch.concurrency,
75
+ "job_ids": list(ids),
76
+ "at": at,
77
+ }
78
+ )
79
+ _append(run_dir, rows)
80
+
81
+
82
+ def append_ended(
83
+ run_dir: Path,
84
+ *,
85
+ sleep: int,
86
+ results: tuple[LaunchResult, ...],
87
+ at: float,
88
+ elapsed_seconds: list[int | None] | None = None,
89
+ ) -> None:
90
+ """One record per job that came back at the wake, keyed to its sleep, with
91
+ how long it ran when the compute could say (aligned with `results`) and
92
+ the last line it printed — captured now, because a later launch with the
93
+ same name overwrites the job dir. A job already recorded as ended under
94
+ this (sleep, name) is not written again: a submitted park's wake and the
95
+ author's wake may both see the same jobs."""
96
+ known = {
97
+ (int(row.get("sleep") or 0), str(row.get("name") or ""))
98
+ for row in read_ledger(run_dir)
99
+ if row.get("event") == "ended"
100
+ }
101
+ _append(
102
+ run_dir,
103
+ [
104
+ {
105
+ "event": "ended",
106
+ "sleep": sleep,
107
+ "name": r.name,
108
+ "exit_code": r.exit_code,
109
+ "state": r.slurm_state,
110
+ "elapsed": (
111
+ elapsed_seconds[i]
112
+ if elapsed_seconds is not None and i < len(elapsed_seconds)
113
+ else None
114
+ ),
115
+ "last_line": last_line(r.stdout_tail),
116
+ "at": at,
117
+ }
118
+ for i, r in enumerate(results)
119
+ if (sleep, r.name) not in known
120
+ ],
121
+ )
122
+
123
+
124
+ def last_line(text: str, cap: int = 160) -> str:
125
+ """The last non-empty line of a job's output — its result line, as the
126
+ wake shows the author — on one line and bounded. "" when there is none."""
127
+ for line in reversed(text.splitlines()):
128
+ flat = " ".join(line.split())
129
+ if flat:
130
+ return flat[:cap]
131
+ return ""
132
+
133
+
134
+ def read_ledger(run_dir: Path) -> list[dict[str, Any]]:
135
+ """Every record, oldest first; a malformed line is skipped, a missing file
136
+ is an empty history."""
137
+ try:
138
+ with (run_dir / LEDGER).open("rb") as fh:
139
+ raw = fh.read(MAX_LEDGER_BYTES)
140
+ except OSError:
141
+ return []
142
+ rows: list[dict[str, Any]] = []
143
+ for line in raw.decode("utf-8", "replace").splitlines():
144
+ try:
145
+ row = json.loads(line)
146
+ except ValueError:
147
+ continue
148
+ if isinstance(row, dict):
149
+ rows.append(row)
150
+ return rows
151
+
152
+
153
+ def history(run_dir: Path) -> list[dict[str, Any]]:
154
+ """Submitted launches in order, each with the ended records of its jobs.
155
+ The identity is (sleep, name): a name is unique within one sleep only."""
156
+ rows = read_ledger(run_dir)
157
+ entries: dict[tuple[int, str], dict[str, Any]] = {}
158
+ for row in rows:
159
+ if row.get("event") != "submitted":
160
+ continue
161
+ key = (int(row.get("sleep") or 0), str(row.get("name") or ""))
162
+ entries[key] = {
163
+ "sleep": row.get("sleep"),
164
+ "name": row.get("name"),
165
+ "why": row.get("why", ""),
166
+ "minutes": row.get("minutes"),
167
+ "array": row.get("array", 1),
168
+ "concurrency": row.get("concurrency", 0),
169
+ "job_ids": list(row.get("job_ids") or []),
170
+ "submitted_at": row.get("at"),
171
+ "jobs": [],
172
+ }
173
+ for row in rows:
174
+ if row.get("event") != "ended":
175
+ continue
176
+ # an array member is `<name>.<i>`; the dot is outside the name alphabet
177
+ launch_name = str(row.get("name") or "").split(".", 1)[0]
178
+ entry = entries.get((int(row.get("sleep") or 0), launch_name))
179
+ if entry is not None:
180
+ entry["jobs"].append(
181
+ {
182
+ "name": row.get("name"),
183
+ "exit_code": row.get("exit_code"),
184
+ "state": row.get("state", ""),
185
+ "elapsed": row.get("elapsed"),
186
+ "last_line": str(row.get("last_line") or ""),
187
+ "ended_at": row.get("at"),
188
+ }
189
+ )
190
+ return list(entries.values())
191
+
192
+
193
+ def experiments_rows(run_dir: Path) -> list[dict[str, Any]]:
194
+ """The pull request's experiments table: one row per job of every launch
195
+ this run — its sleep, name and why, how it ended and how long it ran, and
196
+ the last line it printed, all from the ledger (the job dir is overwritten
197
+ by a later launch of the same name). A job not back yet says so."""
198
+ rows: list[dict[str, Any]] = []
199
+ for entry in history(run_dir):
200
+ ended = {str(j.get("name")): j for j in entry["jobs"]}
201
+ array = int(entry.get("array") or 1)
202
+ name = str(entry.get("name") or "")
203
+ jobs = [name] if array <= 1 else [f"{name}.{k}" for k in range(array)]
204
+ for job in jobs:
205
+ j = ended.get(job)
206
+ rows.append(
207
+ {
208
+ "sleep": entry.get("sleep"),
209
+ "launch": name,
210
+ "why": str(entry.get("why") or ""),
211
+ "array": array,
212
+ "concurrency": int(entry.get("concurrency") or 0),
213
+ "job": job,
214
+ "back": j is not None,
215
+ "exit_code": j.get("exit_code") if j else None,
216
+ "state": str(j.get("state") or "") if j else "",
217
+ "elapsed": j.get("elapsed") if j else None,
218
+ "result": str(j.get("last_line") or "") if j else "",
219
+ }
220
+ )
221
+ return rows
222
+
223
+
224
+ def why_by_job(run_dir: Path) -> dict[str, dict[str, Any]]:
225
+ """Slurm job id -> the launch it belongs to (name, why, sleep): how the
226
+ queue view labels a launch job for every agent."""
227
+ out: dict[str, dict[str, Any]] = {}
228
+ for row in read_ledger(run_dir):
229
+ if row.get("event") != "submitted":
230
+ continue
231
+ for job_id in row.get("job_ids") or []:
232
+ out[str(job_id)] = {
233
+ "name": row.get("name", ""),
234
+ "why": row.get("why", ""),
235
+ "sleep": row.get("sleep"),
236
+ "array": int(row.get("array") or 1),
237
+ "concurrency": int(row.get("concurrency") or 0),
238
+ }
239
+ return out
outerloop/maintain.py ADDED
@@ -0,0 +1,325 @@
1
+ """The maintenance scan: a read-only agent session over a checkout of a
2
+ repository's default branch that records cleanup, upgrade, test-health and
3
+ performance items as findings, and the digest those findings render into.
4
+
5
+ It reuses the reviewer's machinery — the FINDINGS_SCHEMA verdict through the
6
+ syscall tool, lenses fanned out and merged by the summarizer, the emit/post
7
+ split — and differs in three places: the brief scans a tree instead of a
8
+ diff, nothing is blocking, and the destination is one rolling issue
9
+ (docs/design/reviewer-infra.md, "Maintenance scan"). Any repository can run
10
+ it from the reusable workflow; the brief assumes nothing about this one."""
11
+
12
+ from __future__ import annotations
13
+
14
+ import contextlib
15
+ import logging
16
+ import urllib.parse
17
+ from collections.abc import Iterable
18
+ from datetime import UTC, datetime
19
+ from pathlib import Path
20
+
21
+ from outerloop.harness import Harness, backend_id
22
+ from outerloop.markers import marker
23
+ from outerloop.posting import EXPECTED_FAILURES
24
+ from outerloop.review import DEFAULT_SYSCALL_CMD, Finding, ReviewResult, sanitize
25
+ from outerloop.review_agent import emit_envelope
26
+ from outerloop.role_runner import run_role
27
+ from outerloop.rolespec import RoleSpec
28
+
29
+ log = logging.getLogger(__name__)
30
+
31
+ MARKER = marker("maintenance-digest")
32
+ DIGEST_TITLE = "Maintainer digest"
33
+ ADVISORY = (
34
+ "*Advisory findings from `outerloop`. The maintainer decides: items marked "
35
+ "**Decision** need a call before any change; the rest are mechanical and may "
36
+ "be taken as work orders. The scan edits nothing.*"
37
+ )
38
+
39
+ # Each lens is one section of the digest; `general` is the whole checklist.
40
+ # The library lives here; which lenses run is the caller workflow's matrix.
41
+ MAINTENANCE_LENSES: dict[str, str] = {
42
+ "pathways": (
43
+ "LENS — dead and unused pathways: symbols, CLI flags, config keys and "
44
+ "environment knobs with no caller or no documentation; compatibility "
45
+ "shims and what each still guards; test-only code living in the "
46
+ "package. Grep across source, tests, scripts and docs before calling "
47
+ "anything unused."
48
+ ),
49
+ "duplication": (
50
+ "LENS — logic with more than one owner: the same rule implemented in "
51
+ "two places (two parsers of one file format, two copies of one "
52
+ "sequence), private helpers imported across modules, argument groups "
53
+ "or fixtures copied between entry points or test files, version pins "
54
+ "repeated in several files."
55
+ ),
56
+ "structure": (
57
+ "LENS — size and shape: the largest modules and longest functions "
58
+ "(measure them), import cycles and the in-function imports that hide "
59
+ "them, templates or data embedded in code, and the natural seams a "
60
+ "split would follow."
61
+ ),
62
+ "upgrades": (
63
+ "LENS — dependencies and tooling: pinned versions against the latest "
64
+ "available (the package index, GitHub releases), CI action versions, "
65
+ "runner images, linter and type-checker settings that could be "
66
+ "tightened cheaply, interpreter versions exercised. Name the pin and "
67
+ "the current upstream for each."
68
+ ),
69
+ "tests": (
70
+ "LENS — test-suite health: the slowest tests and why, real sleeps and "
71
+ "real subprocesses where a fake would do, fixtures and fakes defined "
72
+ "several times, tests that no longer pin the behavior they name, "
73
+ "markers declared but unused."
74
+ ),
75
+ "performance": (
76
+ "LENS — repeated work on the hot path: find the loop or entry point "
77
+ "the repository runs most often and count what it re-reads, re-lists "
78
+ "or re-fetches per iteration and per record; caching that is missing, "
79
+ "network calls without conditional requests, files parsed more than "
80
+ "once."
81
+ ),
82
+ "docs": (
83
+ "LENS — documentation drift: comments that narrate history instead of "
84
+ "intent, changelog sections to consolidate, roadmap or design notes "
85
+ "whose status no longer matches the code, knobs and flags the docs "
86
+ "never name, wording that disagrees between two documents."
87
+ ),
88
+ }
89
+
90
+ SYSTEM_PROMPT = (
91
+ "You are the maintainer's periodic scan of this repository. You read the whole "
92
+ "tree, measure rather than guess, and record each item worth doing as a finding. "
93
+ "Nothing you find blocks anything: the maintainer reads the digest and decides.\n\n"
94
+ "What to record: cleanup, simplification, upgrade, test-health and performance "
95
+ "items — what a careful maintainer would put on their own list after a week away. "
96
+ "Skip style nits a linter already reports and work the repository's own roadmap "
97
+ "already tracks as planned.\n\n"
98
+ "Evidence: every item names a file and a line, and a measured fact where one "
99
+ "exists (a line count, a call count, the pinned and the latest version, a test "
100
+ "duration). Say what you ran.\n\n"
101
+ "Shape of each finding:\n"
102
+ "- --kind change: mechanical, safe for an agent to do in a pull request without a "
103
+ "design call.\n"
104
+ "- --kind question: needs the maintainer's decision first (when to drop a "
105
+ "compatibility path, whether to re-verify a pinned tool, which module owns a "
106
+ "duplicated rule).\n"
107
+ "- --kind note: worth knowing, not worth a change.\n"
108
+ "- --category: the digest section the item belongs to, one of {sections}.\n"
109
+ '- --detail: start with effort and risk, for example "S, low." (S is under an '
110
+ "hour, M an afternoon, L a day or more; risk is what could break), then the "
111
+ "evidence.\n"
112
+ "- never --blocking.\n\n"
113
+ "Your concluding notes open the digest: one line on what is healthy, then the "
114
+ "three items most worth doing, one sentence each."
115
+ )
116
+
117
+
118
+ def _investigation(ref: str, syscall_cmd: str) -> str:
119
+ return (
120
+ f"The repository is checked out in your working directory at commit {ref}. "
121
+ "Use Read, Grep and Glob, and run read-only commands in the shell: line "
122
+ "counts, the test suite with durations, the package manager's outdated "
123
+ "list, the linter's statistics. Do not modify the tree, do not install "
124
+ "anything beyond what its own lockfile describes, and do not push or "
125
+ "post anything — your only product is the verdict.\n\n"
126
+ "Record each item as you confirm it, one command per item:\n"
127
+ f" {syscall_cmd} finding --file <path> [--line N] "
128
+ "--confidence <low|medium|high> --category <section> --summary <one line> "
129
+ "--detail <effort, risk, then the evidence> --kind <change|question|note>\n"
130
+ "When you are done, commit your verdict and end your turn:\n"
131
+ f" {syscall_cmd} conclude --notes <what is healthy; the three items most worth doing>\n"
132
+ "The verdict you commit is your final answer — do not also restate it in a message."
133
+ )
134
+
135
+
136
+ def build_maintenance_brief(
137
+ repo: str,
138
+ ref: str,
139
+ today: str | None = None,
140
+ *,
141
+ syscall_cmd: str = DEFAULT_SYSCALL_CMD,
142
+ lens: str = "",
143
+ ) -> str:
144
+ """The scan brief: the standing prompt, the lens (or every section for
145
+ `general`), the investigation instruction and the repository line. An
146
+ unknown lens fails loudly, as the reviewer's does."""
147
+ if lens and lens != "general" and lens not in MAINTENANCE_LENSES:
148
+ raise ValueError(f"unknown maintenance lens {lens!r} (have: {sorted(MAINTENANCE_LENSES)})")
149
+ sections = ", ".join(MAINTENANCE_LENSES)
150
+ if lens and lens != "general":
151
+ focus = MAINTENANCE_LENSES[lens]
152
+ else:
153
+ focus = "Cover every section:\n\n" + "\n\n".join(MAINTENANCE_LENSES.values())
154
+ header = f"Today's date: {today}\n" if today else ""
155
+ header += f"Repository: {repo} at {ref}"
156
+ return (
157
+ f"{SYSTEM_PROMPT.format(sections=sections)}\n\n{focus}\n\n"
158
+ f"{_investigation(ref, syscall_cmd)}\n\n{header}\n"
159
+ )
160
+
161
+
162
+ _KIND_ORDER = {"question": 0, "change": 1, "suggestion": 1, "note": 2}
163
+ _CONFIDENCE_ORDER = {"high": 0, "medium": 1, "low": 2}
164
+ _LABEL = {"question": "**Decision.** ", "note": "*Note.* "}
165
+
166
+
167
+ def _item(finding: Finding, repo: str, ref: str) -> str:
168
+ # backticks stripped: a file value containing one would close the code
169
+ # span and render model markdown inline (same rule as the review body)
170
+ safe_file = finding.file.replace("`", "")
171
+ where = f"`{safe_file}`" + (f":{finding.line}" if finding.line else "")
172
+ link = (
173
+ f"https://github.com/{repo}/blob/{urllib.parse.quote(ref)}/{urllib.parse.quote(safe_file)}"
174
+ )
175
+ if finding.line:
176
+ link += f"#L{finding.line}"
177
+ summary = finding.summary.rstrip(".!?…")
178
+ if summary.count("`") % 2:
179
+ summary += "`"
180
+ detail = finding.detail + ("`" if finding.detail.count("`") % 2 else "")
181
+ label = _LABEL.get(finding.kind, "")
182
+ return f"- {label}**{summary}.** {detail} ([{where}]({link}); {finding.confidence})"
183
+
184
+
185
+ def render_digest(
186
+ result: ReviewResult,
187
+ *,
188
+ repo: str,
189
+ ref: str,
190
+ today: str,
191
+ reviewed_by: str,
192
+ ) -> str:
193
+ """The rolling issue's body: marker first, the header and the advisory
194
+ line, the counts, the scan's own summary, then one section per category
195
+ with decisions first. Every string in `result` is already sanitized by
196
+ `result_from_data`; the marker leads so the poster can find the issue."""
197
+ findings = result.findings
198
+ decisions = sum(1 for f in findings if f.kind == "question")
199
+ notes = sum(1 for f in findings if f.kind == "note")
200
+ mechanical = len(findings) - decisions - notes
201
+ who = sanitize(reviewed_by, 120) or "unattributed"
202
+ lines = [
203
+ MARKER,
204
+ f"**{DIGEST_TITLE}** — {repo} at `{ref[:8]}` on {today}; scanned by `{who}`.",
205
+ "",
206
+ ADVISORY,
207
+ "",
208
+ f"{len(findings)} items: {decisions} need a decision, {mechanical} are "
209
+ f"mechanical, {notes} are notes.",
210
+ "",
211
+ ]
212
+ if result.notes:
213
+ lines += [result.notes, ""]
214
+ by_section: dict[str, list[Finding]] = {}
215
+ for f in findings:
216
+ section = f.category if f.category in MAINTENANCE_LENSES else "other"
217
+ by_section.setdefault(section, []).append(f)
218
+ for section in [*MAINTENANCE_LENSES, "other"]:
219
+ items = by_section.get(section)
220
+ if not items:
221
+ continue
222
+ items.sort(key=lambda f: (_KIND_ORDER.get(f.kind, 2), _CONFIDENCE_ORDER[f.confidence]))
223
+ lines += [f"### {section}", ""]
224
+ lines += [_item(f, repo, ref) for f in items]
225
+ lines.append("")
226
+ lines.append(
227
+ "_Each scan replaces this body; earlier digests are in the edit history. "
228
+ "Run a scan by hand from the Actions tab (maintenance → Run workflow)._"
229
+ )
230
+ return "\n".join(lines).rstrip() + "\n"
231
+
232
+
233
+ def render_stub(detail: str, *, repo: str, ref: str, today: str, who: str) -> str:
234
+ """What the poster writes when the scan could not run: the reason, on the
235
+ digest issue, never silence."""
236
+ reason = sanitize(detail, 300)
237
+ by = f" ({sanitize(who, 120)})" if who else ""
238
+ return (
239
+ f"{MARKER}\n**{DIGEST_TITLE}** — the scan of {repo} at `{ref[:8]}` on {today} "
240
+ f"could not run{by}: {reason}"
241
+ )
242
+
243
+
244
+ def run_maintenance_scan(
245
+ repo: str,
246
+ ref: str,
247
+ harness: Harness,
248
+ workspace: Path,
249
+ *,
250
+ spec: RoleSpec | None = None,
251
+ emit_path: Path,
252
+ today: str | None = None,
253
+ lens: str = "",
254
+ ) -> str | None:
255
+ """One lens session over `workspace` (a default-branch checkout the caller
256
+ prepared and sanitized). EVERY outcome writes an envelope for the posting
257
+ job — findings, or a skip-stub naming why — so a missing artifact always
258
+ means a broken session. Returns "emitted", or None when it could not
259
+ produce a verdict. Advisory: never raises the expected failures."""
260
+ from outerloop.roles import maintainer_spec
261
+
262
+ spec = spec or maintainer_spec()
263
+ today = today or datetime.now(UTC).date().isoformat()
264
+ try:
265
+ from outerloop.syscall import tool_command
266
+
267
+ brief = build_maintenance_brief(
268
+ repo, ref, today, syscall_cmd=tool_command(workspace), lens=lens
269
+ )
270
+ role_result = run_role(spec, harness, brief, workspace)
271
+ if not role_result.ok or role_result.data is None:
272
+ detail = role_result.error or role_result.session.stop_reason
273
+ log.warning("maintenance scan produced no verdict on %s (%s): %s", repo, lens, detail)
274
+ emit_envelope(
275
+ emit_path,
276
+ repo,
277
+ 0,
278
+ kind="skip-stub",
279
+ detail=detail,
280
+ reviewed_by=backend_id(harness),
281
+ lens=lens,
282
+ )
283
+ return None
284
+ emit_envelope(
285
+ emit_path,
286
+ repo,
287
+ 0,
288
+ kind="findings",
289
+ data=role_result.data,
290
+ reviewed_by=backend_id(harness),
291
+ lens=lens,
292
+ )
293
+ cost = role_result.session.cost_usd
294
+ log.info(
295
+ "emitted maintenance findings for %s (%s; cost=%s turns=%d)",
296
+ repo,
297
+ lens or "general",
298
+ f"${cost:.2f}" if cost else "unreported",
299
+ role_result.session.num_turns,
300
+ )
301
+ return "emitted"
302
+ except EXPECTED_FAILURES as exc: # advisory: never red the repository's Actions
303
+ log.warning("maintenance scan did not complete: %s: %s", type(exc).__name__, exc)
304
+ with contextlib.suppress(Exception):
305
+ emit_envelope(
306
+ emit_path,
307
+ repo,
308
+ 0,
309
+ kind="skip-stub",
310
+ detail=f"{type(exc).__name__}: {exc}",
311
+ reviewed_by=backend_id(harness),
312
+ lens=lens,
313
+ )
314
+ return None
315
+
316
+
317
+ def lens_names(lenses: Iterable[str]) -> list[str]:
318
+ """The lens names a caller configured, `general` included, unknown ones
319
+ refused — so a misspelled matrix entry fails at configuration time."""
320
+ out = []
321
+ for name in lenses:
322
+ if name != "general" and name not in MAINTENANCE_LENSES:
323
+ raise ValueError(f"unknown maintenance lens {name!r}")
324
+ out.append(name)
325
+ return out